diff --git a/graphify/__main__.py b/graphify/__main__.py index 7db7652..d521e0e 100644 --- a/graphify/__main__.py +++ b/graphify/__main__.py @@ -1075,6 +1075,7 @@ def main() -> None: print(" (also: GRAPHIFY_FORCE=1 env var; use after refactors that delete code)") print(" cluster-only rerun clustering on an existing graph.json and regenerate report") print(" --no-viz skip graph.html generation (useful for >5000 node graphs / CI)") + print(" --graph path to graph.json (default /graphify-out/graph.json)") print(" query \"\" BFS traversal of graph.json for a question") print(" --dfs use depth-first instead of breadth-first") print(" --context C explicit edge-context filter (repeatable)") @@ -1494,11 +1495,30 @@ def main() -> None: sys.exit(1) elif cmd == "cluster-only": - watch_path = Path(sys.argv[2]) if len(sys.argv) > 2 else Path(".") + # Mirror the tree/export arg-parsing pattern: walk argv so flags and + # the optional positional path can appear in any order (#724). no_viz = "--no-viz" in sys.argv _min_cs_arg = next((a for a in sys.argv if a.startswith("--min-community-size=")), None) min_community_size = int(_min_cs_arg.split("=")[1]) if _min_cs_arg else 3 - graph_json = watch_path / "graphify-out" / "graph.json" + args = sys.argv[2:] + watch_path: Path | None = None + graph_override: Path | None = None + i_arg = 0 + while i_arg < len(args): + a = args[i_arg] + if a == "--graph" and i_arg + 1 < len(args): + graph_override = Path(args[i_arg + 1]); i_arg += 2 + elif a == "--no-viz" or a.startswith("--min-community-size="): + i_arg += 1 + elif a.startswith("--"): + i_arg += 1 + elif watch_path is None: + watch_path = Path(a); i_arg += 1 + else: + i_arg += 1 + if watch_path is None: + watch_path = Path(".") + graph_json = graph_override if graph_override is not None else watch_path / "graphify-out" / "graph.json" if not graph_json.exists(): print(f"error: no graph found at {graph_json} — run /graphify first", file=sys.stderr) sys.exit(1) @@ -2122,6 +2142,10 @@ def main() -> None: f"{merged['output_tokens']:,} out, " f"est. cost: ${cost:.4f}" ) + try: + _save_manifest(files_by_type, manifest_path=str(manifest_path)) + except Exception as exc: + print(f"[graphify extract] warning: could not write manifest: {exc}", file=sys.stderr) sys.exit(0) # Build graph + cluster + score + write. diff --git a/graphify/build.py b/graphify/build.py index 76a8357..2b45520 100644 --- a/graphify/build.py +++ b/graphify/build.py @@ -245,8 +245,19 @@ def build_merge( if d.get("source_file") in prune_sources ] G.remove_nodes_from(to_remove) - if to_remove: - print(f"[graphify] Pruned {len(to_remove)} node(s) from deleted sources.", file=sys.stderr) + n_files = len(prune_sources) + n_nodes = len(to_remove) + if n_nodes: + print( + f"[graphify] Pruned {n_nodes} node(s) from {n_files} deleted source file(s).", + file=sys.stderr, + ) + else: + print( + f"[graphify] {n_files} source file(s) deleted since last run — " + f"no matching nodes in graph, already clean.", + file=sys.stderr, + ) # Safety check: refuse to shrink the graph silently (#479) # Skip when dedup or prune_sources is active — shrinkage is intentional there. diff --git a/graphify/detect.py b/graphify/detect.py index 8708602..50f9e2c 100644 --- a/graphify/detect.py +++ b/graphify/detect.py @@ -33,7 +33,7 @@ FILE_COUNT_UPPER = 200 # files - above this, warn about token cost _SENSITIVE_PATTERNS = [ re.compile(r'(^|[\\/])\.(env|envrc)(\.|$)', re.IGNORECASE), re.compile(r'\.(pem|key|p12|pfx|cert|crt|der|p8)$', re.IGNORECASE), - re.compile(r'(credential|secret|passwd|password|token|private_key)', re.IGNORECASE), + re.compile(r'\b(credential|secret|passwd|password|token|private_key)s?\b', re.IGNORECASE), re.compile(r'(id_rsa|id_dsa|id_ecdsa|id_ed25519)(\.pub)?$'), re.compile(r'(\.netrc|\.pgpass|\.htpasswd)$', re.IGNORECASE), re.compile(r'(aws_credentials|gcloud_credentials|service.account)', re.IGNORECASE), diff --git a/graphify/extract.py b/graphify/extract.py index 42fd78f..6f5a0e8 100644 --- a/graphify/extract.py +++ b/graphify/extract.py @@ -1767,6 +1767,7 @@ def extract_svelte(path: Path) -> dict: elif resolved.suffix == ".jsx": resolved = resolved.with_suffix(".tsx") node_id = _make_id(str(resolved)) + stub_source_file = str(resolved) else: # Check tsconfig.json path aliases (e.g. "$lib/" -> "src/lib/", "@/" -> "src/") # before treating as external. Mirrors _import_js logic so SvelteKit alias @@ -1779,6 +1780,7 @@ def extract_svelte(path: Path) -> dict: break if resolved_alias is not None: node_id = _make_id(str(resolved_alias)) + stub_source_file = str(resolved_alias) else: # Bare/scoped import (node_modules) - use last segment; # build_from_json drops as external if no matching node exists. @@ -1786,6 +1788,7 @@ def extract_svelte(path: Path) -> dict: if not module_name: continue node_id = _make_id(module_name) + stub_source_file = raw if node_id in existing_ids: # Edge target already a real node - just add the edge, don't add a node. result.setdefault("edges", []).append({ @@ -1796,7 +1799,7 @@ def extract_svelte(path: Path) -> dict: continue result.setdefault("nodes", []).append({ "id": node_id, "label": raw, - "file_type": "code", "source_file": str(path), + "file_type": "code", "source_file": stub_source_file, "confidence": "EXTRACTED", }) result.setdefault("edges", []).append({ @@ -1805,6 +1808,65 @@ def extract_svelte(path: Path) -> dict: "source_file": str(path), }) existing_ids.add(node_id) + # Static imports inside