From 9e6192a6c225855084108aa5d42717d978a9e0d9 Mon Sep 17 00:00:00 2001 From: Safi Date: Wed, 20 May 2026 18:04:28 +0100 Subject: [PATCH] fix stale wiki nodes (#936), gitignore fallback and --exclude flag (#945/#947), NAT64 SSRF false-positive Co-Authored-By: Claude Sonnet 4.6 --- graphify/__main__.py | 8 +++++- graphify/detect.py | 17 ++++++++++-- graphify/security.py | 8 ++++++ graphify/wiki.py | 23 ++++++++++++++++ tests/test_detect.py | 64 ++++++++++++++++++++++++++++++++++++++++++++ tests/test_wiki.py | 32 ++++++++++++++++++++++ 6 files changed, 149 insertions(+), 3 deletions(-) diff --git a/graphify/__main__.py b/graphify/__main__.py index 895f626..3765a81 100644 --- a/graphify/__main__.py +++ b/graphify/__main__.py @@ -2431,6 +2431,7 @@ def main() -> None: # Clustering tuning knobs cli_resolution: float = 1.0 cli_exclude_hubs: float | None = None + cli_excludes: list[str] = [] def _parse_int(name: str, raw: str) -> int: try: @@ -2504,6 +2505,10 @@ def main() -> None: cli_exclude_hubs = float(args[i + 1]); i += 2 elif a.startswith("--exclude-hubs="): cli_exclude_hubs = float(a.split("=", 1)[1]); i += 1 + elif a == "--exclude" and i + 1 < len(args): + cli_excludes.append(args[i + 1]); i += 2 + elif a.startswith("--exclude="): + cli_excludes.append(a.split("=", 1)[1]); i += 1 else: i += 1 @@ -2610,10 +2615,11 @@ def main() -> None: target, manifest_path=str(manifest_path), google_workspace=google_workspace or None, + extra_excludes=cli_excludes or None, ) else: print(f"[graphify extract] scanning {target}") - detection = _detect(target, google_workspace=google_workspace or None) + detection = _detect(target, google_workspace=google_workspace or None, extra_excludes=cli_excludes or None) files_by_type = detection.get("files", {}) if incremental_mode: diff --git a/graphify/detect.py b/graphify/detect.py index c1482d0..16951ba 100644 --- a/graphify/detect.py +++ b/graphify/detect.py @@ -400,6 +400,7 @@ _SKIP_DIRS = { ".next", ".nuxt", ".turbo", ".angular", ".idea", ".cache", ".parcel-cache", ".svelte-kit", ".terraform", ".serverless", ".graphify", # graphify's own extraction cache — never index self-generated data + ".worktrees", # git worktree convention (#947) — sibling checkouts, always redundant } # Large generated files that are never useful to extract @@ -486,7 +487,11 @@ def _load_graphifyignore(root: Path) -> list[tuple[Path, str]]: patterns: list[tuple[Path, str]] = [] for d in dirs: + # Prefer .graphifyignore; fall back to .gitignore so projects that already + # maintain a .gitignore get sensible defaults without duplicating it (#945). ignore_file = d / ".graphifyignore" + if not ignore_file.exists(): + ignore_file = d / ".gitignore" if ignore_file.exists(): for raw in ignore_file.read_text(encoding="utf-8", errors="ignore").splitlines(): line = _parse_gitignore_line(raw) @@ -701,7 +706,7 @@ def _auto_follow_symlinks(root: Path) -> bool: return False -def detect(root: Path, *, follow_symlinks: bool | None = None, google_workspace: bool | None = None) -> dict: +def detect(root: Path, *, follow_symlinks: bool | None = None, google_workspace: bool | None = None, extra_excludes: list[str] | None = None) -> dict: root = root.resolve() if follow_symlinks is None: follow_symlinks = _auto_follow_symlinks(root) @@ -717,6 +722,13 @@ def detect(root: Path, *, follow_symlinks: bool | None = None, google_workspace: skipped_sensitive: list[str] = [] ignore_patterns = _load_graphifyignore(root) + # CLI --exclude patterns are anchored at the scan root and appended last + # so they win over any .graphifyignore/.gitignore rules (#947). + if extra_excludes: + for pat in extra_excludes: + line = _parse_gitignore_line(pat) + if line: + ignore_patterns.append((root, line)) include_patterns = _load_graphifyinclude(root) # Always include graphify-out/memory/ - query results filed back into the graph @@ -933,6 +945,7 @@ def detect_incremental( follow_symlinks: bool | None = None, google_workspace: bool | None = None, kind: str = "semantic", + extra_excludes: list[str] | None = None, ) -> dict: """Like detect(), but returns only new or modified files since the last run. @@ -957,7 +970,7 @@ def detect_incremental( incremental runs. ``None`` (default) means auto-detect: ``True`` when ``root`` contains at least one direct symlinked child, ``False`` otherwise. """ - full = detect(root, follow_symlinks=follow_symlinks, google_workspace=google_workspace) + full = detect(root, follow_symlinks=follow_symlinks, google_workspace=google_workspace, extra_excludes=extra_excludes) manifest = load_manifest(manifest_path) if not manifest: diff --git a/graphify/security.py b/graphify/security.py index cf1904d..a594af3 100644 --- a/graphify/security.py +++ b/graphify/security.py @@ -22,6 +22,10 @@ _BLOCKED_HOSTS = {"metadata.google.internal", "metadata.google.com"} # RFC 6598 Shared Address Space (CGN) -- is_private misses this on Python <3.11 _CGN_NETWORK = ipaddress.ip_network("100.64.0.0/10") +# RFC 6052 NAT64 Well-Known Prefix -- is_reserved=True in Python but these embed +# public IPv4 addresses and are legitimate public internet traffic, not SSRF vectors. +_NAT64_WKP = ipaddress.ip_network("64:ff9b::/96") + # --------------------------------------------------------------------------- # URL validation @@ -57,6 +61,10 @@ def validate_url(url: str) -> str: for info in infos: addr = info[4][0] ip = ipaddress.ip_address(addr) + # For NAT64 addresses, check the embedded IPv4 instead of the wrapper + if isinstance(ip, ipaddress.IPv6Address) and ip in _NAT64_WKP: + embedded = ipaddress.ip_address(int(ip) & 0xFFFFFFFF) + ip = embedded if ip.is_private or ip.is_reserved or ip.is_loopback or ip.is_link_local or ip in _CGN_NETWORK: raise ValueError( f"Blocked private/internal IP {addr} (resolved from '{hostname}'). " diff --git a/graphify/wiki.py b/graphify/wiki.py index 53ed625..b9a6b83 100644 --- a/graphify/wiki.py +++ b/graphify/wiki.py @@ -204,6 +204,29 @@ def to_wiki( "Run `graphify extract .` or `graphify cluster-only .` first." ) + # Filter stale node IDs that exist in communities but not in G. + # Analysis JSON can drift from the graph after dedup / re-extract / update. + # NetworkX 3.x returns DegreeView({}) for missing nodes instead of raising, + # which crashes sorted() with TypeError; G.neighbors()/G.nodes[] also raise. + import sys as _sys + _g_nodes = set(G.nodes) + _orig_total = sum(len(ns) for ns in communities.values()) + communities = {cid: [n for n in nodes if n in _g_nodes] for cid, nodes in communities.items()} + communities = {cid: nodes for cid, nodes in communities.items() if nodes} + _kept_total = sum(len(ns) for ns in communities.values()) + if _kept_total < _orig_total: + print( + f"wiki: dropped {_orig_total - _kept_total} stale node ID(s) not in graph " + f"({len(communities)} communities remaining)", + file=_sys.stderr, + ) + + if not communities: + raise ValueError( + "all community node IDs are stale — none exist in the graph. " + "Re-run `graphify extract .` to regenerate .graphify_analysis.json." + ) + # Clear stale .md files from previous runs to prevent orphan accumulation. # Community labels are LLM-generated (per skill.md Step 5) and non-deterministic # across runs — the same conceptual community may be named differently each time diff --git a/tests/test_detect.py b/tests/test_detect.py index 0337a77..7bf8546 100644 --- a/tests/test_detect.py +++ b/tests/test_detect.py @@ -608,3 +608,67 @@ def test_save_manifest_without_filter_unchanged_for_code(tmp_path): manifest = json.loads(Path(manifest_path).read_text()) assert str(py) in manifest assert manifest[str(py)]["ast_hash"] != "" + + +# Regression tests for #945 - .gitignore fallback when no .graphifyignore exists + +def test_gitignore_fallback_when_no_graphifyignore(tmp_path): + """When no .graphifyignore exists, .gitignore patterns are honored (#945).""" + (tmp_path / ".git").mkdir() + (tmp_path / ".gitignore").write_text("vendor/\n*.generated.py\n") + vendor = tmp_path / "vendor" + vendor.mkdir() + (vendor / "lib.py").write_text("x = 1") + (tmp_path / "main.py").write_text("print('hi')") + (tmp_path / "schema.generated.py").write_text("x = 1") + + result = detect(tmp_path) + code = result["files"]["code"] + assert any("main.py" in f for f in code) + assert not any("vendor" in f for f in code) + assert not any("generated" in f for f in code) + + +def test_graphifyignore_takes_precedence_over_gitignore(tmp_path): + """When both exist, .graphifyignore is used and .gitignore is ignored (#945).""" + (tmp_path / ".git").mkdir() + # .gitignore would exclude main.py; .graphifyignore excludes only other.py + (tmp_path / ".gitignore").write_text("main.py\n") + (tmp_path / ".graphifyignore").write_text("other.py\n") + (tmp_path / "main.py").write_text("x = 1") + (tmp_path / "other.py").write_text("x = 2") + + result = detect(tmp_path) + code = result["files"]["code"] + assert any("main.py" in f for f in code) # gitignore NOT applied + assert not any("other.py" in f for f in code) # graphifyignore IS applied + + +# Regression tests for #947 - .worktrees/ skipped and --exclude flag + +def test_detect_skips_worktrees_dir(tmp_path): + """Files inside .worktrees/ are never indexed (#947).""" + wt = tmp_path / ".worktrees" / "feature-branch" + wt.mkdir(parents=True) + (wt / "main.py").write_text("x = 1") + (tmp_path / "app.py").write_text("y = 2") + + result = detect(tmp_path) + code = result["files"]["code"] + assert any("app.py" in f for f in code) + assert not any(".worktrees" in f for f in code) + + +def test_detect_extra_excludes_pattern(tmp_path): + """extra_excludes patterns exclude matching files from detect() (#947).""" + (tmp_path / "main.py").write_text("x = 1") + (tmp_path / "secret.py").write_text("API_KEY = 'abc'") + subdir = tmp_path / "legacy" + subdir.mkdir() + (subdir / "old.py").write_text("y = 2") + + result = detect(tmp_path, extra_excludes=["secret.py", "legacy/"]) + code = result["files"]["code"] + assert any("main.py" in f for f in code) + assert not any("secret.py" in f for f in code) + assert not any("legacy" in f for f in code) diff --git a/tests/test_wiki.py b/tests/test_wiki.py index 2eb5bc8..8826f94 100644 --- a/tests/test_wiki.py +++ b/tests/test_wiki.py @@ -165,3 +165,35 @@ def test_god_node_article_community_without_node_attr(tmp_path): to_wiki(G, communities, tmp_path, community_labels=labels, god_nodes_data=god_nodes) article = (tmp_path / "parse.md").read_text() assert "[[Core Logic]]" in article + + +# Regression tests for #936 - stale community node IDs crash to_wiki after dedup/re-extract + +def test_to_wiki_drops_stale_community_nodes(tmp_path): + """Stale node IDs in communities dict are silently dropped without crash (#936).""" + G = _make_graph() + # Add a stale ID that exists in communities but not in G + communities = {0: ["n1", "n2", "stale_ghost"], 1: ["n3", "n4"]} + n = to_wiki(G, communities, tmp_path, community_labels=LABELS) + assert n == 2 # both community articles still written + article = (tmp_path / "Parsing_Layer.md").read_text() + assert "parse" in article + assert "stale_ghost" not in article + + +def test_to_wiki_all_stale_raises(tmp_path): + """If every community node is stale, raise ValueError with a helpful message (#936).""" + G = _make_graph() + all_stale = {0: ["ghost1", "ghost2"], 1: ["ghost3"]} + with pytest.raises(ValueError, match="stale"): + to_wiki(G, all_stale, tmp_path, community_labels=LABELS) + + +def test_to_wiki_stale_nodes_prints_warning(tmp_path, capsys): + """Stale node IDs trigger a stderr warning showing the drop count (#936).""" + G = _make_graph() + communities = {0: ["n1", "stale1", "stale2"], 1: ["n3", "n4"]} + to_wiki(G, communities, tmp_path, community_labels=LABELS) + err = capsys.readouterr().err + assert "2" in err # dropped count + assert "stale" in err.lower()