fix stale wiki nodes (#936), gitignore fallback and --exclude flag (#945/#947), NAT64 SSRF false-positive

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
Safi
2026-05-20 18:04:28 +01:00
co-authored by Claude Sonnet 4.6
parent 6939494b3e
commit 9e6192a6c2
6 changed files with 149 additions and 3 deletions
+7 -1
View File
@@ -2431,6 +2431,7 @@ def main() -> None:
# Clustering tuning knobs
cli_resolution: float = 1.0
cli_exclude_hubs: float | None = None
cli_excludes: list[str] = []
def _parse_int(name: str, raw: str) -> int:
try:
@@ -2504,6 +2505,10 @@ def main() -> None:
cli_exclude_hubs = float(args[i + 1]); i += 2
elif a.startswith("--exclude-hubs="):
cli_exclude_hubs = float(a.split("=", 1)[1]); i += 1
elif a == "--exclude" and i + 1 < len(args):
cli_excludes.append(args[i + 1]); i += 2
elif a.startswith("--exclude="):
cli_excludes.append(a.split("=", 1)[1]); i += 1
else:
i += 1
@@ -2610,10 +2615,11 @@ def main() -> None:
target,
manifest_path=str(manifest_path),
google_workspace=google_workspace or None,
extra_excludes=cli_excludes or None,
)
else:
print(f"[graphify extract] scanning {target}")
detection = _detect(target, google_workspace=google_workspace or None)
detection = _detect(target, google_workspace=google_workspace or None, extra_excludes=cli_excludes or None)
files_by_type = detection.get("files", {})
if incremental_mode:
+15 -2
View File
@@ -400,6 +400,7 @@ _SKIP_DIRS = {
".next", ".nuxt", ".turbo", ".angular",
".idea", ".cache", ".parcel-cache", ".svelte-kit", ".terraform", ".serverless",
".graphify", # graphify's own extraction cache — never index self-generated data
".worktrees", # git worktree convention (#947) — sibling checkouts, always redundant
}
# Large generated files that are never useful to extract
@@ -486,7 +487,11 @@ def _load_graphifyignore(root: Path) -> list[tuple[Path, str]]:
patterns: list[tuple[Path, str]] = []
for d in dirs:
# Prefer .graphifyignore; fall back to .gitignore so projects that already
# maintain a .gitignore get sensible defaults without duplicating it (#945).
ignore_file = d / ".graphifyignore"
if not ignore_file.exists():
ignore_file = d / ".gitignore"
if ignore_file.exists():
for raw in ignore_file.read_text(encoding="utf-8", errors="ignore").splitlines():
line = _parse_gitignore_line(raw)
@@ -701,7 +706,7 @@ def _auto_follow_symlinks(root: Path) -> bool:
return False
def detect(root: Path, *, follow_symlinks: bool | None = None, google_workspace: bool | None = None) -> dict:
def detect(root: Path, *, follow_symlinks: bool | None = None, google_workspace: bool | None = None, extra_excludes: list[str] | None = None) -> dict:
root = root.resolve()
if follow_symlinks is None:
follow_symlinks = _auto_follow_symlinks(root)
@@ -717,6 +722,13 @@ def detect(root: Path, *, follow_symlinks: bool | None = None, google_workspace:
skipped_sensitive: list[str] = []
ignore_patterns = _load_graphifyignore(root)
# CLI --exclude patterns are anchored at the scan root and appended last
# so they win over any .graphifyignore/.gitignore rules (#947).
if extra_excludes:
for pat in extra_excludes:
line = _parse_gitignore_line(pat)
if line:
ignore_patterns.append((root, line))
include_patterns = _load_graphifyinclude(root)
# Always include graphify-out/memory/ - query results filed back into the graph
@@ -933,6 +945,7 @@ def detect_incremental(
follow_symlinks: bool | None = None,
google_workspace: bool | None = None,
kind: str = "semantic",
extra_excludes: list[str] | None = None,
) -> dict:
"""Like detect(), but returns only new or modified files since the last run.
@@ -957,7 +970,7 @@ def detect_incremental(
incremental runs. ``None`` (default) means auto-detect: ``True`` when ``root``
contains at least one direct symlinked child, ``False`` otherwise.
"""
full = detect(root, follow_symlinks=follow_symlinks, google_workspace=google_workspace)
full = detect(root, follow_symlinks=follow_symlinks, google_workspace=google_workspace, extra_excludes=extra_excludes)
manifest = load_manifest(manifest_path)
if not manifest:
+8
View File
@@ -22,6 +22,10 @@ _BLOCKED_HOSTS = {"metadata.google.internal", "metadata.google.com"}
# RFC 6598 Shared Address Space (CGN) -- is_private misses this on Python <3.11
_CGN_NETWORK = ipaddress.ip_network("100.64.0.0/10")
# RFC 6052 NAT64 Well-Known Prefix -- is_reserved=True in Python but these embed
# public IPv4 addresses and are legitimate public internet traffic, not SSRF vectors.
_NAT64_WKP = ipaddress.ip_network("64:ff9b::/96")
# ---------------------------------------------------------------------------
# URL validation
@@ -57,6 +61,10 @@ def validate_url(url: str) -> str:
for info in infos:
addr = info[4][0]
ip = ipaddress.ip_address(addr)
# For NAT64 addresses, check the embedded IPv4 instead of the wrapper
if isinstance(ip, ipaddress.IPv6Address) and ip in _NAT64_WKP:
embedded = ipaddress.ip_address(int(ip) & 0xFFFFFFFF)
ip = embedded
if ip.is_private or ip.is_reserved or ip.is_loopback or ip.is_link_local or ip in _CGN_NETWORK:
raise ValueError(
f"Blocked private/internal IP {addr} (resolved from '{hostname}'). "
+23
View File
@@ -204,6 +204,29 @@ def to_wiki(
"Run `graphify extract .` or `graphify cluster-only .` first."
)
# Filter stale node IDs that exist in communities but not in G.
# Analysis JSON can drift from the graph after dedup / re-extract / update.
# NetworkX 3.x returns DegreeView({}) for missing nodes instead of raising,
# which crashes sorted() with TypeError; G.neighbors()/G.nodes[] also raise.
import sys as _sys
_g_nodes = set(G.nodes)
_orig_total = sum(len(ns) for ns in communities.values())
communities = {cid: [n for n in nodes if n in _g_nodes] for cid, nodes in communities.items()}
communities = {cid: nodes for cid, nodes in communities.items() if nodes}
_kept_total = sum(len(ns) for ns in communities.values())
if _kept_total < _orig_total:
print(
f"wiki: dropped {_orig_total - _kept_total} stale node ID(s) not in graph "
f"({len(communities)} communities remaining)",
file=_sys.stderr,
)
if not communities:
raise ValueError(
"all community node IDs are stale — none exist in the graph. "
"Re-run `graphify extract .` to regenerate .graphify_analysis.json."
)
# Clear stale .md files from previous runs to prevent orphan accumulation.
# Community labels are LLM-generated (per skill.md Step 5) and non-deterministic
# across runs — the same conceptual community may be named differently each time
+64
View File
@@ -608,3 +608,67 @@ def test_save_manifest_without_filter_unchanged_for_code(tmp_path):
manifest = json.loads(Path(manifest_path).read_text())
assert str(py) in manifest
assert manifest[str(py)]["ast_hash"] != ""
# Regression tests for #945 - .gitignore fallback when no .graphifyignore exists
def test_gitignore_fallback_when_no_graphifyignore(tmp_path):
"""When no .graphifyignore exists, .gitignore patterns are honored (#945)."""
(tmp_path / ".git").mkdir()
(tmp_path / ".gitignore").write_text("vendor/\n*.generated.py\n")
vendor = tmp_path / "vendor"
vendor.mkdir()
(vendor / "lib.py").write_text("x = 1")
(tmp_path / "main.py").write_text("print('hi')")
(tmp_path / "schema.generated.py").write_text("x = 1")
result = detect(tmp_path)
code = result["files"]["code"]
assert any("main.py" in f for f in code)
assert not any("vendor" in f for f in code)
assert not any("generated" in f for f in code)
def test_graphifyignore_takes_precedence_over_gitignore(tmp_path):
"""When both exist, .graphifyignore is used and .gitignore is ignored (#945)."""
(tmp_path / ".git").mkdir()
# .gitignore would exclude main.py; .graphifyignore excludes only other.py
(tmp_path / ".gitignore").write_text("main.py\n")
(tmp_path / ".graphifyignore").write_text("other.py\n")
(tmp_path / "main.py").write_text("x = 1")
(tmp_path / "other.py").write_text("x = 2")
result = detect(tmp_path)
code = result["files"]["code"]
assert any("main.py" in f for f in code) # gitignore NOT applied
assert not any("other.py" in f for f in code) # graphifyignore IS applied
# Regression tests for #947 - .worktrees/ skipped and --exclude flag
def test_detect_skips_worktrees_dir(tmp_path):
"""Files inside .worktrees/ are never indexed (#947)."""
wt = tmp_path / ".worktrees" / "feature-branch"
wt.mkdir(parents=True)
(wt / "main.py").write_text("x = 1")
(tmp_path / "app.py").write_text("y = 2")
result = detect(tmp_path)
code = result["files"]["code"]
assert any("app.py" in f for f in code)
assert not any(".worktrees" in f for f in code)
def test_detect_extra_excludes_pattern(tmp_path):
"""extra_excludes patterns exclude matching files from detect() (#947)."""
(tmp_path / "main.py").write_text("x = 1")
(tmp_path / "secret.py").write_text("API_KEY = 'abc'")
subdir = tmp_path / "legacy"
subdir.mkdir()
(subdir / "old.py").write_text("y = 2")
result = detect(tmp_path, extra_excludes=["secret.py", "legacy/"])
code = result["files"]["code"]
assert any("main.py" in f for f in code)
assert not any("secret.py" in f for f in code)
assert not any("legacy" in f for f in code)
+32
View File
@@ -165,3 +165,35 @@ def test_god_node_article_community_without_node_attr(tmp_path):
to_wiki(G, communities, tmp_path, community_labels=labels, god_nodes_data=god_nodes)
article = (tmp_path / "parse.md").read_text()
assert "[[Core Logic]]" in article
# Regression tests for #936 - stale community node IDs crash to_wiki after dedup/re-extract
def test_to_wiki_drops_stale_community_nodes(tmp_path):
"""Stale node IDs in communities dict are silently dropped without crash (#936)."""
G = _make_graph()
# Add a stale ID that exists in communities but not in G
communities = {0: ["n1", "n2", "stale_ghost"], 1: ["n3", "n4"]}
n = to_wiki(G, communities, tmp_path, community_labels=LABELS)
assert n == 2 # both community articles still written
article = (tmp_path / "Parsing_Layer.md").read_text()
assert "parse" in article
assert "stale_ghost" not in article
def test_to_wiki_all_stale_raises(tmp_path):
"""If every community node is stale, raise ValueError with a helpful message (#936)."""
G = _make_graph()
all_stale = {0: ["ghost1", "ghost2"], 1: ["ghost3"]}
with pytest.raises(ValueError, match="stale"):
to_wiki(G, all_stale, tmp_path, community_labels=LABELS)
def test_to_wiki_stale_nodes_prints_warning(tmp_path, capsys):
"""Stale node IDs trigger a stderr warning showing the drop count (#936)."""
G = _make_graph()
communities = {0: ["n1", "stale1", "stale2"], 1: ["n3", "n4"]}
to_wiki(G, communities, tmp_path, community_labels=LABELS)
err = capsys.readouterr().err
assert "2" in err # dropped count
assert "stale" in err.lower()