fix stale wiki nodes (#936), gitignore fallback and --exclude flag (#945/#947), NAT64 SSRF false-positive
Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Sonnet 4.6
parent
6939494b3e
commit
9e6192a6c2
@@ -2431,6 +2431,7 @@ def main() -> None:
|
||||
# Clustering tuning knobs
|
||||
cli_resolution: float = 1.0
|
||||
cli_exclude_hubs: float | None = None
|
||||
cli_excludes: list[str] = []
|
||||
|
||||
def _parse_int(name: str, raw: str) -> int:
|
||||
try:
|
||||
@@ -2504,6 +2505,10 @@ def main() -> None:
|
||||
cli_exclude_hubs = float(args[i + 1]); i += 2
|
||||
elif a.startswith("--exclude-hubs="):
|
||||
cli_exclude_hubs = float(a.split("=", 1)[1]); i += 1
|
||||
elif a == "--exclude" and i + 1 < len(args):
|
||||
cli_excludes.append(args[i + 1]); i += 2
|
||||
elif a.startswith("--exclude="):
|
||||
cli_excludes.append(a.split("=", 1)[1]); i += 1
|
||||
else:
|
||||
i += 1
|
||||
|
||||
@@ -2610,10 +2615,11 @@ def main() -> None:
|
||||
target,
|
||||
manifest_path=str(manifest_path),
|
||||
google_workspace=google_workspace or None,
|
||||
extra_excludes=cli_excludes or None,
|
||||
)
|
||||
else:
|
||||
print(f"[graphify extract] scanning {target}")
|
||||
detection = _detect(target, google_workspace=google_workspace or None)
|
||||
detection = _detect(target, google_workspace=google_workspace or None, extra_excludes=cli_excludes or None)
|
||||
|
||||
files_by_type = detection.get("files", {})
|
||||
if incremental_mode:
|
||||
|
||||
+15
-2
@@ -400,6 +400,7 @@ _SKIP_DIRS = {
|
||||
".next", ".nuxt", ".turbo", ".angular",
|
||||
".idea", ".cache", ".parcel-cache", ".svelte-kit", ".terraform", ".serverless",
|
||||
".graphify", # graphify's own extraction cache — never index self-generated data
|
||||
".worktrees", # git worktree convention (#947) — sibling checkouts, always redundant
|
||||
}
|
||||
|
||||
# Large generated files that are never useful to extract
|
||||
@@ -486,7 +487,11 @@ def _load_graphifyignore(root: Path) -> list[tuple[Path, str]]:
|
||||
|
||||
patterns: list[tuple[Path, str]] = []
|
||||
for d in dirs:
|
||||
# Prefer .graphifyignore; fall back to .gitignore so projects that already
|
||||
# maintain a .gitignore get sensible defaults without duplicating it (#945).
|
||||
ignore_file = d / ".graphifyignore"
|
||||
if not ignore_file.exists():
|
||||
ignore_file = d / ".gitignore"
|
||||
if ignore_file.exists():
|
||||
for raw in ignore_file.read_text(encoding="utf-8", errors="ignore").splitlines():
|
||||
line = _parse_gitignore_line(raw)
|
||||
@@ -701,7 +706,7 @@ def _auto_follow_symlinks(root: Path) -> bool:
|
||||
return False
|
||||
|
||||
|
||||
def detect(root: Path, *, follow_symlinks: bool | None = None, google_workspace: bool | None = None) -> dict:
|
||||
def detect(root: Path, *, follow_symlinks: bool | None = None, google_workspace: bool | None = None, extra_excludes: list[str] | None = None) -> dict:
|
||||
root = root.resolve()
|
||||
if follow_symlinks is None:
|
||||
follow_symlinks = _auto_follow_symlinks(root)
|
||||
@@ -717,6 +722,13 @@ def detect(root: Path, *, follow_symlinks: bool | None = None, google_workspace:
|
||||
|
||||
skipped_sensitive: list[str] = []
|
||||
ignore_patterns = _load_graphifyignore(root)
|
||||
# CLI --exclude patterns are anchored at the scan root and appended last
|
||||
# so they win over any .graphifyignore/.gitignore rules (#947).
|
||||
if extra_excludes:
|
||||
for pat in extra_excludes:
|
||||
line = _parse_gitignore_line(pat)
|
||||
if line:
|
||||
ignore_patterns.append((root, line))
|
||||
include_patterns = _load_graphifyinclude(root)
|
||||
|
||||
# Always include graphify-out/memory/ - query results filed back into the graph
|
||||
@@ -933,6 +945,7 @@ def detect_incremental(
|
||||
follow_symlinks: bool | None = None,
|
||||
google_workspace: bool | None = None,
|
||||
kind: str = "semantic",
|
||||
extra_excludes: list[str] | None = None,
|
||||
) -> dict:
|
||||
"""Like detect(), but returns only new or modified files since the last run.
|
||||
|
||||
@@ -957,7 +970,7 @@ def detect_incremental(
|
||||
incremental runs. ``None`` (default) means auto-detect: ``True`` when ``root``
|
||||
contains at least one direct symlinked child, ``False`` otherwise.
|
||||
"""
|
||||
full = detect(root, follow_symlinks=follow_symlinks, google_workspace=google_workspace)
|
||||
full = detect(root, follow_symlinks=follow_symlinks, google_workspace=google_workspace, extra_excludes=extra_excludes)
|
||||
manifest = load_manifest(manifest_path)
|
||||
|
||||
if not manifest:
|
||||
|
||||
@@ -22,6 +22,10 @@ _BLOCKED_HOSTS = {"metadata.google.internal", "metadata.google.com"}
|
||||
# RFC 6598 Shared Address Space (CGN) -- is_private misses this on Python <3.11
|
||||
_CGN_NETWORK = ipaddress.ip_network("100.64.0.0/10")
|
||||
|
||||
# RFC 6052 NAT64 Well-Known Prefix -- is_reserved=True in Python but these embed
|
||||
# public IPv4 addresses and are legitimate public internet traffic, not SSRF vectors.
|
||||
_NAT64_WKP = ipaddress.ip_network("64:ff9b::/96")
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# URL validation
|
||||
@@ -57,6 +61,10 @@ def validate_url(url: str) -> str:
|
||||
for info in infos:
|
||||
addr = info[4][0]
|
||||
ip = ipaddress.ip_address(addr)
|
||||
# For NAT64 addresses, check the embedded IPv4 instead of the wrapper
|
||||
if isinstance(ip, ipaddress.IPv6Address) and ip in _NAT64_WKP:
|
||||
embedded = ipaddress.ip_address(int(ip) & 0xFFFFFFFF)
|
||||
ip = embedded
|
||||
if ip.is_private or ip.is_reserved or ip.is_loopback or ip.is_link_local or ip in _CGN_NETWORK:
|
||||
raise ValueError(
|
||||
f"Blocked private/internal IP {addr} (resolved from '{hostname}'). "
|
||||
|
||||
@@ -204,6 +204,29 @@ def to_wiki(
|
||||
"Run `graphify extract .` or `graphify cluster-only .` first."
|
||||
)
|
||||
|
||||
# Filter stale node IDs that exist in communities but not in G.
|
||||
# Analysis JSON can drift from the graph after dedup / re-extract / update.
|
||||
# NetworkX 3.x returns DegreeView({}) for missing nodes instead of raising,
|
||||
# which crashes sorted() with TypeError; G.neighbors()/G.nodes[] also raise.
|
||||
import sys as _sys
|
||||
_g_nodes = set(G.nodes)
|
||||
_orig_total = sum(len(ns) for ns in communities.values())
|
||||
communities = {cid: [n for n in nodes if n in _g_nodes] for cid, nodes in communities.items()}
|
||||
communities = {cid: nodes for cid, nodes in communities.items() if nodes}
|
||||
_kept_total = sum(len(ns) for ns in communities.values())
|
||||
if _kept_total < _orig_total:
|
||||
print(
|
||||
f"wiki: dropped {_orig_total - _kept_total} stale node ID(s) not in graph "
|
||||
f"({len(communities)} communities remaining)",
|
||||
file=_sys.stderr,
|
||||
)
|
||||
|
||||
if not communities:
|
||||
raise ValueError(
|
||||
"all community node IDs are stale — none exist in the graph. "
|
||||
"Re-run `graphify extract .` to regenerate .graphify_analysis.json."
|
||||
)
|
||||
|
||||
# Clear stale .md files from previous runs to prevent orphan accumulation.
|
||||
# Community labels are LLM-generated (per skill.md Step 5) and non-deterministic
|
||||
# across runs — the same conceptual community may be named differently each time
|
||||
|
||||
@@ -608,3 +608,67 @@ def test_save_manifest_without_filter_unchanged_for_code(tmp_path):
|
||||
manifest = json.loads(Path(manifest_path).read_text())
|
||||
assert str(py) in manifest
|
||||
assert manifest[str(py)]["ast_hash"] != ""
|
||||
|
||||
|
||||
# Regression tests for #945 - .gitignore fallback when no .graphifyignore exists
|
||||
|
||||
def test_gitignore_fallback_when_no_graphifyignore(tmp_path):
|
||||
"""When no .graphifyignore exists, .gitignore patterns are honored (#945)."""
|
||||
(tmp_path / ".git").mkdir()
|
||||
(tmp_path / ".gitignore").write_text("vendor/\n*.generated.py\n")
|
||||
vendor = tmp_path / "vendor"
|
||||
vendor.mkdir()
|
||||
(vendor / "lib.py").write_text("x = 1")
|
||||
(tmp_path / "main.py").write_text("print('hi')")
|
||||
(tmp_path / "schema.generated.py").write_text("x = 1")
|
||||
|
||||
result = detect(tmp_path)
|
||||
code = result["files"]["code"]
|
||||
assert any("main.py" in f for f in code)
|
||||
assert not any("vendor" in f for f in code)
|
||||
assert not any("generated" in f for f in code)
|
||||
|
||||
|
||||
def test_graphifyignore_takes_precedence_over_gitignore(tmp_path):
|
||||
"""When both exist, .graphifyignore is used and .gitignore is ignored (#945)."""
|
||||
(tmp_path / ".git").mkdir()
|
||||
# .gitignore would exclude main.py; .graphifyignore excludes only other.py
|
||||
(tmp_path / ".gitignore").write_text("main.py\n")
|
||||
(tmp_path / ".graphifyignore").write_text("other.py\n")
|
||||
(tmp_path / "main.py").write_text("x = 1")
|
||||
(tmp_path / "other.py").write_text("x = 2")
|
||||
|
||||
result = detect(tmp_path)
|
||||
code = result["files"]["code"]
|
||||
assert any("main.py" in f for f in code) # gitignore NOT applied
|
||||
assert not any("other.py" in f for f in code) # graphifyignore IS applied
|
||||
|
||||
|
||||
# Regression tests for #947 - .worktrees/ skipped and --exclude flag
|
||||
|
||||
def test_detect_skips_worktrees_dir(tmp_path):
|
||||
"""Files inside .worktrees/ are never indexed (#947)."""
|
||||
wt = tmp_path / ".worktrees" / "feature-branch"
|
||||
wt.mkdir(parents=True)
|
||||
(wt / "main.py").write_text("x = 1")
|
||||
(tmp_path / "app.py").write_text("y = 2")
|
||||
|
||||
result = detect(tmp_path)
|
||||
code = result["files"]["code"]
|
||||
assert any("app.py" in f for f in code)
|
||||
assert not any(".worktrees" in f for f in code)
|
||||
|
||||
|
||||
def test_detect_extra_excludes_pattern(tmp_path):
|
||||
"""extra_excludes patterns exclude matching files from detect() (#947)."""
|
||||
(tmp_path / "main.py").write_text("x = 1")
|
||||
(tmp_path / "secret.py").write_text("API_KEY = 'abc'")
|
||||
subdir = tmp_path / "legacy"
|
||||
subdir.mkdir()
|
||||
(subdir / "old.py").write_text("y = 2")
|
||||
|
||||
result = detect(tmp_path, extra_excludes=["secret.py", "legacy/"])
|
||||
code = result["files"]["code"]
|
||||
assert any("main.py" in f for f in code)
|
||||
assert not any("secret.py" in f for f in code)
|
||||
assert not any("legacy" in f for f in code)
|
||||
|
||||
@@ -165,3 +165,35 @@ def test_god_node_article_community_without_node_attr(tmp_path):
|
||||
to_wiki(G, communities, tmp_path, community_labels=labels, god_nodes_data=god_nodes)
|
||||
article = (tmp_path / "parse.md").read_text()
|
||||
assert "[[Core Logic]]" in article
|
||||
|
||||
|
||||
# Regression tests for #936 - stale community node IDs crash to_wiki after dedup/re-extract
|
||||
|
||||
def test_to_wiki_drops_stale_community_nodes(tmp_path):
|
||||
"""Stale node IDs in communities dict are silently dropped without crash (#936)."""
|
||||
G = _make_graph()
|
||||
# Add a stale ID that exists in communities but not in G
|
||||
communities = {0: ["n1", "n2", "stale_ghost"], 1: ["n3", "n4"]}
|
||||
n = to_wiki(G, communities, tmp_path, community_labels=LABELS)
|
||||
assert n == 2 # both community articles still written
|
||||
article = (tmp_path / "Parsing_Layer.md").read_text()
|
||||
assert "parse" in article
|
||||
assert "stale_ghost" not in article
|
||||
|
||||
|
||||
def test_to_wiki_all_stale_raises(tmp_path):
|
||||
"""If every community node is stale, raise ValueError with a helpful message (#936)."""
|
||||
G = _make_graph()
|
||||
all_stale = {0: ["ghost1", "ghost2"], 1: ["ghost3"]}
|
||||
with pytest.raises(ValueError, match="stale"):
|
||||
to_wiki(G, all_stale, tmp_path, community_labels=LABELS)
|
||||
|
||||
|
||||
def test_to_wiki_stale_nodes_prints_warning(tmp_path, capsys):
|
||||
"""Stale node IDs trigger a stderr warning showing the drop count (#936)."""
|
||||
G = _make_graph()
|
||||
communities = {0: ["n1", "stale1", "stale2"], 1: ["n3", "n4"]}
|
||||
to_wiki(G, communities, tmp_path, community_labels=LABELS)
|
||||
err = capsys.readouterr().err
|
||||
assert "2" in err # dropped count
|
||||
assert "stale" in err.lower()
|
||||
|
||||
Reference in New Issue
Block a user