fix(extract): incremental --no-cluster merges instead of overwriting the graph (#2169)

An incremental `extract --no-cluster` wrote only the changed files over
graph.json with no merge, dropping every node/edge owned by an unchanged
file; and the id-canonicalization pass only learned batch files, so the
changed file's cross-file edges kept absolute-path target ids and
dangled. The raw path now merges the existing graph forward with the same
replace/prune semantics as the clustered path (new merge_raw_extraction
helper in build.py, shared loader), refuses to overwrite a corrupt
existing graph, and the remap pass now also learns in-root edge
target_file paths (existence-gated) so cross-file targets canonicalize.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
safishamsi
2026-07-25 22:53:43 +01:00
co-authored by Claude Opus 4.8
parent 05ee568969
commit 137dcf23fe
5 changed files with 358 additions and 26 deletions
+134 -21
View File
@@ -1041,6 +1041,137 @@ def deduplicate_by_label(nodes: list[dict], edges: list[dict]) -> tuple[list[dic
return deduped_nodes, deduped_edges
def _load_existing_graph(graph_path: Path) -> "tuple[list, list, list] | None":
"""Load (nodes, edges, hyperedges) from an existing graph.json for an
incremental merge, accepting both the ``links`` and ``edges`` spellings.
Reads the JSON directly instead of going through node_link_graph().
The latter rebuilds an undirected nx.Graph and then enumerating
edges() yields endpoints based on node insertion order, which
silently flips directional edges (e.g. `calls`) when the callee
was inserted before the caller. The _src/_tgt direction-preserving
attrs are popped before saving in export.py, so going through the
NetworkX round-trip loses direction permanently (#760).
Returns None when the file does not exist. Raises RuntimeError when it
exists but cannot be parsed — callers must refuse to overwrite rather
than silently replace a possibly-recoverable graph.
"""
if not graph_path.exists():
return None
from graphify.security import check_graph_file_size_cap
check_graph_file_size_cap(graph_path)
try:
data = json.loads(graph_path.read_text(encoding="utf-8"))
except (json.JSONDecodeError, OSError) as exc:
raise RuntimeError(
f"Cannot read {graph_path} for incremental merge: {exc}. "
"Delete the file and run a full rebuild."
) from exc
links_key = "links" if "links" in data else "edges"
return (
list(data.get("nodes", [])),
list(data.get(links_key, [])),
list(data.get("hyperedges", [])),
)
def merge_raw_extraction(
new: dict,
graph_path: str | Path,
prune_sources: "list[str] | None" = None,
root: "str | Path | None" = None,
) -> dict:
"""Merge the existing raw graph.json forward into a fresh raw extraction
(the ``extract --no-cluster`` incremental path, #2169).
Replace/prune semantics mirror :func:`build_merge` exactly, so the raw and
clustered incremental paths can't drift:
- sources re-extracted this run REPLACE their prior contribution — existing
nodes/edges/hyperedges owned by them are dropped, matched in both raw and
:func:`_norm_source_file` form (#1007);
- ``prune_sources`` (deleted / excluded / graph-stale files) are dropped,
with the ``_abs_identity`` third-form fallback (#2012), and "replace" wins
over a contradictory "delete" of a re-extracted source (#1796);
- everything else — nodes/edges/hyperedges owned by unchanged files — is
carried forward unchanged.
Survivors are PREPENDED to ``new``'s lists (existing-first), so the caller's
``dedupe_nodes`` last-writer-wins keeps fresh attributes for re-extracted
nodes while ``dedupe_edges`` first-wins never resurrects a replaced edge
(replaced sources' edges were already dropped above). Token counters and
every other key of ``new`` are left untouched. Returns ``new``, mutated in
place. Raises RuntimeError (via :func:`_load_existing_graph`) when the
existing graph is present but unparseable — the caller must refuse to
overwrite it. No-op when ``graph_path`` does not exist.
"""
graph_path = Path(graph_path)
loaded = _load_existing_graph(graph_path)
if loaded is None:
return new
existing_nodes, existing_edges, existing_hyperedges = loaded
_eff_root = (
str(Path(root).resolve()) if root is not None
else _infer_merge_root(graph_path)
)
new_sources: set[str] = set()
for n in new.get("nodes", []):
if not isinstance(n, dict):
continue
sf = n.get("source_file")
if not sf:
continue
new_sources.add(sf)
norm = _norm_source_file(sf, _eff_root)
if norm:
new_sources.add(norm)
prune_set: set[str] = set()
prune_abs: set[str] = set()
for p in (prune_sources or []):
if not p:
continue
prune_set.add(p)
norm = _norm_source_file(p, _eff_root)
if norm:
prune_set.add(norm)
a = _abs_identity(p, _eff_root)
if a:
prune_abs.add(a)
# "Replace" wins over a contradictory "delete" of the same source (#1796),
# in both string and absolute-identity space (#2012) — as in build_merge.
prune_set -= new_sources
new_abs = {_abs_identity(s, _eff_root) for s in new_sources}
new_abs.discard(None)
prune_abs -= new_abs
def _dropped(item: dict) -> bool:
if not isinstance(item, dict):
return True
sf = item.get("source_file")
if sf in new_sources or _norm_source_file(sf, _eff_root) in new_sources:
return True # re-extracted this run — replaced by the new chunk
if not sf:
return False # unowned — carry forward
if sf in prune_set:
return True
norm = _norm_source_file(sf, _eff_root)
if norm and norm in prune_set:
return True
a = _abs_identity(sf, _eff_root)
return bool(a) and a in prune_abs
new["nodes"] = [n for n in existing_nodes if not _dropped(n)] + list(new.get("nodes", []))
new["edges"] = [e for e in existing_edges if not _dropped(e)] + list(new.get("edges", []))
carried_hyper = [he for he in existing_hyperedges if not _dropped(he)]
if carried_hyper or new.get("hyperedges"):
new["hyperedges"] = carried_hyper + list(new.get("hyperedges", []))
return new
def build_merge(
new_chunks: list[dict],
graph_path: str | Path | None = None,
@@ -1061,27 +1192,9 @@ def build_merge(
root: if given, absolute source_file paths in new_chunks are made relative (#932).
"""
graph_path = Path(graph_path if graph_path is not None else _default_graph_json())
if graph_path.exists():
# Read JSON directly instead of going through node_link_graph().
# The latter rebuilds an undirected nx.Graph and then enumerating
# edges() yields endpoints based on node insertion order, which
# silently flips directional edges (e.g. `calls`) when the callee
# was inserted before the caller. The _src/_tgt direction-preserving
# attrs are popped before saving in export.py, so going through the
# NetworkX round-trip loses direction permanently (#760).
from graphify.security import check_graph_file_size_cap
check_graph_file_size_cap(graph_path)
try:
data = json.loads(graph_path.read_text(encoding="utf-8"))
except (json.JSONDecodeError, OSError) as exc:
raise RuntimeError(
f"Cannot read {graph_path} for incremental merge: {exc}. "
"Delete the file and run a full rebuild."
) from exc
links_key = "links" if "links" in data else "edges"
existing_nodes = list(data.get("nodes", []))
existing_edges = list(data.get(links_key, []))
existing_hyperedges = list(data.get("hyperedges", []))
_loaded = _load_existing_graph(graph_path)
if _loaded is not None:
existing_nodes, existing_edges, existing_hyperedges = _loaded
had_graph = True
else:
existing_nodes = []
+27
View File
@@ -3264,6 +3264,33 @@ def dispatch_command(cmd: str) -> None:
stages.total()
sys.exit(0)
if incremental_mode:
# #2169: this raw path used to write ONLY this run's extraction
# over graph.json — on an incremental run that is just the
# changed files, silently dropping every node/edge owned by an
# unchanged file. Merge the existing graph forward first, with
# the same replace/prune semantics as the clustered path's
# build_merge: re-extracted sources replaced, deleted +
# excluded + graph-stale sources pruned, everything else
# carried. Survivors are prepended, so the dedupe below keeps
# this run's fresh attributes for re-extracted nodes.
from graphify.build import merge_raw_extraction as _merge_raw_extraction
_raw_prune_sources: list[str] = list(deleted_files)
for _src in list(excluded_files) + graph_stale_sources:
if _src not in _raw_prune_sources:
_raw_prune_sources.append(_src)
try:
merged = _merge_raw_extraction(
merged,
graph_path=existing_graph_path,
prune_sources=_raw_prune_sources or None,
root=target,
)
except RuntimeError as exc:
# Existing graph present but unparseable: refuse to
# raw-dump this run's partial extraction over it.
print(f"error: {exc}", file=sys.stderr)
sys.exit(1)
merged["nodes"] = _dedupe_nodes(merged["nodes"])
merged["edges"] = _dedupe_edges(merged["edges"])
# Disambiguate colliding-basename file-node labels (#2032). This raw
+41 -1
View File
@@ -4657,7 +4657,47 @@ def extract(
# barrel repoint below (#1983). Unlike prefix_remap this records ALL
# in-root files, not just those whose prefix changed.
stem_forms: dict[Path, tuple[str, list[str]]] = {}
for path in paths:
# Canonicalize edge-target files too, not just this batch's inputs (#2169).
# On an incremental run `paths` is only the CHANGED files, so a changed
# file's cross-file import/re-export edges keep absolute-path-derived
# target ids the remap below never learns — they match no node in the
# merged graph and silently dangle. The target_file stamp (set at edge
# emit time) names each resolved target, so registering id_remap /
# stem_forms for those in-root files as well lets the edge remap and the
# target_file-guided repoint pass fix them exactly as on a full scan.
remap_paths: list[Path] = list(paths)
_remap_seen: set[Path] = set()
for _p in paths:
try:
_remap_seen.add(_p.resolve())
except (OSError, RuntimeError):
pass
for _e in all_edges:
_tf = _e.get("target_file")
if not _tf:
continue
try:
_tp = Path(_tf).resolve()
except (OSError, RuntimeError):
continue
if _tp in _remap_seen:
continue
_remap_seen.add(_tp)
try:
_tp.relative_to(root)
except ValueError:
continue # out-of-root target: leave its ids alone
try:
if not _tp.is_file():
# Speculatively-resolved target that doesn't exist (e.g. an
# import of a not-yet-created sibling): keep its raw id
# dangling, exactly as before, so no false canonical edge is
# fabricated toward a nonexistent file.
continue
except OSError:
continue
remap_paths.append(_tp)
for path in remap_paths:
old_id = _make_id(str(path))
try:
rel = path.relative_to(root)
+48 -4
View File
@@ -146,8 +146,12 @@ def _arm_no_cluster(monkeypatch, tmp_path, *, extra_argv=()):
def test_no_cluster_incomplete_build_refuses_to_shrink(tmp_path, monkeypatch, capsys):
# --force: the non-incremental raw-dump path, where the shrink guard is the
# only thing standing between a partial 1-node extraction and the existing
# complete 5-node graph. (Incremental runs merge the existing graph forward
# first — #2169 — so a partial run no longer shrinks there; see below.)
import json
graph = _arm_no_cluster(monkeypatch, tmp_path)
graph = _arm_no_cluster(monkeypatch, tmp_path, extra_argv=["--force"])
with pytest.raises(SystemExit) as exc:
mainmod.main()
@@ -158,9 +162,29 @@ def test_no_cluster_incomplete_build_refuses_to_shrink(tmp_path, monkeypatch, ca
assert len(json.loads(graph.read_text())["nodes"]) == 5
def test_no_cluster_incremental_incomplete_build_carries_existing_nodes(
tmp_path, monkeypatch
):
"""#2169: an INCREMENTAL --no-cluster run merges the existing graph forward,
so even an incomplete extraction does not shrink the graph — the existing
nodes are carried and this run's partial chunk is added, no guard refusal."""
import json
graph = _arm_no_cluster(monkeypatch, tmp_path)
with pytest.raises(SystemExit) as exc:
mainmod.main()
assert exc.value.code == 0 # the raw --no-cluster path exits 0 on success
ids = {n["id"] for n in json.loads(graph.read_text())["nodes"]}
assert {f"keep{i}" for i in range(5)} <= ids, ids
assert "s1" in ids, ids
def test_no_cluster_allow_partial_overwrites(tmp_path, monkeypatch):
import json
graph = _arm_no_cluster(monkeypatch, tmp_path, extra_argv=["--allow-partial"])
graph = _arm_no_cluster(
monkeypatch, tmp_path, extra_argv=["--force", "--allow-partial"]
)
with pytest.raises(SystemExit) as exc:
mainmod.main()
@@ -175,8 +199,10 @@ def test_no_cluster_incomplete_build_fails_closed_on_malformed_existing_graph(
"""A present-but-unparseable existing graph.json (corrupt or mid-write) could
be hiding a complete graph, so an incomplete --no-cluster build must refuse
to overwrite it — matching to_json's #479 fail-closed handling, not the
fail-open 'proceed when we can't count' path."""
graph = _arm_no_cluster(monkeypatch, tmp_path)
fail-open 'proceed when we can't count' path. --force: the non-incremental
raw-dump path (the incremental path fails even earlier, at the forward
merge — see the test below)."""
graph = _arm_no_cluster(monkeypatch, tmp_path, extra_argv=["--force"])
graph.write_text("{corrupt json", encoding="utf-8") # non-empty, unparseable
with pytest.raises(SystemExit) as exc:
@@ -186,3 +212,21 @@ def test_no_cluster_incomplete_build_fails_closed_on_malformed_existing_graph(
assert "unparseable" in capsys.readouterr().err
# The corrupt file is left untouched rather than clobbered by the partial build.
assert graph.read_text() == "{corrupt json"
def test_no_cluster_incremental_malformed_existing_graph_refuses_merge(
tmp_path, monkeypatch, capsys
):
"""#2169: an incremental --no-cluster run must hard-fail on an unparseable
existing graph.json (build_merge's message) instead of raw-dumping this
run's chunks over it."""
graph = _arm_no_cluster(monkeypatch, tmp_path)
graph.write_text("{corrupt json", encoding="utf-8") # non-empty, unparseable
with pytest.raises(SystemExit) as exc:
mainmod.main()
assert exc.value.code == 1
assert "Cannot read" in capsys.readouterr().err
# The corrupt file is left untouched rather than clobbered.
assert graph.read_text() == "{corrupt json"
+108
View File
@@ -103,6 +103,114 @@ def _edges(graph_json: Path) -> list[dict]:
return g.get("links", g.get("edges", []))
def test_extract_no_cluster_incremental_changed_file_preserves_unchanged_files(tmp_path):
"""#2169: an incremental --no-cluster extract of ONE changed file must merge
into the existing graph, not overwrite graph.json with just that file's
chunk — and the changed file's cross-file import edges must keep pointing at
the unchanged target file's canonical node ids, not dangling
absolute-path-derived ones."""
proj = tmp_path / "proj"
(proj / "app" / "add").mkdir(parents=True)
(proj / "src" / "components").mkdir(parents=True)
(proj / "src" / "components" / "ScanScreen.tsx").write_text(
"export function ScanScreen() {\n return null;\n}\n", encoding="utf-8"
)
scan_tsx = proj / "app" / "add" / "scan.tsx"
scan_tsx.write_text(
"import {ScanScreen} from '../../src/components/ScanScreen';\n"
"export default ScanScreen;\n",
encoding="utf-8",
)
first = _run(["extract", str(proj), "--code-only", "--no-cluster"], tmp_path)
assert first.returncode == 0, first.stderr
gj = proj / "graphify-out" / "graph.json"
base = json.loads(gj.read_text(encoding="utf-8"))
base_ids = {n["id"] for n in base["nodes"]}
# Sanity: importer file, target file, and target symbol all present.
assert {
"app_add_scan",
"src_components_scanscreen",
"src_components_scanscreen_scanscreen",
} <= base_ids, base_ids
# Change ONLY scan.tsx (harmless comment), then re-run the same command.
scan_tsx.write_text(
scan_tsx.read_text(encoding="utf-8") + "\n// touched\n", encoding="utf-8"
)
second = _run(["extract", str(proj), "--code-only", "--no-cluster"], tmp_path)
assert second.returncode == 0, second.stderr
# Guard against a silent full rescan masking the merge bug.
assert "incremental scan" in second.stdout.lower(), second.stdout
after = json.loads(gj.read_text(encoding="utf-8"))
after_ids = {n["id"] for n in after["nodes"]}
# The unchanged file's nodes must survive the incremental raw write.
assert after_ids == base_ids, (
f"incremental --no-cluster dropped/changed nodes: "
f"missing={base_ids - after_ids}, extra={after_ids - base_ids}"
)
after_edges = after.get("links", after.get("edges", []))
# The unchanged file's own edge survives.
assert any(
e.get("relation") == "contains"
and e.get("source") == "src_components_scanscreen"
and e.get("target") == "src_components_scanscreen_scanscreen"
for e in after_edges
), after_edges
# No dangling endpoints on cross-file edges: the changed file's re-extracted
# imports/re-exports must resolve to the unchanged target's canonical ids,
# not absolute-path-derived ghosts (the extract.py half of #2169).
for e in after_edges:
if e.get("relation") in ("imports_from", "re_exports", "contains", "imports"):
assert e.get("source") in after_ids, f"dangling source: {e}"
assert e.get("target") in after_ids, f"dangling target: {e}"
def test_extract_no_cluster_incremental_code_only_preserves_doc_nodes(tmp_path):
"""#2169: an incremental --code-only --no-cluster run over a mixed corpus
must carry forward doc-sourced nodes it did not re-extract."""
proj = tmp_path / "proj"
proj.mkdir()
util = proj / "util.py"
util.write_text("def alpha():\n return 1\n", encoding="utf-8")
(proj / "notes.md").write_text("# Notes\nSome prose.\n", encoding="utf-8")
first = _run(["extract", str(proj), "--code-only", "--no-cluster"], tmp_path)
assert first.returncode == 0, first.stderr
gj = proj / "graphify-out" / "graph.json"
g = json.loads(gj.read_text(encoding="utf-8"))
assert g.get("nodes"), "first run should produce a non-empty code graph"
# Seed a doc-sourced node, as a prior (LLM-backed) run would have written.
g["nodes"].append({
"id": "notes",
"label": "notes.md",
"type": "document",
"source_file": "notes.md",
})
gj.write_text(json.dumps(g), encoding="utf-8")
# Change only the code file; the doc node must survive the incremental run.
util.write_text(
"def alpha():\n return 1\n\ndef beta():\n return 2\n",
encoding="utf-8",
)
second = _run(["extract", str(proj), "--code-only", "--no-cluster"], tmp_path)
assert second.returncode == 0, second.stderr
assert "incremental scan" in second.stdout.lower(), second.stdout
after = json.loads(gj.read_text(encoding="utf-8"))
after_by_id = {n["id"]: n for n in after["nodes"]}
assert "notes" in after_by_id, (
f"doc node dropped by incremental --code-only --no-cluster: "
f"{sorted(after_by_id)}"
)
assert after_by_id["notes"].get("source_file") == "notes.md"
# And the changed code file was actually re-extracted.
assert any("beta" in i for i in after_by_id), sorted(after_by_id)
def test_update_prunes_a_removed_imports_edge(tmp_path):
"""#1521: when an import is deleted from a file, `graphify update` must prune
the edge it produced — preserving it (keyed only on endpoint membership) left a