diff --git a/graphify/dedup.py b/graphify/dedup.py index d5b8276..b2885fe 100644 --- a/graphify/dedup.py +++ b/graphify/dedup.py @@ -186,7 +186,11 @@ def deduplicate_entities( for node in group: sf = node.get("source_file") or "" by_file[sf].append(node) - for file_group in by_file.values(): + for sf, file_group in by_file.items(): + if not sf: + # No source_file — cannot prove same symbol; skip to avoid + # collapsing distinct nodes that happen to share a label (#1178). + continue if len(file_group) > 1: winner = _pick_winner(file_group) for node in file_group: diff --git a/graphify/export.py b/graphify/export.py index 425cced..e9f7b50 100644 --- a/graphify/export.py +++ b/graphify/export.py @@ -493,9 +493,12 @@ def to_json(G: nx.Graph, communities: dict[int, list[str]], output_path: str, *, import sys as _sys print( f"[graphify] WARNING: new graph has {new_n} nodes but existing " - f"graph.json has {existing_n}. Refusing to overwrite — you may be " - f"missing chunk files from a previous session. " - f"Pass force=True to override.", + f"graph.json has {existing_n} (net -{existing_n - new_n}). " + f"Refusing to overwrite. Possible causes: missing chunk files from " + f"a previous session, or fuzzy dedup collapsed same-named symbols " + f"across files during an --update on an already-current graph. " + f"Run a full rebuild (/graphify .) to be safe, or pass force=True " + f"only if you have verified the reduction is legitimate.", file=_sys.stderr, ) return False diff --git a/graphify/skills/amp/references/update.md b/graphify/skills/amp/references/update.md index f53974a..d35b665 100644 --- a/graphify/skills/amp/references/update.md +++ b/graphify/skills/amp/references/update.md @@ -93,13 +93,18 @@ from graphify.detect import save_manifest new_extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) incremental = json.loads(Path('graphify-out/.graphify_incremental.json').read_text(encoding=\"utf-8\")) deleted = list(incremental.get('deleted_files', [])) +# Also prune old nodes for re-extracted (changed) files before inserting fresh AST. +# Without this, build_merge's dedup pass tries to reconcile old and new versions of +# the same file's nodes and can collapse same-named symbols across files (#1178). +changed = [f for files in incremental.get('new_files', {}).values() for f in files] +prune = list(dict.fromkeys(deleted + changed)) or None # Use build_merge() — reads graph.json directly without NetworkX round-trip # so edge direction (calls, implements, imports) is always preserved (#801). G = build_merge( [new_extraction], graph_path='graphify-out/graph.json', - prune_sources=deleted or None, + prune_sources=prune, ) print(f'[graphify update] Merged: {G.number_of_nodes()} nodes, {G.number_of_edges()} edges') diff --git a/graphify/skills/claude/references/update.md b/graphify/skills/claude/references/update.md index f53974a..d35b665 100644 --- a/graphify/skills/claude/references/update.md +++ b/graphify/skills/claude/references/update.md @@ -93,13 +93,18 @@ from graphify.detect import save_manifest new_extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) incremental = json.loads(Path('graphify-out/.graphify_incremental.json').read_text(encoding=\"utf-8\")) deleted = list(incremental.get('deleted_files', [])) +# Also prune old nodes for re-extracted (changed) files before inserting fresh AST. +# Without this, build_merge's dedup pass tries to reconcile old and new versions of +# the same file's nodes and can collapse same-named symbols across files (#1178). +changed = [f for files in incremental.get('new_files', {}).values() for f in files] +prune = list(dict.fromkeys(deleted + changed)) or None # Use build_merge() — reads graph.json directly without NetworkX round-trip # so edge direction (calls, implements, imports) is always preserved (#801). G = build_merge( [new_extraction], graph_path='graphify-out/graph.json', - prune_sources=deleted or None, + prune_sources=prune, ) print(f'[graphify update] Merged: {G.number_of_nodes()} nodes, {G.number_of_edges()} edges') diff --git a/graphify/skills/claw/references/update.md b/graphify/skills/claw/references/update.md index f53974a..d35b665 100644 --- a/graphify/skills/claw/references/update.md +++ b/graphify/skills/claw/references/update.md @@ -93,13 +93,18 @@ from graphify.detect import save_manifest new_extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) incremental = json.loads(Path('graphify-out/.graphify_incremental.json').read_text(encoding=\"utf-8\")) deleted = list(incremental.get('deleted_files', [])) +# Also prune old nodes for re-extracted (changed) files before inserting fresh AST. +# Without this, build_merge's dedup pass tries to reconcile old and new versions of +# the same file's nodes and can collapse same-named symbols across files (#1178). +changed = [f for files in incremental.get('new_files', {}).values() for f in files] +prune = list(dict.fromkeys(deleted + changed)) or None # Use build_merge() — reads graph.json directly without NetworkX round-trip # so edge direction (calls, implements, imports) is always preserved (#801). G = build_merge( [new_extraction], graph_path='graphify-out/graph.json', - prune_sources=deleted or None, + prune_sources=prune, ) print(f'[graphify update] Merged: {G.number_of_nodes()} nodes, {G.number_of_edges()} edges') diff --git a/graphify/skills/codex/references/update.md b/graphify/skills/codex/references/update.md index f53974a..d35b665 100644 --- a/graphify/skills/codex/references/update.md +++ b/graphify/skills/codex/references/update.md @@ -93,13 +93,18 @@ from graphify.detect import save_manifest new_extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) incremental = json.loads(Path('graphify-out/.graphify_incremental.json').read_text(encoding=\"utf-8\")) deleted = list(incremental.get('deleted_files', [])) +# Also prune old nodes for re-extracted (changed) files before inserting fresh AST. +# Without this, build_merge's dedup pass tries to reconcile old and new versions of +# the same file's nodes and can collapse same-named symbols across files (#1178). +changed = [f for files in incremental.get('new_files', {}).values() for f in files] +prune = list(dict.fromkeys(deleted + changed)) or None # Use build_merge() — reads graph.json directly without NetworkX round-trip # so edge direction (calls, implements, imports) is always preserved (#801). G = build_merge( [new_extraction], graph_path='graphify-out/graph.json', - prune_sources=deleted or None, + prune_sources=prune, ) print(f'[graphify update] Merged: {G.number_of_nodes()} nodes, {G.number_of_edges()} edges') diff --git a/graphify/skills/copilot/references/update.md b/graphify/skills/copilot/references/update.md index f53974a..d35b665 100644 --- a/graphify/skills/copilot/references/update.md +++ b/graphify/skills/copilot/references/update.md @@ -93,13 +93,18 @@ from graphify.detect import save_manifest new_extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) incremental = json.loads(Path('graphify-out/.graphify_incremental.json').read_text(encoding=\"utf-8\")) deleted = list(incremental.get('deleted_files', [])) +# Also prune old nodes for re-extracted (changed) files before inserting fresh AST. +# Without this, build_merge's dedup pass tries to reconcile old and new versions of +# the same file's nodes and can collapse same-named symbols across files (#1178). +changed = [f for files in incremental.get('new_files', {}).values() for f in files] +prune = list(dict.fromkeys(deleted + changed)) or None # Use build_merge() — reads graph.json directly without NetworkX round-trip # so edge direction (calls, implements, imports) is always preserved (#801). G = build_merge( [new_extraction], graph_path='graphify-out/graph.json', - prune_sources=deleted or None, + prune_sources=prune, ) print(f'[graphify update] Merged: {G.number_of_nodes()} nodes, {G.number_of_edges()} edges') diff --git a/graphify/skills/droid/references/update.md b/graphify/skills/droid/references/update.md index f53974a..d35b665 100644 --- a/graphify/skills/droid/references/update.md +++ b/graphify/skills/droid/references/update.md @@ -93,13 +93,18 @@ from graphify.detect import save_manifest new_extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) incremental = json.loads(Path('graphify-out/.graphify_incremental.json').read_text(encoding=\"utf-8\")) deleted = list(incremental.get('deleted_files', [])) +# Also prune old nodes for re-extracted (changed) files before inserting fresh AST. +# Without this, build_merge's dedup pass tries to reconcile old and new versions of +# the same file's nodes and can collapse same-named symbols across files (#1178). +changed = [f for files in incremental.get('new_files', {}).values() for f in files] +prune = list(dict.fromkeys(deleted + changed)) or None # Use build_merge() — reads graph.json directly without NetworkX round-trip # so edge direction (calls, implements, imports) is always preserved (#801). G = build_merge( [new_extraction], graph_path='graphify-out/graph.json', - prune_sources=deleted or None, + prune_sources=prune, ) print(f'[graphify update] Merged: {G.number_of_nodes()} nodes, {G.number_of_edges()} edges') diff --git a/graphify/skills/kilo/references/update.md b/graphify/skills/kilo/references/update.md index f53974a..d35b665 100644 --- a/graphify/skills/kilo/references/update.md +++ b/graphify/skills/kilo/references/update.md @@ -93,13 +93,18 @@ from graphify.detect import save_manifest new_extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) incremental = json.loads(Path('graphify-out/.graphify_incremental.json').read_text(encoding=\"utf-8\")) deleted = list(incremental.get('deleted_files', [])) +# Also prune old nodes for re-extracted (changed) files before inserting fresh AST. +# Without this, build_merge's dedup pass tries to reconcile old and new versions of +# the same file's nodes and can collapse same-named symbols across files (#1178). +changed = [f for files in incremental.get('new_files', {}).values() for f in files] +prune = list(dict.fromkeys(deleted + changed)) or None # Use build_merge() — reads graph.json directly without NetworkX round-trip # so edge direction (calls, implements, imports) is always preserved (#801). G = build_merge( [new_extraction], graph_path='graphify-out/graph.json', - prune_sources=deleted or None, + prune_sources=prune, ) print(f'[graphify update] Merged: {G.number_of_nodes()} nodes, {G.number_of_edges()} edges') diff --git a/graphify/skills/kiro/references/update.md b/graphify/skills/kiro/references/update.md index f53974a..d35b665 100644 --- a/graphify/skills/kiro/references/update.md +++ b/graphify/skills/kiro/references/update.md @@ -93,13 +93,18 @@ from graphify.detect import save_manifest new_extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) incremental = json.loads(Path('graphify-out/.graphify_incremental.json').read_text(encoding=\"utf-8\")) deleted = list(incremental.get('deleted_files', [])) +# Also prune old nodes for re-extracted (changed) files before inserting fresh AST. +# Without this, build_merge's dedup pass tries to reconcile old and new versions of +# the same file's nodes and can collapse same-named symbols across files (#1178). +changed = [f for files in incremental.get('new_files', {}).values() for f in files] +prune = list(dict.fromkeys(deleted + changed)) or None # Use build_merge() — reads graph.json directly without NetworkX round-trip # so edge direction (calls, implements, imports) is always preserved (#801). G = build_merge( [new_extraction], graph_path='graphify-out/graph.json', - prune_sources=deleted or None, + prune_sources=prune, ) print(f'[graphify update] Merged: {G.number_of_nodes()} nodes, {G.number_of_edges()} edges') diff --git a/graphify/skills/opencode/references/update.md b/graphify/skills/opencode/references/update.md index f53974a..d35b665 100644 --- a/graphify/skills/opencode/references/update.md +++ b/graphify/skills/opencode/references/update.md @@ -93,13 +93,18 @@ from graphify.detect import save_manifest new_extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) incremental = json.loads(Path('graphify-out/.graphify_incremental.json').read_text(encoding=\"utf-8\")) deleted = list(incremental.get('deleted_files', [])) +# Also prune old nodes for re-extracted (changed) files before inserting fresh AST. +# Without this, build_merge's dedup pass tries to reconcile old and new versions of +# the same file's nodes and can collapse same-named symbols across files (#1178). +changed = [f for files in incremental.get('new_files', {}).values() for f in files] +prune = list(dict.fromkeys(deleted + changed)) or None # Use build_merge() — reads graph.json directly without NetworkX round-trip # so edge direction (calls, implements, imports) is always preserved (#801). G = build_merge( [new_extraction], graph_path='graphify-out/graph.json', - prune_sources=deleted or None, + prune_sources=prune, ) print(f'[graphify update] Merged: {G.number_of_nodes()} nodes, {G.number_of_edges()} edges') diff --git a/graphify/skills/pi/references/update.md b/graphify/skills/pi/references/update.md index f53974a..d35b665 100644 --- a/graphify/skills/pi/references/update.md +++ b/graphify/skills/pi/references/update.md @@ -93,13 +93,18 @@ from graphify.detect import save_manifest new_extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) incremental = json.loads(Path('graphify-out/.graphify_incremental.json').read_text(encoding=\"utf-8\")) deleted = list(incremental.get('deleted_files', [])) +# Also prune old nodes for re-extracted (changed) files before inserting fresh AST. +# Without this, build_merge's dedup pass tries to reconcile old and new versions of +# the same file's nodes and can collapse same-named symbols across files (#1178). +changed = [f for files in incremental.get('new_files', {}).values() for f in files] +prune = list(dict.fromkeys(deleted + changed)) or None # Use build_merge() — reads graph.json directly without NetworkX round-trip # so edge direction (calls, implements, imports) is always preserved (#801). G = build_merge( [new_extraction], graph_path='graphify-out/graph.json', - prune_sources=deleted or None, + prune_sources=prune, ) print(f'[graphify update] Merged: {G.number_of_nodes()} nodes, {G.number_of_edges()} edges') diff --git a/graphify/skills/trae/references/update.md b/graphify/skills/trae/references/update.md index f53974a..d35b665 100644 --- a/graphify/skills/trae/references/update.md +++ b/graphify/skills/trae/references/update.md @@ -93,13 +93,18 @@ from graphify.detect import save_manifest new_extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) incremental = json.loads(Path('graphify-out/.graphify_incremental.json').read_text(encoding=\"utf-8\")) deleted = list(incremental.get('deleted_files', [])) +# Also prune old nodes for re-extracted (changed) files before inserting fresh AST. +# Without this, build_merge's dedup pass tries to reconcile old and new versions of +# the same file's nodes and can collapse same-named symbols across files (#1178). +changed = [f for files in incremental.get('new_files', {}).values() for f in files] +prune = list(dict.fromkeys(deleted + changed)) or None # Use build_merge() — reads graph.json directly without NetworkX round-trip # so edge direction (calls, implements, imports) is always preserved (#801). G = build_merge( [new_extraction], graph_path='graphify-out/graph.json', - prune_sources=deleted or None, + prune_sources=prune, ) print(f'[graphify update] Merged: {G.number_of_nodes()} nodes, {G.number_of_edges()} edges') diff --git a/graphify/skills/vscode/references/update.md b/graphify/skills/vscode/references/update.md index f53974a..d35b665 100644 --- a/graphify/skills/vscode/references/update.md +++ b/graphify/skills/vscode/references/update.md @@ -93,13 +93,18 @@ from graphify.detect import save_manifest new_extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) incremental = json.loads(Path('graphify-out/.graphify_incremental.json').read_text(encoding=\"utf-8\")) deleted = list(incremental.get('deleted_files', [])) +# Also prune old nodes for re-extracted (changed) files before inserting fresh AST. +# Without this, build_merge's dedup pass tries to reconcile old and new versions of +# the same file's nodes and can collapse same-named symbols across files (#1178). +changed = [f for files in incremental.get('new_files', {}).values() for f in files] +prune = list(dict.fromkeys(deleted + changed)) or None # Use build_merge() — reads graph.json directly without NetworkX round-trip # so edge direction (calls, implements, imports) is always preserved (#801). G = build_merge( [new_extraction], graph_path='graphify-out/graph.json', - prune_sources=deleted or None, + prune_sources=prune, ) print(f'[graphify update] Merged: {G.number_of_nodes()} nodes, {G.number_of_edges()} edges') diff --git a/graphify/skills/windows/references/update.md b/graphify/skills/windows/references/update.md index f53974a..d35b665 100644 --- a/graphify/skills/windows/references/update.md +++ b/graphify/skills/windows/references/update.md @@ -93,13 +93,18 @@ from graphify.detect import save_manifest new_extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) incremental = json.loads(Path('graphify-out/.graphify_incremental.json').read_text(encoding=\"utf-8\")) deleted = list(incremental.get('deleted_files', [])) +# Also prune old nodes for re-extracted (changed) files before inserting fresh AST. +# Without this, build_merge's dedup pass tries to reconcile old and new versions of +# the same file's nodes and can collapse same-named symbols across files (#1178). +changed = [f for files in incremental.get('new_files', {}).values() for f in files] +prune = list(dict.fromkeys(deleted + changed)) or None # Use build_merge() — reads graph.json directly without NetworkX round-trip # so edge direction (calls, implements, imports) is always preserved (#801). G = build_merge( [new_extraction], graph_path='graphify-out/graph.json', - prune_sources=deleted or None, + prune_sources=prune, ) print(f'[graphify update] Merged: {G.number_of_nodes()} nodes, {G.number_of_edges()} edges') diff --git a/tools/skillgen/expected/graphify__skills__amp__references__update.md b/tools/skillgen/expected/graphify__skills__amp__references__update.md index f53974a..d35b665 100644 --- a/tools/skillgen/expected/graphify__skills__amp__references__update.md +++ b/tools/skillgen/expected/graphify__skills__amp__references__update.md @@ -93,13 +93,18 @@ from graphify.detect import save_manifest new_extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) incremental = json.loads(Path('graphify-out/.graphify_incremental.json').read_text(encoding=\"utf-8\")) deleted = list(incremental.get('deleted_files', [])) +# Also prune old nodes for re-extracted (changed) files before inserting fresh AST. +# Without this, build_merge's dedup pass tries to reconcile old and new versions of +# the same file's nodes and can collapse same-named symbols across files (#1178). +changed = [f for files in incremental.get('new_files', {}).values() for f in files] +prune = list(dict.fromkeys(deleted + changed)) or None # Use build_merge() — reads graph.json directly without NetworkX round-trip # so edge direction (calls, implements, imports) is always preserved (#801). G = build_merge( [new_extraction], graph_path='graphify-out/graph.json', - prune_sources=deleted or None, + prune_sources=prune, ) print(f'[graphify update] Merged: {G.number_of_nodes()} nodes, {G.number_of_edges()} edges') diff --git a/tools/skillgen/expected/graphify__skills__claude__references__update.md b/tools/skillgen/expected/graphify__skills__claude__references__update.md index f53974a..d35b665 100644 --- a/tools/skillgen/expected/graphify__skills__claude__references__update.md +++ b/tools/skillgen/expected/graphify__skills__claude__references__update.md @@ -93,13 +93,18 @@ from graphify.detect import save_manifest new_extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) incremental = json.loads(Path('graphify-out/.graphify_incremental.json').read_text(encoding=\"utf-8\")) deleted = list(incremental.get('deleted_files', [])) +# Also prune old nodes for re-extracted (changed) files before inserting fresh AST. +# Without this, build_merge's dedup pass tries to reconcile old and new versions of +# the same file's nodes and can collapse same-named symbols across files (#1178). +changed = [f for files in incremental.get('new_files', {}).values() for f in files] +prune = list(dict.fromkeys(deleted + changed)) or None # Use build_merge() — reads graph.json directly without NetworkX round-trip # so edge direction (calls, implements, imports) is always preserved (#801). G = build_merge( [new_extraction], graph_path='graphify-out/graph.json', - prune_sources=deleted or None, + prune_sources=prune, ) print(f'[graphify update] Merged: {G.number_of_nodes()} nodes, {G.number_of_edges()} edges') diff --git a/tools/skillgen/expected/graphify__skills__claw__references__update.md b/tools/skillgen/expected/graphify__skills__claw__references__update.md index f53974a..d35b665 100644 --- a/tools/skillgen/expected/graphify__skills__claw__references__update.md +++ b/tools/skillgen/expected/graphify__skills__claw__references__update.md @@ -93,13 +93,18 @@ from graphify.detect import save_manifest new_extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) incremental = json.loads(Path('graphify-out/.graphify_incremental.json').read_text(encoding=\"utf-8\")) deleted = list(incremental.get('deleted_files', [])) +# Also prune old nodes for re-extracted (changed) files before inserting fresh AST. +# Without this, build_merge's dedup pass tries to reconcile old and new versions of +# the same file's nodes and can collapse same-named symbols across files (#1178). +changed = [f for files in incremental.get('new_files', {}).values() for f in files] +prune = list(dict.fromkeys(deleted + changed)) or None # Use build_merge() — reads graph.json directly without NetworkX round-trip # so edge direction (calls, implements, imports) is always preserved (#801). G = build_merge( [new_extraction], graph_path='graphify-out/graph.json', - prune_sources=deleted or None, + prune_sources=prune, ) print(f'[graphify update] Merged: {G.number_of_nodes()} nodes, {G.number_of_edges()} edges') diff --git a/tools/skillgen/expected/graphify__skills__codex__references__update.md b/tools/skillgen/expected/graphify__skills__codex__references__update.md index f53974a..d35b665 100644 --- a/tools/skillgen/expected/graphify__skills__codex__references__update.md +++ b/tools/skillgen/expected/graphify__skills__codex__references__update.md @@ -93,13 +93,18 @@ from graphify.detect import save_manifest new_extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) incremental = json.loads(Path('graphify-out/.graphify_incremental.json').read_text(encoding=\"utf-8\")) deleted = list(incremental.get('deleted_files', [])) +# Also prune old nodes for re-extracted (changed) files before inserting fresh AST. +# Without this, build_merge's dedup pass tries to reconcile old and new versions of +# the same file's nodes and can collapse same-named symbols across files (#1178). +changed = [f for files in incremental.get('new_files', {}).values() for f in files] +prune = list(dict.fromkeys(deleted + changed)) or None # Use build_merge() — reads graph.json directly without NetworkX round-trip # so edge direction (calls, implements, imports) is always preserved (#801). G = build_merge( [new_extraction], graph_path='graphify-out/graph.json', - prune_sources=deleted or None, + prune_sources=prune, ) print(f'[graphify update] Merged: {G.number_of_nodes()} nodes, {G.number_of_edges()} edges') diff --git a/tools/skillgen/expected/graphify__skills__copilot__references__update.md b/tools/skillgen/expected/graphify__skills__copilot__references__update.md index f53974a..d35b665 100644 --- a/tools/skillgen/expected/graphify__skills__copilot__references__update.md +++ b/tools/skillgen/expected/graphify__skills__copilot__references__update.md @@ -93,13 +93,18 @@ from graphify.detect import save_manifest new_extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) incremental = json.loads(Path('graphify-out/.graphify_incremental.json').read_text(encoding=\"utf-8\")) deleted = list(incremental.get('deleted_files', [])) +# Also prune old nodes for re-extracted (changed) files before inserting fresh AST. +# Without this, build_merge's dedup pass tries to reconcile old and new versions of +# the same file's nodes and can collapse same-named symbols across files (#1178). +changed = [f for files in incremental.get('new_files', {}).values() for f in files] +prune = list(dict.fromkeys(deleted + changed)) or None # Use build_merge() — reads graph.json directly without NetworkX round-trip # so edge direction (calls, implements, imports) is always preserved (#801). G = build_merge( [new_extraction], graph_path='graphify-out/graph.json', - prune_sources=deleted or None, + prune_sources=prune, ) print(f'[graphify update] Merged: {G.number_of_nodes()} nodes, {G.number_of_edges()} edges') diff --git a/tools/skillgen/expected/graphify__skills__droid__references__update.md b/tools/skillgen/expected/graphify__skills__droid__references__update.md index f53974a..d35b665 100644 --- a/tools/skillgen/expected/graphify__skills__droid__references__update.md +++ b/tools/skillgen/expected/graphify__skills__droid__references__update.md @@ -93,13 +93,18 @@ from graphify.detect import save_manifest new_extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) incremental = json.loads(Path('graphify-out/.graphify_incremental.json').read_text(encoding=\"utf-8\")) deleted = list(incremental.get('deleted_files', [])) +# Also prune old nodes for re-extracted (changed) files before inserting fresh AST. +# Without this, build_merge's dedup pass tries to reconcile old and new versions of +# the same file's nodes and can collapse same-named symbols across files (#1178). +changed = [f for files in incremental.get('new_files', {}).values() for f in files] +prune = list(dict.fromkeys(deleted + changed)) or None # Use build_merge() — reads graph.json directly without NetworkX round-trip # so edge direction (calls, implements, imports) is always preserved (#801). G = build_merge( [new_extraction], graph_path='graphify-out/graph.json', - prune_sources=deleted or None, + prune_sources=prune, ) print(f'[graphify update] Merged: {G.number_of_nodes()} nodes, {G.number_of_edges()} edges') diff --git a/tools/skillgen/expected/graphify__skills__kilo__references__update.md b/tools/skillgen/expected/graphify__skills__kilo__references__update.md index f53974a..d35b665 100644 --- a/tools/skillgen/expected/graphify__skills__kilo__references__update.md +++ b/tools/skillgen/expected/graphify__skills__kilo__references__update.md @@ -93,13 +93,18 @@ from graphify.detect import save_manifest new_extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) incremental = json.loads(Path('graphify-out/.graphify_incremental.json').read_text(encoding=\"utf-8\")) deleted = list(incremental.get('deleted_files', [])) +# Also prune old nodes for re-extracted (changed) files before inserting fresh AST. +# Without this, build_merge's dedup pass tries to reconcile old and new versions of +# the same file's nodes and can collapse same-named symbols across files (#1178). +changed = [f for files in incremental.get('new_files', {}).values() for f in files] +prune = list(dict.fromkeys(deleted + changed)) or None # Use build_merge() — reads graph.json directly without NetworkX round-trip # so edge direction (calls, implements, imports) is always preserved (#801). G = build_merge( [new_extraction], graph_path='graphify-out/graph.json', - prune_sources=deleted or None, + prune_sources=prune, ) print(f'[graphify update] Merged: {G.number_of_nodes()} nodes, {G.number_of_edges()} edges') diff --git a/tools/skillgen/expected/graphify__skills__kiro__references__update.md b/tools/skillgen/expected/graphify__skills__kiro__references__update.md index f53974a..d35b665 100644 --- a/tools/skillgen/expected/graphify__skills__kiro__references__update.md +++ b/tools/skillgen/expected/graphify__skills__kiro__references__update.md @@ -93,13 +93,18 @@ from graphify.detect import save_manifest new_extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) incremental = json.loads(Path('graphify-out/.graphify_incremental.json').read_text(encoding=\"utf-8\")) deleted = list(incremental.get('deleted_files', [])) +# Also prune old nodes for re-extracted (changed) files before inserting fresh AST. +# Without this, build_merge's dedup pass tries to reconcile old and new versions of +# the same file's nodes and can collapse same-named symbols across files (#1178). +changed = [f for files in incremental.get('new_files', {}).values() for f in files] +prune = list(dict.fromkeys(deleted + changed)) or None # Use build_merge() — reads graph.json directly without NetworkX round-trip # so edge direction (calls, implements, imports) is always preserved (#801). G = build_merge( [new_extraction], graph_path='graphify-out/graph.json', - prune_sources=deleted or None, + prune_sources=prune, ) print(f'[graphify update] Merged: {G.number_of_nodes()} nodes, {G.number_of_edges()} edges') diff --git a/tools/skillgen/expected/graphify__skills__opencode__references__update.md b/tools/skillgen/expected/graphify__skills__opencode__references__update.md index f53974a..d35b665 100644 --- a/tools/skillgen/expected/graphify__skills__opencode__references__update.md +++ b/tools/skillgen/expected/graphify__skills__opencode__references__update.md @@ -93,13 +93,18 @@ from graphify.detect import save_manifest new_extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) incremental = json.loads(Path('graphify-out/.graphify_incremental.json').read_text(encoding=\"utf-8\")) deleted = list(incremental.get('deleted_files', [])) +# Also prune old nodes for re-extracted (changed) files before inserting fresh AST. +# Without this, build_merge's dedup pass tries to reconcile old and new versions of +# the same file's nodes and can collapse same-named symbols across files (#1178). +changed = [f for files in incremental.get('new_files', {}).values() for f in files] +prune = list(dict.fromkeys(deleted + changed)) or None # Use build_merge() — reads graph.json directly without NetworkX round-trip # so edge direction (calls, implements, imports) is always preserved (#801). G = build_merge( [new_extraction], graph_path='graphify-out/graph.json', - prune_sources=deleted or None, + prune_sources=prune, ) print(f'[graphify update] Merged: {G.number_of_nodes()} nodes, {G.number_of_edges()} edges') diff --git a/tools/skillgen/expected/graphify__skills__pi__references__update.md b/tools/skillgen/expected/graphify__skills__pi__references__update.md index f53974a..d35b665 100644 --- a/tools/skillgen/expected/graphify__skills__pi__references__update.md +++ b/tools/skillgen/expected/graphify__skills__pi__references__update.md @@ -93,13 +93,18 @@ from graphify.detect import save_manifest new_extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) incremental = json.loads(Path('graphify-out/.graphify_incremental.json').read_text(encoding=\"utf-8\")) deleted = list(incremental.get('deleted_files', [])) +# Also prune old nodes for re-extracted (changed) files before inserting fresh AST. +# Without this, build_merge's dedup pass tries to reconcile old and new versions of +# the same file's nodes and can collapse same-named symbols across files (#1178). +changed = [f for files in incremental.get('new_files', {}).values() for f in files] +prune = list(dict.fromkeys(deleted + changed)) or None # Use build_merge() — reads graph.json directly without NetworkX round-trip # so edge direction (calls, implements, imports) is always preserved (#801). G = build_merge( [new_extraction], graph_path='graphify-out/graph.json', - prune_sources=deleted or None, + prune_sources=prune, ) print(f'[graphify update] Merged: {G.number_of_nodes()} nodes, {G.number_of_edges()} edges') diff --git a/tools/skillgen/expected/graphify__skills__trae__references__update.md b/tools/skillgen/expected/graphify__skills__trae__references__update.md index f53974a..d35b665 100644 --- a/tools/skillgen/expected/graphify__skills__trae__references__update.md +++ b/tools/skillgen/expected/graphify__skills__trae__references__update.md @@ -93,13 +93,18 @@ from graphify.detect import save_manifest new_extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) incremental = json.loads(Path('graphify-out/.graphify_incremental.json').read_text(encoding=\"utf-8\")) deleted = list(incremental.get('deleted_files', [])) +# Also prune old nodes for re-extracted (changed) files before inserting fresh AST. +# Without this, build_merge's dedup pass tries to reconcile old and new versions of +# the same file's nodes and can collapse same-named symbols across files (#1178). +changed = [f for files in incremental.get('new_files', {}).values() for f in files] +prune = list(dict.fromkeys(deleted + changed)) or None # Use build_merge() — reads graph.json directly without NetworkX round-trip # so edge direction (calls, implements, imports) is always preserved (#801). G = build_merge( [new_extraction], graph_path='graphify-out/graph.json', - prune_sources=deleted or None, + prune_sources=prune, ) print(f'[graphify update] Merged: {G.number_of_nodes()} nodes, {G.number_of_edges()} edges') diff --git a/tools/skillgen/expected/graphify__skills__vscode__references__update.md b/tools/skillgen/expected/graphify__skills__vscode__references__update.md index f53974a..d35b665 100644 --- a/tools/skillgen/expected/graphify__skills__vscode__references__update.md +++ b/tools/skillgen/expected/graphify__skills__vscode__references__update.md @@ -93,13 +93,18 @@ from graphify.detect import save_manifest new_extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) incremental = json.loads(Path('graphify-out/.graphify_incremental.json').read_text(encoding=\"utf-8\")) deleted = list(incremental.get('deleted_files', [])) +# Also prune old nodes for re-extracted (changed) files before inserting fresh AST. +# Without this, build_merge's dedup pass tries to reconcile old and new versions of +# the same file's nodes and can collapse same-named symbols across files (#1178). +changed = [f for files in incremental.get('new_files', {}).values() for f in files] +prune = list(dict.fromkeys(deleted + changed)) or None # Use build_merge() — reads graph.json directly without NetworkX round-trip # so edge direction (calls, implements, imports) is always preserved (#801). G = build_merge( [new_extraction], graph_path='graphify-out/graph.json', - prune_sources=deleted or None, + prune_sources=prune, ) print(f'[graphify update] Merged: {G.number_of_nodes()} nodes, {G.number_of_edges()} edges') diff --git a/tools/skillgen/expected/graphify__skills__windows__references__update.md b/tools/skillgen/expected/graphify__skills__windows__references__update.md index f53974a..d35b665 100644 --- a/tools/skillgen/expected/graphify__skills__windows__references__update.md +++ b/tools/skillgen/expected/graphify__skills__windows__references__update.md @@ -93,13 +93,18 @@ from graphify.detect import save_manifest new_extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) incremental = json.loads(Path('graphify-out/.graphify_incremental.json').read_text(encoding=\"utf-8\")) deleted = list(incremental.get('deleted_files', [])) +# Also prune old nodes for re-extracted (changed) files before inserting fresh AST. +# Without this, build_merge's dedup pass tries to reconcile old and new versions of +# the same file's nodes and can collapse same-named symbols across files (#1178). +changed = [f for files in incremental.get('new_files', {}).values() for f in files] +prune = list(dict.fromkeys(deleted + changed)) or None # Use build_merge() — reads graph.json directly without NetworkX round-trip # so edge direction (calls, implements, imports) is always preserved (#801). G = build_merge( [new_extraction], graph_path='graphify-out/graph.json', - prune_sources=deleted or None, + prune_sources=prune, ) print(f'[graphify update] Merged: {G.number_of_nodes()} nodes, {G.number_of_edges()} edges') diff --git a/tools/skillgen/fragments/references/shared/update.md b/tools/skillgen/fragments/references/shared/update.md index f53974a..d35b665 100644 --- a/tools/skillgen/fragments/references/shared/update.md +++ b/tools/skillgen/fragments/references/shared/update.md @@ -93,13 +93,18 @@ from graphify.detect import save_manifest new_extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) incremental = json.loads(Path('graphify-out/.graphify_incremental.json').read_text(encoding=\"utf-8\")) deleted = list(incremental.get('deleted_files', [])) +# Also prune old nodes for re-extracted (changed) files before inserting fresh AST. +# Without this, build_merge's dedup pass tries to reconcile old and new versions of +# the same file's nodes and can collapse same-named symbols across files (#1178). +changed = [f for files in incremental.get('new_files', {}).values() for f in files] +prune = list(dict.fromkeys(deleted + changed)) or None # Use build_merge() — reads graph.json directly without NetworkX round-trip # so edge direction (calls, implements, imports) is always preserved (#801). G = build_merge( [new_extraction], graph_path='graphify-out/graph.json', - prune_sources=deleted or None, + prune_sources=prune, ) print(f'[graphify update] Merged: {G.number_of_nodes()} nodes, {G.number_of_edges()} edges')