Collapse Swift module imports to one shared node

Adopts the approach from #1330 (thanks @duncan-daydream) on top of the v0.8.40
Swift import fix: _import_swift returns (id,label) module pairs, the extractor
materializes a type=module anchor node per import, and _disambiguate_colliding_node_ids
exempts type=module nodes so the same module imported from N files collapses to
one shared node (enables reverse traversal "what imports CoreKit"). The --no-cluster
writer now dedupes nodes by id and edges to match the clustered build_from_json path.

Replaces the interim _import_label/synthesize_import_module_nodes mechanism.
Adds tests/test_swift_import_resolution.py (cross-file collapse, build survival)
and dedupe_nodes coverage. Refs #1327, #1330.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
Safi
2026-06-16 03:06:43 +01:00
co-authored by Claude Opus 4.8
parent fd152add64
commit 4fbb33301c
8 changed files with 174 additions and 33 deletions
+2
View File
@@ -4,6 +4,8 @@ Full release notes with details on each version: [GitHub Releases](https://githu
## Unreleased
- Fix: Swift imports of the same module from multiple files now collapse to a single shared `type=module` node instead of N path-qualified duplicates. The import target is tagged `type=module` and exempted from id-disambiguation, so reverse traversal ("what imports CoreKit?") works; the `--no-cluster` writer also now dedupes nodes by id (and edges) to match the clustered `build_from_json` path. Builds on the v0.8.40 Swift-import fix (#1327, #1330; thanks @duncan-daydream).
## 0.8.40 (2026-06-16)
- Feat: custom OpenAI- and Anthropic-compatible endpoints via `OPENAI_BASE_URL`/`OPENAI_MODEL` and `ANTHROPIC_BASE_URL`/`ANTHROPIC_MODEL`. Point either backend at a self-hosted or proxy server (vLLM, llama.cpp, LM Studio, LiteLLM, gateways); defaults still resolve to `api.openai.com` / `api.anthropic.com`, and `GRAPHIFY_OPENAI_MODEL` keeps precedence over `OPENAI_MODEL`. Wired through both the extraction path (`_call_claude`) and community labeling (#1273).
+6 -3
View File
@@ -4425,10 +4425,13 @@ def main() -> None:
if no_cluster:
# --no-cluster: dump the raw merged extraction as graph.json.
# No NetworkX, no community detection, no analysis sidecar.
# Dedupe parallel edges so counts match the clustered path (whose
# DiGraph collapses them) and stay deterministic across modes (#1317).
from graphify.build import dedupe_edges as _dedupe_edges
# Dedupe nodes (by id) and parallel edges so the raw output matches the
# clustered path (whose DiGraph collapses both) and stays deterministic
# across modes (#1317; node dedup also collapses shared Swift module
# anchors emitted per importing file, #1327).
from graphify.build import dedupe_edges as _dedupe_edges, dedupe_nodes as _dedupe_nodes
from graphify.export import backup_if_protected as _backup
merged["nodes"] = _dedupe_nodes(merged["nodes"])
merged["edges"] = _dedupe_edges(merged["edges"])
_backup(graphify_out)
graph_json_path.write_text(
+19
View File
@@ -104,6 +104,25 @@ def edge_datas(G: nx.Graph, u: str, v: str) -> list[dict]:
return [raw]
def dedupe_nodes(nodes: list[dict]) -> list[dict]:
"""Collapse nodes sharing an ``id``, last-writer-wins on attributes.
Mirrors what ``build_from_json``'s ``G.add_node`` does implicitly (idempotent;
a later node overwrites an earlier one's attributes). The ``--no-cluster``
write path dumps the raw node list without building a graph, so same-id nodes
— e.g. a Swift ``type=module`` anchor emitted once per importing file (#1327)
— would otherwise appear as duplicates. Insertion order follows each id's
first appearance; the retained dict is the last one seen.
"""
by_id: dict = {}
for n in nodes:
nid = n.get("id")
if nid is None:
continue
by_id[nid] = n
return list(by_id.values())
def dedupe_edges(edges: list[dict]) -> list[dict]:
"""Collapse exact parallel edges by ``(source, target, relation)``, keeping the
first occurrence.
+40 -27
View File
@@ -466,12 +466,6 @@ class LanguageConfig:
# Extra walk hook called after generic dispatch (for JS arrow functions, C# namespaces, etc.)
extra_walk_fn: Callable | None = None
# When True, synthesize a node for each `imports` edge target an import_handler
# emits (carrying an `_import_label` key). Languages whose imports name modules
# rather than resolvable files (e.g. Swift `import CoreKit`) need this, else the
# edge is pruned in build.py for pointing at a non-existent target node.
synthesize_import_module_nodes: bool = False
# ── Generic helpers ───────────────────────────────────────────────────────────
@@ -2221,7 +2215,16 @@ _LUA_CONFIG = LanguageConfig(
)
def _import_swift(node, source: bytes, file_nid: str, stem: str, edges: list, str_path: str) -> None:
def _import_swift(node, source: bytes, file_nid: str, stem: str, edges: list, str_path: str) -> list[tuple[str, str]]:
"""Emit module-level ``imports`` edges and report the imported modules.
A Swift ``import CoreKit`` names a module, not a file path, so — unlike the
file-resolving JS/TS handlers — there is no existing node for the edge to
point at. The returned ``(id, label)`` pairs let the extractor materialize a
``type=module`` anchor node so the edge survives; without it ``build_from_json``
prunes every Swift import edge as a dangling/external reference (#1327).
"""
modules: list[tuple[str, str]] = []
for child in node.children:
if child.type == "identifier":
raw = _read_text(child, source)
@@ -2235,11 +2238,10 @@ def _import_swift(node, source: bytes, file_nid: str, stem: str, edges: list, st
"source_file": str_path,
"source_location": f"L{node.start_point[0] + 1}",
"weight": 1.0,
# Consumed by import-node synthesis at the walk() call site
# (LanguageConfig.synthesize_import_module_nodes); see #1327.
"_import_label": raw,
})
modules.append((tgt_nid, raw))
break
return modules
def _read_csharp_type_name(node, source: bytes) -> str | None:
@@ -2276,7 +2278,6 @@ _SWIFT_CONFIG = LanguageConfig(
body_fallback_child_types=("class_body", "protocol_body", "function_body", "enum_class_body"),
function_boundary_types=frozenset({"function_declaration", "init_declaration", "deinit_declaration", "subscript_declaration"}),
import_handler=_import_swift,
synthesize_import_module_nodes=True,
)
# ── Generic extractor ─────────────────────────────────────────────────────────
@@ -2382,24 +2383,27 @@ def _extract_generic(path: Path, config: LanguageConfig) -> dict:
# Import types
if t in config.import_types:
if config.import_handler:
_imp_before = len(edges)
config.import_handler(node, source, file_nid, stem, edges, str_path)
if config.synthesize_import_module_nodes:
# Imports that name a module (not a resolvable file) point at a
# synthetic target node; create it so build.py keeps the edge (#1327).
for _e in edges[_imp_before:]:
_lbl = _e.pop("_import_label", None)
if _lbl is None or _e.get("relation") != "imports":
continue
_tgt = _e["target"]
if _tgt not in seen_ids:
seen_ids.add(_tgt)
imported_modules = config.import_handler(node, source, file_nid, stem, edges, str_path)
# Module-level import handlers (Swift) name a module, not a file
# path, so there is no pre-existing node to anchor the edge to.
# They return (id, label) pairs for which we materialize a
# `type=module` node; otherwise build_from_json prunes every such
# import edge as a dangling/external reference. The same module
# imported from N files shares one id (file_type=code keeps
# build.py validation happy; `type=module` exempts it from
# id-disambiguation) so it collapses to one shared node (#1327).
if imported_modules:
line = node.start_point[0] + 1
for mod_nid, mod_label in imported_modules:
if mod_nid not in seen_ids:
seen_ids.add(mod_nid)
nodes.append({
"id": _tgt,
"label": _lbl,
"id": mod_nid,
"label": mod_label,
"file_type": "code",
"type": "module",
"source_file": str_path,
"source_location": _e.get("source_location", "L1"),
"source_location": f"L{line}",
})
# For export_statement: only return (skip children) if it's a re-export
# (has a `from` source). Otherwise fall through to walk children which may
@@ -7161,9 +7165,18 @@ def _disambiguate_colliding_node_ids(
raw_calls: list[dict],
root: Path,
) -> None:
"""Rewrite only colliding node IDs, using source path as the disambiguator."""
"""Rewrite only colliding node IDs, using source path as the disambiguator.
Module anchor nodes (#1327) are exempt: ``import CoreKit`` from three files
yields three ``type=module`` nodes with the same id but different
source_files. Those are the *same* module, not distinct same-named symbols,
so they must collapse to one shared node — disambiguating them by path would
scatter a single module across N file-qualified duplicates.
"""
by_id: dict[str, list[dict]] = {}
for node in nodes:
if node.get("type") == "module":
continue
nid = node.get("id")
if isinstance(nid, str) and nid:
by_id.setdefault(nid, []).append(node)
+3 -2
View File
@@ -587,9 +587,10 @@ def _rebuild_code(
# Dedupe parallel edges (the clustered path's DiGraph collapses them implicitly);
# without it, --no-cluster + repeated `update` accumulate duplicates and edge
# counts diverge across build modes (#1317).
from graphify.build import dedupe_edges as _dedupe_edges
from graphify.build import dedupe_edges as _dedupe_edges, dedupe_nodes as _dedupe_nodes
candidate_graph_data = {
**{k: v for k, v in result.items() if k != "edges"},
**{k: v for k, v in result.items() if k not in ("edges", "nodes")},
"nodes": _dedupe_nodes(result.get("nodes", [])),
"links": _dedupe_edges(result.get("edges", [])),
}
candidate_graph_text = _json_text(candidate_graph_data)
+16 -1
View File
@@ -2,7 +2,7 @@ import json
from pathlib import Path
import networkx as nx
from networkx.readwrite import json_graph
from graphify.build import build_from_json, build, build_merge, edge_data, edge_datas, dedupe_edges
from graphify.build import build_from_json, build, build_merge, edge_data, edge_datas, dedupe_edges, dedupe_nodes
FIXTURES = Path(__file__).parent / "fixtures"
@@ -32,6 +32,21 @@ def test_dedupe_edges_is_idempotent():
assert len(once) == 1
assert len(twice) == 1
def test_dedupe_nodes_collapses_by_id_last_wins():
# #1327: a shared module anchor is emitted once per importing file; the
# --no-cluster raw writer must collapse same-id node dicts (#1317).
nodes = [
{"id": "foundation", "label": "Foundation", "type": "module", "source_file": "A.swift"},
{"id": "akit", "label": "AKit", "file_type": "code"},
{"id": "foundation", "label": "Foundation", "type": "module", "source_file": "B.swift"},
]
out = dedupe_nodes(nodes)
ids = [n["id"] for n in out]
assert ids == ["foundation", "akit"] # first-appearance order
# last writer wins on attributes
assert next(n for n in out if n["id"] == "foundation")["source_file"] == "B.swift"
def load_extraction():
return json.loads((FIXTURES / "extraction.json").read_text())
+3
View File
@@ -610,6 +610,9 @@ def test_swift_imports_survive_build():
node_ids = {n["id"] for n in r["nodes"]}
for e in import_edges:
assert e["target"] in node_ids # synthesized module node exists
# Imported modules are tagged type=module (anchor nodes, #1327/#1330).
module_labels = {n["label"] for n in r["nodes"] if n.get("type") == "module"}
assert {"Foundation", "UIKit"} <= module_labels
# No private bookkeeping key should leak into output edges.
assert all("_import_label" not in e for e in r["edges"])
# Edges must survive the build (which prunes edges with unknown endpoints).
+85
View File
@@ -0,0 +1,85 @@
from __future__ import annotations
from pathlib import Path
from graphify.build import build_from_json
from graphify.extract import extract
def _write(path: Path, text: str) -> Path:
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(text, encoding="utf-8")
return path
def _module_nodes(result: dict, label: str) -> list[dict]:
return [
n for n in result["nodes"]
if n.get("type") == "module" and n.get("label") == label
]
def _import_edges(result: dict) -> list[dict]:
return [e for e in result["edges"] if e.get("relation") == "imports"]
def test_swift_import_resolves_to_module_node(tmp_path: Path):
# #1327: `import CoreKit` must anchor to a node, or build_from_json prunes
# the edge as a dangling/external reference.
core = _write(tmp_path / "Sources/CoreKit/CoreKit.swift", "public struct CoreKit {}\n")
feature = _write(
tmp_path / "Sources/FeatureKit/FeatureKit.swift",
"import CoreKit\n\npublic struct FeatureKit {}\n",
)
result = extract([core, feature], cache_root=tmp_path)
node_ids = {n["id"] for n in result["nodes"]}
imports = _import_edges(result)
assert imports
for e in imports:
assert e["target"] in node_ids
assert _module_nodes(result, "CoreKit")
def test_swift_same_module_imported_twice_collapses_to_one_node(tmp_path: Path):
# #1327: the same module imported from multiple files is ONE module, not N
# file-qualified duplicates — collision-disambiguation must exempt modules.
core = _write(tmp_path / "Sources/CoreKit/CoreKit.swift", "public struct CoreKit {}\n")
a = _write(
tmp_path / "Sources/AKit/AKit.swift",
"import CoreKit\n\npublic struct AKit {}\n",
)
b = _write(
tmp_path / "Sources/BKit/BKit.swift",
"import CoreKit\n\npublic struct BKit {}\n",
)
result = extract([core, a, b], cache_root=tmp_path)
# Each importing file contributes a module-node dict, but they must share a
# single id (NOT be split into path-qualified duplicates) so build_from_json
# collapses them into one shared node.
core_modules = _module_nodes(result, "CoreKit")
module_ids = {n["id"] for n in core_modules}
assert len(module_ids) == 1
# Both importers point at that single shared module id.
import_targets = {e["target"] for e in _import_edges(result)}
assert import_targets == module_ids
def test_swift_import_edges_survive_build(tmp_path: Path):
# #1327: edges must remain after graph assembly, deduped to one module node.
core = _write(tmp_path / "Sources/CoreKit/CoreKit.swift", "public struct CoreKit {}\n")
a = _write(tmp_path / "Sources/AKit/AKit.swift", "import CoreKit\n")
b = _write(tmp_path / "Sources/BKit/BKit.swift", "import CoreKit\n")
result = extract([core, a, b], cache_root=tmp_path)
G = build_from_json(result, directed=True)
import_edges = [
(u, v) for u, v, d in G.edges(data=True) if d.get("relation") == "imports"
]
assert len(import_edges) == 2
# Both edges land on the same CoreKit module node.
assert len({v for _, v in import_edges}) == 1