Collapse Swift module imports to one shared node
Adopts the approach from #1330 (thanks @duncan-daydream) on top of the v0.8.40 Swift import fix: _import_swift returns (id,label) module pairs, the extractor materializes a type=module anchor node per import, and _disambiguate_colliding_node_ids exempts type=module nodes so the same module imported from N files collapses to one shared node (enables reverse traversal "what imports CoreKit"). The --no-cluster writer now dedupes nodes by id and edges to match the clustered build_from_json path. Replaces the interim _import_label/synthesize_import_module_nodes mechanism. Adds tests/test_swift_import_resolution.py (cross-file collapse, build survival) and dedupe_nodes coverage. Refs #1327, #1330. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
@@ -4,6 +4,8 @@ Full release notes with details on each version: [GitHub Releases](https://githu
|
||||
|
||||
## Unreleased
|
||||
|
||||
- Fix: Swift imports of the same module from multiple files now collapse to a single shared `type=module` node instead of N path-qualified duplicates. The import target is tagged `type=module` and exempted from id-disambiguation, so reverse traversal ("what imports CoreKit?") works; the `--no-cluster` writer also now dedupes nodes by id (and edges) to match the clustered `build_from_json` path. Builds on the v0.8.40 Swift-import fix (#1327, #1330; thanks @duncan-daydream).
|
||||
|
||||
## 0.8.40 (2026-06-16)
|
||||
|
||||
- Feat: custom OpenAI- and Anthropic-compatible endpoints via `OPENAI_BASE_URL`/`OPENAI_MODEL` and `ANTHROPIC_BASE_URL`/`ANTHROPIC_MODEL`. Point either backend at a self-hosted or proxy server (vLLM, llama.cpp, LM Studio, LiteLLM, gateways); defaults still resolve to `api.openai.com` / `api.anthropic.com`, and `GRAPHIFY_OPENAI_MODEL` keeps precedence over `OPENAI_MODEL`. Wired through both the extraction path (`_call_claude`) and community labeling (#1273).
|
||||
|
||||
@@ -4425,10 +4425,13 @@ def main() -> None:
|
||||
if no_cluster:
|
||||
# --no-cluster: dump the raw merged extraction as graph.json.
|
||||
# No NetworkX, no community detection, no analysis sidecar.
|
||||
# Dedupe parallel edges so counts match the clustered path (whose
|
||||
# DiGraph collapses them) and stay deterministic across modes (#1317).
|
||||
from graphify.build import dedupe_edges as _dedupe_edges
|
||||
# Dedupe nodes (by id) and parallel edges so the raw output matches the
|
||||
# clustered path (whose DiGraph collapses both) and stays deterministic
|
||||
# across modes (#1317; node dedup also collapses shared Swift module
|
||||
# anchors emitted per importing file, #1327).
|
||||
from graphify.build import dedupe_edges as _dedupe_edges, dedupe_nodes as _dedupe_nodes
|
||||
from graphify.export import backup_if_protected as _backup
|
||||
merged["nodes"] = _dedupe_nodes(merged["nodes"])
|
||||
merged["edges"] = _dedupe_edges(merged["edges"])
|
||||
_backup(graphify_out)
|
||||
graph_json_path.write_text(
|
||||
|
||||
@@ -104,6 +104,25 @@ def edge_datas(G: nx.Graph, u: str, v: str) -> list[dict]:
|
||||
return [raw]
|
||||
|
||||
|
||||
def dedupe_nodes(nodes: list[dict]) -> list[dict]:
|
||||
"""Collapse nodes sharing an ``id``, last-writer-wins on attributes.
|
||||
|
||||
Mirrors what ``build_from_json``'s ``G.add_node`` does implicitly (idempotent;
|
||||
a later node overwrites an earlier one's attributes). The ``--no-cluster``
|
||||
write path dumps the raw node list without building a graph, so same-id nodes
|
||||
— e.g. a Swift ``type=module`` anchor emitted once per importing file (#1327)
|
||||
— would otherwise appear as duplicates. Insertion order follows each id's
|
||||
first appearance; the retained dict is the last one seen.
|
||||
"""
|
||||
by_id: dict = {}
|
||||
for n in nodes:
|
||||
nid = n.get("id")
|
||||
if nid is None:
|
||||
continue
|
||||
by_id[nid] = n
|
||||
return list(by_id.values())
|
||||
|
||||
|
||||
def dedupe_edges(edges: list[dict]) -> list[dict]:
|
||||
"""Collapse exact parallel edges by ``(source, target, relation)``, keeping the
|
||||
first occurrence.
|
||||
|
||||
+40
-27
@@ -466,12 +466,6 @@ class LanguageConfig:
|
||||
# Extra walk hook called after generic dispatch (for JS arrow functions, C# namespaces, etc.)
|
||||
extra_walk_fn: Callable | None = None
|
||||
|
||||
# When True, synthesize a node for each `imports` edge target an import_handler
|
||||
# emits (carrying an `_import_label` key). Languages whose imports name modules
|
||||
# rather than resolvable files (e.g. Swift `import CoreKit`) need this, else the
|
||||
# edge is pruned in build.py for pointing at a non-existent target node.
|
||||
synthesize_import_module_nodes: bool = False
|
||||
|
||||
|
||||
# ── Generic helpers ───────────────────────────────────────────────────────────
|
||||
|
||||
@@ -2221,7 +2215,16 @@ _LUA_CONFIG = LanguageConfig(
|
||||
)
|
||||
|
||||
|
||||
def _import_swift(node, source: bytes, file_nid: str, stem: str, edges: list, str_path: str) -> None:
|
||||
def _import_swift(node, source: bytes, file_nid: str, stem: str, edges: list, str_path: str) -> list[tuple[str, str]]:
|
||||
"""Emit module-level ``imports`` edges and report the imported modules.
|
||||
|
||||
A Swift ``import CoreKit`` names a module, not a file path, so — unlike the
|
||||
file-resolving JS/TS handlers — there is no existing node for the edge to
|
||||
point at. The returned ``(id, label)`` pairs let the extractor materialize a
|
||||
``type=module`` anchor node so the edge survives; without it ``build_from_json``
|
||||
prunes every Swift import edge as a dangling/external reference (#1327).
|
||||
"""
|
||||
modules: list[tuple[str, str]] = []
|
||||
for child in node.children:
|
||||
if child.type == "identifier":
|
||||
raw = _read_text(child, source)
|
||||
@@ -2235,11 +2238,10 @@ def _import_swift(node, source: bytes, file_nid: str, stem: str, edges: list, st
|
||||
"source_file": str_path,
|
||||
"source_location": f"L{node.start_point[0] + 1}",
|
||||
"weight": 1.0,
|
||||
# Consumed by import-node synthesis at the walk() call site
|
||||
# (LanguageConfig.synthesize_import_module_nodes); see #1327.
|
||||
"_import_label": raw,
|
||||
})
|
||||
modules.append((tgt_nid, raw))
|
||||
break
|
||||
return modules
|
||||
|
||||
|
||||
def _read_csharp_type_name(node, source: bytes) -> str | None:
|
||||
@@ -2276,7 +2278,6 @@ _SWIFT_CONFIG = LanguageConfig(
|
||||
body_fallback_child_types=("class_body", "protocol_body", "function_body", "enum_class_body"),
|
||||
function_boundary_types=frozenset({"function_declaration", "init_declaration", "deinit_declaration", "subscript_declaration"}),
|
||||
import_handler=_import_swift,
|
||||
synthesize_import_module_nodes=True,
|
||||
)
|
||||
|
||||
# ── Generic extractor ─────────────────────────────────────────────────────────
|
||||
@@ -2382,24 +2383,27 @@ def _extract_generic(path: Path, config: LanguageConfig) -> dict:
|
||||
# Import types
|
||||
if t in config.import_types:
|
||||
if config.import_handler:
|
||||
_imp_before = len(edges)
|
||||
config.import_handler(node, source, file_nid, stem, edges, str_path)
|
||||
if config.synthesize_import_module_nodes:
|
||||
# Imports that name a module (not a resolvable file) point at a
|
||||
# synthetic target node; create it so build.py keeps the edge (#1327).
|
||||
for _e in edges[_imp_before:]:
|
||||
_lbl = _e.pop("_import_label", None)
|
||||
if _lbl is None or _e.get("relation") != "imports":
|
||||
continue
|
||||
_tgt = _e["target"]
|
||||
if _tgt not in seen_ids:
|
||||
seen_ids.add(_tgt)
|
||||
imported_modules = config.import_handler(node, source, file_nid, stem, edges, str_path)
|
||||
# Module-level import handlers (Swift) name a module, not a file
|
||||
# path, so there is no pre-existing node to anchor the edge to.
|
||||
# They return (id, label) pairs for which we materialize a
|
||||
# `type=module` node; otherwise build_from_json prunes every such
|
||||
# import edge as a dangling/external reference. The same module
|
||||
# imported from N files shares one id (file_type=code keeps
|
||||
# build.py validation happy; `type=module` exempts it from
|
||||
# id-disambiguation) so it collapses to one shared node (#1327).
|
||||
if imported_modules:
|
||||
line = node.start_point[0] + 1
|
||||
for mod_nid, mod_label in imported_modules:
|
||||
if mod_nid not in seen_ids:
|
||||
seen_ids.add(mod_nid)
|
||||
nodes.append({
|
||||
"id": _tgt,
|
||||
"label": _lbl,
|
||||
"id": mod_nid,
|
||||
"label": mod_label,
|
||||
"file_type": "code",
|
||||
"type": "module",
|
||||
"source_file": str_path,
|
||||
"source_location": _e.get("source_location", "L1"),
|
||||
"source_location": f"L{line}",
|
||||
})
|
||||
# For export_statement: only return (skip children) if it's a re-export
|
||||
# (has a `from` source). Otherwise fall through to walk children which may
|
||||
@@ -7161,9 +7165,18 @@ def _disambiguate_colliding_node_ids(
|
||||
raw_calls: list[dict],
|
||||
root: Path,
|
||||
) -> None:
|
||||
"""Rewrite only colliding node IDs, using source path as the disambiguator."""
|
||||
"""Rewrite only colliding node IDs, using source path as the disambiguator.
|
||||
|
||||
Module anchor nodes (#1327) are exempt: ``import CoreKit`` from three files
|
||||
yields three ``type=module`` nodes with the same id but different
|
||||
source_files. Those are the *same* module, not distinct same-named symbols,
|
||||
so they must collapse to one shared node — disambiguating them by path would
|
||||
scatter a single module across N file-qualified duplicates.
|
||||
"""
|
||||
by_id: dict[str, list[dict]] = {}
|
||||
for node in nodes:
|
||||
if node.get("type") == "module":
|
||||
continue
|
||||
nid = node.get("id")
|
||||
if isinstance(nid, str) and nid:
|
||||
by_id.setdefault(nid, []).append(node)
|
||||
|
||||
+3
-2
@@ -587,9 +587,10 @@ def _rebuild_code(
|
||||
# Dedupe parallel edges (the clustered path's DiGraph collapses them implicitly);
|
||||
# without it, --no-cluster + repeated `update` accumulate duplicates and edge
|
||||
# counts diverge across build modes (#1317).
|
||||
from graphify.build import dedupe_edges as _dedupe_edges
|
||||
from graphify.build import dedupe_edges as _dedupe_edges, dedupe_nodes as _dedupe_nodes
|
||||
candidate_graph_data = {
|
||||
**{k: v for k, v in result.items() if k != "edges"},
|
||||
**{k: v for k, v in result.items() if k not in ("edges", "nodes")},
|
||||
"nodes": _dedupe_nodes(result.get("nodes", [])),
|
||||
"links": _dedupe_edges(result.get("edges", [])),
|
||||
}
|
||||
candidate_graph_text = _json_text(candidate_graph_data)
|
||||
|
||||
+16
-1
@@ -2,7 +2,7 @@ import json
|
||||
from pathlib import Path
|
||||
import networkx as nx
|
||||
from networkx.readwrite import json_graph
|
||||
from graphify.build import build_from_json, build, build_merge, edge_data, edge_datas, dedupe_edges
|
||||
from graphify.build import build_from_json, build, build_merge, edge_data, edge_datas, dedupe_edges, dedupe_nodes
|
||||
|
||||
FIXTURES = Path(__file__).parent / "fixtures"
|
||||
|
||||
@@ -32,6 +32,21 @@ def test_dedupe_edges_is_idempotent():
|
||||
assert len(once) == 1
|
||||
assert len(twice) == 1
|
||||
|
||||
|
||||
def test_dedupe_nodes_collapses_by_id_last_wins():
|
||||
# #1327: a shared module anchor is emitted once per importing file; the
|
||||
# --no-cluster raw writer must collapse same-id node dicts (#1317).
|
||||
nodes = [
|
||||
{"id": "foundation", "label": "Foundation", "type": "module", "source_file": "A.swift"},
|
||||
{"id": "akit", "label": "AKit", "file_type": "code"},
|
||||
{"id": "foundation", "label": "Foundation", "type": "module", "source_file": "B.swift"},
|
||||
]
|
||||
out = dedupe_nodes(nodes)
|
||||
ids = [n["id"] for n in out]
|
||||
assert ids == ["foundation", "akit"] # first-appearance order
|
||||
# last writer wins on attributes
|
||||
assert next(n for n in out if n["id"] == "foundation")["source_file"] == "B.swift"
|
||||
|
||||
def load_extraction():
|
||||
return json.loads((FIXTURES / "extraction.json").read_text())
|
||||
|
||||
|
||||
@@ -610,6 +610,9 @@ def test_swift_imports_survive_build():
|
||||
node_ids = {n["id"] for n in r["nodes"]}
|
||||
for e in import_edges:
|
||||
assert e["target"] in node_ids # synthesized module node exists
|
||||
# Imported modules are tagged type=module (anchor nodes, #1327/#1330).
|
||||
module_labels = {n["label"] for n in r["nodes"] if n.get("type") == "module"}
|
||||
assert {"Foundation", "UIKit"} <= module_labels
|
||||
# No private bookkeeping key should leak into output edges.
|
||||
assert all("_import_label" not in e for e in r["edges"])
|
||||
# Edges must survive the build (which prunes edges with unknown endpoints).
|
||||
|
||||
@@ -0,0 +1,85 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
from graphify.build import build_from_json
|
||||
from graphify.extract import extract
|
||||
|
||||
|
||||
def _write(path: Path, text: str) -> Path:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
path.write_text(text, encoding="utf-8")
|
||||
return path
|
||||
|
||||
|
||||
def _module_nodes(result: dict, label: str) -> list[dict]:
|
||||
return [
|
||||
n for n in result["nodes"]
|
||||
if n.get("type") == "module" and n.get("label") == label
|
||||
]
|
||||
|
||||
|
||||
def _import_edges(result: dict) -> list[dict]:
|
||||
return [e for e in result["edges"] if e.get("relation") == "imports"]
|
||||
|
||||
|
||||
def test_swift_import_resolves_to_module_node(tmp_path: Path):
|
||||
# #1327: `import CoreKit` must anchor to a node, or build_from_json prunes
|
||||
# the edge as a dangling/external reference.
|
||||
core = _write(tmp_path / "Sources/CoreKit/CoreKit.swift", "public struct CoreKit {}\n")
|
||||
feature = _write(
|
||||
tmp_path / "Sources/FeatureKit/FeatureKit.swift",
|
||||
"import CoreKit\n\npublic struct FeatureKit {}\n",
|
||||
)
|
||||
|
||||
result = extract([core, feature], cache_root=tmp_path)
|
||||
|
||||
node_ids = {n["id"] for n in result["nodes"]}
|
||||
imports = _import_edges(result)
|
||||
assert imports
|
||||
for e in imports:
|
||||
assert e["target"] in node_ids
|
||||
assert _module_nodes(result, "CoreKit")
|
||||
|
||||
|
||||
def test_swift_same_module_imported_twice_collapses_to_one_node(tmp_path: Path):
|
||||
# #1327: the same module imported from multiple files is ONE module, not N
|
||||
# file-qualified duplicates — collision-disambiguation must exempt modules.
|
||||
core = _write(tmp_path / "Sources/CoreKit/CoreKit.swift", "public struct CoreKit {}\n")
|
||||
a = _write(
|
||||
tmp_path / "Sources/AKit/AKit.swift",
|
||||
"import CoreKit\n\npublic struct AKit {}\n",
|
||||
)
|
||||
b = _write(
|
||||
tmp_path / "Sources/BKit/BKit.swift",
|
||||
"import CoreKit\n\npublic struct BKit {}\n",
|
||||
)
|
||||
|
||||
result = extract([core, a, b], cache_root=tmp_path)
|
||||
|
||||
# Each importing file contributes a module-node dict, but they must share a
|
||||
# single id (NOT be split into path-qualified duplicates) so build_from_json
|
||||
# collapses them into one shared node.
|
||||
core_modules = _module_nodes(result, "CoreKit")
|
||||
module_ids = {n["id"] for n in core_modules}
|
||||
assert len(module_ids) == 1
|
||||
# Both importers point at that single shared module id.
|
||||
import_targets = {e["target"] for e in _import_edges(result)}
|
||||
assert import_targets == module_ids
|
||||
|
||||
|
||||
def test_swift_import_edges_survive_build(tmp_path: Path):
|
||||
# #1327: edges must remain after graph assembly, deduped to one module node.
|
||||
core = _write(tmp_path / "Sources/CoreKit/CoreKit.swift", "public struct CoreKit {}\n")
|
||||
a = _write(tmp_path / "Sources/AKit/AKit.swift", "import CoreKit\n")
|
||||
b = _write(tmp_path / "Sources/BKit/BKit.swift", "import CoreKit\n")
|
||||
|
||||
result = extract([core, a, b], cache_root=tmp_path)
|
||||
G = build_from_json(result, directed=True)
|
||||
|
||||
import_edges = [
|
||||
(u, v) for u, v, d in G.edges(data=True) if d.get("relation") == "imports"
|
||||
]
|
||||
assert len(import_edges) == 2
|
||||
# Both edges land on the same CoreKit module node.
|
||||
assert len({v for _, v in import_edges}) == 1
|
||||
Reference in New Issue
Block a user