diff --git a/graphify/extract.py b/graphify/extract.py
index dbd441c..357e430 100644
--- a/graphify/extract.py
+++ b/graphify/extract.py
@@ -18,6 +18,48 @@ def _make_id(*parts: str) -> str:
return cleaned.strip("_").lower()
+def _file_stem(path: Path) -> str:
+ """Return a stem qualified with the parent directory name to avoid ID collisions
+ when multiple files share the same filename in different directories (#550)."""
+ parent = path.parent.name
+ if parent and parent not in (".", ""):
+ return f"{parent}.{path.stem}"
+ return path.stem
+
+
+_TSCONFIG_ALIAS_CACHE: dict[str, dict[str, str]] = {}
+
+
+def _load_tsconfig_aliases(start_dir: Path) -> dict[str, str]:
+ """Walk up from start_dir to find tsconfig.json and return compilerOptions.paths aliases.
+
+ Returns a dict mapping alias prefix (e.g. "@/") to resolved base dir (e.g. "src/").
+ Result is cached by tsconfig path string.
+ """
+ current = start_dir.resolve()
+ for candidate in [current, *current.parents]:
+ tsconfig = candidate / "tsconfig.json"
+ if tsconfig.exists():
+ key = str(tsconfig)
+ if key not in _TSCONFIG_ALIAS_CACHE:
+ try:
+ data = json.loads(tsconfig.read_text(encoding="utf-8"))
+ paths = data.get("compilerOptions", {}).get("paths", {})
+ aliases: dict[str, str] = {}
+ for alias, targets in paths.items():
+ if not targets:
+ continue
+ # Strip trailing /* from alias and target
+ alias_prefix = alias.rstrip("/*")
+ target_base = targets[0].rstrip("/*")
+ aliases[alias_prefix] = str(candidate / target_base)
+ _TSCONFIG_ALIAS_CACHE[key] = aliases
+ except Exception:
+ _TSCONFIG_ALIAS_CACHE[key] = {}
+ return _TSCONFIG_ALIAS_CACHE[key]
+ return {}
+
+
# ── LanguageConfig dataclass ─────────────────────────────────────────────────
@dataclass
@@ -156,11 +198,22 @@ def _import_js(node, source: bytes, file_nid: str, stem: str, edges: list, str_p
resolved = resolved.with_suffix(".tsx")
tgt_nid = _make_id(str(resolved))
else:
- # Bare/scoped import (node_modules) - use last segment; dropped as external
- module_name = raw.split("/")[-1]
- if not module_name:
- break
- tgt_nid = _make_id(module_name)
+ # Check tsconfig.json path aliases (e.g. "@/" → "src/") before treating as external (#575)
+ aliases = _load_tsconfig_aliases(Path(str_path).parent)
+ resolved_alias = None
+ for alias_prefix, alias_base in aliases.items():
+ if raw == alias_prefix or raw.startswith(alias_prefix + "/"):
+ rest = raw[len(alias_prefix):].lstrip("/")
+ resolved_alias = Path(os.path.normpath(Path(alias_base) / rest))
+ break
+ if resolved_alias is not None:
+ tgt_nid = _make_id(str(resolved_alias))
+ else:
+ # Bare/scoped import (node_modules) - use last segment; dropped as external
+ module_name = raw.split("/")[-1]
+ if not module_name:
+ break
+ tgt_nid = _make_id(module_name)
edges.append({
"source": file_nid,
"target": tgt_nid,
@@ -676,7 +729,7 @@ def _extract_generic(path: Path, config: LanguageConfig) -> dict:
except Exception as e:
return {"nodes": [], "edges": [], "error": str(e)}
- stem = path.stem
+ stem = _file_stem(path)
str_path = str(path)
nodes: list[dict] = []
edges: list[dict] = []
@@ -1301,7 +1354,7 @@ def _extract_python_rationale(path: Path, result: dict) -> None:
except Exception:
return
- stem = path.stem
+ stem = _file_stem(path)
str_path = str(path)
nodes = result["nodes"]
edges = result["edges"]
@@ -1560,7 +1613,7 @@ def extract_verilog(path: Path) -> dict:
except Exception as e:
return {"nodes": [], "edges": [], "error": str(e)}
- stem = path.stem
+ stem = _file_stem(path)
str_path = str(path)
nodes: list[dict] = []
edges: list[dict] = []
@@ -1676,7 +1729,7 @@ def extract_julia(path: Path) -> dict:
except Exception as e:
return {"nodes": [], "edges": [], "error": str(e)}
- stem = path.stem
+ stem = _file_stem(path)
str_path = str(path)
nodes: list[dict] = []
edges: list[dict] = []
@@ -1888,7 +1941,7 @@ def extract_go(path: Path) -> dict:
except Exception as e:
return {"nodes": [], "edges": [], "error": str(e)}
- stem = path.stem
+ stem = _file_stem(path)
# Use directory name as package scope so methods on the same type across
# multiple files in a package share one canonical type node.
pkg_scope = path.parent.name or stem
@@ -2087,7 +2140,7 @@ def extract_rust(path: Path) -> dict:
except Exception as e:
return {"nodes": [], "edges": [], "error": str(e)}
- stem = path.stem
+ stem = _file_stem(path)
str_path = str(path)
nodes: list[dict] = []
edges: list[dict] = []
@@ -2264,7 +2317,7 @@ def extract_zig(path: Path) -> dict:
except Exception as e:
return {"nodes": [], "edges": [], "error": str(e)}
- stem = path.stem
+ stem = _file_stem(path)
str_path = str(path)
nodes: list[dict] = []
edges: list[dict] = []
@@ -2427,7 +2480,7 @@ def extract_powershell(path: Path) -> dict:
except Exception as e:
return {"nodes": [], "edges": [], "error": str(e)}
- stem = path.stem
+ stem = _file_stem(path)
str_path = str(path)
nodes: list[dict] = []
edges: list[dict] = []
@@ -2621,7 +2674,7 @@ def _resolve_cross_file_imports(
stem_to_path: dict[str, Path] = {p.stem: p for p in paths}
for file_result, path in zip(per_file, paths):
- stem = path.stem
+ stem = _file_stem(path)
str_path = str(path)
# Find all classes defined in this file (the importers)
@@ -2745,7 +2798,7 @@ def _resolve_cross_file_java_imports(
new_edges: list[dict] = []
seen_pairs: set[tuple[str, str]] = set()
for path in paths:
- file_nid = _make_id(path.stem)
+ file_nid = _make_id(str(path))
try:
source = path.read_bytes()
tree = parser.parse(source)
@@ -2809,7 +2862,7 @@ def extract_objc(path: Path) -> dict:
except Exception as e:
return {"nodes": [], "edges": [], "error": str(e)}
- stem = path.stem
+ stem = _file_stem(path)
str_path = str(path)
nodes: list[dict] = []
edges: list[dict] = []
@@ -3007,7 +3060,7 @@ def extract_elixir(path: Path) -> dict:
except Exception as e:
return {"nodes": [], "edges": [], "error": str(e)}
- stem = path.stem
+ stem = _file_stem(path)
str_path = str(path)
nodes: list[dict] = []
edges: list[dict] = []
@@ -3373,6 +3426,19 @@ def extract(paths: list[Path], cache_root: Path | None = None) -> dict:
"weight": 1.0,
})
+ # Relativize source_file fields so paths are portable across machines (#555)
+ for item in all_nodes + all_edges:
+ sf = item.get("source_file")
+ if not sf:
+ continue
+ sf_path = Path(sf)
+ if not sf_path.is_absolute():
+ continue
+ try:
+ item["source_file"] = str(sf_path.relative_to(root))
+ except ValueError:
+ pass
+
return {
"nodes": all_nodes,
"edges": all_edges,
diff --git a/graphify/skill-aider.md b/graphify/skill-aider.md
index 7c745f3..fa7007f 100644
--- a/graphify/skill-aider.md
+++ b/graphify/skill-aider.md
@@ -235,7 +235,7 @@ Process each file one at a time. For each file:
- INFERRED: reasonable inference (shared structure, implied dependency)
- AMBIGUOUS: uncertain — flag it, do not omit
- Code files: semantic edges AST cannot find. Do not re-extract imports.
- - Doc/paper files: named concepts, entities, citations, and rationale nodes (WHY decisions were made → `rationale_for` edges)
+ - Doc/paper files: named concepts, entities, citations. Store rationale (WHY decisions were made) as a `rationale` attribute on the relevant node, not as a separate node. When adding `calls` edges: source is caller, target is callee.
- Image files: use vision — understand what the image IS, not just OCR
- DEEP_MODE (if --mode deep): be aggressive with INFERRED edges
- Semantic similarity: if two concepts solve the same problem without a structural link, add `semantically_similar_to` INFERRED edge (confidence 0.6-0.95). Non-obvious cross-file links only.
diff --git a/graphify/skill-claw.md b/graphify/skill-claw.md
index 2f3b3a7..3d84acb 100644
--- a/graphify/skill-claw.md
+++ b/graphify/skill-claw.md
@@ -235,7 +235,7 @@ Process each file one at a time. For each file:
- INFERRED: reasonable inference (shared structure, implied dependency)
- AMBIGUOUS: uncertain — flag it, do not omit
- Code files: semantic edges AST cannot find. Do not re-extract imports.
- - Doc/paper files: named concepts, entities, citations, and rationale nodes (WHY decisions were made → `rationale_for` edges)
+ - Doc/paper files: named concepts, entities, citations. Store rationale (WHY decisions were made) as a `rationale` attribute on the relevant node, not as a separate node. When adding `calls` edges: source is caller, target is callee.
- Image files: use vision — understand what the image IS, not just OCR
- DEEP_MODE (if --mode deep): be aggressive with INFERRED edges
- Semantic similarity: if two concepts solve the same problem without a structural link, add `semantically_similar_to` INFERRED edge (confidence 0.6-0.95). Non-obvious cross-file links only.
diff --git a/graphify/skill-codex.md b/graphify/skill-codex.md
index f3c4408..b2e79e2 100644
--- a/graphify/skill-codex.md
+++ b/graphify/skill-codex.md
@@ -263,7 +263,8 @@ Rules:
Code files: focus on semantic edges AST cannot find (call relationships, shared data, arch patterns).
Do not re-extract imports - AST already has those.
-Doc/paper files: extract named concepts, entities, citations. Also extract rationale — sections that explain WHY a decision was made, trade-offs chosen, or design intent. These become nodes with `rationale_for` edges pointing to the concept they explain.
+Doc/paper files: extract named concepts, entities, citations. For rationale (WHY decisions were made, trade-offs, design intent): store as a `rationale` attribute on the relevant concept node — do NOT create a separate rationale node or fragment node. Only create a node for something that is itself a named entity or concept.
+Code files: when adding `calls` edges, source MUST be the caller (the function/class doing the calling), target MUST be the callee. Never reverse this direction.
Image files: use vision to understand what the image IS - do not just OCR.
UI screenshot: layout patterns, design decisions, key elements, purpose.
Chart: metric, trend/insight, data source.
diff --git a/graphify/skill-copilot.md b/graphify/skill-copilot.md
index cfccee5..397669a 100644
--- a/graphify/skill-copilot.md
+++ b/graphify/skill-copilot.md
@@ -259,7 +259,8 @@ Rules:
Code files: focus on semantic edges AST cannot find (call relationships, shared data, arch patterns).
Do not re-extract imports - AST already has those.
-Doc/paper files: extract named concepts, entities, citations. Also extract rationale — sections that explain WHY a decision was made, trade-offs chosen, or design intent. These become nodes with `rationale_for` edges pointing to the concept they explain.
+Doc/paper files: extract named concepts, entities, citations. For rationale (WHY decisions were made, trade-offs, design intent): store as a `rationale` attribute on the relevant concept node — do NOT create a separate rationale node or fragment node. Only create a node for something that is itself a named entity or concept.
+Code files: when adding `calls` edges, source MUST be the caller (the function/class doing the calling), target MUST be the callee. Never reverse this direction.
Image files: use vision to understand what the image IS - do not just OCR.
UI screenshot: layout patterns, design decisions, key elements, purpose.
Chart: metric, trend/insight, data source.
diff --git a/graphify/skill-droid.md b/graphify/skill-droid.md
index 1f25735..6cde935 100644
--- a/graphify/skill-droid.md
+++ b/graphify/skill-droid.md
@@ -260,7 +260,8 @@ Rules:
Code files: focus on semantic edges AST cannot find (call relationships, shared data, arch patterns).
Do not re-extract imports - AST already has those.
-Doc/paper files: extract named concepts, entities, citations. Also extract rationale — sections that explain WHY a decision was made, trade-offs chosen, or design intent. These become nodes with `rationale_for` edges pointing to the concept they explain.
+Doc/paper files: extract named concepts, entities, citations. For rationale (WHY decisions were made, trade-offs, design intent): store as a `rationale` attribute on the relevant concept node — do NOT create a separate rationale node or fragment node. Only create a node for something that is itself a named entity or concept.
+Code files: when adding `calls` edges, source MUST be the caller (the function/class doing the calling), target MUST be the callee. Never reverse this direction.
Image files: use vision to understand what the image IS - do not just OCR.
UI screenshot: layout patterns, design decisions, key elements, purpose.
Chart: metric, trend/insight, data source.
diff --git a/graphify/skill-kiro.md b/graphify/skill-kiro.md
index f04bc0f..b3db443 100644
--- a/graphify/skill-kiro.md
+++ b/graphify/skill-kiro.md
@@ -234,7 +234,7 @@ Process each file one at a time. For each file:
- INFERRED: reasonable inference (shared structure, implied dependency)
- AMBIGUOUS: uncertain — flag it, do not omit
- Code files: semantic edges AST cannot find. Do not re-extract imports.
- - Doc/paper files: named concepts, entities, citations, and rationale nodes (WHY decisions were made → `rationale_for` edges)
+ - Doc/paper files: named concepts, entities, citations. Store rationale (WHY decisions were made) as a `rationale` attribute on the relevant node, not as a separate node. When adding `calls` edges: source is caller, target is callee.
- Image files: use vision — understand what the image IS, not just OCR
- DEEP_MODE (if --mode deep): be aggressive with INFERRED edges
- Semantic similarity: if two concepts solve the same problem without a structural link, add `semantically_similar_to` INFERRED edge (confidence 0.6-0.95). Non-obvious cross-file links only.
diff --git a/graphify/skill-opencode.md b/graphify/skill-opencode.md
index 479b677..32819c8 100644
--- a/graphify/skill-opencode.md
+++ b/graphify/skill-opencode.md
@@ -261,7 +261,8 @@ Rules:
Code files: focus on semantic edges AST cannot find (call relationships, shared data, arch patterns).
Do not re-extract imports - AST already has those.
-Doc/paper files: extract named concepts, entities, citations. Also extract rationale — sections that explain WHY a decision was made, trade-offs chosen, or design intent. These become nodes with `rationale_for` edges pointing to the concept they explain.
+Doc/paper files: extract named concepts, entities, citations. For rationale (WHY decisions were made, trade-offs, design intent): store as a `rationale` attribute on the relevant concept node — do NOT create a separate rationale node or fragment node. Only create a node for something that is itself a named entity or concept.
+Code files: when adding `calls` edges, source MUST be the caller (the function/class doing the calling), target MUST be the callee. Never reverse this direction.
Image files: use vision to understand what the image IS - do not just OCR.
UI screenshot: layout patterns, design decisions, key elements, purpose.
Chart: metric, trend/insight, data source.
diff --git a/graphify/skill-trae.md b/graphify/skill-trae.md
index 5dfdc2d..2b5b401 100644
--- a/graphify/skill-trae.md
+++ b/graphify/skill-trae.md
@@ -250,7 +250,8 @@ Rules:
Code files: focus on semantic edges AST cannot find (call relationships, shared data, arch patterns).
Do not re-extract imports - AST already has those.
-Doc/paper files: extract named concepts, entities, citations. Also extract rationale — sections that explain WHY a decision was made, trade-offs chosen, or design intent. These become nodes with `rationale_for` edges pointing to the concept they explain.
+Doc/paper files: extract named concepts, entities, citations. For rationale (WHY decisions were made, trade-offs, design intent): store as a `rationale` attribute on the relevant concept node — do NOT create a separate rationale node or fragment node. Only create a node for something that is itself a named entity or concept.
+Code files: when adding `calls` edges, source MUST be the caller (the function/class doing the calling), target MUST be the callee. Never reverse this direction.
Image files: use vision to understand what the image IS - do not just OCR.
UI screenshot: layout patterns, design decisions, key elements, purpose.
Chart: metric, trend/insight, data source.
diff --git a/graphify/skill-windows.md b/graphify/skill-windows.md
index 8aa0482..9984022 100644
--- a/graphify/skill-windows.md
+++ b/graphify/skill-windows.md
@@ -249,7 +249,8 @@ Rules:
Code files: focus on semantic edges AST cannot find (call relationships, shared data, arch patterns).
Do not re-extract imports - AST already has those.
-Doc/paper files: extract named concepts, entities, citations. Also extract rationale — sections that explain WHY a decision was made, trade-offs chosen, or design intent. These become nodes with `rationale_for` edges pointing to the concept they explain.
+Doc/paper files: extract named concepts, entities, citations. For rationale (WHY decisions were made, trade-offs, design intent): store as a `rationale` attribute on the relevant concept node — do NOT create a separate rationale node or fragment node. Only create a node for something that is itself a named entity or concept.
+Code files: when adding `calls` edges, source MUST be the caller (the function/class doing the calling), target MUST be the callee. Never reverse this direction.
Image files: use vision to understand what the image IS - do not just OCR.
UI screenshot: layout patterns, design decisions, key elements, purpose.
Chart: metric, trend/insight, data source.
diff --git a/graphify/skill.md b/graphify/skill.md
index 60d2422..be1e7db 100644
--- a/graphify/skill.md
+++ b/graphify/skill.md
@@ -300,7 +300,8 @@ Rules:
Code files: focus on semantic edges AST cannot find (call relationships, shared data, arch patterns).
Do not re-extract imports - AST already has those.
-Doc/paper files: extract named concepts, entities, citations. Also extract rationale — sections that explain WHY a decision was made, trade-offs chosen, or design intent. These become nodes with `rationale_for` edges pointing to the concept they explain.
+Doc/paper files: extract named concepts, entities, citations. For rationale (WHY decisions were made, trade-offs, design intent): store as a `rationale` attribute on the relevant concept node — do NOT create a separate rationale node or fragment node. Only create a node for something that is itself a named entity or concept.
+Code files: when adding `calls` edges, source MUST be the caller (the function/class doing the calling), target MUST be the callee. Never reverse this direction.
Image files: use vision to understand what the image IS - do not just OCR.
UI screenshot: layout patterns, design decisions, key elements, purpose.
Chart: metric, trend/insight, data source.
diff --git a/graphify/watch.py b/graphify/watch.py
index a09dd51..b0e8e74 100644
--- a/graphify/watch.py
+++ b/graphify/watch.py
@@ -99,10 +99,13 @@ def _rebuild_code(watch_path: Path, *, follow_symlinks: bool = False) -> bool:
out.mkdir(exist_ok=True)
+ json_written = to_json(G, communities, str(out / "graph.json"))
+ if not json_written:
+ return False
+
report = generate(G, communities, cohesion, labels, gods, surprises, detection,
{"input": 0, "output": 0}, report_root, suggested_questions=questions)
(out / "GRAPH_REPORT.md").write_text(report, encoding="utf-8")
- to_json(G, communities, str(out / "graph.json"))
# to_html raises ValueError for graphs > MAX_NODES_FOR_VIZ (5000).
# Wrap so core outputs (graph.json + GRAPH_REPORT.md) always land.