Fix 6 bugs: ID collisions, path portability, alias resolution, HTML controls, desync guard, rationale prompt

- #550: _file_stem() includes parent dir to prevent node ID collisions for same-named files
- #555: extract() relativizes source_file paths before returning for cross-machine portability
- #562: to_json() returns bool; _rebuild_code() writes report/html only if json succeeded
- #563: skill prompts store rationale as node attribute, not separate node; enforce calls direction
- #566: Show All / Hide All buttons added to HTML community panel
- #575: _import_js() resolves tsconfig.json compilerOptions.paths aliases before external fallback

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
Safi
2026-04-27 21:44:22 +01:00
co-authored by Claude Sonnet 4.6
parent 6175e0a8ea
commit 4563b043f2
13 changed files with 126 additions and 30 deletions
+22 -2
View File
@@ -55,6 +55,9 @@ def _html_styles() -> str:
.legend-label { flex: 1; overflow: hidden; text-overflow: ellipsis; white-space: nowrap; }
.legend-count { color: #666; font-size: 11px; }
#stats { padding: 10px 14px; border-top: 1px solid #2a2a4e; font-size: 11px; color: #555; }
#legend-controls { display: flex; gap: 6px; margin-bottom: 8px; }
#legend-controls button { flex: 1; background: #0f0f1a; border: 1px solid #3a3a5e; color: #aaa; padding: 4px 0; border-radius: 4px; font-size: 11px; cursor: pointer; }
#legend-controls button:hover { border-color: #4E79A7; color: #e0e0e0; }
</style>"""
@@ -240,6 +243,18 @@ document.addEventListener('click', e => {{
}});
const hiddenCommunities = new Set();
function toggleAllCommunities(hide) {{
document.querySelectorAll('.legend-item').forEach(item => {{
hide ? item.classList.add('dimmed') : item.classList.remove('dimmed');
}});
LEGEND.forEach(c => {{
if (hide) hiddenCommunities.add(c.cid); else hiddenCommunities.delete(c.cid);
}});
const updates = RAW_NODES.map(n => ({{ id: n.id, hidden: hide }}));
nodesDS.update(updates);
}}
const legendEl = document.getElementById('legend');
LEGEND.forEach(c => {{
const item = document.createElement('div');
@@ -279,7 +294,7 @@ def attach_hyperedges(G: nx.Graph, hyperedges: list) -> None:
G.graph["hyperedges"] = existing
def to_json(G: nx.Graph, communities: dict[int, list[str]], output_path: str, *, force: bool = False) -> None:
def to_json(G: nx.Graph, communities: dict[int, list[str]], output_path: str, *, force: bool = False) -> bool:
# Safety check: refuse to silently shrink an existing graph (#479)
existing_path = Path(output_path)
if not force and existing_path.exists():
@@ -296,7 +311,7 @@ def to_json(G: nx.Graph, communities: dict[int, list[str]], output_path: str, *,
f"Pass force=True to override.",
file=_sys.stderr,
)
return
return False
except Exception:
pass # unreadable existing file — proceed with write
@@ -315,6 +330,7 @@ def to_json(G: nx.Graph, communities: dict[int, list[str]], output_path: str, *,
data["hyperedges"] = getattr(G, "graph", {}).get("hyperedges", [])
with open(output_path, "w", encoding="utf-8") as f: # nosec
json.dump(data, f, indent=2)
return True
def prune_dangling_edges(graph_data: dict) -> tuple[dict, int]:
@@ -471,6 +487,10 @@ def to_html(
</div>
<div id="legend-wrap">
<h3>Communities</h3>
<div id="legend-controls">
<button onclick="toggleAllCommunities(false)">Show All</button>
<button onclick="toggleAllCommunities(true)">Hide All</button>
</div>
<div id="legend"></div>
</div>
<div id="stats">{stats}</div>
+83 -17
View File
@@ -18,6 +18,48 @@ def _make_id(*parts: str) -> str:
return cleaned.strip("_").lower()
def _file_stem(path: Path) -> str:
"""Return a stem qualified with the parent directory name to avoid ID collisions
when multiple files share the same filename in different directories (#550)."""
parent = path.parent.name
if parent and parent not in (".", ""):
return f"{parent}.{path.stem}"
return path.stem
_TSCONFIG_ALIAS_CACHE: dict[str, dict[str, str]] = {}
def _load_tsconfig_aliases(start_dir: Path) -> dict[str, str]:
"""Walk up from start_dir to find tsconfig.json and return compilerOptions.paths aliases.
Returns a dict mapping alias prefix (e.g. "@/") to resolved base dir (e.g. "src/").
Result is cached by tsconfig path string.
"""
current = start_dir.resolve()
for candidate in [current, *current.parents]:
tsconfig = candidate / "tsconfig.json"
if tsconfig.exists():
key = str(tsconfig)
if key not in _TSCONFIG_ALIAS_CACHE:
try:
data = json.loads(tsconfig.read_text(encoding="utf-8"))
paths = data.get("compilerOptions", {}).get("paths", {})
aliases: dict[str, str] = {}
for alias, targets in paths.items():
if not targets:
continue
# Strip trailing /* from alias and target
alias_prefix = alias.rstrip("/*")
target_base = targets[0].rstrip("/*")
aliases[alias_prefix] = str(candidate / target_base)
_TSCONFIG_ALIAS_CACHE[key] = aliases
except Exception:
_TSCONFIG_ALIAS_CACHE[key] = {}
return _TSCONFIG_ALIAS_CACHE[key]
return {}
# ── LanguageConfig dataclass ─────────────────────────────────────────────────
@dataclass
@@ -156,11 +198,22 @@ def _import_js(node, source: bytes, file_nid: str, stem: str, edges: list, str_p
resolved = resolved.with_suffix(".tsx")
tgt_nid = _make_id(str(resolved))
else:
# Bare/scoped import (node_modules) - use last segment; dropped as external
module_name = raw.split("/")[-1]
if not module_name:
break
tgt_nid = _make_id(module_name)
# Check tsconfig.json path aliases (e.g. "@/" → "src/") before treating as external (#575)
aliases = _load_tsconfig_aliases(Path(str_path).parent)
resolved_alias = None
for alias_prefix, alias_base in aliases.items():
if raw == alias_prefix or raw.startswith(alias_prefix + "/"):
rest = raw[len(alias_prefix):].lstrip("/")
resolved_alias = Path(os.path.normpath(Path(alias_base) / rest))
break
if resolved_alias is not None:
tgt_nid = _make_id(str(resolved_alias))
else:
# Bare/scoped import (node_modules) - use last segment; dropped as external
module_name = raw.split("/")[-1]
if not module_name:
break
tgt_nid = _make_id(module_name)
edges.append({
"source": file_nid,
"target": tgt_nid,
@@ -676,7 +729,7 @@ def _extract_generic(path: Path, config: LanguageConfig) -> dict:
except Exception as e:
return {"nodes": [], "edges": [], "error": str(e)}
stem = path.stem
stem = _file_stem(path)
str_path = str(path)
nodes: list[dict] = []
edges: list[dict] = []
@@ -1301,7 +1354,7 @@ def _extract_python_rationale(path: Path, result: dict) -> None:
except Exception:
return
stem = path.stem
stem = _file_stem(path)
str_path = str(path)
nodes = result["nodes"]
edges = result["edges"]
@@ -1560,7 +1613,7 @@ def extract_verilog(path: Path) -> dict:
except Exception as e:
return {"nodes": [], "edges": [], "error": str(e)}
stem = path.stem
stem = _file_stem(path)
str_path = str(path)
nodes: list[dict] = []
edges: list[dict] = []
@@ -1676,7 +1729,7 @@ def extract_julia(path: Path) -> dict:
except Exception as e:
return {"nodes": [], "edges": [], "error": str(e)}
stem = path.stem
stem = _file_stem(path)
str_path = str(path)
nodes: list[dict] = []
edges: list[dict] = []
@@ -1888,7 +1941,7 @@ def extract_go(path: Path) -> dict:
except Exception as e:
return {"nodes": [], "edges": [], "error": str(e)}
stem = path.stem
stem = _file_stem(path)
# Use directory name as package scope so methods on the same type across
# multiple files in a package share one canonical type node.
pkg_scope = path.parent.name or stem
@@ -2087,7 +2140,7 @@ def extract_rust(path: Path) -> dict:
except Exception as e:
return {"nodes": [], "edges": [], "error": str(e)}
stem = path.stem
stem = _file_stem(path)
str_path = str(path)
nodes: list[dict] = []
edges: list[dict] = []
@@ -2264,7 +2317,7 @@ def extract_zig(path: Path) -> dict:
except Exception as e:
return {"nodes": [], "edges": [], "error": str(e)}
stem = path.stem
stem = _file_stem(path)
str_path = str(path)
nodes: list[dict] = []
edges: list[dict] = []
@@ -2427,7 +2480,7 @@ def extract_powershell(path: Path) -> dict:
except Exception as e:
return {"nodes": [], "edges": [], "error": str(e)}
stem = path.stem
stem = _file_stem(path)
str_path = str(path)
nodes: list[dict] = []
edges: list[dict] = []
@@ -2621,7 +2674,7 @@ def _resolve_cross_file_imports(
stem_to_path: dict[str, Path] = {p.stem: p for p in paths}
for file_result, path in zip(per_file, paths):
stem = path.stem
stem = _file_stem(path)
str_path = str(path)
# Find all classes defined in this file (the importers)
@@ -2745,7 +2798,7 @@ def _resolve_cross_file_java_imports(
new_edges: list[dict] = []
seen_pairs: set[tuple[str, str]] = set()
for path in paths:
file_nid = _make_id(path.stem)
file_nid = _make_id(str(path))
try:
source = path.read_bytes()
tree = parser.parse(source)
@@ -2809,7 +2862,7 @@ def extract_objc(path: Path) -> dict:
except Exception as e:
return {"nodes": [], "edges": [], "error": str(e)}
stem = path.stem
stem = _file_stem(path)
str_path = str(path)
nodes: list[dict] = []
edges: list[dict] = []
@@ -3007,7 +3060,7 @@ def extract_elixir(path: Path) -> dict:
except Exception as e:
return {"nodes": [], "edges": [], "error": str(e)}
stem = path.stem
stem = _file_stem(path)
str_path = str(path)
nodes: list[dict] = []
edges: list[dict] = []
@@ -3373,6 +3426,19 @@ def extract(paths: list[Path], cache_root: Path | None = None) -> dict:
"weight": 1.0,
})
# Relativize source_file fields so paths are portable across machines (#555)
for item in all_nodes + all_edges:
sf = item.get("source_file")
if not sf:
continue
sf_path = Path(sf)
if not sf_path.is_absolute():
continue
try:
item["source_file"] = str(sf_path.relative_to(root))
except ValueError:
pass
return {
"nodes": all_nodes,
"edges": all_edges,
+1 -1
View File
@@ -235,7 +235,7 @@ Process each file one at a time. For each file:
- INFERRED: reasonable inference (shared structure, implied dependency)
- AMBIGUOUS: uncertain — flag it, do not omit
- Code files: semantic edges AST cannot find. Do not re-extract imports.
- Doc/paper files: named concepts, entities, citations, and rationale nodes (WHY decisions were made → `rationale_for` edges)
- Doc/paper files: named concepts, entities, citations. Store rationale (WHY decisions were made) as a `rationale` attribute on the relevant node, not as a separate node. When adding `calls` edges: source is caller, target is callee.
- Image files: use vision — understand what the image IS, not just OCR
- DEEP_MODE (if --mode deep): be aggressive with INFERRED edges
- Semantic similarity: if two concepts solve the same problem without a structural link, add `semantically_similar_to` INFERRED edge (confidence 0.6-0.95). Non-obvious cross-file links only.
+1 -1
View File
@@ -235,7 +235,7 @@ Process each file one at a time. For each file:
- INFERRED: reasonable inference (shared structure, implied dependency)
- AMBIGUOUS: uncertain — flag it, do not omit
- Code files: semantic edges AST cannot find. Do not re-extract imports.
- Doc/paper files: named concepts, entities, citations, and rationale nodes (WHY decisions were made → `rationale_for` edges)
- Doc/paper files: named concepts, entities, citations. Store rationale (WHY decisions were made) as a `rationale` attribute on the relevant node, not as a separate node. When adding `calls` edges: source is caller, target is callee.
- Image files: use vision — understand what the image IS, not just OCR
- DEEP_MODE (if --mode deep): be aggressive with INFERRED edges
- Semantic similarity: if two concepts solve the same problem without a structural link, add `semantically_similar_to` INFERRED edge (confidence 0.6-0.95). Non-obvious cross-file links only.
+2 -1
View File
@@ -263,7 +263,8 @@ Rules:
Code files: focus on semantic edges AST cannot find (call relationships, shared data, arch patterns).
Do not re-extract imports - AST already has those.
Doc/paper files: extract named concepts, entities, citations. Also extract rationale — sections that explain WHY a decision was made, trade-offs chosen, or design intent. These become nodes with `rationale_for` edges pointing to the concept they explain.
Doc/paper files: extract named concepts, entities, citations. For rationale (WHY decisions were made, trade-offs, design intent): store as a `rationale` attribute on the relevant concept node — do NOT create a separate rationale node or fragment node. Only create a node for something that is itself a named entity or concept.
Code files: when adding `calls` edges, source MUST be the caller (the function/class doing the calling), target MUST be the callee. Never reverse this direction.
Image files: use vision to understand what the image IS - do not just OCR.
UI screenshot: layout patterns, design decisions, key elements, purpose.
Chart: metric, trend/insight, data source.
+2 -1
View File
@@ -259,7 +259,8 @@ Rules:
Code files: focus on semantic edges AST cannot find (call relationships, shared data, arch patterns).
Do not re-extract imports - AST already has those.
Doc/paper files: extract named concepts, entities, citations. Also extract rationale — sections that explain WHY a decision was made, trade-offs chosen, or design intent. These become nodes with `rationale_for` edges pointing to the concept they explain.
Doc/paper files: extract named concepts, entities, citations. For rationale (WHY decisions were made, trade-offs, design intent): store as a `rationale` attribute on the relevant concept node — do NOT create a separate rationale node or fragment node. Only create a node for something that is itself a named entity or concept.
Code files: when adding `calls` edges, source MUST be the caller (the function/class doing the calling), target MUST be the callee. Never reverse this direction.
Image files: use vision to understand what the image IS - do not just OCR.
UI screenshot: layout patterns, design decisions, key elements, purpose.
Chart: metric, trend/insight, data source.
+2 -1
View File
@@ -260,7 +260,8 @@ Rules:
Code files: focus on semantic edges AST cannot find (call relationships, shared data, arch patterns).
Do not re-extract imports - AST already has those.
Doc/paper files: extract named concepts, entities, citations. Also extract rationale — sections that explain WHY a decision was made, trade-offs chosen, or design intent. These become nodes with `rationale_for` edges pointing to the concept they explain.
Doc/paper files: extract named concepts, entities, citations. For rationale (WHY decisions were made, trade-offs, design intent): store as a `rationale` attribute on the relevant concept node — do NOT create a separate rationale node or fragment node. Only create a node for something that is itself a named entity or concept.
Code files: when adding `calls` edges, source MUST be the caller (the function/class doing the calling), target MUST be the callee. Never reverse this direction.
Image files: use vision to understand what the image IS - do not just OCR.
UI screenshot: layout patterns, design decisions, key elements, purpose.
Chart: metric, trend/insight, data source.
+1 -1
View File
@@ -234,7 +234,7 @@ Process each file one at a time. For each file:
- INFERRED: reasonable inference (shared structure, implied dependency)
- AMBIGUOUS: uncertain — flag it, do not omit
- Code files: semantic edges AST cannot find. Do not re-extract imports.
- Doc/paper files: named concepts, entities, citations, and rationale nodes (WHY decisions were made → `rationale_for` edges)
- Doc/paper files: named concepts, entities, citations. Store rationale (WHY decisions were made) as a `rationale` attribute on the relevant node, not as a separate node. When adding `calls` edges: source is caller, target is callee.
- Image files: use vision — understand what the image IS, not just OCR
- DEEP_MODE (if --mode deep): be aggressive with INFERRED edges
- Semantic similarity: if two concepts solve the same problem without a structural link, add `semantically_similar_to` INFERRED edge (confidence 0.6-0.95). Non-obvious cross-file links only.
+2 -1
View File
@@ -261,7 +261,8 @@ Rules:
Code files: focus on semantic edges AST cannot find (call relationships, shared data, arch patterns).
Do not re-extract imports - AST already has those.
Doc/paper files: extract named concepts, entities, citations. Also extract rationale — sections that explain WHY a decision was made, trade-offs chosen, or design intent. These become nodes with `rationale_for` edges pointing to the concept they explain.
Doc/paper files: extract named concepts, entities, citations. For rationale (WHY decisions were made, trade-offs, design intent): store as a `rationale` attribute on the relevant concept node — do NOT create a separate rationale node or fragment node. Only create a node for something that is itself a named entity or concept.
Code files: when adding `calls` edges, source MUST be the caller (the function/class doing the calling), target MUST be the callee. Never reverse this direction.
Image files: use vision to understand what the image IS - do not just OCR.
UI screenshot: layout patterns, design decisions, key elements, purpose.
Chart: metric, trend/insight, data source.
+2 -1
View File
@@ -250,7 +250,8 @@ Rules:
Code files: focus on semantic edges AST cannot find (call relationships, shared data, arch patterns).
Do not re-extract imports - AST already has those.
Doc/paper files: extract named concepts, entities, citations. Also extract rationale — sections that explain WHY a decision was made, trade-offs chosen, or design intent. These become nodes with `rationale_for` edges pointing to the concept they explain.
Doc/paper files: extract named concepts, entities, citations. For rationale (WHY decisions were made, trade-offs, design intent): store as a `rationale` attribute on the relevant concept node — do NOT create a separate rationale node or fragment node. Only create a node for something that is itself a named entity or concept.
Code files: when adding `calls` edges, source MUST be the caller (the function/class doing the calling), target MUST be the callee. Never reverse this direction.
Image files: use vision to understand what the image IS - do not just OCR.
UI screenshot: layout patterns, design decisions, key elements, purpose.
Chart: metric, trend/insight, data source.
+2 -1
View File
@@ -249,7 +249,8 @@ Rules:
Code files: focus on semantic edges AST cannot find (call relationships, shared data, arch patterns).
Do not re-extract imports - AST already has those.
Doc/paper files: extract named concepts, entities, citations. Also extract rationale — sections that explain WHY a decision was made, trade-offs chosen, or design intent. These become nodes with `rationale_for` edges pointing to the concept they explain.
Doc/paper files: extract named concepts, entities, citations. For rationale (WHY decisions were made, trade-offs, design intent): store as a `rationale` attribute on the relevant concept node — do NOT create a separate rationale node or fragment node. Only create a node for something that is itself a named entity or concept.
Code files: when adding `calls` edges, source MUST be the caller (the function/class doing the calling), target MUST be the callee. Never reverse this direction.
Image files: use vision to understand what the image IS - do not just OCR.
UI screenshot: layout patterns, design decisions, key elements, purpose.
Chart: metric, trend/insight, data source.
+2 -1
View File
@@ -300,7 +300,8 @@ Rules:
Code files: focus on semantic edges AST cannot find (call relationships, shared data, arch patterns).
Do not re-extract imports - AST already has those.
Doc/paper files: extract named concepts, entities, citations. Also extract rationale — sections that explain WHY a decision was made, trade-offs chosen, or design intent. These become nodes with `rationale_for` edges pointing to the concept they explain.
Doc/paper files: extract named concepts, entities, citations. For rationale (WHY decisions were made, trade-offs, design intent): store as a `rationale` attribute on the relevant concept node — do NOT create a separate rationale node or fragment node. Only create a node for something that is itself a named entity or concept.
Code files: when adding `calls` edges, source MUST be the caller (the function/class doing the calling), target MUST be the callee. Never reverse this direction.
Image files: use vision to understand what the image IS - do not just OCR.
UI screenshot: layout patterns, design decisions, key elements, purpose.
Chart: metric, trend/insight, data source.
+4 -1
View File
@@ -99,10 +99,13 @@ def _rebuild_code(watch_path: Path, *, follow_symlinks: bool = False) -> bool:
out.mkdir(exist_ok=True)
json_written = to_json(G, communities, str(out / "graph.json"))
if not json_written:
return False
report = generate(G, communities, cohesion, labels, gods, surprises, detection,
{"input": 0, "output": 0}, report_root, suggested_questions=questions)
(out / "GRAPH_REPORT.md").write_text(report, encoding="utf-8")
to_json(G, communities, str(out / "graph.json"))
# to_html raises ValueError for graphs > MAX_NODES_FOR_VIZ (5000).
# Wrap so core outputs (graph.json + GRAPH_REPORT.md) always land.