diff --git a/graphify/export.py b/graphify/export.py index 1decae2..b14812f 100644 --- a/graphify/export.py +++ b/graphify/export.py @@ -55,6 +55,9 @@ def _html_styles() -> str: .legend-label { flex: 1; overflow: hidden; text-overflow: ellipsis; white-space: nowrap; } .legend-count { color: #666; font-size: 11px; } #stats { padding: 10px 14px; border-top: 1px solid #2a2a4e; font-size: 11px; color: #555; } + #legend-controls { display: flex; gap: 6px; margin-bottom: 8px; } + #legend-controls button { flex: 1; background: #0f0f1a; border: 1px solid #3a3a5e; color: #aaa; padding: 4px 0; border-radius: 4px; font-size: 11px; cursor: pointer; } + #legend-controls button:hover { border-color: #4E79A7; color: #e0e0e0; } """ @@ -240,6 +243,18 @@ document.addEventListener('click', e => {{ }}); const hiddenCommunities = new Set(); + +function toggleAllCommunities(hide) {{ + document.querySelectorAll('.legend-item').forEach(item => {{ + hide ? item.classList.add('dimmed') : item.classList.remove('dimmed'); + }}); + LEGEND.forEach(c => {{ + if (hide) hiddenCommunities.add(c.cid); else hiddenCommunities.delete(c.cid); + }}); + const updates = RAW_NODES.map(n => ({{ id: n.id, hidden: hide }})); + nodesDS.update(updates); +}} + const legendEl = document.getElementById('legend'); LEGEND.forEach(c => {{ const item = document.createElement('div'); @@ -279,7 +294,7 @@ def attach_hyperedges(G: nx.Graph, hyperedges: list) -> None: G.graph["hyperedges"] = existing -def to_json(G: nx.Graph, communities: dict[int, list[str]], output_path: str, *, force: bool = False) -> None: +def to_json(G: nx.Graph, communities: dict[int, list[str]], output_path: str, *, force: bool = False) -> bool: # Safety check: refuse to silently shrink an existing graph (#479) existing_path = Path(output_path) if not force and existing_path.exists(): @@ -296,7 +311,7 @@ def to_json(G: nx.Graph, communities: dict[int, list[str]], output_path: str, *, f"Pass force=True to override.", file=_sys.stderr, ) - return + return False except Exception: pass # unreadable existing file — proceed with write @@ -315,6 +330,7 @@ def to_json(G: nx.Graph, communities: dict[int, list[str]], output_path: str, *, data["hyperedges"] = getattr(G, "graph", {}).get("hyperedges", []) with open(output_path, "w", encoding="utf-8") as f: # nosec json.dump(data, f, indent=2) + return True def prune_dangling_edges(graph_data: dict) -> tuple[dict, int]: @@ -471,6 +487,10 @@ def to_html(

Communities

+
+ + +
{stats}
diff --git a/graphify/extract.py b/graphify/extract.py index dbd441c..357e430 100644 --- a/graphify/extract.py +++ b/graphify/extract.py @@ -18,6 +18,48 @@ def _make_id(*parts: str) -> str: return cleaned.strip("_").lower() +def _file_stem(path: Path) -> str: + """Return a stem qualified with the parent directory name to avoid ID collisions + when multiple files share the same filename in different directories (#550).""" + parent = path.parent.name + if parent and parent not in (".", ""): + return f"{parent}.{path.stem}" + return path.stem + + +_TSCONFIG_ALIAS_CACHE: dict[str, dict[str, str]] = {} + + +def _load_tsconfig_aliases(start_dir: Path) -> dict[str, str]: + """Walk up from start_dir to find tsconfig.json and return compilerOptions.paths aliases. + + Returns a dict mapping alias prefix (e.g. "@/") to resolved base dir (e.g. "src/"). + Result is cached by tsconfig path string. + """ + current = start_dir.resolve() + for candidate in [current, *current.parents]: + tsconfig = candidate / "tsconfig.json" + if tsconfig.exists(): + key = str(tsconfig) + if key not in _TSCONFIG_ALIAS_CACHE: + try: + data = json.loads(tsconfig.read_text(encoding="utf-8")) + paths = data.get("compilerOptions", {}).get("paths", {}) + aliases: dict[str, str] = {} + for alias, targets in paths.items(): + if not targets: + continue + # Strip trailing /* from alias and target + alias_prefix = alias.rstrip("/*") + target_base = targets[0].rstrip("/*") + aliases[alias_prefix] = str(candidate / target_base) + _TSCONFIG_ALIAS_CACHE[key] = aliases + except Exception: + _TSCONFIG_ALIAS_CACHE[key] = {} + return _TSCONFIG_ALIAS_CACHE[key] + return {} + + # ── LanguageConfig dataclass ───────────────────────────────────────────────── @dataclass @@ -156,11 +198,22 @@ def _import_js(node, source: bytes, file_nid: str, stem: str, edges: list, str_p resolved = resolved.with_suffix(".tsx") tgt_nid = _make_id(str(resolved)) else: - # Bare/scoped import (node_modules) - use last segment; dropped as external - module_name = raw.split("/")[-1] - if not module_name: - break - tgt_nid = _make_id(module_name) + # Check tsconfig.json path aliases (e.g. "@/" → "src/") before treating as external (#575) + aliases = _load_tsconfig_aliases(Path(str_path).parent) + resolved_alias = None + for alias_prefix, alias_base in aliases.items(): + if raw == alias_prefix or raw.startswith(alias_prefix + "/"): + rest = raw[len(alias_prefix):].lstrip("/") + resolved_alias = Path(os.path.normpath(Path(alias_base) / rest)) + break + if resolved_alias is not None: + tgt_nid = _make_id(str(resolved_alias)) + else: + # Bare/scoped import (node_modules) - use last segment; dropped as external + module_name = raw.split("/")[-1] + if not module_name: + break + tgt_nid = _make_id(module_name) edges.append({ "source": file_nid, "target": tgt_nid, @@ -676,7 +729,7 @@ def _extract_generic(path: Path, config: LanguageConfig) -> dict: except Exception as e: return {"nodes": [], "edges": [], "error": str(e)} - stem = path.stem + stem = _file_stem(path) str_path = str(path) nodes: list[dict] = [] edges: list[dict] = [] @@ -1301,7 +1354,7 @@ def _extract_python_rationale(path: Path, result: dict) -> None: except Exception: return - stem = path.stem + stem = _file_stem(path) str_path = str(path) nodes = result["nodes"] edges = result["edges"] @@ -1560,7 +1613,7 @@ def extract_verilog(path: Path) -> dict: except Exception as e: return {"nodes": [], "edges": [], "error": str(e)} - stem = path.stem + stem = _file_stem(path) str_path = str(path) nodes: list[dict] = [] edges: list[dict] = [] @@ -1676,7 +1729,7 @@ def extract_julia(path: Path) -> dict: except Exception as e: return {"nodes": [], "edges": [], "error": str(e)} - stem = path.stem + stem = _file_stem(path) str_path = str(path) nodes: list[dict] = [] edges: list[dict] = [] @@ -1888,7 +1941,7 @@ def extract_go(path: Path) -> dict: except Exception as e: return {"nodes": [], "edges": [], "error": str(e)} - stem = path.stem + stem = _file_stem(path) # Use directory name as package scope so methods on the same type across # multiple files in a package share one canonical type node. pkg_scope = path.parent.name or stem @@ -2087,7 +2140,7 @@ def extract_rust(path: Path) -> dict: except Exception as e: return {"nodes": [], "edges": [], "error": str(e)} - stem = path.stem + stem = _file_stem(path) str_path = str(path) nodes: list[dict] = [] edges: list[dict] = [] @@ -2264,7 +2317,7 @@ def extract_zig(path: Path) -> dict: except Exception as e: return {"nodes": [], "edges": [], "error": str(e)} - stem = path.stem + stem = _file_stem(path) str_path = str(path) nodes: list[dict] = [] edges: list[dict] = [] @@ -2427,7 +2480,7 @@ def extract_powershell(path: Path) -> dict: except Exception as e: return {"nodes": [], "edges": [], "error": str(e)} - stem = path.stem + stem = _file_stem(path) str_path = str(path) nodes: list[dict] = [] edges: list[dict] = [] @@ -2621,7 +2674,7 @@ def _resolve_cross_file_imports( stem_to_path: dict[str, Path] = {p.stem: p for p in paths} for file_result, path in zip(per_file, paths): - stem = path.stem + stem = _file_stem(path) str_path = str(path) # Find all classes defined in this file (the importers) @@ -2745,7 +2798,7 @@ def _resolve_cross_file_java_imports( new_edges: list[dict] = [] seen_pairs: set[tuple[str, str]] = set() for path in paths: - file_nid = _make_id(path.stem) + file_nid = _make_id(str(path)) try: source = path.read_bytes() tree = parser.parse(source) @@ -2809,7 +2862,7 @@ def extract_objc(path: Path) -> dict: except Exception as e: return {"nodes": [], "edges": [], "error": str(e)} - stem = path.stem + stem = _file_stem(path) str_path = str(path) nodes: list[dict] = [] edges: list[dict] = [] @@ -3007,7 +3060,7 @@ def extract_elixir(path: Path) -> dict: except Exception as e: return {"nodes": [], "edges": [], "error": str(e)} - stem = path.stem + stem = _file_stem(path) str_path = str(path) nodes: list[dict] = [] edges: list[dict] = [] @@ -3373,6 +3426,19 @@ def extract(paths: list[Path], cache_root: Path | None = None) -> dict: "weight": 1.0, }) + # Relativize source_file fields so paths are portable across machines (#555) + for item in all_nodes + all_edges: + sf = item.get("source_file") + if not sf: + continue + sf_path = Path(sf) + if not sf_path.is_absolute(): + continue + try: + item["source_file"] = str(sf_path.relative_to(root)) + except ValueError: + pass + return { "nodes": all_nodes, "edges": all_edges, diff --git a/graphify/skill-aider.md b/graphify/skill-aider.md index 7c745f3..fa7007f 100644 --- a/graphify/skill-aider.md +++ b/graphify/skill-aider.md @@ -235,7 +235,7 @@ Process each file one at a time. For each file: - INFERRED: reasonable inference (shared structure, implied dependency) - AMBIGUOUS: uncertain — flag it, do not omit - Code files: semantic edges AST cannot find. Do not re-extract imports. - - Doc/paper files: named concepts, entities, citations, and rationale nodes (WHY decisions were made → `rationale_for` edges) + - Doc/paper files: named concepts, entities, citations. Store rationale (WHY decisions were made) as a `rationale` attribute on the relevant node, not as a separate node. When adding `calls` edges: source is caller, target is callee. - Image files: use vision — understand what the image IS, not just OCR - DEEP_MODE (if --mode deep): be aggressive with INFERRED edges - Semantic similarity: if two concepts solve the same problem without a structural link, add `semantically_similar_to` INFERRED edge (confidence 0.6-0.95). Non-obvious cross-file links only. diff --git a/graphify/skill-claw.md b/graphify/skill-claw.md index 2f3b3a7..3d84acb 100644 --- a/graphify/skill-claw.md +++ b/graphify/skill-claw.md @@ -235,7 +235,7 @@ Process each file one at a time. For each file: - INFERRED: reasonable inference (shared structure, implied dependency) - AMBIGUOUS: uncertain — flag it, do not omit - Code files: semantic edges AST cannot find. Do not re-extract imports. - - Doc/paper files: named concepts, entities, citations, and rationale nodes (WHY decisions were made → `rationale_for` edges) + - Doc/paper files: named concepts, entities, citations. Store rationale (WHY decisions were made) as a `rationale` attribute on the relevant node, not as a separate node. When adding `calls` edges: source is caller, target is callee. - Image files: use vision — understand what the image IS, not just OCR - DEEP_MODE (if --mode deep): be aggressive with INFERRED edges - Semantic similarity: if two concepts solve the same problem without a structural link, add `semantically_similar_to` INFERRED edge (confidence 0.6-0.95). Non-obvious cross-file links only. diff --git a/graphify/skill-codex.md b/graphify/skill-codex.md index f3c4408..b2e79e2 100644 --- a/graphify/skill-codex.md +++ b/graphify/skill-codex.md @@ -263,7 +263,8 @@ Rules: Code files: focus on semantic edges AST cannot find (call relationships, shared data, arch patterns). Do not re-extract imports - AST already has those. -Doc/paper files: extract named concepts, entities, citations. Also extract rationale — sections that explain WHY a decision was made, trade-offs chosen, or design intent. These become nodes with `rationale_for` edges pointing to the concept they explain. +Doc/paper files: extract named concepts, entities, citations. For rationale (WHY decisions were made, trade-offs, design intent): store as a `rationale` attribute on the relevant concept node — do NOT create a separate rationale node or fragment node. Only create a node for something that is itself a named entity or concept. +Code files: when adding `calls` edges, source MUST be the caller (the function/class doing the calling), target MUST be the callee. Never reverse this direction. Image files: use vision to understand what the image IS - do not just OCR. UI screenshot: layout patterns, design decisions, key elements, purpose. Chart: metric, trend/insight, data source. diff --git a/graphify/skill-copilot.md b/graphify/skill-copilot.md index cfccee5..397669a 100644 --- a/graphify/skill-copilot.md +++ b/graphify/skill-copilot.md @@ -259,7 +259,8 @@ Rules: Code files: focus on semantic edges AST cannot find (call relationships, shared data, arch patterns). Do not re-extract imports - AST already has those. -Doc/paper files: extract named concepts, entities, citations. Also extract rationale — sections that explain WHY a decision was made, trade-offs chosen, or design intent. These become nodes with `rationale_for` edges pointing to the concept they explain. +Doc/paper files: extract named concepts, entities, citations. For rationale (WHY decisions were made, trade-offs, design intent): store as a `rationale` attribute on the relevant concept node — do NOT create a separate rationale node or fragment node. Only create a node for something that is itself a named entity or concept. +Code files: when adding `calls` edges, source MUST be the caller (the function/class doing the calling), target MUST be the callee. Never reverse this direction. Image files: use vision to understand what the image IS - do not just OCR. UI screenshot: layout patterns, design decisions, key elements, purpose. Chart: metric, trend/insight, data source. diff --git a/graphify/skill-droid.md b/graphify/skill-droid.md index 1f25735..6cde935 100644 --- a/graphify/skill-droid.md +++ b/graphify/skill-droid.md @@ -260,7 +260,8 @@ Rules: Code files: focus on semantic edges AST cannot find (call relationships, shared data, arch patterns). Do not re-extract imports - AST already has those. -Doc/paper files: extract named concepts, entities, citations. Also extract rationale — sections that explain WHY a decision was made, trade-offs chosen, or design intent. These become nodes with `rationale_for` edges pointing to the concept they explain. +Doc/paper files: extract named concepts, entities, citations. For rationale (WHY decisions were made, trade-offs, design intent): store as a `rationale` attribute on the relevant concept node — do NOT create a separate rationale node or fragment node. Only create a node for something that is itself a named entity or concept. +Code files: when adding `calls` edges, source MUST be the caller (the function/class doing the calling), target MUST be the callee. Never reverse this direction. Image files: use vision to understand what the image IS - do not just OCR. UI screenshot: layout patterns, design decisions, key elements, purpose. Chart: metric, trend/insight, data source. diff --git a/graphify/skill-kiro.md b/graphify/skill-kiro.md index f04bc0f..b3db443 100644 --- a/graphify/skill-kiro.md +++ b/graphify/skill-kiro.md @@ -234,7 +234,7 @@ Process each file one at a time. For each file: - INFERRED: reasonable inference (shared structure, implied dependency) - AMBIGUOUS: uncertain — flag it, do not omit - Code files: semantic edges AST cannot find. Do not re-extract imports. - - Doc/paper files: named concepts, entities, citations, and rationale nodes (WHY decisions were made → `rationale_for` edges) + - Doc/paper files: named concepts, entities, citations. Store rationale (WHY decisions were made) as a `rationale` attribute on the relevant node, not as a separate node. When adding `calls` edges: source is caller, target is callee. - Image files: use vision — understand what the image IS, not just OCR - DEEP_MODE (if --mode deep): be aggressive with INFERRED edges - Semantic similarity: if two concepts solve the same problem without a structural link, add `semantically_similar_to` INFERRED edge (confidence 0.6-0.95). Non-obvious cross-file links only. diff --git a/graphify/skill-opencode.md b/graphify/skill-opencode.md index 479b677..32819c8 100644 --- a/graphify/skill-opencode.md +++ b/graphify/skill-opencode.md @@ -261,7 +261,8 @@ Rules: Code files: focus on semantic edges AST cannot find (call relationships, shared data, arch patterns). Do not re-extract imports - AST already has those. -Doc/paper files: extract named concepts, entities, citations. Also extract rationale — sections that explain WHY a decision was made, trade-offs chosen, or design intent. These become nodes with `rationale_for` edges pointing to the concept they explain. +Doc/paper files: extract named concepts, entities, citations. For rationale (WHY decisions were made, trade-offs, design intent): store as a `rationale` attribute on the relevant concept node — do NOT create a separate rationale node or fragment node. Only create a node for something that is itself a named entity or concept. +Code files: when adding `calls` edges, source MUST be the caller (the function/class doing the calling), target MUST be the callee. Never reverse this direction. Image files: use vision to understand what the image IS - do not just OCR. UI screenshot: layout patterns, design decisions, key elements, purpose. Chart: metric, trend/insight, data source. diff --git a/graphify/skill-trae.md b/graphify/skill-trae.md index 5dfdc2d..2b5b401 100644 --- a/graphify/skill-trae.md +++ b/graphify/skill-trae.md @@ -250,7 +250,8 @@ Rules: Code files: focus on semantic edges AST cannot find (call relationships, shared data, arch patterns). Do not re-extract imports - AST already has those. -Doc/paper files: extract named concepts, entities, citations. Also extract rationale — sections that explain WHY a decision was made, trade-offs chosen, or design intent. These become nodes with `rationale_for` edges pointing to the concept they explain. +Doc/paper files: extract named concepts, entities, citations. For rationale (WHY decisions were made, trade-offs, design intent): store as a `rationale` attribute on the relevant concept node — do NOT create a separate rationale node or fragment node. Only create a node for something that is itself a named entity or concept. +Code files: when adding `calls` edges, source MUST be the caller (the function/class doing the calling), target MUST be the callee. Never reverse this direction. Image files: use vision to understand what the image IS - do not just OCR. UI screenshot: layout patterns, design decisions, key elements, purpose. Chart: metric, trend/insight, data source. diff --git a/graphify/skill-windows.md b/graphify/skill-windows.md index 8aa0482..9984022 100644 --- a/graphify/skill-windows.md +++ b/graphify/skill-windows.md @@ -249,7 +249,8 @@ Rules: Code files: focus on semantic edges AST cannot find (call relationships, shared data, arch patterns). Do not re-extract imports - AST already has those. -Doc/paper files: extract named concepts, entities, citations. Also extract rationale — sections that explain WHY a decision was made, trade-offs chosen, or design intent. These become nodes with `rationale_for` edges pointing to the concept they explain. +Doc/paper files: extract named concepts, entities, citations. For rationale (WHY decisions were made, trade-offs, design intent): store as a `rationale` attribute on the relevant concept node — do NOT create a separate rationale node or fragment node. Only create a node for something that is itself a named entity or concept. +Code files: when adding `calls` edges, source MUST be the caller (the function/class doing the calling), target MUST be the callee. Never reverse this direction. Image files: use vision to understand what the image IS - do not just OCR. UI screenshot: layout patterns, design decisions, key elements, purpose. Chart: metric, trend/insight, data source. diff --git a/graphify/skill.md b/graphify/skill.md index 60d2422..be1e7db 100644 --- a/graphify/skill.md +++ b/graphify/skill.md @@ -300,7 +300,8 @@ Rules: Code files: focus on semantic edges AST cannot find (call relationships, shared data, arch patterns). Do not re-extract imports - AST already has those. -Doc/paper files: extract named concepts, entities, citations. Also extract rationale — sections that explain WHY a decision was made, trade-offs chosen, or design intent. These become nodes with `rationale_for` edges pointing to the concept they explain. +Doc/paper files: extract named concepts, entities, citations. For rationale (WHY decisions were made, trade-offs, design intent): store as a `rationale` attribute on the relevant concept node — do NOT create a separate rationale node or fragment node. Only create a node for something that is itself a named entity or concept. +Code files: when adding `calls` edges, source MUST be the caller (the function/class doing the calling), target MUST be the callee. Never reverse this direction. Image files: use vision to understand what the image IS - do not just OCR. UI screenshot: layout patterns, design decisions, key elements, purpose. Chart: metric, trend/insight, data source. diff --git a/graphify/watch.py b/graphify/watch.py index a09dd51..b0e8e74 100644 --- a/graphify/watch.py +++ b/graphify/watch.py @@ -99,10 +99,13 @@ def _rebuild_code(watch_path: Path, *, follow_symlinks: bool = False) -> bool: out.mkdir(exist_ok=True) + json_written = to_json(G, communities, str(out / "graph.json")) + if not json_written: + return False + report = generate(G, communities, cohesion, labels, gods, surprises, detection, {"input": 0, "output": 0}, report_root, suggested_questions=questions) (out / "GRAPH_REPORT.md").write_text(report, encoding="utf-8") - to_json(G, communities, str(out / "graph.json")) # to_html raises ValueError for graphs > MAX_NODES_FOR_VIZ (5000). # Wrap so core outputs (graph.json + GRAPH_REPORT.md) always land.