diff --git a/.gitignore b/.gitignore index 9d2498c..b6215f8 100644 --- a/.gitignore +++ b/.gitignore @@ -13,3 +13,5 @@ build/ *.so *.egg .graphify/ +graphify-out/ +.graphify_*.json diff --git a/README.md b/README.md index c2d76d8..06bcbb0 100644 --- a/README.md +++ b/README.md @@ -12,6 +12,7 @@ ``` graphify-out/ +├── graph.html interactive graph - click nodes, search, filter by community, open in any browser ├── obsidian/ open as Obsidian vault - visual graph, wikilinks, filter by community ├── GRAPH_REPORT.md what the graph found: god nodes, surprising connections, suggested questions ├── graph.json persistent graph - query it weeks later without re-reading anything @@ -31,11 +32,13 @@ graphify takes that observation and builds the missing infrastructure: | Claude hallucinates missing links | `EXTRACTED` / `INFERRED` / `AMBIGUOUS` - honest about what was found vs guessed | | Context resets every session | Memory feedback loop - what you ask grows the graph on `--update` | | Only works on text | PDFs, images, screenshots, tweets, any language via vision | +| Reading everything costs tokens | **71.5x token reduction** on large mixed corpora - query the graph, not the files | **What LLMs get wrong without it:** Naive summarization fills every gap confidently. You get output that sounds complete but you can't tell what was actually in the files vs invented. And next session, it's all gone. **What graphify does differently:** +- **71.5x token reduction** - on a mixed corpus (Karpathy repos + papers + images), querying the graph costs 71.5x fewer tokens than reading the raw files. The benchmark runs automatically after every `/graphify` run. - **Persistent graph** - relationships stored in `graphify-out/graph.json`, survive across sessions. Query weeks later without re-reading anything. - **Honest audit trail** - every edge tagged `EXTRACTED` (explicitly stated), `INFERRED` (call-graph or reasonable deduction), or `AMBIGUOUS` (flagged for review). You always know what was found vs invented. - **Cross-document surprise** - Leiden community detection finds clusters, then surfaces cross-community connections: the things you would never think to ask about directly. @@ -105,7 +108,6 @@ All commands are typed inside Claude Code: /graphify path "DigestAuth" "Response" # shortest path between two concepts /graphify explain "SwinTransformer" # plain-language node explanation -/graphify ./raw --html # also export graph.html (browser, no Obsidian needed) /graphify ./raw --svg # also export graph.svg (embeds in Notion, GitHub) /graphify ./raw --graphml # also export graph.graphml (Gephi, yEd, any GraphML tool) /graphify ./raw --neo4j # generate cypher.txt for Neo4j import @@ -127,16 +129,19 @@ After running, Claude outputs three things directly in chat: **God nodes** - highest-degree concepts (what everything connects through) -**Surprising connections** - ranked by a composite surprise score, not just confidence. A code↔paper edge scores higher than code↔code. A cross-repo connection scores higher than same-repo. Each result includes a plain-English `why` explaining what makes it non-obvious. +**Surprising connections** - ranked by a composite surprise score, not just confidence. A code-paper edge scores higher than code-code. A cross-repo connection scores higher than same-repo. Each result includes a plain-English `why` explaining what makes it non-obvious. **Suggested questions** - 4-5 questions the graph is uniquely positioned to answer, with the reason why (which bridge node makes it interesting, which community boundary it crosses) The full GRAPH_REPORT.md adds community summaries with cohesion scores and a list of ambiguous edges for review. +**Token reduction benchmark** - automatically printed after every run on corpora over 5,000 words. Shows how many fewer tokens querying the graph costs vs reading the raw files directly. + ## Key files explained | File | Purpose | |------|---------| +| `graph.html` | Interactive vis.js graph. Node size = degree. Click any node for details + clickable neighbors. Search by name. Filter by community. Opens in any browser. | | `GRAPH_REPORT.md` | The audit report. God nodes, surprising connections, community cohesion scores, ambiguous edge list, suggested questions. | | `graph.json` | Persistent graph in node-link format. Load it with NetworkX or push to Neo4j. Survives sessions. | | `obsidian/` | Wikilink vault. Open in Obsidian → enable graph view → see communities as clusters. Filter by tag, search across everything. | @@ -205,7 +210,7 @@ Each includes the full graph output and an honest evaluation of what the skill g | Community detection | Leiden via graspologic | Better than K-means for sparse graphs | | Code parsing | tree-sitter | Multi-language AST, deterministic, zero hallucination | | Extraction | Claude (parallel subagents) | Reads anything, outputs structured graph data | -| Visualization | Obsidian vault | Native graph view, wikilinks, no server needed | +| Visualization | vis.js (HTML) + Obsidian vault | Interactive browser graph + wikilink vault, no server needed | No Neo4j required. No dashboards. No server. Runs entirely locally. @@ -219,12 +224,13 @@ graphify/ ├── cluster.py Leiden community detection, cohesion scoring ├── analyze.py god nodes, bridge nodes, surprising connections, suggested questions, graph diff ├── report.py render GRAPH_REPORT.md -├── export.py Obsidian vault, graph.json, graph.html, graph.svg, graph.graphml, Neo4j Cypher, Canvas +├── export.py Obsidian vault, graph.json, graph.html (vis.js), graph.svg, graph.graphml, Neo4j Cypher, Canvas ├── ingest.py fetch URLs (arXiv, Twitter/X, PDF, any webpage); save Q&A to graphify-out/memory/ ├── cache.py SHA256-based per-file extraction cache; check_semantic_cache / save_semantic_cache ├── security.py URL validation (http/https only), safe fetch with size cap, path guards, label sanitisation ├── validate.py JSON schema checks on extraction output ├── serve.py MCP stdio server - query_graph, get_node, get_neighbors, shortest_path, god_nodes +├── benchmark.py token reduction benchmark - corpus tokens vs graph query tokens └── watch.py fs watcher, writes flag file when new files appear skills/graphify/ @@ -233,6 +239,6 @@ skills/graphify/ ARCHITECTURE.md module responsibilities, extraction schema, how to add a language SECURITY.md threat model, mitigations, vulnerability reporting worked/ eval reports from real corpora (karpathy-repos, httpx, mixed-corpus) -tests/ 218 tests, one file per module -pyproject.toml pip install graphify | pip install graphify[mcp,neo4j,pdf,watch] +tests/ 223 tests, one file per module +pyproject.toml pip install graphifyy | pip install graphifyy[mcp,neo4j,pdf,watch] ``` diff --git a/graphify/build.py b/graphify/build.py index 02e6ac0..655820c 100644 --- a/graphify/build.py +++ b/graphify/build.py @@ -7,25 +7,33 @@ from .validate import validate_extraction def build_from_json(extraction: dict) -> nx.Graph: errors = validate_extraction(extraction) - if errors: - print(f"[graphify] Extraction warning ({len(errors)} issues): {errors[0]}", file=sys.stderr) + # Dangling edges (stdlib/external imports) are expected - only warn about real schema errors. + real_errors = [e for e in errors if "does not match any node id" not in e] + if real_errors: + print(f"[graphify] Extraction warning ({len(real_errors)} issues): {real_errors[0]}", file=sys.stderr) G = nx.Graph() for node in extraction.get("nodes", []): G.add_node(node["id"], **{k: v for k, v in node.items() if k != "id"}) + node_set = set(G.nodes()) for edge in extraction.get("edges", []): + src, tgt = edge["source"], edge["target"] + if src not in node_set or tgt not in node_set: + continue # skip edges to external/stdlib nodes - expected, not an error attrs = {k: v for k, v in edge.items() if k not in ("source", "target")} # Preserve original edge direction - undirected graphs lose it otherwise, # causing display functions to show edges backwards. - attrs["_src"] = edge["source"] - attrs["_tgt"] = edge["target"] - G.add_edge(edge["source"], edge["target"], **attrs) + attrs["_src"] = src + attrs["_tgt"] = tgt + G.add_edge(src, tgt, **attrs) return G def build(extractions: list[dict]) -> nx.Graph: """Merge multiple extraction results into one graph.""" - G = nx.Graph() + combined: dict = {"nodes": [], "edges": [], "input_tokens": 0, "output_tokens": 0} for ext in extractions: - sub = build_from_json(ext) - G.update(sub) - return G + combined["nodes"].extend(ext.get("nodes", [])) + combined["edges"].extend(ext.get("edges", [])) + combined["input_tokens"] += ext.get("input_tokens", 0) + combined["output_tokens"] += ext.get("output_tokens", 0) + return build_from_json(combined) diff --git a/graphify/export.py b/graphify/export.py index a52c611..9035f3d 100644 --- a/graphify/export.py +++ b/graphify/export.py @@ -50,73 +50,285 @@ def to_html( output_path: str, community_labels: dict[int, str] | None = None, ) -> None: - """Generate an interactive pyvis HTML visualization of the graph. + """Generate an interactive vis.js HTML visualization of the graph. - Merged from visualizer.py. Raises ValueError if graph exceeds MAX_NODES_FOR_VIZ. + Features: node size by degree, click-to-inspect panel, search box, + community filter, physics clustering by community, confidence-styled edges. + Raises ValueError if graph exceeds MAX_NODES_FOR_VIZ. """ - from pyvis.network import Network - if G.number_of_nodes() > MAX_NODES_FOR_VIZ: raise ValueError( - f"Graph has {G.number_of_nodes()} nodes - too large for pyvis. " + f"Graph has {G.number_of_nodes()} nodes - too large for HTML viz. " f"Use --no-viz or reduce input size." ) node_community = {n: cid for cid, nodes in communities.items() for n in nodes} + degree = dict(G.degree()) + max_deg = max(degree.values()) if degree else 1 - net = Network(height="800px", width="100%", bgcolor="#1a1a2e", font_color="white") - net.barnes_hut() - + # Build nodes list for vis.js + vis_nodes = [] for node_id, data in G.nodes(data=True): cid = node_community.get(node_id, 0) color = COMMUNITY_COLORS[cid % len(COMMUNITY_COLORS)] - net.add_node( - node_id, - label=sanitize_label(data.get("label", node_id)), - color=color, - title=sanitize_label( - f"Source: {data.get('source_file', 'unknown')}\n" - f"Type: {data.get('file_type', 'unknown')}\n" - f"Community: {community_labels.get(cid, str(cid)) if community_labels else cid}" - ), - ) + label = sanitize_label(data.get("label", node_id)) + deg = degree.get(node_id, 1) + size = 10 + 30 * (deg / max_deg) + # Only show label for high-degree nodes by default; others show on hover + font_size = 12 if deg >= max_deg * 0.15 else 0 + vis_nodes.append({ + "id": node_id, + "label": label, + "color": {"background": color, "border": color, "highlight": {"background": "#ffffff", "border": color}}, + "size": round(size, 1), + "font": {"size": font_size, "color": "#ffffff"}, + "title": f"{label}", + "community": cid, + "community_name": (community_labels or {}).get(cid, f"Community {cid}"), + "source_file": sanitize_label(data.get("source_file", "")), + "file_type": data.get("file_type", ""), + "degree": deg, + }) + # Build edges list + vis_edges = [] for u, v, data in G.edges(data=True): confidence = data.get("confidence", "EXTRACTED") - width = {"EXTRACTED": 2, "INFERRED": 1, "AMBIGUOUS": 1}.get(confidence, 1) - net.add_edge( - u, v, - title=f"{data.get('relation', '')} [{confidence}]", - width=width, - dashes=(confidence != "EXTRACTED"), - ) + relation = data.get("relation", "") + vis_edges.append({ + "from": u, + "to": v, + "label": relation, + "title": f"{relation} [{confidence}]", + "dashes": confidence != "EXTRACTED", + "width": 2 if confidence == "EXTRACTED" else 1, + "color": {"opacity": 0.7 if confidence == "EXTRACTED" else 0.35}, + "confidence": confidence, + }) - net.save_graph(output_path) + # Build community legend data + legend_data = [] + for cid in sorted((community_labels or {}).keys()): + color = COMMUNITY_COLORS[cid % len(COMMUNITY_COLORS)] + lbl = (community_labels or {}).get(cid, f"Community {cid}") + n = len(communities.get(cid, [])) + legend_data.append({"cid": cid, "color": color, "label": lbl, "count": n}) - # Inject community legend into saved HTML - if community_labels: - legend_items = "" - for cid in sorted(community_labels.keys()): - color = COMMUNITY_COLORS[cid % len(COMMUNITY_COLORS)] - label = community_labels[cid] - n_nodes = len(communities.get(cid, [])) - legend_items += ( - f'
", legend_html + "\n") - Path(output_path).write_text(content) + nodes_json = json.dumps(vis_nodes) + edges_json = json.dumps(vis_edges) + legend_json = json.dumps(legend_data) + title = sanitize_label(str(output_path)) + + html = f""" + +
+ +
+ + + +
+