From 87083334847b3e1c932766e7c3b25b4ac648aa8b Mon Sep 17 00:00:00 2001
From: Safi
Date: Sun, 5 Apr 2026 00:16:54 +0100
Subject: [PATCH] feat: vis.js HTML graph, token reduction benchmark, repo
cleanup
- Replace pyvis with custom vis.js renderer: node size by degree,
click-to-inspect panel with clickable neighbors, search box,
community filter, physics clustering by community
- HTML graph generated by default on every run (no --html flag needed)
- Token reduction benchmark auto-runs after every /graphify on corpora >5k words
- Fix 292 edge warnings: silently skip stdlib/external edges in build.py
- Fix build() to merge extractions before building (cross-extraction edges were dropped)
- Add 5 HTML renderer tests (223 total)
- Remove unnecessary files: lib/, tests/eval_attention.py, misplaced eval reports
- Add graphify-out/ and .graphify_*.json to .gitignore
- Bump version to 0.1.4, remove pyvis dependency
- README: token reduction as top-level selling point, vis.js in tech stack,
graph.html in output listing, correct test count and install command
---
.gitignore | 2 +
README.md | 18 +-
graphify/build.py | 26 ++-
graphify/export.py | 314 +++++++++++++++++++++++-----
graphify/skill.md | 51 ++++-
pyproject.toml | 3 +-
skills/graphify/skill.md | 32 ++-
tests/EVAL_httpx.md | 401 ------------------------------------
tests/EVAL_mixed_corpus.md | 176 ----------------
tests/GRAPH_REPORT_httpx.md | 62 ------
tests/eval_attention.py | 147 -------------
tests/test_export.py | 48 ++++-
12 files changed, 408 insertions(+), 872 deletions(-)
delete mode 100644 tests/EVAL_httpx.md
delete mode 100644 tests/EVAL_mixed_corpus.md
delete mode 100644 tests/GRAPH_REPORT_httpx.md
delete mode 100644 tests/eval_attention.py
diff --git a/.gitignore b/.gitignore
index 9d2498c..b6215f8 100644
--- a/.gitignore
+++ b/.gitignore
@@ -13,3 +13,5 @@ build/
*.so
*.egg
.graphify/
+graphify-out/
+.graphify_*.json
diff --git a/README.md b/README.md
index c2d76d8..06bcbb0 100644
--- a/README.md
+++ b/README.md
@@ -12,6 +12,7 @@
```
graphify-out/
+├── graph.html interactive graph - click nodes, search, filter by community, open in any browser
├── obsidian/ open as Obsidian vault - visual graph, wikilinks, filter by community
├── GRAPH_REPORT.md what the graph found: god nodes, surprising connections, suggested questions
├── graph.json persistent graph - query it weeks later without re-reading anything
@@ -31,11 +32,13 @@ graphify takes that observation and builds the missing infrastructure:
| Claude hallucinates missing links | `EXTRACTED` / `INFERRED` / `AMBIGUOUS` - honest about what was found vs guessed |
| Context resets every session | Memory feedback loop - what you ask grows the graph on `--update` |
| Only works on text | PDFs, images, screenshots, tweets, any language via vision |
+| Reading everything costs tokens | **71.5x token reduction** on large mixed corpora - query the graph, not the files |
**What LLMs get wrong without it:** Naive summarization fills every gap confidently. You get output that sounds complete but you can't tell what was actually in the files vs invented. And next session, it's all gone.
**What graphify does differently:**
+- **71.5x token reduction** - on a mixed corpus (Karpathy repos + papers + images), querying the graph costs 71.5x fewer tokens than reading the raw files. The benchmark runs automatically after every `/graphify` run.
- **Persistent graph** - relationships stored in `graphify-out/graph.json`, survive across sessions. Query weeks later without re-reading anything.
- **Honest audit trail** - every edge tagged `EXTRACTED` (explicitly stated), `INFERRED` (call-graph or reasonable deduction), or `AMBIGUOUS` (flagged for review). You always know what was found vs invented.
- **Cross-document surprise** - Leiden community detection finds clusters, then surfaces cross-community connections: the things you would never think to ask about directly.
@@ -105,7 +108,6 @@ All commands are typed inside Claude Code:
/graphify path "DigestAuth" "Response" # shortest path between two concepts
/graphify explain "SwinTransformer" # plain-language node explanation
-/graphify ./raw --html # also export graph.html (browser, no Obsidian needed)
/graphify ./raw --svg # also export graph.svg (embeds in Notion, GitHub)
/graphify ./raw --graphml # also export graph.graphml (Gephi, yEd, any GraphML tool)
/graphify ./raw --neo4j # generate cypher.txt for Neo4j import
@@ -127,16 +129,19 @@ After running, Claude outputs three things directly in chat:
**God nodes** - highest-degree concepts (what everything connects through)
-**Surprising connections** - ranked by a composite surprise score, not just confidence. A code↔paper edge scores higher than code↔code. A cross-repo connection scores higher than same-repo. Each result includes a plain-English `why` explaining what makes it non-obvious.
+**Surprising connections** - ranked by a composite surprise score, not just confidence. A code-paper edge scores higher than code-code. A cross-repo connection scores higher than same-repo. Each result includes a plain-English `why` explaining what makes it non-obvious.
**Suggested questions** - 4-5 questions the graph is uniquely positioned to answer, with the reason why (which bridge node makes it interesting, which community boundary it crosses)
The full GRAPH_REPORT.md adds community summaries with cohesion scores and a list of ambiguous edges for review.
+**Token reduction benchmark** - automatically printed after every run on corpora over 5,000 words. Shows how many fewer tokens querying the graph costs vs reading the raw files directly.
+
## Key files explained
| File | Purpose |
|------|---------|
+| `graph.html` | Interactive vis.js graph. Node size = degree. Click any node for details + clickable neighbors. Search by name. Filter by community. Opens in any browser. |
| `GRAPH_REPORT.md` | The audit report. God nodes, surprising connections, community cohesion scores, ambiguous edge list, suggested questions. |
| `graph.json` | Persistent graph in node-link format. Load it with NetworkX or push to Neo4j. Survives sessions. |
| `obsidian/` | Wikilink vault. Open in Obsidian → enable graph view → see communities as clusters. Filter by tag, search across everything. |
@@ -205,7 +210,7 @@ Each includes the full graph output and an honest evaluation of what the skill g
| Community detection | Leiden via graspologic | Better than K-means for sparse graphs |
| Code parsing | tree-sitter | Multi-language AST, deterministic, zero hallucination |
| Extraction | Claude (parallel subagents) | Reads anything, outputs structured graph data |
-| Visualization | Obsidian vault | Native graph view, wikilinks, no server needed |
+| Visualization | vis.js (HTML) + Obsidian vault | Interactive browser graph + wikilink vault, no server needed |
No Neo4j required. No dashboards. No server. Runs entirely locally.
@@ -219,12 +224,13 @@ graphify/
├── cluster.py Leiden community detection, cohesion scoring
├── analyze.py god nodes, bridge nodes, surprising connections, suggested questions, graph diff
├── report.py render GRAPH_REPORT.md
-├── export.py Obsidian vault, graph.json, graph.html, graph.svg, graph.graphml, Neo4j Cypher, Canvas
+├── export.py Obsidian vault, graph.json, graph.html (vis.js), graph.svg, graph.graphml, Neo4j Cypher, Canvas
├── ingest.py fetch URLs (arXiv, Twitter/X, PDF, any webpage); save Q&A to graphify-out/memory/
├── cache.py SHA256-based per-file extraction cache; check_semantic_cache / save_semantic_cache
├── security.py URL validation (http/https only), safe fetch with size cap, path guards, label sanitisation
├── validate.py JSON schema checks on extraction output
├── serve.py MCP stdio server - query_graph, get_node, get_neighbors, shortest_path, god_nodes
+├── benchmark.py token reduction benchmark - corpus tokens vs graph query tokens
└── watch.py fs watcher, writes flag file when new files appear
skills/graphify/
@@ -233,6 +239,6 @@ skills/graphify/
ARCHITECTURE.md module responsibilities, extraction schema, how to add a language
SECURITY.md threat model, mitigations, vulnerability reporting
worked/ eval reports from real corpora (karpathy-repos, httpx, mixed-corpus)
-tests/ 218 tests, one file per module
-pyproject.toml pip install graphify | pip install graphify[mcp,neo4j,pdf,watch]
+tests/ 223 tests, one file per module
+pyproject.toml pip install graphifyy | pip install graphifyy[mcp,neo4j,pdf,watch]
```
diff --git a/graphify/build.py b/graphify/build.py
index 02e6ac0..655820c 100644
--- a/graphify/build.py
+++ b/graphify/build.py
@@ -7,25 +7,33 @@ from .validate import validate_extraction
def build_from_json(extraction: dict) -> nx.Graph:
errors = validate_extraction(extraction)
- if errors:
- print(f"[graphify] Extraction warning ({len(errors)} issues): {errors[0]}", file=sys.stderr)
+ # Dangling edges (stdlib/external imports) are expected - only warn about real schema errors.
+ real_errors = [e for e in errors if "does not match any node id" not in e]
+ if real_errors:
+ print(f"[graphify] Extraction warning ({len(real_errors)} issues): {real_errors[0]}", file=sys.stderr)
G = nx.Graph()
for node in extraction.get("nodes", []):
G.add_node(node["id"], **{k: v for k, v in node.items() if k != "id"})
+ node_set = set(G.nodes())
for edge in extraction.get("edges", []):
+ src, tgt = edge["source"], edge["target"]
+ if src not in node_set or tgt not in node_set:
+ continue # skip edges to external/stdlib nodes - expected, not an error
attrs = {k: v for k, v in edge.items() if k not in ("source", "target")}
# Preserve original edge direction - undirected graphs lose it otherwise,
# causing display functions to show edges backwards.
- attrs["_src"] = edge["source"]
- attrs["_tgt"] = edge["target"]
- G.add_edge(edge["source"], edge["target"], **attrs)
+ attrs["_src"] = src
+ attrs["_tgt"] = tgt
+ G.add_edge(src, tgt, **attrs)
return G
def build(extractions: list[dict]) -> nx.Graph:
"""Merge multiple extraction results into one graph."""
- G = nx.Graph()
+ combined: dict = {"nodes": [], "edges": [], "input_tokens": 0, "output_tokens": 0}
for ext in extractions:
- sub = build_from_json(ext)
- G.update(sub)
- return G
+ combined["nodes"].extend(ext.get("nodes", []))
+ combined["edges"].extend(ext.get("edges", []))
+ combined["input_tokens"] += ext.get("input_tokens", 0)
+ combined["output_tokens"] += ext.get("output_tokens", 0)
+ return build_from_json(combined)
diff --git a/graphify/export.py b/graphify/export.py
index a52c611..9035f3d 100644
--- a/graphify/export.py
+++ b/graphify/export.py
@@ -50,73 +50,285 @@ def to_html(
output_path: str,
community_labels: dict[int, str] | None = None,
) -> None:
- """Generate an interactive pyvis HTML visualization of the graph.
+ """Generate an interactive vis.js HTML visualization of the graph.
- Merged from visualizer.py. Raises ValueError if graph exceeds MAX_NODES_FOR_VIZ.
+ Features: node size by degree, click-to-inspect panel, search box,
+ community filter, physics clustering by community, confidence-styled edges.
+ Raises ValueError if graph exceeds MAX_NODES_FOR_VIZ.
"""
- from pyvis.network import Network
-
if G.number_of_nodes() > MAX_NODES_FOR_VIZ:
raise ValueError(
- f"Graph has {G.number_of_nodes()} nodes - too large for pyvis. "
+ f"Graph has {G.number_of_nodes()} nodes - too large for HTML viz. "
f"Use --no-viz or reduce input size."
)
node_community = {n: cid for cid, nodes in communities.items() for n in nodes}
+ degree = dict(G.degree())
+ max_deg = max(degree.values()) if degree else 1
- net = Network(height="800px", width="100%", bgcolor="#1a1a2e", font_color="white")
- net.barnes_hut()
-
+ # Build nodes list for vis.js
+ vis_nodes = []
for node_id, data in G.nodes(data=True):
cid = node_community.get(node_id, 0)
color = COMMUNITY_COLORS[cid % len(COMMUNITY_COLORS)]
- net.add_node(
- node_id,
- label=sanitize_label(data.get("label", node_id)),
- color=color,
- title=sanitize_label(
- f"Source: {data.get('source_file', 'unknown')}\n"
- f"Type: {data.get('file_type', 'unknown')}\n"
- f"Community: {community_labels.get(cid, str(cid)) if community_labels else cid}"
- ),
- )
+ label = sanitize_label(data.get("label", node_id))
+ deg = degree.get(node_id, 1)
+ size = 10 + 30 * (deg / max_deg)
+ # Only show label for high-degree nodes by default; others show on hover
+ font_size = 12 if deg >= max_deg * 0.15 else 0
+ vis_nodes.append({
+ "id": node_id,
+ "label": label,
+ "color": {"background": color, "border": color, "highlight": {"background": "#ffffff", "border": color}},
+ "size": round(size, 1),
+ "font": {"size": font_size, "color": "#ffffff"},
+ "title": f"{label}",
+ "community": cid,
+ "community_name": (community_labels or {}).get(cid, f"Community {cid}"),
+ "source_file": sanitize_label(data.get("source_file", "")),
+ "file_type": data.get("file_type", ""),
+ "degree": deg,
+ })
+ # Build edges list
+ vis_edges = []
for u, v, data in G.edges(data=True):
confidence = data.get("confidence", "EXTRACTED")
- width = {"EXTRACTED": 2, "INFERRED": 1, "AMBIGUOUS": 1}.get(confidence, 1)
- net.add_edge(
- u, v,
- title=f"{data.get('relation', '')} [{confidence}]",
- width=width,
- dashes=(confidence != "EXTRACTED"),
- )
+ relation = data.get("relation", "")
+ vis_edges.append({
+ "from": u,
+ "to": v,
+ "label": relation,
+ "title": f"{relation} [{confidence}]",
+ "dashes": confidence != "EXTRACTED",
+ "width": 2 if confidence == "EXTRACTED" else 1,
+ "color": {"opacity": 0.7 if confidence == "EXTRACTED" else 0.35},
+ "confidence": confidence,
+ })
- net.save_graph(output_path)
+ # Build community legend data
+ legend_data = []
+ for cid in sorted((community_labels or {}).keys()):
+ color = COMMUNITY_COLORS[cid % len(COMMUNITY_COLORS)]
+ lbl = (community_labels or {}).get(cid, f"Community {cid}")
+ n = len(communities.get(cid, []))
+ legend_data.append({"cid": cid, "color": color, "label": lbl, "count": n})
- # Inject community legend into saved HTML
- if community_labels:
- legend_items = ""
- for cid in sorted(community_labels.keys()):
- color = COMMUNITY_COLORS[cid % len(COMMUNITY_COLORS)]
- label = community_labels[cid]
- n_nodes = len(communities.get(cid, []))
- legend_items += (
- f''
- f'■ '
- f'{label} ({n_nodes})'
- f'
'
- )
- legend_html = (
- ''
- 'Communities
'
- + legend_items +
- '
'
- )
- content = Path(output_path).read_text()
- content = content.replace("
+
+
+
+
+", legend_html + "\n")
- Path(output_path).write_text(content)
+ nodes_json = json.dumps(vis_nodes)
+ edges_json = json.dumps(vis_edges)
+ legend_json = json.dumps(legend_data)
+ title = sanitize_label(str(output_path))
+
+ html = f"""
+
+