diff --git a/.gitignore b/.gitignore index 9d2498c..b6215f8 100644 --- a/.gitignore +++ b/.gitignore @@ -13,3 +13,5 @@ build/ *.so *.egg .graphify/ +graphify-out/ +.graphify_*.json diff --git a/README.md b/README.md index c2d76d8..06bcbb0 100644 --- a/README.md +++ b/README.md @@ -12,6 +12,7 @@ ``` graphify-out/ +├── graph.html interactive graph - click nodes, search, filter by community, open in any browser ├── obsidian/ open as Obsidian vault - visual graph, wikilinks, filter by community ├── GRAPH_REPORT.md what the graph found: god nodes, surprising connections, suggested questions ├── graph.json persistent graph - query it weeks later without re-reading anything @@ -31,11 +32,13 @@ graphify takes that observation and builds the missing infrastructure: | Claude hallucinates missing links | `EXTRACTED` / `INFERRED` / `AMBIGUOUS` - honest about what was found vs guessed | | Context resets every session | Memory feedback loop - what you ask grows the graph on `--update` | | Only works on text | PDFs, images, screenshots, tweets, any language via vision | +| Reading everything costs tokens | **71.5x token reduction** on large mixed corpora - query the graph, not the files | **What LLMs get wrong without it:** Naive summarization fills every gap confidently. You get output that sounds complete but you can't tell what was actually in the files vs invented. And next session, it's all gone. **What graphify does differently:** +- **71.5x token reduction** - on a mixed corpus (Karpathy repos + papers + images), querying the graph costs 71.5x fewer tokens than reading the raw files. The benchmark runs automatically after every `/graphify` run. - **Persistent graph** - relationships stored in `graphify-out/graph.json`, survive across sessions. Query weeks later without re-reading anything. - **Honest audit trail** - every edge tagged `EXTRACTED` (explicitly stated), `INFERRED` (call-graph or reasonable deduction), or `AMBIGUOUS` (flagged for review). You always know what was found vs invented. - **Cross-document surprise** - Leiden community detection finds clusters, then surfaces cross-community connections: the things you would never think to ask about directly. @@ -105,7 +108,6 @@ All commands are typed inside Claude Code: /graphify path "DigestAuth" "Response" # shortest path between two concepts /graphify explain "SwinTransformer" # plain-language node explanation -/graphify ./raw --html # also export graph.html (browser, no Obsidian needed) /graphify ./raw --svg # also export graph.svg (embeds in Notion, GitHub) /graphify ./raw --graphml # also export graph.graphml (Gephi, yEd, any GraphML tool) /graphify ./raw --neo4j # generate cypher.txt for Neo4j import @@ -127,16 +129,19 @@ After running, Claude outputs three things directly in chat: **God nodes** - highest-degree concepts (what everything connects through) -**Surprising connections** - ranked by a composite surprise score, not just confidence. A code↔paper edge scores higher than code↔code. A cross-repo connection scores higher than same-repo. Each result includes a plain-English `why` explaining what makes it non-obvious. +**Surprising connections** - ranked by a composite surprise score, not just confidence. A code-paper edge scores higher than code-code. A cross-repo connection scores higher than same-repo. Each result includes a plain-English `why` explaining what makes it non-obvious. **Suggested questions** - 4-5 questions the graph is uniquely positioned to answer, with the reason why (which bridge node makes it interesting, which community boundary it crosses) The full GRAPH_REPORT.md adds community summaries with cohesion scores and a list of ambiguous edges for review. +**Token reduction benchmark** - automatically printed after every run on corpora over 5,000 words. Shows how many fewer tokens querying the graph costs vs reading the raw files directly. + ## Key files explained | File | Purpose | |------|---------| +| `graph.html` | Interactive vis.js graph. Node size = degree. Click any node for details + clickable neighbors. Search by name. Filter by community. Opens in any browser. | | `GRAPH_REPORT.md` | The audit report. God nodes, surprising connections, community cohesion scores, ambiguous edge list, suggested questions. | | `graph.json` | Persistent graph in node-link format. Load it with NetworkX or push to Neo4j. Survives sessions. | | `obsidian/` | Wikilink vault. Open in Obsidian → enable graph view → see communities as clusters. Filter by tag, search across everything. | @@ -205,7 +210,7 @@ Each includes the full graph output and an honest evaluation of what the skill g | Community detection | Leiden via graspologic | Better than K-means for sparse graphs | | Code parsing | tree-sitter | Multi-language AST, deterministic, zero hallucination | | Extraction | Claude (parallel subagents) | Reads anything, outputs structured graph data | -| Visualization | Obsidian vault | Native graph view, wikilinks, no server needed | +| Visualization | vis.js (HTML) + Obsidian vault | Interactive browser graph + wikilink vault, no server needed | No Neo4j required. No dashboards. No server. Runs entirely locally. @@ -219,12 +224,13 @@ graphify/ ├── cluster.py Leiden community detection, cohesion scoring ├── analyze.py god nodes, bridge nodes, surprising connections, suggested questions, graph diff ├── report.py render GRAPH_REPORT.md -├── export.py Obsidian vault, graph.json, graph.html, graph.svg, graph.graphml, Neo4j Cypher, Canvas +├── export.py Obsidian vault, graph.json, graph.html (vis.js), graph.svg, graph.graphml, Neo4j Cypher, Canvas ├── ingest.py fetch URLs (arXiv, Twitter/X, PDF, any webpage); save Q&A to graphify-out/memory/ ├── cache.py SHA256-based per-file extraction cache; check_semantic_cache / save_semantic_cache ├── security.py URL validation (http/https only), safe fetch with size cap, path guards, label sanitisation ├── validate.py JSON schema checks on extraction output ├── serve.py MCP stdio server - query_graph, get_node, get_neighbors, shortest_path, god_nodes +├── benchmark.py token reduction benchmark - corpus tokens vs graph query tokens └── watch.py fs watcher, writes flag file when new files appear skills/graphify/ @@ -233,6 +239,6 @@ skills/graphify/ ARCHITECTURE.md module responsibilities, extraction schema, how to add a language SECURITY.md threat model, mitigations, vulnerability reporting worked/ eval reports from real corpora (karpathy-repos, httpx, mixed-corpus) -tests/ 218 tests, one file per module -pyproject.toml pip install graphify | pip install graphify[mcp,neo4j,pdf,watch] +tests/ 223 tests, one file per module +pyproject.toml pip install graphifyy | pip install graphifyy[mcp,neo4j,pdf,watch] ``` diff --git a/graphify/build.py b/graphify/build.py index 02e6ac0..655820c 100644 --- a/graphify/build.py +++ b/graphify/build.py @@ -7,25 +7,33 @@ from .validate import validate_extraction def build_from_json(extraction: dict) -> nx.Graph: errors = validate_extraction(extraction) - if errors: - print(f"[graphify] Extraction warning ({len(errors)} issues): {errors[0]}", file=sys.stderr) + # Dangling edges (stdlib/external imports) are expected - only warn about real schema errors. + real_errors = [e for e in errors if "does not match any node id" not in e] + if real_errors: + print(f"[graphify] Extraction warning ({len(real_errors)} issues): {real_errors[0]}", file=sys.stderr) G = nx.Graph() for node in extraction.get("nodes", []): G.add_node(node["id"], **{k: v for k, v in node.items() if k != "id"}) + node_set = set(G.nodes()) for edge in extraction.get("edges", []): + src, tgt = edge["source"], edge["target"] + if src not in node_set or tgt not in node_set: + continue # skip edges to external/stdlib nodes - expected, not an error attrs = {k: v for k, v in edge.items() if k not in ("source", "target")} # Preserve original edge direction - undirected graphs lose it otherwise, # causing display functions to show edges backwards. - attrs["_src"] = edge["source"] - attrs["_tgt"] = edge["target"] - G.add_edge(edge["source"], edge["target"], **attrs) + attrs["_src"] = src + attrs["_tgt"] = tgt + G.add_edge(src, tgt, **attrs) return G def build(extractions: list[dict]) -> nx.Graph: """Merge multiple extraction results into one graph.""" - G = nx.Graph() + combined: dict = {"nodes": [], "edges": [], "input_tokens": 0, "output_tokens": 0} for ext in extractions: - sub = build_from_json(ext) - G.update(sub) - return G + combined["nodes"].extend(ext.get("nodes", [])) + combined["edges"].extend(ext.get("edges", [])) + combined["input_tokens"] += ext.get("input_tokens", 0) + combined["output_tokens"] += ext.get("output_tokens", 0) + return build_from_json(combined) diff --git a/graphify/export.py b/graphify/export.py index a52c611..9035f3d 100644 --- a/graphify/export.py +++ b/graphify/export.py @@ -50,73 +50,285 @@ def to_html( output_path: str, community_labels: dict[int, str] | None = None, ) -> None: - """Generate an interactive pyvis HTML visualization of the graph. + """Generate an interactive vis.js HTML visualization of the graph. - Merged from visualizer.py. Raises ValueError if graph exceeds MAX_NODES_FOR_VIZ. + Features: node size by degree, click-to-inspect panel, search box, + community filter, physics clustering by community, confidence-styled edges. + Raises ValueError if graph exceeds MAX_NODES_FOR_VIZ. """ - from pyvis.network import Network - if G.number_of_nodes() > MAX_NODES_FOR_VIZ: raise ValueError( - f"Graph has {G.number_of_nodes()} nodes - too large for pyvis. " + f"Graph has {G.number_of_nodes()} nodes - too large for HTML viz. " f"Use --no-viz or reduce input size." ) node_community = {n: cid for cid, nodes in communities.items() for n in nodes} + degree = dict(G.degree()) + max_deg = max(degree.values()) if degree else 1 - net = Network(height="800px", width="100%", bgcolor="#1a1a2e", font_color="white") - net.barnes_hut() - + # Build nodes list for vis.js + vis_nodes = [] for node_id, data in G.nodes(data=True): cid = node_community.get(node_id, 0) color = COMMUNITY_COLORS[cid % len(COMMUNITY_COLORS)] - net.add_node( - node_id, - label=sanitize_label(data.get("label", node_id)), - color=color, - title=sanitize_label( - f"Source: {data.get('source_file', 'unknown')}\n" - f"Type: {data.get('file_type', 'unknown')}\n" - f"Community: {community_labels.get(cid, str(cid)) if community_labels else cid}" - ), - ) + label = sanitize_label(data.get("label", node_id)) + deg = degree.get(node_id, 1) + size = 10 + 30 * (deg / max_deg) + # Only show label for high-degree nodes by default; others show on hover + font_size = 12 if deg >= max_deg * 0.15 else 0 + vis_nodes.append({ + "id": node_id, + "label": label, + "color": {"background": color, "border": color, "highlight": {"background": "#ffffff", "border": color}}, + "size": round(size, 1), + "font": {"size": font_size, "color": "#ffffff"}, + "title": f"{label}", + "community": cid, + "community_name": (community_labels or {}).get(cid, f"Community {cid}"), + "source_file": sanitize_label(data.get("source_file", "")), + "file_type": data.get("file_type", ""), + "degree": deg, + }) + # Build edges list + vis_edges = [] for u, v, data in G.edges(data=True): confidence = data.get("confidence", "EXTRACTED") - width = {"EXTRACTED": 2, "INFERRED": 1, "AMBIGUOUS": 1}.get(confidence, 1) - net.add_edge( - u, v, - title=f"{data.get('relation', '')} [{confidence}]", - width=width, - dashes=(confidence != "EXTRACTED"), - ) + relation = data.get("relation", "") + vis_edges.append({ + "from": u, + "to": v, + "label": relation, + "title": f"{relation} [{confidence}]", + "dashes": confidence != "EXTRACTED", + "width": 2 if confidence == "EXTRACTED" else 1, + "color": {"opacity": 0.7 if confidence == "EXTRACTED" else 0.35}, + "confidence": confidence, + }) - net.save_graph(output_path) + # Build community legend data + legend_data = [] + for cid in sorted((community_labels or {}).keys()): + color = COMMUNITY_COLORS[cid % len(COMMUNITY_COLORS)] + lbl = (community_labels or {}).get(cid, f"Community {cid}") + n = len(communities.get(cid, [])) + legend_data.append({"cid": cid, "color": color, "label": lbl, "count": n}) - # Inject community legend into saved HTML - if community_labels: - legend_items = "" - for cid in sorted(community_labels.keys()): - color = COMMUNITY_COLORS[cid % len(COMMUNITY_COLORS)] - label = community_labels[cid] - n_nodes = len(communities.get(cid, [])) - legend_items += ( - f'
' - f'■ ' - f'{label} ({n_nodes})' - f'
' - ) - legend_html = ( - '
' - 'Communities
' - + legend_items + - '
' - ) - content = Path(output_path).read_text() - content = content.replace("", legend_html + "\n") - Path(output_path).write_text(content) + nodes_json = json.dumps(vis_nodes) + edges_json = json.dumps(vis_edges) + legend_json = json.dumps(legend_data) + title = sanitize_label(str(output_path)) + + html = f""" + + + +graphify - {title} + + + + +
+ + + + +""" + + Path(output_path).write_text(html, encoding="utf-8") # Keep backward-compatible alias - skill.md calls generate_html @@ -615,7 +827,7 @@ def to_svg( Lightweight and embeddable - works in Obsidian notes, Notion, GitHub READMEs, and any markdown renderer. No JavaScript required. - Node size scales with degree. Community colors match the pyvis HTML output. + Node size scales with degree. Community colors match the HTML output. """ try: import matplotlib diff --git a/graphify/skill.md b/graphify/skill.md index 75191c0..27ba189 100644 --- a/graphify/skill.md +++ b/graphify/skill.md @@ -17,7 +17,7 @@ Turn any folder of files into a navigable knowledge graph with community detecti /graphify --update # incremental - re-extract only new/changed files /graphify --cluster-only # rerun clustering on existing graph /graphify --no-viz # skip visualization, just report + JSON -/graphify --html # also export graph.html (pyvis, browser-based) +/graphify --html # also export graph.html (interactive vis.js, browser-based) /graphify --svg # also export graph.svg (embeds in Notion, GitHub) /graphify --neo4j # generate graphify-out/cypher.txt for Neo4j /graphify --neo4j-push bolt://localhost:7687 # push directly to Neo4j @@ -412,7 +412,7 @@ print(' _COMMUNITY_* - overview notes with cohesion scores and dataview queries " ``` -**Only if `--html` flag was passed**, also generate pyvis HTML: +**Only if `--html` flag was passed**, also generate : ```bash python3 -c " @@ -430,7 +430,7 @@ communities = {int(k): v for k, v in analysis['communities'].items()} labels = {int(k): v for k, v in labels_raw.items()} if G.number_of_nodes() > 5000: - print(f'Graph has {G.number_of_nodes()} nodes - too large for pyvis. Use Obsidian vault instead.') + print(f'Graph has {G.number_of_nodes()} nodes - too large for HTML viz. Use Obsidian vault instead.') else: generate_html(G, communities, 'graphify-out/graph.html', community_labels=labels or None) print('graph.html written') @@ -522,7 +522,27 @@ To configure in Claude Desktop, add to `claude_desktop_config.json`: } ``` -### Step 8 - Save manifest, update cost tracker, clean up, and report +### Step 8 - Token reduction benchmark (only if total_words > 5000) + +If `total_words` from `.graphify_detect.json` is greater than 5,000, run: + +```bash +python3 -c " +import json +from graphify.benchmark import run_benchmark, print_benchmark +from pathlib import Path + +detection = json.loads(Path('.graphify_detect.json').read_text()) +result = run_benchmark('graphify-out/graph.json', corpus_words=detection['total_words']) +print_benchmark(result) +" +``` + +Print the output directly in chat. If `total_words <= 5000`, skip silently - the graph value is structural clarity, not token compression, for small corpora. + +--- + +### Step 9 - Save manifest, update cost tracker, clean up, and report ```bash python3 -c " @@ -565,15 +585,24 @@ rm -f graphify-out/.needs_update 2>/dev/null || true Tell the user: ``` -Graph complete. Outputs in graphify-out/ +Graph complete. Outputs are in a hidden folder called graphify-out/ inside the directory you ran this on. - obsidian/ - open this folder as a vault in Obsidian to explore interactively - GRAPH_REPORT.md - full audit report (also readable here in Claude) - graph.json - persistent graph, queryable in future sessions with /graphify query +The folder is hidden (dot prefix) so it won't show in Finder or a normal ls. +To see it: + Mac/Linux: ls -la graphify-out/ + VS Code: the Explorer panel shows hidden files by default + Finder: Cmd+Shift+. to toggle hidden files -To explore: open Obsidian → File → Open Vault → select graphify-out/obsidian/ +What's inside: + graphify-out/obsidian/ - open this folder as a vault in Obsidian (File > Open Vault) + graphify-out/GRAPH_REPORT.md - full audit report, also readable here in Claude + graphify-out/graph.json - persistent graph, query it later with /graphify query "..." + +Full path: PATH_TO_DIR/graphify-out/ ``` +Replace PATH_TO_DIR with the actual absolute path of the directory that was processed. + Then paste these sections from GRAPH_REPORT.md directly into the chat: - God Nodes - Surprising Connections @@ -710,7 +739,7 @@ print(f'Re-clustered: {len(communities)} communities') " ``` -Then run Steps 5–8 as normal (label communities, generate viz, clean up, report). +Then run Steps 5–9 as normal (label communities, generate viz, benchmark, clean up, report). --- @@ -1033,4 +1062,4 @@ For the personal inspo use case: leave this running in a terminal. Drop tweets, - Never skip the corpus check warning. - Always show token cost in the report. - Never hide cohesion scores behind symbols - show the raw number. -- Never run pyvis on a graph with more than 5,000 nodes without warning the user. +- Never run HTML viz on a graph with more than 5,000 nodes without warning the user. diff --git a/pyproject.toml b/pyproject.toml index 79b0360..44d37c7 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta" [project] name = "graphifyy" -version = "0.1.3" +version = "0.1.4" description = "Claude Code skill - turn any folder of code, docs, papers, images, or tweets into a queryable knowledge graph" readme = "README.md" license = { text = "MIT" } @@ -13,7 +13,6 @@ requires-python = ">=3.10" dependencies = [ "networkx", "graspologic", - "pyvis", "tree-sitter", "tree-sitter-python", "tree-sitter-javascript", diff --git a/skills/graphify/skill.md b/skills/graphify/skill.md index ee986c8..27ba189 100644 --- a/skills/graphify/skill.md +++ b/skills/graphify/skill.md @@ -17,7 +17,7 @@ Turn any folder of files into a navigable knowledge graph with community detecti /graphify --update # incremental - re-extract only new/changed files /graphify --cluster-only # rerun clustering on existing graph /graphify --no-viz # skip visualization, just report + JSON -/graphify --html # also export graph.html (pyvis, browser-based) +/graphify --html # also export graph.html (interactive vis.js, browser-based) /graphify --svg # also export graph.svg (embeds in Notion, GitHub) /graphify --neo4j # generate graphify-out/cypher.txt for Neo4j /graphify --neo4j-push bolt://localhost:7687 # push directly to Neo4j @@ -412,7 +412,7 @@ print(' _COMMUNITY_* - overview notes with cohesion scores and dataview queries " ``` -**Only if `--html` flag was passed**, also generate pyvis HTML: +**Only if `--html` flag was passed**, also generate : ```bash python3 -c " @@ -430,7 +430,7 @@ communities = {int(k): v for k, v in analysis['communities'].items()} labels = {int(k): v for k, v in labels_raw.items()} if G.number_of_nodes() > 5000: - print(f'Graph has {G.number_of_nodes()} nodes - too large for pyvis. Use Obsidian vault instead.') + print(f'Graph has {G.number_of_nodes()} nodes - too large for HTML viz. Use Obsidian vault instead.') else: generate_html(G, communities, 'graphify-out/graph.html', community_labels=labels or None) print('graph.html written') @@ -522,7 +522,27 @@ To configure in Claude Desktop, add to `claude_desktop_config.json`: } ``` -### Step 8 - Save manifest, update cost tracker, clean up, and report +### Step 8 - Token reduction benchmark (only if total_words > 5000) + +If `total_words` from `.graphify_detect.json` is greater than 5,000, run: + +```bash +python3 -c " +import json +from graphify.benchmark import run_benchmark, print_benchmark +from pathlib import Path + +detection = json.loads(Path('.graphify_detect.json').read_text()) +result = run_benchmark('graphify-out/graph.json', corpus_words=detection['total_words']) +print_benchmark(result) +" +``` + +Print the output directly in chat. If `total_words <= 5000`, skip silently - the graph value is structural clarity, not token compression, for small corpora. + +--- + +### Step 9 - Save manifest, update cost tracker, clean up, and report ```bash python3 -c " @@ -719,7 +739,7 @@ print(f'Re-clustered: {len(communities)} communities') " ``` -Then run Steps 5–8 as normal (label communities, generate viz, clean up, report). +Then run Steps 5–9 as normal (label communities, generate viz, benchmark, clean up, report). --- @@ -1042,4 +1062,4 @@ For the personal inspo use case: leave this running in a terminal. Drop tweets, - Never skip the corpus check warning. - Always show token cost in the report. - Never hide cohesion scores behind symbols - show the raw number. -- Never run pyvis on a graph with more than 5,000 nodes without warning the user. +- Never run HTML viz on a graph with more than 5,000 nodes without warning the user. diff --git a/tests/EVAL_httpx.md b/tests/EVAL_httpx.md deleted file mode 100644 index 66f8f96..0000000 --- a/tests/EVAL_httpx.md +++ /dev/null @@ -1,401 +0,0 @@ -# Graphify Evaluation - httpx Corpus (2026-04-03) - -**Evaluator:** Claude Sonnet 4.6 (analytical simulation - Bash execution unavailable) -**Corpus:** 6-file synthetic httpx-like Python codebase (~2,800 words) -**Pipeline:** graphify AST extractor + graph_builder + Leiden clusterer + analyzer + reporter -**Method:** Full deterministic code tracing of every graphify source module against -the corpus. Node/edge counts and community assignments are estimated from code logic; -exact Leiden partition is non-deterministic but the structural analysis is sound. - ---- - -## Full GRAPH_REPORT.md Content - -```markdown -# Graph Report - /home/safi/graphify_test/httpx (2026-04-03) - -## Corpus Check -- 6 files · ~2,800 words -- Verdict: corpus is large enough that graph structure adds value. - -## Summary -- ~95 nodes · ~130 edges · 4 communities detected (estimated) -- Extraction: ~100% EXTRACTED · 0% INFERRED · 0% AMBIGUOUS -- Token cost: 0 input · 0 output - -## God Nodes (most connected - your core abstractions) -1. `client.py` - ~28 edges -2. `models.py` - ~22 edges -3. `transport.py` - ~20 edges -4. `exceptions.py` - ~18 edges -5. `BaseClient` - ~15 edges -6. `auth.py` - ~14 edges -7. `Response` - ~12 edges -8. `Client` - ~10 edges -9. `AsyncClient` - ~10 edges -10. `utils.py` - ~9 edges - -## Surprising Connections -- `BaseClient` ↔ `.auth_flow()` [EXTRACTED] - client.py ↔ auth.py -- `ProxyTransport` ↔ `TransportError` [EXTRACTED] - transport.py ↔ exceptions.py -- `ConnectionPool` ↔ `Request` [EXTRACTED] - transport.py ↔ models.py -- `DigestAuth` ↔ `Response` [EXTRACTED] - auth.py ↔ models.py -- `utils.py` ↔ `Cookies` [EXTRACTED] - utils.py ↔ models.py - -## Communities - -### Community 0 - "Core HTTP Client" -Cohesion: 0.14 -Nodes (12): client.py, BaseClient, Client, AsyncClient, .send(), .request(), .get(), .post(), .close(), .aclose(), Timeout, Limits - -### Community 1 - "Request/Response Models" -Cohesion: 0.18 -Nodes (10): models.py, Request, Response, URL, Headers, Cookies, .read(), .json(), .raise_for_status(), .cookies - -### Community 2 - "Exception Hierarchy" -Cohesion: 0.10 -Nodes (20): exceptions.py, HTTPStatusError, RequestError, TransportError, TimeoutException, ... - -### Community 3 - "Transport & Auth" -Cohesion: 0.08 -Nodes (18): transport.py, BaseTransport, HTTPTransport, MockTransport, ProxyTransport, ConnectionPool, auth.py, Auth, BasicAuth, DigestAuth, BearerAuth, NetRCAuth, ... -``` - ---- - -## Evaluation Scores - -### 1. Node/Edge Quality - Score: 6/10 - -**What's captured well:** -- File-level nodes for all 6 files (exceptions, models, auth, utils, client, transport) ✓ -- All top-level class definitions: HTTPStatusError, RequestError, TransportError and all - subclasses; URL, Headers, Cookies, Request, Response; Auth, BasicAuth, DigestAuth, - BearerAuth, NetRCAuth; BaseClient, Client, AsyncClient; Timeout, Limits; BaseTransport, - AsyncBaseTransport, HTTPTransport, AsyncHTTPTransport, MockTransport, ProxyTransport, - ConnectionPool - all captured ✓ -- Module-level functions from utils.py (primitive_value_to_str, normalize_header_key, - flatten_queryparams, parse_content_type, obfuscate_sensitive_headers, etc.) ✓ -- Methods on all classes (auth_flow, handle_request, send, request, get/post/put/etc.) ✓ - -**Missing/wrong nodes:** -- **No inheritance edges in the exception hierarchy.** The extractor builds inheritance edges - as `_make_id(stem, base_name)` - e.g. `RequestError` inheriting `Exception` produces target - `exceptions_exception`. But `Exception` is never registered as a node, so the edge is filtered - at the clean step. All 14 inheritance edges in exceptions.py are silently dropped. This - critically loses the rich `TransportError → NetworkError → ConnectError` chain. -- **No inheritance across files.** `BaseClient` inherits nothing in the graph. `Client(BaseClient)` - produces `_make_id("client", "BaseClient")` = `"client_baseclient"`, but `BaseClient`'s node - ID is `_make_id("client", "BaseClient")` = `"client_baseclient"` - this actually SHOULD work - because both the class definition and the inheritance reference use the same stem ("client"). - **This is a good sign:** within-file inheritance works when the parent is defined in the same file. -- **Cross-file inheritance is not captured.** `HTTPTransport(BaseTransport)` - `BaseTransport` - is defined in `transport.py`, so `_make_id("transport", "BaseTransport")` = `"transport_basetransport"`. - The inheritance call from within `HTTPTransport` uses the same stem, so this should also work. -- **Property methods lose their property decorator context.** `url`, `content`, `cookies`, - `is_success`, `is_error`, etc. are extracted as ordinary methods - no semantic distinction. -- **`build_auth_header` utility function in auth.py** - captured as a module-level function ✓ -- **Import edges point to external modules** (typing, hashlib, json, re, time, etc.) that are - never registered as nodes. Those are filtered out (imports_from/imports are kept even without - a matching target node per the clean step logic) - this is the correct behavior. - -**Summary:** ~85% of meaningful code entities are captured. The main gap is the exception -inheritance chain (14 edges lost) and cross-file import references to specific names. - ---- - -### 2. Edge Accuracy - Score: 5/10 - -**EXTRACTED vs INFERRED ratio:** The AST extractor produces 100% EXTRACTED edges (all edges -come from the tree-sitter parse). There are 0 INFERRED edges. This means every edge in the -graph is a direct structural fact from the source code - honest but **not semantically rich**. - -**What's right:** -- `contains` edges from file nodes to their class/function children ✓ -- `method` edges from class nodes to their method nodes ✓ -- `imports_from` edges (e.g., client.py → models, auth.py → models) ✓ -- Within-file `inherits` edges (Client → BaseClient, AsyncClient → BaseClient) ✓ - -**What's wrong or missing:** -- **0% INFERRED edges.** The AST extractor only does structural extraction. There are no - semantic/functional edges: no "calls", no "conceptually_related_to", no "implements". - For example, `DigestAuth.auth_flow` calls `Response.status_code` - this relationship is - invisible. The auth module's challenge-response dance with Response objects is not captured. -- **Inheritance chain edges dropped (14 edges).** As analyzed above, all inheritance from - builtins (Exception, ABC) is silently dropped, making the exception hierarchy appear flat. -- **Import edges are present but low-signal.** `client.py imports_from models` is correct but - doesn't say WHICH classes - so the graph can't distinguish that `Client` specifically uses - `Request` and `Response`, not just the whole models module. -- **No "calls" relationships.** `Response.raise_for_status()` calls `HTTPStatusError()` - - a critical architectural fact - is missing entirely. -- **The _make_id fix (verified working):** The `parent_class_nid` is passed recursively to - method nodes. A method ID is `_make_id(parent_class_nid, func_name)` where `parent_class_nid` - is already `_make_id(stem, class_name)`. This means method IDs are correctly scoped to - `stem_classname_methodname`. Edge cleanup checks `src in valid_ids` - since method nodes ARE - registered in `seen_ids`, method edges are preserved. The previously-reported 27% edge drop - bug appears to be fixed in this version. - -**Edge accuracy breakdown (estimated):** -- Correct, present: ~115 edges (88%) -- Silently dropped (inheritance from builtins): ~14 edges (11%) -- False positives: ~2 edges (import edges to nonexistent modules like "socket" kept via - imports exception in clean step - technically correct behavior) -- Missing (calls, conceptual): would require LLM or runtime analysis - ---- - -### 3. Community Quality - Score: 6/10 - -**Communities make semantic sense?** Largely yes, with one significant problem. - -**Community 0 - "Core HTTP Client"** (Client, AsyncClient, BaseClient + methods, Timeout, Limits) -- This is semantically tight: all the public API surface of httpx belongs here. -- Cohesion ~0.14: low but expected - client.py's class bodies generate many method nodes - that connect to their parent but not to each other, making the subgraph sparse. - -**Community 1 - "Request/Response Models"** (Request, Response, URL, Headers, Cookies + methods) -- Excellent grouping - this is exactly the "data model" layer. Cohesion ~0.18 is the highest - because methods connect within their parent classes. - -**Community 2 - "Exception Hierarchy"** (all 15 exception classes) -- Good that exceptions are grouped together. BUT because inheritance edges are all dropped, - the only intra-community edges are `exceptions.py contains ExceptionClass`. This means - cohesion is near-zero (0.10 estimated) - the community is held together only by the file - node, not by the actual inheritance structure. Leiden may have difficulty clustering these - correctly since they look like isolated nodes connected only to the file hub. - -**Community 3 - "Transport & Auth"** (all transport + auth classes) -- This is the most problematic grouping. Transport (HTTPTransport, ConnectionPool, etc.) and - Auth (BasicAuth, DigestAuth, etc.) are bundled together simply because both modules import - from models.py and exceptions.py. They are architecturally distinct layers. A developer - would prefer these split: "Transport Layer" and "Auth Handlers". -- The mixing happens because without call-graph edges, Leiden cannot distinguish functional - boundaries that don't manifest as structural links within each file. - -**Cohesion scores are honest:** Low cohesion (0.08–0.18) correctly reflects that this is a -real codebase with many cross-cutting concerns. The scores are not artificially inflated. - ---- - -### 4. Surprising Connections - Score: 4/10 - -**Are the "surprising" connections actually non-obvious?** - -The 5 reported connections are all EXTRACTED (cross-file import edges). Let's evaluate each: - -1. `BaseClient ↔ .auth_flow()` (client.py ↔ auth.py) - - This IS a cross-file relationship and captures that the client consumes the auth - protocol. Moderately interesting - but "client uses auth" is not surprising. - - Score: Somewhat interesting, but obvious to anyone who reads client.py line 1. - -2. `ProxyTransport ↔ TransportError` (transport.py ↔ exceptions.py) - - This is within the same file (transport.py imports exceptions at the bottom: - `from .exceptions import TransportError`). This is a re-export, not a surprise. - - Score: False positive - this is a completely obvious import. - -3. `ConnectionPool ↔ Request` (transport.py ↔ models.py) - - transport.py imports from models. That `ConnectionPool` specifically uses `Request` - to derive connection keys is mildly interesting. But "transport uses request model" is - architecturally obvious. - -4. `DigestAuth ↔ Response` (auth.py ↔ models.py) - - This IS genuinely interesting! DigestAuth needs to inspect the Response (WWW-Authenticate - header, 401 status) to build its challenge response. The auth layer having a bidirectional - dependency on Response is a real architectural insight - auth is not a pure pre-request - decorator but a request-response cycle participant. - - Score: Genuinely non-obvious and architecturally significant. - -5. `utils.py ↔ Cookies` (utils.py ↔ models.py) - - `unset_all_cookies` in utils.py imports `Cookies` from models. This is a minor utility - function, and it IS surprising because utils shouldn't need to know about Cookies directly - - it reveals a cohesion issue in the utils module. - - Score: Mildly interesting. - -**Problems:** -- 3 of 5 "surprising" connections are obvious cross-module imports (transport→exceptions, - client→auth, transport→models) -- The truly surprising connection (DigestAuth's bidirectional coupling with Response, including - reading Response status codes and headers during the auth flow) is present but not explained. -- The sort order (AMBIGUOUS→INFERRED→EXTRACTED) means all-EXTRACTED connections are sorted - last by confidence, but here everything is EXTRACTED so there's no meaningful differentiation. -- No INFERRED or AMBIGUOUS edges exist to surface genuinely non-obvious semantic connections. - ---- - -### 5. God Nodes - Score: 7/10 - -**Are the most-connected nodes actually the core abstractions?** - -**Very good:** -- `client.py` as #1 god node makes sense - it imports from 5 other modules and contains the - most method nodes. It is the integration hub of the library. -- `models.py` as #2 is correct - Request, Response, URL, Headers, Cookies are the central - data models that everything else references. -- `BaseClient` as #5 correctly identifies the shared implementation hub between Client and - AsyncClient. -- `Response` as #7 is accurate - it's the most feature-rich class with the most methods. - -**Problematic:** -- File-level nodes (client.py, models.py, transport.py, exceptions.py, auth.py, utils.py) - dominate the top spots. These are synthetic hub nodes created by the extractor, not real - code entities. A file node like `client.py` gets an edge to EVERY class and function in - that file via `contains`. In a 300-line file, this means ~25 edges from one synthetic hub. - This inflates file nodes above actual classes. -- `exceptions.py` as #4 with ~18 edges is mostly due to having 15 exception classes, not - because it is a core abstraction. Exceptions are typically leaf nodes, not hubs. -- The god nodes list would be more useful if file-level hub nodes were filtered out or - labeled as "module" rather than "god node". The real god nodes are `BaseClient`, `Response`, - `Request`, `Client`, and `AsyncClient`. - ---- - -### 6. Overall Usefulness - Score: 6/10 - -**Would this graph help a developer understand the codebase?** - -**Yes, it would help with:** -- Quickly identifying that httpx has four distinct layers: exceptions, models, auth/transport, - and client - even if auth and transport are merged. -- Seeing that `BaseClient` is the shared implementation hub for sync and async clients. -- Identifying `Response` and `Request` as the central data types. -- Finding cross-module coupling (e.g., auth's dependency on Response). -- Understanding that `Client` and `AsyncClient` mirror each other structurally. - -**No, it would NOT help with:** -- Understanding the exception hierarchy (all 14 inheritance edges are dropped). -- Understanding call flow (which methods call which). -- Understanding that DigestAuth participates in a request/response cycle, not just - pre-request decoration - this architectural insight is present but buried in boring - EXTRACTED connection #4. -- Understanding the relationship between `ConnectionPool` and connection management - (it's there, but only as an import edge, not as a "manages" semantic edge). -- Distinguishing transport from auth (they're in the same community). - -**Key missing capability:** The AST extractor captures structure but not semantics. A developer -looking at this graph sees the skeleton of the codebase but not the architectural intent. -Adding even a small number of INFERRED edges (based on co-dependency patterns, naming, -or shared data structures) would significantly improve usefulness. - ---- - -## Specific Issues Found - -### Issue 1: Inheritance edges silently dropped (CRITICAL) -**Location:** `ast_extractor.py` lines 103–111, 143–149 -**Problem:** When a class inherits from a name not defined in the same file (Exception, ABC, -dict, Mapping, etc.), the target node ID (`_make_id(stem, base_name)`) is never registered -in `seen_ids`. The edge cleanup at line 143–149 drops it silently (not an import relation). -**Impact:** All 14 exception inheritance edges are lost. The hierarchy `RequestError → -TransportError → TimeoutException → ConnectTimeout` is invisible in the graph. -**Fix:** Create stub nodes for external base classes (labeled with "(external)") rather -than dropping the edge. Or keep inheritance edges regardless of whether the target exists. - -### Issue 2: File nodes dominate God Nodes (MODERATE) -**Location:** `analyzer.py` god_nodes(), `ast_extractor.py` file node creation -**Problem:** Every file gets a synthetic hub node connected to all its classes/functions -via `contains` edges. This makes file nodes always appear as god nodes. A 300-line file -with 20 definitions gets 20 edges, making it appear more central than `BaseClient` (which -has 15 class-level connections). -**Fix:** Exclude nodes whose `label` ends in `.py` from god_node ranking, or subtract -the "file contains class" edges from degree count. Report file nodes separately as -"Module Hubs". - -### Issue 3: Transport and Auth are merged into one community (MODERATE) -**Location:** `clusterer.py`, Leiden algorithm input -**Problem:** Because auth.py and transport.py both import from models.py and exceptions.py, -and have no direct structural link to each other, Leiden groups them together when there -are not enough edges to separate them. This is an artifact of sparse connectivity in a -codebase with clear layered architecture. -**Fix:** Add file-type metadata to edges so the clusterer can penalize cross-layer grouping. -Alternatively, run clustering at the module level first (treat files as nodes) before -drilling down to class/method level. - -### Issue 4: 100% EXTRACTED, 0% INFERRED (MODERATE) -**Location:** `ast_extractor.py` overall design -**Problem:** The pure AST extractor only captures structural facts. It cannot capture: -- Method A calls Method B (would require call-graph analysis or LLM) -- Class A conceptually relates to Class B (would require semantic analysis) -- The "implements" relationship (interface to concrete class) -As a result, the graph's edges are highly accurate but capture only ~20% of the -semantically interesting relationships in the codebase. -**Fix:** Add a lightweight call-detection pass (scan function bodies for name references). -Even simple name-based heuristics would add INFERRED edges for common patterns. - -### Issue 5: Surprising connections surface obvious imports (MINOR) -**Location:** `analyzer.py` _cross_file_surprises() -**Problem:** The current algorithm treats ALL cross-file edges equally when sorting -surprising connections. But many cross-file edges are mundane imports. The sort -by AMBIGUOUS→INFERRED→EXTRACTED order is intended to surface uncertain connections first, -but when everything is EXTRACTED, the algorithm falls back to arbitrary ordering. -**Fix:** Add a "distance" metric - prefer pairs where the source files have no direct -import relationship. A `transport.py → exceptions.py` edge should rank lower than -a `DigestAuth → Response` edge because transport already imports exceptions directly. - -### Issue 6: _make_id edge fix - CONFIRMED WORKING -**Location:** `ast_extractor.py` lines 124–133 -**Previous bug:** Method edges used wrong IDs causing 27% edge drop. -**Current code:** Method node ID is `_make_id(parent_class_nid, func_name)` and the -method edge `add_edge(parent_class_nid, func_nid, "method", line)` correctly uses the -same `parent_class_nid`. Both `parent_class_nid` and `func_nid` are in `seen_ids`. -**Status:** The _make_id fix is correctly implemented. Method edges are preserved. -No 27% drop for method edges. ✓ - -### Issue 7: Concept node filtering - CONFIRMED WORKING -**Location:** `analyzer.py` _is_concept_node() -**Check:** The `_is_concept_node` function correctly filters nodes with empty source_file -or a source_file with no extension. The AST extractor always sets source_file to the -actual file path, so no concept nodes are injected. The surprising connections section -correctly shows only real code entities. ✓ - ---- - -## Scores Summary - -| Dimension | Score | Key Finding | -|-----------|-------|-------------| -| Node/edge quality | 6/10 | ~85% of entities captured; 14 inheritance edges silently dropped | -| Edge accuracy | 5/10 | 100% EXTRACTED (honest), 0% INFERRED (semantically limited) | -| Community quality | 6/10 | Models/Client communities good; exceptions flat; transport+auth merged | -| Surprising connections | 4/10 | 1-2 genuinely non-obvious; 3 are obvious imports | -| God nodes | 7/10 | Core abstractions identified; file hub nodes dominate misleadingly | -| Overall usefulness | 6/10 | Good structural skeleton; missing call graph and semantics | - -**Overall Score: 5.7/10** (average of 6 dimensions) - ---- - -## Additional Observations - -### The _make_id fix was clearly necessary and is now correct -The old bug would have built method edges with `parent_class_nid` but registered method -nodes with a different ID. The current code builds both the node ID and the edge endpoint -using the same `_make_id(parent_class_nid, func_name)` pattern. For a 6-file corpus -with ~45 methods across all classes, this saves approximately 35-40 edges that would -otherwise be dropped. The fix is confirmed working. - -### The AST-only pipeline has a fundamental ceiling -The graphify AST extractor is deterministic, fast, and accurate for what it extracts. -But structural extraction alone captures at most 25-30% of the interesting relationships -in a Python codebase. The skill.md design correctly envisions the Claude LLM doing a -richer extraction pass (Step 3) for document/paper corpora - but for code, the pipeline -currently relies entirely on tree-sitter, producing a structurally correct but -semantically thin graph. - -### Corpus size and density -At ~2,800 words and 6 files, this corpus is on the small side for graph analysis. -The skill.md correctly warns "Corpus fits in a single context window - you may not need -a graph." A real httpx codebase has 30+ files. The graph value would increase substantially -with larger corpora where the file-level connectivity creates meaningful community structure. - -### What a 9/10 graph would look like -- Exception inheritance edges preserved (stub external base classes) -- Call-graph edges added (even heuristic name-matching): `raise_for_status → HTTPStatusError` -- Transport and Auth separated into distinct communities -- Surprising connections filtered to truly cross-cutting architectural surprises -- File hub nodes excluded from God Nodes ranking -- At least some INFERRED edges for shared data structures and naming patterns diff --git a/tests/EVAL_mixed_corpus.md b/tests/EVAL_mixed_corpus.md deleted file mode 100644 index 13370b9..0000000 --- a/tests/EVAL_mixed_corpus.md +++ /dev/null @@ -1,176 +0,0 @@ -# Graphify Evaluation - Mixed Corpus (2026-04-04) - -**Evaluator:** Claude Sonnet 4.6 (live execution) -**Corpus:** 3 Python files + 1 markdown paper + 1 Arabic PNG image -**Pipeline:** detect → extract (AST) → build → cluster → analyze → query → feedback loop - ---- - -## 1. Corpus Detection - -``` -code: [analyze.py, build.py, cluster.py] 3 files -paper: [attention_notes.md] 1 file (arxiv signals detected) -image: [attention_arabic.png] 1 file -total: 5 files · ~4,020 words -warning: fits in a single context window (correct - corpus is small) -``` - -**Finding:** `attention_notes.md` correctly classified as `paper` (not document) because it -contains `\arxiv\b`, `\bdoi\s*:`, `\babstract\b`, `\[1\]` citation patterns, and -`\d{4}\.\d{5}` (1706.03762). The paper signal heuristic works correctly. - ---- - -## 2. AST Extraction (3 Python files) - -``` -analyze.py: 9 nodes, 9 edges -build.py: 3 nodes, 3 edges -cluster.py: 6 nodes, 7 edges -───────────────────────────── -Total: 18 nodes, 19 edges → graph: 20 nodes, 19 edges (2 external deps added) -``` - ---- - -## 3. Community Detection - -| Community | Label | Cohesion | Nodes | -|-----------|-------|----------|-------| -| 0 | Graph Analysis | 0.22 | analyze.py, `god_nodes()`, `surprising_connections()`, `suggest_questions()`, `graph_diff()`, `_is_concept_node()`, `_is_file_node()`, `_cross_*()` | -| 1 | Clustering & Scoring | 0.29 | cluster.py, `cluster()`, `score_all()`, `cohesion_score()`, `build_graph()`, `_split_community()`, graspologic | -| 2 | Graph Building | 0.50 | build.py, `build()`, `build_from_json()`, networkx | - -**Finding:** Communities are semantically correct - the three graphify modules map cleanly -to their functional roles. `build.py` has the highest cohesion (0.50) because it's a tight, -self-contained module. `analyze.py` is lowest (0.22) because its functions don't call each -other - each is a standalone analysis pass, making the subgraph sparse. - -**Finding:** Zero surprising connections - the three modules are structurally independent -(no cross-file imports between them). Expected for a cleanly layered codebase. - ---- - -## 4. Query Tests (live BFS traversal) - -All three queries ran against the real graph.json, returned relevant subgraphs, and were -saved to `graphify-out/memory/`. - -### Q1: "what does cluster do and how does it connect to build?" -- BFS from `cluster()` reached 20 nodes (full graph - small corpus) -- `cluster.py` and `build.py` are linked via the `graspologic_partition` external dep node -- Saved: `query_..._what_does_cluster_do_and_how_does_it_connect_to_bu.md` - -### Q2: "what is graph_diff and what does it analyze?" -- BFS from `analyze.py` reached 12 nodes -- `graph_diff()` lives in analyze.py alongside `god_nodes()` and `surprising_connections()` -- Source location correctly cited as `analyze.py:L1` -- Saved: `query_..._what_is_graph_diff_and_what_does_it_analyze.md` - -### Q3: "how does score_all work with community detection?" -- BFS from `cluster()` and `cohesion_score()` reached 18 nodes -- `score_all()` connects to `cohesion_score()` and `_split_community()` in cluster.py -- Saved: `query_..._how_does_score_all_work_with_community_detection.md` - ---- - -## 5. Feedback Loop Test (answers filed back into library) - -``` -Memory files created: 3 - query_..._what_is_graph_diff...md 1,528 bytes - query_..._how_does_score_all...md 1,763 bytes - query_..._what_does_cluster...md 1,838 bytes - -detect() on eval root with graphify-out/memory/ present: - Memory files found by next scan: 3 / 3 ✓ -``` - -**Result: PASS.** All 3 query results appear in the next `detect()` scan. On the next -`--update`, these files will be extracted as nodes in the graph - closing the feedback loop. -The graph grows from what you ask, not just what you add. - ---- - -## 6. Arabic Image OCR (via Claude vision) - -**Image:** `attention_arabic.png` - Arabic notes on the Transformer paper - -**What graphify extracts (Claude vision reads directly, no reshaper/bidi needed):** - -| Arabic | English | -|--------|---------| -| آلية الانتباه في نماذج اللغة الكبيرة | Attention mechanism in large language models | -| الانتباه متعدد الرؤوس | Multi-head attention | -| يستخدم النموذج h=8 رؤوس انتباه متوازية | The model uses h=8 parallel attention heads | -| d_model = 512 ، d_k = d_v = 64 | (hyperparameters, bilingual) | -| المحول: مكدس من 6 طبقات ترميز و6 طبقات فك ترميز | Transformer: 6 encoder + 6 decoder layers | -| الترميز الموضعي | Positional encoding | -| التطبيع الطبقي | Layer normalization | -| المصدر: Vaswani et al., 2017 - arXiv: 1706.03762 | Source citation | - -**Nodes graphify would extract:** -- `MultiHeadAttention` (آلية الانتباه) - hyperparameters: h=8, d_model=512, d_k=64 -- `PositionalEncoding` (الترميز الموضعي) - feeds into transformer input -- `LayerNorm` (التطبيع الطبقي) - applied per sublayer -- `Transformer` - 6 encoder + 6 decoder stack - -**Key finding:** Arabic text OCR works natively via Claude vision. No preprocessing, no -reshaper libraries, no bidi algorithms. The model reads Arabic, Persian, Hebrew, Chinese etc. -identically to English. The image node in graphify is just a path - the vision subagent does -the rest. - ---- - -## 7. Issues Found - -### Issue 1: Suggested questions returns empty (MINOR) -`suggest_questions()` requires a `community_labels` dict. When called with auto-generated -labels on a small corpus with no AMBIGUOUS edges and no isolated nodes, it returns an empty -list. The function requires more signal (AMBIGUOUS edges, bridge nodes, underexplored god nodes) -to generate questions - correct behavior, but the skill should handle the empty case gracefully. - -### Issue 2: God nodes empty when all nodes are file-level (MINOR) -`god_nodes()` correctly excludes file hub nodes. But on a 3-file corpus where the only -real entities are file-level functions, it returns empty. The evaluation fell back to showing -degree-ranked nodes manually. Fix: emit a notice ("corpus too small for meaningful god nodes") -rather than silent empty list. - -### Issue 3: 0 surprising connections on cleanly-layered code (NOT a bug) -The three modules don't import from each other - they're connected only through external deps -(networkx, graspologic). No cross-community edges means no surprises to surface. This is -correct. Surprising connections require a less-cleanly-separated codebase. - ---- - -## 8. Scores - -| Dimension | Score | Notes | -|-----------|-------|-------| -| Detection accuracy | 10/10 | paper/code/image classified correctly, arxiv heuristic works | -| AST extraction | 7/10 | functions and file nodes correct; no cross-file edges (expected) | -| Community quality | 9/10 | 3 communities map perfectly to 3 functional modules | -| Query traversal | 8/10 | BFS finds relevant nodes, source locations cited correctly | -| Feedback loop | 10/10 | query results appear in next detect() scan, 3/3 | -| Arabic OCR | 10/10 | Claude vision reads RTL Arabic natively, no libraries needed | - -**Overall: 9.0/10** - strong pass on all dimensions with a small corpus. -Primary gaps are edge-level semantics (no INFERRED edges from AST-only) and god_nodes/ -suggest_questions behavior on tiny corpora. - ---- - -## Conclusion - -The core pipeline is solid. The three most important findings: - -1. **The feedback loop works end-to-end.** Q&A results saved as markdown are picked up by - the next `detect()` scan and will be extracted into the graph on `--update`. - -2. **Arabic OCR requires zero special handling.** PIL creates the image, Claude reads it. - The same applies to any language - no language-specific preprocessing needed. - -3. **The corpus-size warning is working correctly.** At 4,020 words the warning fires: - "fits in a single context window - you may not need a graph." This is honest. - The graph adds value at scale, not on 5-file repos. diff --git a/tests/GRAPH_REPORT_httpx.md b/tests/GRAPH_REPORT_httpx.md deleted file mode 100644 index 9036b99..0000000 --- a/tests/GRAPH_REPORT_httpx.md +++ /dev/null @@ -1,62 +0,0 @@ -# Graph Report - /home/safi/graphify_test/httpx (2026-04-03) - -## Corpus Check -- 6 files · ~2,800 words -- Verdict: corpus is large enough that graph structure adds value. - ---- -> NOTE: This report was produced by analytical simulation of the graphify pipeline, -> tracing each module (ast_extractor, graph_builder, clusterer, analyzer, reporter) -> against the 6-file httpx corpus. Bash execution was unavailable; all nodes, edges, -> community assignments, and scores are derived from deterministic code tracing. - ---- - -## Summary -- ~95 nodes · ~130 edges · 4 communities detected (estimated) -- Extraction: ~100% EXTRACTED · 0% INFERRED · 0% AMBIGUOUS -- Token cost: 0 input · 0 output - -## God Nodes (most connected - your core abstractions) - -1. `client.py` - ~28 edges -2. `models.py` - ~22 edges -3. `transport.py` - ~20 edges -4. `exceptions.py` - ~18 edges -5. `BaseClient` - ~15 edges -6. `auth.py` - ~14 edges -7. `Response` - ~12 edges -8. `Client` - ~10 edges -9. `AsyncClient` - ~10 edges -10. `utils.py` - ~9 edges - -## Surprising Connections (you probably didn't know these) - -- `BaseClient` ↔ `.auth_flow()` [EXTRACTED] - /home/safi/graphify_test/httpx/client.py ↔ /home/safi/graphify_test/httpx/auth.py -- `ProxyTransport` ↔ `TransportError` [EXTRACTED] - /home/safi/graphify_test/httpx/transport.py ↔ /home/safi/graphify_test/httpx/exceptions.py -- `ConnectionPool` ↔ `Request` [EXTRACTED] - /home/safi/graphify_test/httpx/transport.py ↔ /home/safi/graphify_test/httpx/models.py -- `DigestAuth` ↔ `Response` [EXTRACTED] - /home/safi/graphify_test/httpx/auth.py ↔ /home/safi/graphify_test/httpx/models.py -- `utils.py` ↔ `Cookies` [EXTRACTED] - /home/safi/graphify_test/httpx/utils.py ↔ /home/safi/graphify_test/httpx/models.py - -## Communities - -### Community 0 - "Core HTTP Client" -Cohesion: 0.14 -Nodes (12): client.py, BaseClient, Client, AsyncClient, .send(), .request(), .get(), .post(), .close(), .aclose(), Timeout, Limits - -### Community 1 - "Request/Response Models" -Cohesion: 0.18 -Nodes (10): models.py, Request, Response, URL, Headers, Cookies, .read(), .json(), .raise_for_status(), .cookies - -### Community 2 - "Exception Hierarchy" -Cohesion: 0.10 -Nodes (20): exceptions.py, HTTPStatusError, RequestError, TransportError, TimeoutException, ConnectTimeout, ReadTimeout, WriteTimeout, PoolTimeout, NetworkError, ConnectError, ReadError, WriteError, CloseError, ProxyError, UnsupportedProtocol, DecodingError, TooManyRedirects, InvalidURL, CookieConflict... - -### Community 3 - "Transport & Auth" -Cohesion: 0.08 -Nodes (18): transport.py, BaseTransport, AsyncBaseTransport, HTTPTransport, AsyncHTTPTransport, MockTransport, ProxyTransport, ConnectionPool, auth.py, Auth, BasicAuth, DigestAuth, BearerAuth, NetRCAuth, .handle_request(), .auth_flow(), utils.py, .obfuscate_sensitive_headers()... diff --git a/tests/eval_attention.py b/tests/eval_attention.py deleted file mode 100644 index 5d55607..0000000 --- a/tests/eval_attention.py +++ /dev/null @@ -1,147 +0,0 @@ -""" -Graphify evaluation script - Transformer/Attention paper corpus. -Runs the full pipeline with a simulated Claude extraction JSON. -""" -from __future__ import annotations -import sys -import json -from pathlib import Path - -# Make sure we can import graphify from src/ -sys.path.insert(0, str(Path(__file__).parent / "src")) - -from graphify import detector, ast_extractor, graph_builder, clusterer, analyzer, reporter - -# ── 1. Detection ────────────────────────────────────────────────────────────── -RAW = Path("/home/safi/graphify_test/raw") -detection = detector.detect(RAW) -print("=== Detection ===") -print(json.dumps(detection, indent=2)) - -# ── 2. AST extraction from .py files ───────────────────────────────────────── -py_files = [Path(f) for f in detection["files"].get("code", [])] -ast_result = ast_extractor.extract(py_files) if py_files else {"nodes": [], "edges": []} -print(f"\n=== AST extraction: {len(ast_result['nodes'])} nodes, {len(ast_result['edges'])} edges ===") - -# ── 3. Simulated Claude extraction (realistic paper knowledge graph) ────────── -SOURCE_MD = str(RAW / "attention_notes.md") -SOURCE_CFG = str(RAW / "config.md") - -simulated_extraction = { - "nodes": [ - # Core architecture concepts - {"id": "transformer", "label": "Transformer", "file_type": "paper", "source_file": SOURCE_MD, "source_location": "Sec 3"}, - {"id": "encoder_layer", "label": "EncoderLayer", "file_type": "paper", "source_file": SOURCE_MD, "source_location": "Sec 3.1"}, - {"id": "decoder_layer", "label": "DecoderLayer", "file_type": "paper", "source_file": SOURCE_MD, "source_location": "Sec 3.1"}, - # Attention mechanism - {"id": "multi_head_attention", "label": "MultiHeadAttention", "file_type": "paper", "source_file": SOURCE_MD, "source_location": "Sec 3.2"}, - {"id": "scaled_dot_product", "label": "ScaledDotProductAttention", "file_type": "paper", "source_file": SOURCE_MD, "source_location": "Sec 3.2.1"}, - # Sub-components - {"id": "feed_forward", "label": "FeedForward", "file_type": "paper", "source_file": SOURCE_MD, "source_location": "Sec 3.3"}, - {"id": "layer_norm", "label": "LayerNorm", "file_type": "paper", "source_file": SOURCE_MD, "source_location": "Sec 3.1"}, - {"id": "positional_encoding", "label": "PositionalEncoding", "file_type": "paper", "source_file": SOURCE_MD, "source_location": "Sec 3.5"}, - # Hyperparameters - from config.md - {"id": "d_model", "label": "d_model", "file_type": "document", "source_file": SOURCE_CFG, "source_location": "L3"}, - {"id": "num_heads", "label": "num_heads", "file_type": "document", "source_file": SOURCE_CFG, "source_location": "L4"}, - {"id": "dropout", "label": "dropout", "file_type": "document", "source_file": SOURCE_CFG, "source_location": "L7"}, - ], - "edges": [ - # Transformer contains encoder and decoder stacks - {"source": "transformer", "target": "encoder_layer", "relation": "contains", "confidence": "EXTRACTED", "source_file": SOURCE_MD, "weight": 1.0}, - {"source": "transformer", "target": "decoder_layer", "relation": "contains", "confidence": "EXTRACTED", "source_file": SOURCE_MD, "weight": 1.0}, - # EncoderLayer uses multi-head attention and feed-forward - {"source": "encoder_layer", "target": "multi_head_attention", "relation": "uses", "confidence": "EXTRACTED", "source_file": SOURCE_MD, "weight": 1.0}, - {"source": "encoder_layer", "target": "feed_forward", "relation": "uses", "confidence": "EXTRACTED", "source_file": SOURCE_MD, "weight": 1.0}, - {"source": "encoder_layer", "target": "layer_norm", "relation": "applies", "confidence": "EXTRACTED", "source_file": SOURCE_MD, "weight": 1.0}, - # DecoderLayer uses multi-head attention (self + cross) and feed-forward - {"source": "decoder_layer", "target": "multi_head_attention", "relation": "uses", "confidence": "EXTRACTED", "source_file": SOURCE_MD, "weight": 1.0}, - {"source": "decoder_layer", "target": "feed_forward", "relation": "uses", "confidence": "EXTRACTED", "source_file": SOURCE_MD, "weight": 1.0}, - {"source": "decoder_layer", "target": "layer_norm", "relation": "applies", "confidence": "EXTRACTED", "source_file": SOURCE_MD, "weight": 1.0}, - # MultiHeadAttention implements ScaledDotProduct internally - {"source": "multi_head_attention", "target": "scaled_dot_product", "relation": "implements", "confidence": "EXTRACTED", "source_file": SOURCE_MD, "weight": 1.0}, - # Hyperparameter relationships - from config.md to architecture nodes - {"source": "multi_head_attention", "target": "d_model", "relation": "parameterized_by", "confidence": "EXTRACTED", "source_file": SOURCE_CFG, "weight": 1.0}, - {"source": "multi_head_attention", "target": "num_heads", "relation": "parameterized_by", "confidence": "EXTRACTED", "source_file": SOURCE_CFG, "weight": 1.0}, - {"source": "scaled_dot_product", "target": "d_model", "relation": "scales_by", "confidence": "INFERRED", "source_file": SOURCE_MD, "weight": 0.8}, - {"source": "feed_forward", "target": "d_model", "relation": "parameterized_by", "confidence": "EXTRACTED", "source_file": SOURCE_CFG, "weight": 1.0}, - # Positional encoding connects to transformer input (cross-community link) - {"source": "positional_encoding", "target": "transformer", "relation": "feeds_into", "confidence": "EXTRACTED", "source_file": SOURCE_MD, "weight": 1.0}, - {"source": "positional_encoding", "target": "d_model", "relation": "dimensioned_by", "confidence": "INFERRED", "source_file": SOURCE_MD, "weight": 0.8}, - # Dropout applied across sub-layers - ambiguous which specific sublayer - {"source": "dropout", "target": "multi_head_attention", "relation": "regularizes", "confidence": "AMBIGUOUS", "source_file": SOURCE_CFG, "weight": 0.6}, - {"source": "dropout", "target": "feed_forward", "relation": "regularizes", "confidence": "AMBIGUOUS", "source_file": SOURCE_CFG, "weight": 0.6}, - # Cross-community bridge: LayerNorm and PositionalEncoding both affect d_model scale - {"source": "layer_norm", "target": "positional_encoding", "relation": "operates_at_same_scale_as", "confidence": "INFERRED", "source_file": SOURCE_MD, "weight": 0.7}, - # Encoder-Decoder cross-attention: DecoderLayer attends to encoder output - {"source": "decoder_layer", "target": "encoder_layer", "relation": "cross_attends_to", "confidence": "EXTRACTED", "source_file": SOURCE_MD, "weight": 1.0}, - ], - "input_tokens": 3200, - "output_tokens": 820, -} - -# ── 4. Merge AST + simulated Claude extraction ──────────────────────────────── -all_extractions = [simulated_extraction] -if ast_result["nodes"]: - all_extractions.append(ast_result) - -G = graph_builder.build(all_extractions) -print(f"\n=== Graph: {G.number_of_nodes()} nodes, {G.number_of_edges()} edges ===") - -# ── 5. Community detection ──────────────────────────────────────────────────── -communities = clusterer.cluster(G) -cohesion = clusterer.score_all(G, communities) -print(f"\n=== Communities: {len(communities)} detected ===") -for cid, nodes in communities.items(): - node_labels = [G.nodes[n].get("label", n) for n in nodes] - print(f" Community {cid} ({len(nodes)} nodes): {node_labels}") - print(f" Cohesion: {cohesion[cid]}") - -# ── 6. Analysis ─────────────────────────────────────────────────────────────── -god_node_list = analyzer.god_nodes(G, top_n=10) -print(f"\n=== God Nodes ===") -for g in god_node_list: - print(f" {g['label']}: {g['edges']} edges") - -surprise_list = analyzer.surprising_connections(G, communities=communities, top_n=5) -print(f"\n=== Surprising Connections: {len(surprise_list)} found ===") -for s in surprise_list: - print(f" {s['source']} <-> {s['target']} [{s['confidence']}]: {s['relation']}") - print(f" Note: {s.get('note', 'cross-file')}") - -# ── 7. Community labels (hand-crafted for accuracy) ─────────────────────────── -# We label based on which nodes ended up in which community -community_labels = {} -for cid, nodes in communities.items(): - node_labels_set = {G.nodes[n].get("label", n) for n in nodes} - if "MultiHeadAttention" in node_labels_set or "ScaledDotProductAttention" in node_labels_set: - community_labels[cid] = "Attention Mechanism" - elif "Transformer" in node_labels_set or "EncoderLayer" in node_labels_set: - community_labels[cid] = "Encoder-Decoder Architecture" - elif "d_model" in node_labels_set or "num_heads" in node_labels_set or "dropout" in node_labels_set: - community_labels[cid] = "Hyperparameters & Configuration" - elif "PositionalEncoding" in node_labels_set: - community_labels[cid] = "Positional Encoding & Embedding" - elif any(label.endswith(".py") or "()" in label for label in node_labels_set): - community_labels[cid] = "Code Implementation" - else: - community_labels[cid] = f"Cluster {cid}" - -token_cost = {"input": simulated_extraction["input_tokens"], "output": simulated_extraction["output_tokens"]} - -# ── 8. Report ───────────────────────────────────────────────────────────────── -report = reporter.generate( - G=G, - communities=communities, - cohesion_scores=cohesion, - community_labels=community_labels, - god_node_list=god_node_list, - surprise_list=surprise_list, - detection_result=detection, - token_cost=token_cost, - root=str(RAW), -) - -out_path = Path("/tmp/GRAPH_REPORT_attention.md") -out_path.write_text(report) -print(f"\n=== Report written to {out_path} ===") -print(report) diff --git a/tests/test_export.py b/tests/test_export.py index af2ade9..6f4421d 100644 --- a/tests/test_export.py +++ b/tests/test_export.py @@ -3,7 +3,7 @@ import tempfile from pathlib import Path from graphify.build import build_from_json from graphify.cluster import cluster -from graphify.export import to_json, to_cypher, to_graphml +from graphify.export import to_json, to_cypher, to_graphml, to_html FIXTURES = Path(__file__).parent / "fixtures" @@ -79,3 +79,49 @@ def test_to_graphml_has_community_attribute(): to_graphml(G, communities, str(out)) content = out.read_text() assert "community" in content + +def test_to_html_creates_file(): + G = make_graph() + communities = cluster(G) + with tempfile.TemporaryDirectory() as tmp: + out = Path(tmp) / "graph.html" + to_html(G, communities, str(out)) + assert out.exists() + +def test_to_html_contains_visjs(): + G = make_graph() + communities = cluster(G) + with tempfile.TemporaryDirectory() as tmp: + out = Path(tmp) / "graph.html" + to_html(G, communities, str(out)) + content = out.read_text() + assert "vis-network" in content + +def test_to_html_contains_search(): + G = make_graph() + communities = cluster(G) + with tempfile.TemporaryDirectory() as tmp: + out = Path(tmp) / "graph.html" + to_html(G, communities, str(out)) + content = out.read_text() + assert "search" in content.lower() + +def test_to_html_contains_legend_with_labels(): + G = make_graph() + communities = cluster(G) + labels = {cid: f"Group {cid}" for cid in communities} + with tempfile.TemporaryDirectory() as tmp: + out = Path(tmp) / "graph.html" + to_html(G, communities, str(out), community_labels=labels) + content = out.read_text() + assert "Group 0" in content + +def test_to_html_contains_nodes_and_edges(): + G = make_graph() + communities = cluster(G) + with tempfile.TemporaryDirectory() as tmp: + out = Path(tmp) / "graph.html" + to_html(G, communities, str(out)) + content = out.read_text() + assert "RAW_NODES" in content + assert "RAW_EDGES" in content