From 0b7460a9e3dca374c5d9a30d157793740ea8b973 Mon Sep 17 00:00:00 2001 From: Safi Date: Sat, 4 Apr 2026 14:04:39 +0100 Subject: [PATCH] perf+fix: parallel extraction, faster imports, bug fixes 45x faster cluster import, 135x faster detect, parallel subagent extraction, auto-exclude venvs/caches, suggest_questions fix, manifest fix, install CLI --- .gitignore | 5 + README.md | 190 ++++--- graphify/__init__.py | 32 +- graphify/__main__.py | 73 +++ graphify/analyze.py | 12 + graphify/cache.py | 55 ++ graphify/cluster.py | 22 +- graphify/detect.py | 37 +- graphify/manifest.py | 4 + graphify/report.py | 15 +- graphify/skill.md | 1046 ++++++++++++++++++++++++++++++++++++++ graphify/validate.py | 2 +- pyproject.toml | 20 + skills/graphify/skill.md | 87 ++-- tests/test_serve.py | 151 ++++++ tests/test_watch.py | 68 +++ 16 files changed, 1670 insertions(+), 149 deletions(-) create mode 100644 graphify/__main__.py create mode 100644 graphify/manifest.py create mode 100644 graphify/skill.md create mode 100644 tests/test_serve.py create mode 100644 tests/test_watch.py diff --git a/.gitignore b/.gitignore index 8e90692..9d2498c 100644 --- a/.gitignore +++ b/.gitignore @@ -1,4 +1,6 @@ venv/ +.venv/ +env/ __pycache__/ *.pyc *.egg-info/ @@ -6,5 +8,8 @@ __pycache__/ dist/ build/ .pytest_cache/ +.mypy_cache/ +.ruff_cache/ *.so +*.egg .graphify/ diff --git a/README.md b/README.md index d69e5f8..d1b076c 100644 --- a/README.md +++ b/README.md @@ -1,6 +1,6 @@ # graphify -A Claude Code skill that turns any folder of files into a navigable knowledge graph — then opens it as an Obsidian vault you can explore, filter, and query. + any folder of files → persistent knowledge graph → Obsidian vault, graph.json, audit report ``` /graphify ./raw @@ -8,40 +8,66 @@ A Claude Code skill that turns any folder of files into a navigable knowledge gr ``` .graphify/ -├── obsidian/ open as Obsidian vault to explore the graph visually -├── GRAPH_REPORT.md what the graph found — surprising connections, knowledge gaps, suggested questions -└── graph.json persistent graph — query it weeks later without re-reading anything +├── obsidian/ open as Obsidian vault — visual graph, wikilinks, filter by community +├── GRAPH_REPORT.md what the graph found: god nodes, surprising connections, suggested questions +├── graph.json persistent graph — query it weeks later without re-reading anything +├── cache/ per-file SHA256 cache — re-runs only process changed files +└── memory/ Q&A results filed back in — what you ask grows the graph on next --update ``` -## The problem it solves +[placeholder: animated GIF showing the full pipeline — detect → extract → cluster → report → Obsidian vault] -Andrej Karpathy described it well: he keeps a `/raw` folder where he drops papers, tweets, screenshots, and notes. The problem is that folder becomes opaque. You forget what's in it. You can't see what connects. +## Why this exists -Claude can read any single file. But ask Claude "what connects paper A to the code in repo B?" and it will hallucinate — it hasn't read both, and even if it has, it has no memory of the connection next session. +**The problem:** Andrej Karpathy described it well: he keeps a `/raw` folder where he drops papers, tweets, screenshots, and notes. The problem is that folder becomes opaque. You forget what's in it. You can't see what connects. Ask Claude "what links paper A to the code in repo B?" and it will hallucinate — it hasn't read both, and even if it has, it has no memory of that connection next session. -graphify solves this by: +**What LLMs get wrong:** Naive summarization fills in every gap confidently. You get a summary that sounds complete but you can't tell what was actually in the files vs invented by the model. And next session, it's all gone — no memory of what it extracted. -1. Reading everything once, extracting a persistent graph -2. Tagging every edge as `[EXTRACTED]` (explicitly stated), `[INFERRED]` (reasonable), or `[AMBIGUOUS]` (flagged for review) — you always know what was found vs invented -3. Running community detection to find clusters you didn't know existed -4. Surfacing cross-community connections — the things you would never think to ask about directly -5. Storing the graph in `.graphify/graph.json` so you can query it in any future session without re-extracting +**What graphify does differently:** + +- **Persistent graph** — relationships are stored in `.graphify/graph.json` and survive across sessions. Query weeks later without re-reading anything. +- **Honest audit trail** — every edge is tagged `EXTRACTED` (explicitly stated), `INFERRED` (call-graph or reasonable deduction), or `AMBIGUOUS` (flagged for review). You always know what was found vs invented. +- **Cross-document surprise** — Leiden community detection finds clusters, then surfaces cross-community connections: the things you would never think to ask about directly. +- **Feedback loop** — every query answer is saved to `.graphify/memory/`. On next `--update`, that Q&A becomes a node. The graph grows from what you ask, not just what you add. + +The result: a navigable map of your corpus that is honest about what it knows and what it guessed. ## Install -Copy the skill into your Claude Code skills directory: +```bash +pip install graphify && graphify install +``` + +That's it. This copies the skill file into `~/.claude/skills/graphify/` and registers it in `~/.claude/CLAUDE.md` automatically. The Python package and all dependencies install on first `/graphify` run — you never touch pip manually again. + +Then open Claude Code in any directory and type: + +``` +/graphify . +``` + +
+Manual install (curl) + +**Step 1 — copy the skill file** ```bash mkdir -p ~/.claude/skills/graphify -curl -s https://raw.githubusercontent.com/safishamsi/graphify/v1/skills/graphify/skill.md \ +curl -fsSL https://raw.githubusercontent.com/safishamsi/graphify/v1/skills/graphify/skill.md \ > ~/.claude/skills/graphify/SKILL.md ``` -Add to `~/.claude/CLAUDE.md`: +**Step 2 — register it in Claude Code** + +Add this to `~/.claude/CLAUDE.md` (create the file if it doesn't exist): + ``` - **graphify** (`~/.claude/skills/graphify/SKILL.md`) — any input to knowledge graph. Trigger: `/graphify` +When the user types `/graphify`, invoke the Skill tool with `skill: "graphify"` before doing anything else. ``` +
+ ## Usage ```bash @@ -49,81 +75,107 @@ Add to `~/.claude/CLAUDE.md`: /graphify ./raw # run on a specific folder /graphify ./raw --mode deep # more aggressive INFERRED edge extraction /graphify ./raw --update # re-extract only changed files, merge into existing graph -/graphify ./raw --watch # notify when new files appear (drop files, get pinged) +/graphify ./raw --watch # notify when new files appear /graphify add https://arxiv.org/abs/1706.03762 # fetch a paper, save, update graph /graphify add https://x.com/karpathy/status/... # fetch a tweet -/graphify add --author "Karpathy" --contributor "safi" # tag who wrote it and who added it +/graphify add --author "Karpathy" --contributor "safi" /graphify query "what connects attention to the optimizer?" # BFS — broad context /graphify query "how does the encoder reach the loss?" --dfs # DFS — trace a path /graphify query "..." --budget 1500 # cap at N tokens +/graphify path "DigestAuth" "Response" # shortest path between two concepts +/graphify explain "SwinTransformer" # plain-language node explanation + /graphify ./raw --html # also export graph.html (browser, no Obsidian needed) /graphify ./raw --svg # also export graph.svg (embeds in Notion, GitHub) /graphify ./raw --neo4j # generate cypher.txt for Neo4j import +/graphify ./raw --mcp # start MCP stdio server for agent access ``` Works with any mix of file types in the same folder: | Type | Extensions | How it's extracted | |------|-----------|-------------------| -| Code | `.py .ts .js .go .rs .java .cpp .rb` etc | AST (deterministic) + semantic (Claude) | -| Documents | `.md .txt .rst` | Claude reads and extracts concepts + relationships | +| Code | `.py .ts .tsx .js .go .rs` | AST (deterministic) + call-graph pass (INFERRED) | +| Code | `.java .cpp .c .rb .swift .kt` | Claude semantic extraction | +| Documents | `.md .txt .rst` | Concepts + relationships via Claude | | Papers | `.pdf` | Citation mining + concept extraction | -| Images | `.png .jpg .webp .gif .svg` | Claude vision — reads UI screenshots, charts, tweets, diagrams, whiteboards | +| Images | `.png .jpg .webp .gif .svg` | Claude vision — screenshots, charts, whiteboards, any language | ## What you get -After running, Claude pastes three things directly into the chat: +After running, Claude outputs three things directly in chat: -**God nodes** — the highest-degree concepts (what everything connects through) +**God nodes** — highest-degree concepts (what everything connects through) -**Surprising connections** — cross-community edges; relationships between concepts that live in different clusters. These are what you didn't know to look for. +**Surprising connections** — cross-community edges; relationships between concepts in different clusters that you didn't know to look for **Suggested questions** — 4-5 questions the graph is uniquely positioned to answer, with the reason why (which bridge node makes it interesting, which community boundary it crosses) -The full `GRAPH_REPORT.md` also includes community summaries with cohesion scores and a list of ambiguous edges for your review. +The full GRAPH_REPORT.md adds community summaries with cohesion scores and a list of ambiguous edges for review. -## Use cases +## Key files explained -**New codebase** — run `/graphify` before touching anything. Find the god nodes (what you have to understand first), the community structure (what the major subsystems are), and the surprising connections (what talks to what that you wouldn't expect). +| File | Purpose | +|------|---------| +| `GRAPH_REPORT.md` | The audit report. God nodes, surprising connections, community cohesion scores, ambiguous edge list, suggested questions. | +| `graph.json` | Persistent graph in node-link format. Load it with NetworkX or push to Neo4j. Survives sessions. | +| `obsidian/` | Wikilink vault. Open in Obsidian → enable graph view → see communities as clusters. Filter by tag, search across everything. | +| `.graphify/cache/` | SHA256-based per-file cache. A re-run on an unchanged corpus takes seconds. | +| `.graphify/memory/` | Q&A feedback loop. Every `/graphify query` answer is saved here. Next `--update` extracts it into the graph. | -**Research reading list** — drop papers, tweets, and notes into `/raw`. Run `/graphify ./raw`. Get a graph of how concepts connect across everything you've read. Query it: "what connects sparse autoencoders to superposition?" +## What this skill will NOT do -**Personal knowledge base** — leave `--watch` running on your `/raw` folder. Drop things in throughout the day. The graph grows. Query it weeks later without re-reading anything. +- **Won't invent edges** — `AMBIGUOUS` exists so uncertain relationships are flagged, not hidden. If the connection isn't clear, it's tagged, not fabricated. +- **Won't claim the graph is useful when it isn't** — a corpus over 2M words or 200 files gets a cost warning before proceeding. +- **Won't re-extract unchanged files** — SHA256 cache ensures warm re-runs skip everything that hasn't changed. +- **Won't visualize graphs over 5,000 nodes** — use `--no-viz` or query instead. +- **Won't download datasets or set up infrastructure** — graphify reads your files. What you put in the folder is what it works with. +- **Won't implement baselines or run experiments** — it reads and maps. Analysis is yours. -**Collaborative corpus** — use `--contributor` to tag who added what. The graph knows provenance. "What did safi add that connects to the attention mechanism?" +## Design principles -## What it will NOT do +1. **Extraction quality is everything** — clustering is downstream of it. A bad graph clusters into bad communities. The AST + call-graph pass exists because deterministic beats probabilistic for code. +2. **Show the numbers** — cohesion is `0.91`, not "good". Token cost is always printed. You know what you spent. +3. **The best output is what you didn't know** — Surprising Connections is not optional. God nodes you probably already suspected. Cross-community edges are what you came for. +4. **The graph earns its complexity** — below a certain density, just use Claude directly. The graph adds value when you have more than you can hold in context across sessions. +5. **What you ask grows the graph** — query results are filed back in automatically. The corpus is not static. +6. **Honest uncertainty** — `EXTRACTED`, `INFERRED`, `AMBIGUOUS` are not cosmetic labels. They are the difference between trusting the graph and being misled by it. -- Won't invent edges — `[AMBIGUOUS]` exists so uncertain relationships are flagged, not hidden -- Won't claim the graph is useful when it isn't — corpus under 50K words gets a warning -- Won't re-extract unchanged files — `--update` uses a manifest to skip unchanged files -- Won't visualize graphs over 5,000 nodes — use `--no-viz` or query instead +## Contributing -## Files +**Adding worked examples** -``` -graphify/ -├── detect.py detect file types, auto-exclude venvs/caches/node_modules -├── extract.py parse files into nodes + edges (tree-sitter AST + Claude) -├── build.py assemble NetworkX graph from extraction JSON -├── cluster.py Leiden community detection, cohesion scoring -├── analyze.py god nodes, bridge nodes, surprising connections, suggested questions -├── report.py render GRAPH_REPORT.md -├── export.py Obsidian vault, graph.json, graph.html, graph.svg, Neo4j Cypher -├── ingest.py fetch URLs (arXiv, Twitter/X, PDF, any webpage), save annotated markdown -├── validate.py JSON schema checks on extraction output -├── serve.py MCP stdio server — exposes graph tools to other agents -└── watch.py fs watcher, writes flag file when new files appear +Worked examples are the most trust-building part of this project. To add one: -skills/graphify/ -└── skill.md the Claude Code skill — everything the agent runs +1. Pick a real corpus (people should be able to verify the output) +2. Run the skill: `/graphify ` +3. Save the full output to `worked/{corpus_slug}/` +4. Write a `review.md` that honestly evaluates: + - What the graph got right + - What edges it correctly flagged AMBIGUOUS + - Any mistakes or missed connections + - Any surprising connections that were genuinely surprising +5. Submit a PR with all of the above -tests/ 71 tests, one file per module -pyproject.toml deps: networkx, graspologic, tree-sitter, pyvis -``` +**Improving extraction** + +If you find a file type or language where extraction is poor, open an issue with a minimal reproduction case. The best bug reports include: the input file, the extraction output (`.graphify/cache/` entry), and what was missed or invented. + +**Adding domain knowledge** + +If corpora in your domain consistently contain structures graphify doesn't extract well (e.g., legal documents, lab notebooks, musical scores), open a discussion with examples. + +## Worked examples + +| Corpus | Type | Eval report | +|--------|------|-------------| +| httpx (Python HTTP client) | Codebase | `tests/EVAL_httpx.md` + `tests/GRAPH_REPORT_httpx.md` | +| Mixed corpus (code + paper + Arabic image) | Multi-type | `tests/EVAL_mixed_corpus.md` | + +Each includes the full graph output and an honest evaluation of what the skill got right and wrong. ## Tech stack @@ -133,14 +185,30 @@ pyproject.toml deps: networkx, graspologic, tree-sitter, pyvis | Community detection | Leiden via graspologic | Better than K-means for sparse graphs | | Code parsing | tree-sitter | Multi-language AST, deterministic, zero hallucination | | Extraction | Claude (parallel subagents) | Reads anything, outputs structured graph data | -| Visualization | Obsidian vault | Native graph view, wikilinks, search, no server needed | +| Visualization | Obsidian vault | Native graph view, wikilinks, no server needed | No Neo4j required. No dashboards. No server. Runs entirely locally. -## Design principles +## Files -1. Extraction quality is everything — clustering is downstream of it -2. Show the numbers — cohesion is 0.91, not "good" -3. The best output is what you didn't know — Surprising Connections is not optional -4. Token cost is always visible -5. The graph earns its complexity — corpus under 50K words gets a warning to just use Claude directly +``` +graphify/ +├── detect.py detect file types, auto-exclude venvs/caches/node_modules; scan .graphify/memory/ +├── extract.py AST extraction (Python, TypeScript, JavaScript, Go, Rust) + call-graph pass +├── build.py assemble NetworkX graph from extraction JSON; schema-validates before assembly +├── cluster.py Leiden community detection, cohesion scoring +├── analyze.py god nodes, bridge nodes, surprising connections, suggested questions, graph diff +├── report.py render GRAPH_REPORT.md +├── export.py Obsidian vault, graph.json, graph.html, graph.svg, Neo4j Cypher, Canvas +├── ingest.py fetch URLs (arXiv, Twitter/X, PDF, any webpage); save Q&A to .graphify/memory/ +├── cache.py SHA256-based per-file extraction cache; check_semantic_cache / save_semantic_cache +├── validate.py JSON schema checks on extraction output +├── serve.py MCP stdio server — query_graph, get_node, get_neighbors, shortest_path, god_nodes +└── watch.py fs watcher, writes flag file when new files appear + +skills/graphify/ +└── skill.md the Claude Code skill — the full pipeline the agent runs step by step + +tests/ 142 tests, one file per module +pyproject.toml pip install graphify | pip install graphify[mcp,neo4j,pdf,watch] +``` diff --git a/graphify/__init__.py b/graphify/__init__.py index bad401e..3c12c55 100644 --- a/graphify/__init__.py +++ b/graphify/__init__.py @@ -1,7 +1,27 @@ """graphify — extract · build · cluster · analyze · report.""" -from graphify.extract import extract, collect_files -from graphify.build import build_from_json -from graphify.cluster import cluster, score_all, cohesion_score -from graphify.analyze import god_nodes, surprising_connections, suggest_questions -from graphify.report import generate -from graphify.export import to_json, to_html, to_svg, to_canvas + + +def __getattr__(name): + # Lazy imports so `graphify install` works before heavy deps are in place. + _map = { + "extract": ("graphify.extract", "extract"), + "collect_files": ("graphify.extract", "collect_files"), + "build_from_json": ("graphify.build", "build_from_json"), + "cluster": ("graphify.cluster", "cluster"), + "score_all": ("graphify.cluster", "score_all"), + "cohesion_score": ("graphify.cluster", "cohesion_score"), + "god_nodes": ("graphify.analyze", "god_nodes"), + "surprising_connections": ("graphify.analyze", "surprising_connections"), + "suggest_questions": ("graphify.analyze", "suggest_questions"), + "generate": ("graphify.report", "generate"), + "to_json": ("graphify.export", "to_json"), + "to_html": ("graphify.export", "to_html"), + "to_svg": ("graphify.export", "to_svg"), + "to_canvas": ("graphify.export", "to_canvas"), + } + if name in _map: + import importlib + mod_name, attr = _map[name] + mod = importlib.import_module(mod_name) + return getattr(mod, attr) + raise AttributeError(f"module 'graphify' has no attribute {name!r}") diff --git a/graphify/__main__.py b/graphify/__main__.py new file mode 100644 index 0000000..8f5fa8d --- /dev/null +++ b/graphify/__main__.py @@ -0,0 +1,73 @@ +"""graphify CLI — `graphify install` sets up the Claude Code skill.""" +from __future__ import annotations +import shutil +import sys +from pathlib import Path + +_SKILL_REGISTRATION = ( + "\n# graphify\n" + "- **graphify** (`~/.claude/skills/graphify/SKILL.md`) " + "— any input to knowledge graph. Trigger: `/graphify`\n" + "When the user types `/graphify`, invoke the Skill tool " + "with `skill: \"graphify\"` before doing anything else.\n" +) + + +def _bundled_skill() -> Path: + """Path to the skill.md bundled with this package.""" + return Path(__file__).parent / "skill.md" + + +def install() -> None: + skill_src = _bundled_skill() + if not skill_src.exists(): + print("error: skill.md not found in package — reinstall graphify", file=sys.stderr) + sys.exit(1) + + # Copy skill to ~/.claude/skills/graphify/SKILL.md + skill_dst = Path.home() / ".claude" / "skills" / "graphify" / "SKILL.md" + skill_dst.parent.mkdir(parents=True, exist_ok=True) + shutil.copy(skill_src, skill_dst) + print(f" skill installed → {skill_dst}") + + # Register in ~/.claude/CLAUDE.md + claude_md = Path.home() / ".claude" / "CLAUDE.md" + if claude_md.exists(): + content = claude_md.read_text() + if "graphify" in content: + print(f" CLAUDE.md → already registered (no change)") + else: + claude_md.write_text(content.rstrip() + _SKILL_REGISTRATION) + print(f" CLAUDE.md → skill registered in {claude_md}") + else: + claude_md.parent.mkdir(parents=True, exist_ok=True) + claude_md.write_text(_SKILL_REGISTRATION.lstrip()) + print(f" CLAUDE.md → created at {claude_md}") + + print() + print("Done. Open Claude Code in any directory and type:") + print() + print(" /graphify .") + print() + + +def main() -> None: + if len(sys.argv) < 2 or sys.argv[1] in ("-h", "--help"): + print("Usage: graphify ") + print() + print("Commands:") + print(" install copy skill to ~/.claude/skills/ and register in CLAUDE.md") + print() + return + + cmd = sys.argv[1] + if cmd == "install": + install() + else: + print(f"error: unknown command '{cmd}'", file=sys.stderr) + print("Run 'graphify --help' for usage.", file=sys.stderr) + sys.exit(1) + + +if __name__ == "__main__": + main() diff --git a/graphify/analyze.py b/graphify/analyze.py index b48dd23..c15b162 100644 --- a/graphify/analyze.py +++ b/graphify/analyze.py @@ -332,6 +332,18 @@ def suggest_questions( "why": f"Cohesion score {score} — nodes in this community are weakly interconnected.", }) + if not questions: + return [{ + "type": "no_signal", + "question": None, + "why": ( + "Not enough signal to generate questions. " + "This usually means the corpus has no AMBIGUOUS edges, no bridge nodes, " + "no INFERRED relationships, and all communities are tightly cohesive. " + "Add more files or run with --mode deep to extract richer edges." + ), + }] + return questions[:top_n] diff --git a/graphify/cache.py b/graphify/cache.py index 04f9bc8..acefe67 100644 --- a/graphify/cache.py +++ b/graphify/cache.py @@ -61,3 +61,58 @@ def clear_cache(root: Path = Path(".")) -> None: d = cache_dir(root) for f in d.glob("*.json"): f.unlink() + + +def check_semantic_cache( + files: list[str], + root: Path = Path("."), +) -> tuple[list[dict], list[dict], list[str]]: + """Check semantic extraction cache for a list of absolute file paths. + + Returns (cached_nodes, cached_edges, uncached_files). + Uncached files need Claude extraction; cached files are merged directly. + """ + cached_nodes: list[dict] = [] + cached_edges: list[dict] = [] + uncached: list[str] = [] + + for fpath in files: + result = load_cached(Path(fpath), root) + if result is not None: + cached_nodes.extend(result.get("nodes", [])) + cached_edges.extend(result.get("edges", [])) + else: + uncached.append(fpath) + + return cached_nodes, cached_edges, uncached + + +def save_semantic_cache( + nodes: list[dict], + edges: list[dict], + root: Path = Path("."), +) -> int: + """Save semantic extraction results to cache, keyed by source_file. + + Groups nodes and edges by source_file, then saves one cache entry per file. + Returns the number of files cached. + """ + from collections import defaultdict + + by_file: dict[str, dict] = defaultdict(lambda: {"nodes": [], "edges": []}) + for n in nodes: + src = n.get("source_file", "") + if src: + by_file[src]["nodes"].append(n) + for e in edges: + src = e.get("source_file", "") + if src: + by_file[src]["edges"].append(e) + + saved = 0 + for fpath, result in by_file.items(): + p = Path(fpath) + if p.exists(): + save_cached(p, result, root) + saved += 1 + return saved diff --git a/graphify/cluster.py b/graphify/cluster.py index 3ff0d12..dbbeacc 100644 --- a/graphify/cluster.py +++ b/graphify/cluster.py @@ -1,7 +1,6 @@ """Leiden community detection on NetworkX graphs. Splits oversized communities. Returns cohesion scores.""" from __future__ import annotations import networkx as nx -from graspologic.partition import leiden def build_graph(nodes: list[dict], edges: list[dict]) -> nx.Graph: @@ -37,10 +36,24 @@ def cluster(G: nx.Graph) -> dict[int, list[str]]: if G.number_of_edges() == 0: return {i: [n] for i, n in enumerate(sorted(G.nodes))} - partition: dict[str, int] = leiden(G) + from graspologic.partition import leiden # lazy — avoids 15s numba JIT on import + + # Leiden warns and drops isolates — handle them separately + isolates = [n for n in G.nodes() if G.degree(n) == 0] + connected_nodes = [n for n in G.nodes() if G.degree(n) > 0] + connected = G.subgraph(connected_nodes) + raw: dict[int, list[str]] = {} - for node, cid in partition.items(): - raw.setdefault(cid, []).append(node) + if connected.number_of_nodes() > 0: + partition: dict[str, int] = leiden(connected) + for node, cid in partition.items(): + raw.setdefault(cid, []).append(node) + + # Each isolate becomes its own single-node community + next_cid = max(raw.keys(), default=-1) + 1 + for node in isolates: + raw[next_cid] = [node] + next_cid += 1 # Split oversized communities max_size = max(_MIN_SPLIT_SIZE, int(G.number_of_nodes() * _MAX_COMMUNITY_FRACTION)) @@ -63,6 +76,7 @@ def _split_community(G: nx.Graph, nodes: list[str]) -> list[list[str]]: # No edges — split into individual nodes return [[n] for n in sorted(nodes)] try: + from graspologic.partition import leiden sub_partition: dict[str, int] = leiden(subgraph) sub_communities: dict[int, list[str]] = {} for node, cid in sub_partition.items(): diff --git a/graphify/detect.py b/graphify/detect.py index 7140168..1e8957d 100644 --- a/graphify/detect.py +++ b/graphify/detect.py @@ -152,24 +152,31 @@ def detect(root: Path) -> dict: seen: set[Path] = set() all_files: list[Path] = [] + for scan_root in scan_paths: - for p in sorted(scan_root.rglob("*")): - if p not in seen: - seen.add(p) - all_files.append(p) + in_memory_tree = memory_dir.exists() and str(scan_root).startswith(str(memory_dir)) + import os + for dirpath, dirnames, filenames in os.walk(scan_root): + dp = Path(dirpath) + if not in_memory_tree: + # Prune noise dirs in-place so os.walk never descends into them + dirnames[:] = [ + d for d in dirnames + if not d.startswith(".") and not _is_noise_dir(d) + ] + for fname in filenames: + p = dp / fname + if p not in seen: + seen.add(p) + all_files.append(p) for p in all_files: - if not p.is_file(): - continue - # For memory dir files, don't apply hidden/noise filtering + # For memory dir files, skip hidden/noise filtering in_memory = memory_dir.exists() and str(p).startswith(str(memory_dir)) if not in_memory: - try: - parts = p.relative_to(root).parts - except ValueError: - continue - # Skip hidden dirs and known noise dirs - if any(part.startswith(".") or _is_noise_dir(part) for part in parts): + # Hidden files are already excluded via dir pruning above, + # but catch hidden files at the root level + if p.name.startswith("."): continue if _is_sensitive(p): skipped_sensitive.append(str(p)) @@ -221,8 +228,8 @@ def save_manifest(files: dict[str, list[str]], manifest_path: str = _MANIFEST_PA for f in file_list: try: manifest[f] = Path(f).stat().st_mtime - except Exception: - pass + except OSError: + pass # file deleted between detect() and manifest write — skip it Path(manifest_path).parent.mkdir(parents=True, exist_ok=True) Path(manifest_path).write_text(json.dumps(manifest, indent=2)) diff --git a/graphify/manifest.py b/graphify/manifest.py new file mode 100644 index 0000000..cc74b84 --- /dev/null +++ b/graphify/manifest.py @@ -0,0 +1,4 @@ +# re-export manifest helpers from detect for backwards compatibility +from graphify.detect import save_manifest, load_manifest, detect_incremental + +__all__ = ["save_manifest", "load_manifest", "detect_incremental"] diff --git a/graphify/report.py b/graphify/report.py index bbd882e..885de83 100644 --- a/graphify/report.py +++ b/graphify/report.py @@ -119,10 +119,15 @@ def generate( if suggested_questions: lines += ["", "## Suggested Questions"] - lines.append("_Questions this graph is uniquely positioned to answer:_") - lines.append("") - for q in suggested_questions: - lines.append(f"- **{q['question']}**") - lines.append(f" _{q['why']}_") + no_signal = len(suggested_questions) == 1 and suggested_questions[0].get("type") == "no_signal" + if no_signal: + lines.append(f"_{suggested_questions[0]['why']}_") + else: + lines.append("_Questions this graph is uniquely positioned to answer:_") + lines.append("") + for q in suggested_questions: + if q.get("question"): + lines.append(f"- **{q['question']}**") + lines.append(f" _{q['why']}_") return "\n".join(lines) diff --git a/graphify/skill.md b/graphify/skill.md new file mode 100644 index 0000000..174dc5a --- /dev/null +++ b/graphify/skill.md @@ -0,0 +1,1046 @@ +--- +name: graphify +description: any input (code, docs, papers, images) → knowledge graph → clustered communities → HTML + JSON + audit report +trigger: /graphify +--- + +# /graphify + +Turn any folder of files into a navigable knowledge graph with community detection, an honest audit trail, and three outputs: interactive HTML, GraphRAG-ready JSON, and a plain-language GRAPH_REPORT.md. + +## Usage + +``` +/graphify # full pipeline on current directory → Obsidian vault +/graphify # full pipeline on specific path +/graphify --mode deep # thorough extraction, richer INFERRED edges +/graphify --update # incremental — re-extract only new/changed files +/graphify --cluster-only # rerun clustering on existing graph +/graphify --no-viz # skip visualization, just report + JSON +/graphify --html # also export graph.html (pyvis, browser-based) +/graphify --svg # also export graph.svg (embeds in Notion, GitHub) +/graphify --neo4j # generate .graphify/cypher.txt for Neo4j +/graphify --neo4j-push bolt://localhost:7687 # push directly to Neo4j +/graphify --mcp # start MCP stdio server for agent access +/graphify --watch # watch folder, notify when files change +/graphify add # fetch URL, save to ./raw, update graph +/graphify add --author "Name" # tag who wrote it +/graphify add --contributor "Name" # tag who added it to the corpus +/graphify query "" # BFS traversal — broad context +/graphify query "" --dfs # DFS — trace a specific path +/graphify query "" --budget 1500 # cap answer at N tokens +/graphify path "AuthModule" "Database" # shortest path between two concepts +/graphify explain "SwinTransformer" # plain-language explanation of a node +``` + +## What graphify is for + +graphify is built around Andrej Karpathy's /raw folder workflow: drop anything into a folder — papers, tweets, screenshots, code, notes — and get a structured knowledge graph that shows you what you didn't know was connected. + +Three things it does that Claude alone cannot: +1. **Persistent graph** — relationships are stored in `.graphify/graph.json` and survive across sessions. Ask questions weeks later without re-reading everything. +2. **Honest audit trail** — every edge is tagged EXTRACTED, INFERRED, or AMBIGUOUS. You know what was found vs invented. +3. **Cross-document surprise** — community detection finds connections between concepts in different files that you would never think to ask about directly. + +Use it for: +- A codebase you're new to (understand architecture before touching anything) +- A reading list (papers + tweets + notes → one navigable graph) +- A research corpus (citation graph + concept graph in one) +- Your personal /raw folder (drop everything in, let it grow, query it) + +## What You Must Do When Invoked + +If no path was given, use `.` (current directory). Do not ask the user for a path. + +Follow these steps in order. Do not skip steps. + +### Step 1 — Ensure graphify is installed + +```bash +python3 -c "import graphify" 2>/dev/null || pip install graphify -q --break-system-packages 2>&1 | tail -3 +``` + +If the import succeeds, print nothing and move straight to Step 2. + +### Step 2 — Detect files + +```bash +python3 -c " +import json +from graphify.detect import detect +from pathlib import Path +result = detect(Path('INPUT_PATH')) +print(json.dumps(result)) +" > .graphify_detect.json +``` + +Replace INPUT_PATH with the actual path the user provided. Do NOT cat or print the JSON — read it silently and present a clean summary instead: + +``` +Corpus: X files · ~Y words + code: N files (.py .ts .go ...) + docs: N files (.md .txt ...) + papers: N files (.pdf ...) + images: N files +``` + +Then act on it: +- If `total_files` is 0: stop with "No supported files found in [path]." +- If `skipped_sensitive` is non-empty: mention file count skipped, not the file names. +- If `total_words` > 2,000,000 OR `total_files` > 200: show the warning and the top 5 subdirectories by file count, then ask which subfolder to run on. Wait for the user's answer before proceeding. +- Otherwise: proceed directly to Step 3 — no need to ask anything. + +### Step 3 — Extract entities and relationships + +This step has two parts: **structural extraction** (deterministic, free) then **semantic extraction** (Claude, costs tokens). + +#### Part A — Structural extraction for code files + +For any code files detected, run AST extraction first: + +```bash +python3 -c " +import sys, json +from graphify.extract import collect_files, extract +from pathlib import Path +import json + +code_files = [] +detect = json.loads(Path('.graphify_detect.json').read_text()) +for f in detect.get('files', {}).get('code', []): + code_files.extend(collect_files(Path(f)) if Path(f).is_dir() else [Path(f)]) + +if code_files: + result = extract(code_files) + Path('.graphify_ast.json').write_text(json.dumps(result, indent=2)) + print(f'AST: {len(result[\"nodes\"])} nodes, {len(result[\"edges\"])} edges') +else: + Path('.graphify_ast.json').write_text(json.dumps({'nodes':[],'edges':[],'input_tokens':0,'output_tokens':0})) + print('No code files — skipping AST extraction') +" +``` + +#### Part B — Semantic extraction (parallel subagents) + +**MANDATORY: You MUST use the Agent tool here. Reading files yourself one-by-one is forbidden — it is 5-10x slower. If you do not use the Agent tool you are doing this wrong.** + +Before dispatching subagents, print a cost estimate: +- Load `total_words` from `.graphify_detect.json` +- Estimate: ~(total_words / 750) input tokens per file on average, output ~20% of that +- Print: "Semantic extraction: ~N files, estimated ~X input tokens" + +**Step B0 — Check extraction cache first** + +Before dispatching any subagents, check which files already have cached extraction results: + +```bash +python3 -c " +import json +from graphify.cache import check_semantic_cache +from pathlib import Path + +detect = json.loads(Path('.graphify_detect.json').read_text()) +all_files = [f for files in detect['files'].values() for f in files] + +cached_nodes, cached_edges, uncached = check_semantic_cache(all_files) + +if cached_nodes or cached_edges: + Path('.graphify_cached.json').write_text(json.dumps({'nodes': cached_nodes, 'edges': cached_edges})) +Path('.graphify_uncached.txt').write_text('\n'.join(uncached)) +print(f'Cache: {len(all_files)-len(uncached)} files hit, {len(uncached)} files need extraction') +" +``` + +Only dispatch subagents for files listed in `.graphify_uncached.txt`. If all files are cached, skip to Part C directly. + +**Step B1 — Split into chunks** + +Load files from `.graphify_uncached.txt`. Split into chunks of 12-15 files each. Each image gets its own chunk (vision needs separate context). + +**Step B2 — Dispatch ALL subagents in a single message** + +Call the Agent tool multiple times IN THE SAME RESPONSE — one call per chunk. This is the only way they run in parallel. If you make one Agent call, wait, then make another, you are doing it sequentially and defeating the purpose. + +Concrete example for 3 chunks: +``` +[Agent tool call 1: files 1-15] +[Agent tool call 2: files 16-30] +[Agent tool call 3: files 31-45] +``` +All three in one message. Not three separate messages. + +Each subagent receives this exact prompt (substitute FILE_LIST, CHUNK_NUM, TOTAL_CHUNKS, and DEEP_MODE): + +``` +You are a graphify extraction subagent. Read the files listed and extract a knowledge graph fragment. +Output ONLY valid JSON matching the schema below — no explanation, no markdown fences, no preamble. + +Files (chunk CHUNK_NUM of TOTAL_CHUNKS): +FILE_LIST + +Rules: +- EXTRACTED: relationship explicit in source (import, call, citation, "see §3.2") +- INFERRED: reasonable inference (shared data structure, implied dependency) +- AMBIGUOUS: uncertain — flag for review, do not omit + +Code files: focus on semantic edges AST cannot find (call relationships, shared data, arch patterns). + Do not re-extract imports — AST already has those. +Doc/paper files: extract named concepts, entities, citations. +Image files: use vision to understand what the image IS — do not just OCR. + UI screenshot: layout patterns, design decisions, key elements, purpose. + Chart: metric, trend/insight, data source. + Tweet/post: claim as node, author, concepts mentioned. + Diagram: components and connections. + Research figure: what it demonstrates, method, result. + Handwritten/whiteboard: ideas and arrows, mark uncertain readings AMBIGUOUS. + +DEEP_MODE (if --mode deep was given): be aggressive with INFERRED edges — indirect deps, + shared assumptions, latent couplings. Mark uncertain ones AMBIGUOUS instead of omitting. + +If a file has YAML frontmatter (--- ... ---), copy source_url, captured_at, author, + contributor onto every node from that file. + +Output exactly this JSON (no other text): +{"nodes":[{"id":"filestem_entityname","label":"Human Readable Name","file_type":"code|document|paper|image","source_file":"relative/path","source_location":null,"source_url":null,"captured_at":null,"author":null,"contributor":null}],"edges":[{"source":"node_id","target":"node_id","relation":"calls|implements|references|cites|conceptually_related_to|shares_data_with","confidence":"EXTRACTED|INFERRED|AMBIGUOUS","source_file":"relative/path","source_location":null,"weight":1.0}],"input_tokens":0,"output_tokens":0} +``` + +**Step B3 — Collect, cache, and merge** + +Wait for all subagents. For each result: +- If a subagent returned valid JSON with `nodes` and `edges`, include it and save each file's nodes/edges to the cache +- If a subagent failed or returned invalid JSON, print a warning and skip that chunk — do not abort + +If more than half the chunks failed, stop and tell the user. + +Save new results to cache: +```bash +python3 -c " +import json +from graphify.cache import save_semantic_cache +from pathlib import Path + +new = json.loads(Path('.graphify_semantic_new.json').read_text()) if Path('.graphify_semantic_new.json').exists() else {'nodes':[],'edges':[]} +saved = save_semantic_cache(new.get('nodes', []), new.get('edges', [])) +print(f'Cached {saved} files') +" +``` + +Merge cached + new results into `.graphify_semantic.json`: +```bash +python3 -c " +import json +from pathlib import Path + +cached = json.loads(Path('.graphify_cached.json').read_text()) if Path('.graphify_cached.json').exists() else {'nodes':[],'edges':[]} +new = json.loads(Path('.graphify_semantic_new.json').read_text()) if Path('.graphify_semantic_new.json').exists() else {'nodes':[],'edges':[]} + +all_nodes = cached['nodes'] + new.get('nodes', []) +all_edges = cached['edges'] + new.get('edges', []) +seen = set() +deduped = [] +for n in all_nodes: + if n['id'] not in seen: + seen.add(n['id']) + deduped.append(n) + +merged = { + 'nodes': deduped, + 'edges': all_edges, + 'input_tokens': new.get('input_tokens', 0), + 'output_tokens': new.get('output_tokens', 0), +} +Path('.graphify_semantic.json').write_text(json.dumps(merged, indent=2)) +print(f'Extraction complete — {len(deduped)} nodes, {len(all_edges)} edges ({len(cached[\"nodes\"])} from cache, {len(new.get(\"nodes\",[]))} new)') +" +``` +Clean up temp files: `rm -f .graphify_cached.json .graphify_uncached.txt .graphify_semantic_new.json` + +#### Part C — Merge AST + semantic into final extraction + +```bash +python3 -c " +import sys, json +from pathlib import Path + +ast = json.loads(Path('.graphify_ast.json').read_text()) +sem = json.loads(Path('.graphify_semantic.json').read_text()) + +# Merge: AST nodes first, semantic nodes deduplicated by id +seen = {n['id'] for n in ast['nodes']} +merged_nodes = list(ast['nodes']) +for n in sem['nodes']: + if n['id'] not in seen: + merged_nodes.append(n) + seen.add(n['id']) + +merged_edges = ast['edges'] + sem['edges'] +merged = { + 'nodes': merged_nodes, + 'edges': merged_edges, + 'input_tokens': sem.get('input_tokens', 0), + 'output_tokens': sem.get('output_tokens', 0), +} +Path('.graphify_extract.json').write_text(json.dumps(merged, indent=2)) +total = len(merged_nodes) +edges = len(merged_edges) +print(f'Merged: {total} nodes, {edges} edges ({len(ast[\"nodes\"])} AST + {len(sem[\"nodes\"])} semantic)') +" +``` + +### Step 4 — Build graph, cluster, analyze, generate outputs + +```bash +mkdir -p .graphify +python3 -c " +import sys, json +from graphify.build import build_from_json +from graphify.cluster import cluster, score_all +from graphify.analyze import god_nodes, surprising_connections, suggest_questions +from graphify.report import generate +from graphify.export import to_json +from pathlib import Path + +extraction = json.loads(Path('.graphify_extract.json').read_text()) +detection = json.loads(Path('.graphify_detect.json').read_text()) + +G = build_from_json(extraction) +communities = cluster(G) +cohesion = score_all(G, communities) +tokens = {'input': extraction.get('input_tokens', 0), 'output': extraction.get('output_tokens', 0)} +gods = god_nodes(G) +surprises = surprising_connections(G, communities) +labels = {cid: 'Community ' + str(cid) for cid in communities} +# Placeholder questions — regenerated with real labels in Step 5 +questions = suggest_questions(G, communities, labels) + +report = generate(G, communities, cohesion, labels, gods, surprises, detection, tokens, 'INPUT_PATH', suggested_questions=questions) +Path('.graphify/GRAPH_REPORT.md').write_text(report) +to_json(G, communities, '.graphify/graph.json') + +analysis = { + 'communities': {str(k): v for k, v in communities.items()}, + 'cohesion': {str(k): v for k, v in cohesion.items()}, + 'gods': gods, + 'surprises': surprises, + 'questions': questions, +} +Path('.graphify_analysis.json').write_text(json.dumps(analysis, indent=2)) +if G.number_of_nodes() == 0: + print('ERROR: Graph is empty — extraction produced no nodes.') + print('Possible causes: all files were skipped, binary-only corpus, or extraction failed.') + raise SystemExit(1) +print(f'Graph: {G.number_of_nodes()} nodes, {G.number_of_edges()} edges, {len(communities)} communities') +" +``` + +If this step prints `ERROR: Graph is empty`, stop and tell the user what happened — do not proceed to labeling or visualization. + +Replace INPUT_PATH with the actual path. + +### Step 5 — Label communities + +Read `.graphify_analysis.json`. For each community key, look at its node labels and write a 2-5 word plain-language name (e.g. "Attention Mechanism", "Training Pipeline", "Data Loading"). + +Then regenerate the report and save the labels for the visualizer: + +```bash +python3 -c " +import sys, json +from graphify.build import build_from_json +from graphify.cluster import score_all +from graphify.analyze import god_nodes, surprising_connections, suggest_questions +from graphify.report import generate +from pathlib import Path + +extraction = json.loads(Path('.graphify_extract.json').read_text()) +detection = json.loads(Path('.graphify_detect.json').read_text()) +analysis = json.loads(Path('.graphify_analysis.json').read_text()) + +G = build_from_json(extraction) +communities = {int(k): v for k, v in analysis['communities'].items()} +cohesion = {int(k): v for k, v in analysis['cohesion'].items()} +tokens = {'input': extraction.get('input_tokens', 0), 'output': extraction.get('output_tokens', 0)} + +# LABELS — replace these with the names you chose above +labels = LABELS_DICT + +# Regenerate questions with real community labels (labels affect question phrasing) +questions = suggest_questions(G, communities, labels) + +report = generate(G, communities, cohesion, labels, analysis['gods'], analysis['surprises'], detection, tokens, 'INPUT_PATH', suggested_questions=questions) +Path('.graphify/GRAPH_REPORT.md').write_text(report) +Path('.graphify_labels.json').write_text(json.dumps({str(k): v for k, v in labels.items()})) +print('Report updated with community labels') +" +``` + +Replace `LABELS_DICT` with the actual dict you constructed (e.g. `{0: "Attention Mechanism", 1: "Training Pipeline"}`). +Replace INPUT_PATH with the actual path. + +### Step 6 — Generate Obsidian vault (default) + optional HTML + +**Always generate the Obsidian vault** — it is the primary visualization. Skip only if `--no-viz`. + +```bash +python3 -c " +import sys, json +from graphify.build import build_from_json +from graphify.export import to_obsidian, to_canvas +from pathlib import Path + +extraction = json.loads(Path('.graphify_extract.json').read_text()) +analysis = json.loads(Path('.graphify_analysis.json').read_text()) +labels_raw = json.loads(Path('.graphify_labels.json').read_text()) if Path('.graphify_labels.json').exists() else {} + +G = build_from_json(extraction) +communities = {int(k): v for k, v in analysis['communities'].items()} +cohesion = {int(k): v for k, v in analysis['cohesion'].items()} +labels = {int(k): v for k, v in labels_raw.items()} + +n = to_obsidian(G, communities, '.graphify/obsidian', community_labels=labels or None, cohesion=cohesion) +print(f'Obsidian vault: {n} notes in .graphify/obsidian/') + +to_canvas(G, communities, '.graphify/obsidian/graph.canvas', community_labels=labels or None) +print('Canvas: .graphify/obsidian/graph.canvas — open in Obsidian for structured community layout') +print() +print('Open .graphify/obsidian/ as a vault in Obsidian.') +print(' Graph view — nodes colored by community (set automatically)') +print(' graph.canvas — structured layout with communities as groups') +print(' _COMMUNITY_* — overview notes with cohesion scores and dataview queries') +" +``` + +**Only if `--html` flag was passed**, also generate pyvis HTML: + +```bash +python3 -c " +import sys, json +from graphify.build import build_from_json +from graphify.export import generate_html +from pathlib import Path + +extraction = json.loads(Path('.graphify_extract.json').read_text()) +analysis = json.loads(Path('.graphify_analysis.json').read_text()) +labels_raw = json.loads(Path('.graphify_labels.json').read_text()) if Path('.graphify_labels.json').exists() else {} + +G = build_from_json(extraction) +communities = {int(k): v for k, v in analysis['communities'].items()} +labels = {int(k): v for k, v in labels_raw.items()} + +if G.number_of_nodes() > 5000: + print(f'Graph has {G.number_of_nodes()} nodes — too large for pyvis. Use Obsidian vault instead.') +else: + generate_html(G, communities, '.graphify/graph.html', community_labels=labels or None) + print('graph.html written') +" +``` + +### Step 7 — Neo4j export (only if --neo4j or --neo4j-push flag) + +**If `--neo4j`** — generate a Cypher file for manual import: + +```bash +python3 -c " +import sys, json +from graphify.build import build_from_json +from graphify.export import to_cypher +from pathlib import Path + +G = build_from_json(json.loads(Path('.graphify_extract.json').read_text())) +to_cypher(G, '.graphify/cypher.txt') +print('cypher.txt written — import with: cypher-shell < .graphify/cypher.txt') +" +``` + +**If `--neo4j-push `** — push directly to a running Neo4j instance. Ask the user for credentials if not provided: + +```bash +python3 -c " +import sys, json +from graphify.build import build_from_json +from graphify.cluster import cluster +from graphify.export import push_to_neo4j +from pathlib import Path + +extraction = json.loads(Path('.graphify_extract.json').read_text()) +analysis = json.loads(Path('.graphify_analysis.json').read_text()) +G = build_from_json(extraction) +communities = {int(k): v for k, v in analysis['communities'].items()} + +result = push_to_neo4j(G, uri='NEO4J_URI', user='NEO4J_USER', password='NEO4J_PASSWORD', communities=communities) +print(f'Pushed to Neo4j: {result[\"nodes\"]} nodes, {result[\"edges\"]} edges') +" +``` + +Replace `NEO4J_URI`, `NEO4J_USER`, `NEO4J_PASSWORD` with actual values. Default URI is `bolt://localhost:7687`, default user is `neo4j`. Uses MERGE — safe to re-run without creating duplicates. + +### Step 7b — SVG export (only if --svg flag) + +```bash +python3 -c " +import sys, json +from graphify.build import build_from_json +from graphify.export import to_svg +from pathlib import Path + +extraction = json.loads(Path('.graphify_extract.json').read_text()) +analysis = json.loads(Path('.graphify_analysis.json').read_text()) +labels_raw = json.loads(Path('.graphify_labels.json').read_text()) if Path('.graphify_labels.json').exists() else {} + +G = build_from_json(extraction) +communities = {int(k): v for k, v in analysis['communities'].items()} +labels = {int(k): v for k, v in labels_raw.items()} + +to_svg(G, communities, '.graphify/graph.svg', community_labels=labels or None) +print('graph.svg written — embeds in Obsidian, Notion, GitHub READMEs') +" +``` + +### Step 7c — Obsidian export (only if --obsidian flag) + +```bash +python3 -c " +import sys, json +from graphify.build import build_from_json +from graphify.export import to_obsidian +from pathlib import Path + +extraction = json.loads(Path('.graphify_extract.json').read_text()) +analysis = json.loads(Path('.graphify_analysis.json').read_text()) +labels_raw = json.loads(Path('.graphify_labels.json').read_text()) if Path('.graphify_labels.json').exists() else {} + +G = build_from_json(extraction) +communities = {int(k): v for k, v in analysis['communities'].items()} +labels = {int(k): v for k, v in labels_raw.items()} + +n = to_obsidian(G, communities, '.graphify/obsidian', community_labels=labels or None, cohesion=cohesion) +print(f'Obsidian vault written: {n} notes in .graphify/obsidian/') +print('Open .graphify/obsidian/ as a vault in Obsidian to explore the graph.') +" +``` + +### Step 7d — MCP server (only if --mcp flag) + +```bash +python3 -m graphify.serve .graphify/graph.json +``` + +This starts a stdio MCP server that exposes tools: `query_graph`, `get_node`, `get_neighbors`, `get_community`, `god_nodes`, `graph_stats`. Add to Claude Desktop or any MCP-compatible agent orchestrator so other agents can query the graph live. + +To configure in Claude Desktop, add to `claude_desktop_config.json`: +```json +{ + "mcpServers": { + "graphify": { + "command": "python3", + "args": ["-m", "graphify.serve", "/absolute/path/to/.graphify/graph.json"] + } + } +} +``` + +### Step 8 — Save manifest, update cost tracker, clean up, and report + +```bash +python3 -c " +import json +from pathlib import Path +from datetime import datetime, timezone +from graphify.detect import save_manifest + +# Save manifest for --update +detect = json.loads(Path('.graphify_detect.json').read_text()) +save_manifest(detect['files']) + +# Update cumulative cost tracker +extract = json.loads(Path('.graphify_extract.json').read_text()) +input_tok = extract.get('input_tokens', 0) +output_tok = extract.get('output_tokens', 0) + +cost_path = Path('.graphify/cost.json') +if cost_path.exists(): + cost = json.loads(cost_path.read_text()) +else: + cost = {'runs': [], 'total_input_tokens': 0, 'total_output_tokens': 0} + +cost['runs'].append({ + 'date': datetime.now(timezone.utc).isoformat(), + 'input_tokens': input_tok, + 'output_tokens': output_tok, + 'files': detect.get('total_files', 0), +}) +cost['total_input_tokens'] += input_tok +cost['total_output_tokens'] += output_tok +cost_path.write_text(json.dumps(cost, indent=2)) + +print(f'This run: {input_tok:,} input tokens, {output_tok:,} output tokens') +print(f'All time: {cost[\"total_input_tokens\"]:,} input, {cost[\"total_output_tokens\"]:,} output ({len(cost[\"runs\"])} runs)') +" +rm -f .graphify_detect.json .graphify_extract.json .graphify_ast.json .graphify_semantic.json .graphify_analysis.json .graphify_labels.json +rm -f .graphify/.needs_update 2>/dev/null || true +``` + +Tell the user: +``` +Graph complete. Outputs in .graphify/ + + obsidian/ — open this folder as a vault in Obsidian to explore interactively + GRAPH_REPORT.md — full audit report (also readable here in Claude) + graph.json — persistent graph, queryable in future sessions with /graphify query + +To explore: open Obsidian → File → Open Vault → select .graphify/obsidian/ +``` + +Then paste these sections from GRAPH_REPORT.md directly into the chat: +- God Nodes +- Surprising Connections +- Suggested Questions + +Do NOT paste the full report — just those three sections. Keep it concise. + +--- + +## For --update (incremental re-extraction) + +Use when you've added or modified files since the last run. Only re-extracts changed files — saves tokens and time. + +```bash +python3 -c " +import sys, json +from graphify.detect import detect_incremental, save_manifest +from pathlib import Path + +result = detect_incremental(Path('INPUT_PATH')) +new_total = result.get('new_total', 0) +print(json.dumps(result, indent=2)) +if new_total == 0: + print('No files changed since last run. Nothing to update.') + raise SystemExit(0) +print(f'{new_total} new/changed file(s) to re-extract.') +" +``` + +If new files exist, run **Steps 3A–3C** on `result['new_files']` only (not the full corpus). Then: + +```bash +python3 -c " +import sys, json +from graphify.build import build_from_json +from graphify.export import to_json +from networkx.readwrite import json_graph +import networkx as nx +from pathlib import Path + +# Load existing graph +existing_data = json.loads(Path('.graphify/graph.json').read_text()) +G_existing = json_graph.node_link_graph(existing_data, edges='links') + +# Load new extraction +new_extraction = json.loads(Path('.graphify_extract.json').read_text()) +G_new = build_from_json(new_extraction) + +# Merge: new nodes/edges into existing graph +G_existing.update(G_new) +print(f'Merged: {G_existing.number_of_nodes()} nodes, {G_existing.number_of_edges()} edges') + +# Save manifest so next --update knows what changed +from graphify.detect import save_manifest, detect +detect_result = json.loads(Path('.graphify_detect.json').read_text()) +save_manifest(detect_result['files']) +" +``` + +Then run Steps 4–8 on the merged graph as normal. + +After Step 4, show the graph diff: + +```bash +python3 -c " +import json +from graphify.analyze import graph_diff +from graphify.build import build_from_json +from networkx.readwrite import json_graph +import networkx as nx +from pathlib import Path + +# Load old graph (before update) from backup written before merge +old_data = json.loads(Path('.graphify_old.json').read_text()) if Path('.graphify_old.json').exists() else None +new_extract = json.loads(Path('.graphify_extract.json').read_text()) +G_new = build_from_json(new_extract) + +if old_data: + G_old = json_graph.node_link_graph(old_data, edges='links') + diff = graph_diff(G_old, G_new) + print(diff['summary']) + if diff['new_nodes']: + print('New nodes:', ', '.join(n['label'] for n in diff['new_nodes'][:5])) + if diff['new_edges']: + print('New edges:', len(diff['new_edges'])) +" +``` + +Before the merge step, save the old graph: `cp .graphify/graph.json .graphify_old.json` +Clean up after: `rm -f .graphify_old.json` + +--- + +## For --cluster-only + +Skip Steps 1–3. Load the existing graph from `.graphify/graph.json` and re-run clustering: + +```bash +python3 -c " +import sys, json +from graphify.cluster import cluster, score_all +from graphify.analyze import god_nodes, surprising_connections +from graphify.report import generate +from graphify.export import to_json +from networkx.readwrite import json_graph +import networkx as nx +from pathlib import Path + +data = json.loads(Path('.graphify/graph.json').read_text()) +G = json_graph.node_link_graph(data, edges='links') + +detection = {'total_files': 0, 'total_words': 99999, 'needs_graph': True, 'warning': None, + 'files': {'code': [], 'document': [], 'paper': []}} +tokens = {'input': 0, 'output': 0} + +communities = cluster(G) +cohesion = score_all(G, communities) +gods = god_nodes(G) +surprises = surprising_connections(G, communities) +labels = {cid: 'Community ' + str(cid) for cid in communities} + +report = generate(G, communities, cohesion, labels, gods, surprises, detection, tokens, '.') +Path('.graphify/GRAPH_REPORT.md').write_text(report) +to_json(G, communities, '.graphify/graph.json') + +analysis = { + 'communities': {str(k): v for k, v in communities.items()}, + 'cohesion': {str(k): v for k, v in cohesion.items()}, + 'gods': gods, + 'surprises': surprises, +} +Path('.graphify_analysis.json').write_text(json.dumps(analysis, indent=2)) +print(f'Re-clustered: {len(communities)} communities') +" +``` + +Then run Steps 5–8 as normal (label communities, generate viz, clean up, report). + +--- + +## For /graphify query + +Two traversal modes — choose based on the question: + +| Mode | Flag | Best for | +|------|------|----------| +| BFS (default) | _(none)_ | "What is X connected to?" — broad context, nearest neighbors first | +| DFS | `--dfs` | "How does X reach Y?" — trace a specific chain or dependency path | + +Load `.graphify/graph.json`, then: + +1. Find the 1-3 nodes whose label best matches key terms in the question. +2. Run the appropriate traversal from each starting node. +3. Read the subgraph — node labels, edge relations, confidence tags, source locations. +4. Answer using **only** what the graph contains. Quote `source_location` when citing a specific fact. +5. If the graph lacks enough information, say so — do not hallucinate edges. + +```bash +python3 -c " +import sys, json +from networkx.readwrite import json_graph +import networkx as nx +from pathlib import Path + +data = json.loads(Path('.graphify/graph.json').read_text()) +G = json_graph.node_link_graph(data, edges='links') + +question = 'QUESTION' +mode = 'MODE' # 'bfs' or 'dfs' +terms = [t.lower() for t in question.split() if len(t) > 3] + +# Find best-matching start nodes +scored = [] +for nid, ndata in G.nodes(data=True): + label = ndata.get('label', '').lower() + score = sum(1 for t in terms if t in label) + if score > 0: + scored.append((score, nid)) +scored.sort(reverse=True) +start_nodes = [nid for _, nid in scored[:3]] + +if not start_nodes: + print('No matching nodes found for query terms:', terms) + sys.exit(0) + +subgraph_nodes = set() +subgraph_edges = [] + +if mode == 'dfs': + # DFS: follow one path as deep as possible before backtracking. + # Depth-limited to 6 to avoid traversing the whole graph. + visited = set() + stack = [(n, 0) for n in reversed(start_nodes)] + while stack: + node, depth = stack.pop() + if node in visited or depth > 6: + continue + visited.add(node) + subgraph_nodes.add(node) + for neighbor in G.neighbors(node): + if neighbor not in visited: + stack.append((neighbor, depth + 1)) + subgraph_edges.append((node, neighbor)) +else: + # BFS: explore all neighbors layer by layer up to depth 3. + frontier = set(start_nodes) + subgraph_nodes = set(start_nodes) + for _ in range(3): + next_frontier = set() + for n in frontier: + for neighbor in G.neighbors(n): + if neighbor not in subgraph_nodes: + next_frontier.add(neighbor) + subgraph_edges.append((n, neighbor)) + subgraph_nodes.update(next_frontier) + frontier = next_frontier + +# Token-budget aware output: rank by relevance, cut at budget (~4 chars/token) +token_budget = BUDGET # default 2000 +char_budget = token_budget * 4 + +# Score each node by term overlap for ranked output +def relevance(nid): + label = G.nodes[nid].get('label', '').lower() + return sum(1 for t in terms if t in label) + +ranked_nodes = sorted(subgraph_nodes, key=relevance, reverse=True) + +lines = [f'Traversal: {mode.upper()} | Start: {[G.nodes[n].get(\"label\",n) for n in start_nodes]} | {len(subgraph_nodes)} nodes'] +for nid in ranked_nodes: + d = G.nodes[nid] + lines.append(f' NODE {d.get(\"label\", nid)} [src={d.get(\"source_file\",\"\")} loc={d.get(\"source_location\",\"\")}]') +for u, v in subgraph_edges: + if u in subgraph_nodes and v in subgraph_nodes: + d = G.edges[u, v] + lines.append(f' EDGE {G.nodes[u].get(\"label\",u)} --{d.get(\"relation\",\"\")} [{d.get(\"confidence\",\"\")}]--> {G.nodes[v].get(\"label\",v)}') + +output = '\n'.join(lines) +if len(output) > char_budget: + output = output[:char_budget] + f'\n... (truncated at ~{token_budget} token budget — use --budget N for more)' +print(output) +" +``` + +Replace `QUESTION` with the user's actual question, `MODE` with `bfs` or `dfs`, and `BUDGET` with the token budget (default `2000`, or whatever `--budget N` specifies). Then answer based on the subgraph output above. + +After writing the answer, save it back into the graph so it improves future queries: + +```bash +python3 -c " +from graphify.ingest import save_query_result +from pathlib import Path +save_query_result( + question='QUESTION', + answer='ANSWER', + memory_dir=Path('.graphify/memory'), + query_type='query', + source_nodes=SOURCE_NODES, # list of node labels cited, or [] +) +print('Query result saved to .graphify/memory/') +" +``` + +Replace `QUESTION` with the question, `ANSWER` with your full answer text, `SOURCE_NODES` with the list of node labels you cited. This closes the feedback loop: the next `--update` will extract this Q&A as a node in the graph. + +--- + +## For /graphify path + +Find the shortest path between two named concepts in the graph. + +```bash +python3 -c " +import json, sys +import networkx as nx +from networkx.readwrite import json_graph +from pathlib import Path + +data = json.loads(Path('.graphify/graph.json').read_text()) +G = json_graph.node_link_graph(data, edges='links') + +a_term = 'NODE_A' +b_term = 'NODE_B' + +def find_node(term): + term = term.lower() + scored = sorted( + [(sum(1 for w in term.split() if w in G.nodes[n].get('label','').lower()), n) + for n in G.nodes()], + reverse=True + ) + return scored[0][1] if scored and scored[0][0] > 0 else None + +src = find_node(a_term) +tgt = find_node(b_term) + +if not src or not tgt: + print(f'Could not find nodes matching: {a_term!r} or {b_term!r}') + sys.exit(0) + +try: + path = nx.shortest_path(G, src, tgt) + print(f'Shortest path ({len(path)-1} hops):') + for i, nid in enumerate(path): + label = G.nodes[nid].get('label', nid) + if i < len(path) - 1: + edge = G.edges[nid, path[i+1]] + rel = edge.get('relation', '') + conf = edge.get('confidence', '') + print(f' {label} --{rel}--> [{conf}]') + else: + print(f' {label}') +except nx.NetworkXNoPath: + print(f'No path found between {a_term!r} and {b_term!r}') +except nx.NodeNotFound as e: + print(f'Node not found: {e}') +" +``` + +Replace `NODE_A` and `NODE_B` with the actual concept names from the user. Then explain the path in plain language — what each hop means, why it's significant. + +After writing the explanation, save it back: + +```bash +python3 -c " +from graphify.ingest import save_query_result +from pathlib import Path +save_query_result( + question='Path from NODE_A to NODE_B', + answer='ANSWER', + memory_dir=Path('.graphify/memory'), + query_type='path_query', + source_nodes=PATH_NODES, # list of node labels on the path +) +print('Path result saved to .graphify/memory/') +" +``` + +--- + +## For /graphify explain + +Give a plain-language explanation of a single node — everything connected to it. + +```bash +python3 -c " +import json, sys +import networkx as nx +from networkx.readwrite import json_graph +from pathlib import Path + +data = json.loads(Path('.graphify/graph.json').read_text()) +G = json_graph.node_link_graph(data, edges='links') + +term = 'NODE_NAME' +term_lower = term.lower() + +# Find best matching node +scored = sorted( + [(sum(1 for w in term_lower.split() if w in G.nodes[n].get('label','').lower()), n) + for n in G.nodes()], + reverse=True +) +if not scored or scored[0][0] == 0: + print(f'No node matching {term!r}') + sys.exit(0) + +nid = scored[0][1] +data_n = G.nodes[nid] +print(f'NODE: {data_n.get(\"label\", nid)}') +print(f' source: {data_n.get(\"source_file\",\"unknown\")}') +print(f' type: {data_n.get(\"file_type\",\"unknown\")}') +print(f' degree: {G.degree(nid)}') +print() +print('CONNECTIONS:') +for neighbor in G.neighbors(nid): + edge = G.edges[nid, neighbor] + nlabel = G.nodes[neighbor].get('label', neighbor) + rel = edge.get('relation', '') + conf = edge.get('confidence', '') + src_file = G.nodes[neighbor].get('source_file', '') + print(f' --{rel}--> {nlabel} [{conf}] ({src_file})') +" +``` + +Replace `NODE_NAME` with the concept the user asked about. Then write a 3-5 sentence explanation of what this node is, what it connects to, and why those connections are significant. Use the source locations as citations. + +After writing the explanation, save it back: + +```bash +python3 -c " +from graphify.ingest import save_query_result +from pathlib import Path +save_query_result( + question='Explain NODE_NAME', + answer='ANSWER', + memory_dir=Path('.graphify/memory'), + query_type='explain', + source_nodes=['NODE_NAME'], +) +print('Explanation saved to .graphify/memory/') +" +``` + +--- + +## For /graphify add + +Fetch a URL and add it to the corpus, then update the graph. + +```bash +python3 -c " +import sys +from graphify.ingest import ingest +from pathlib import Path + +out = ingest('URL', Path('./raw'), author='AUTHOR', contributor='CONTRIBUTOR') +print(f'Saved to {out}') +" +``` + +Replace `URL` with the actual URL, `AUTHOR` with the user's name if provided, `CONTRIBUTOR` likewise. After saving, automatically run the `--update` pipeline on `./raw` to merge the new file into the existing graph. + +Supported URL types (auto-detected): +- Twitter/X → fetched via oEmbed, saved as `.md` with tweet text and author +- arXiv → abstract + metadata saved as `.md` +- PDF → downloaded as `.pdf` +- Images (.png/.jpg/.webp) → downloaded, Claude vision extracts on next run +- Any webpage → converted to markdown via html2text + +--- + +## For --watch + +Start a background watcher that monitors a folder and auto-reruns `--update` when files change. + +```bash +python3 -m graphify.watch INPUT_PATH --debounce 3 +``` + +Replace INPUT_PATH with the folder to watch. Every time a supported file is added or modified, graphify waits `debounce` seconds (default 3) after the last change, then runs the `--update` pipeline automatically. Press Ctrl+C to stop. + +For the personal inspo use case: leave this running in a terminal. Drop tweets, screenshots, papers, and notes into the folder throughout the day — the graph updates itself. + +--- + +## Honesty Rules + +- Never invent an edge. If unsure, use AMBIGUOUS. +- Never skip the corpus check warning. +- Always show token cost in the report. +- Never hide cohesion scores behind symbols — show the raw number. +- Never run pyvis on a graph with more than 5,000 nodes without warning the user. diff --git a/graphify/validate.py b/graphify/validate.py index b6f4b5b..4029e66 100644 --- a/graphify/validate.py +++ b/graphify/validate.py @@ -1,7 +1,7 @@ # validate extraction JSON against the graphify schema before graph assembly from __future__ import annotations -VALID_FILE_TYPES = {"code", "document", "paper"} +VALID_FILE_TYPES = {"code", "document", "paper", "image"} VALID_CONFIDENCES = {"EXTRACTED", "INFERRED", "AMBIGUOUS"} REQUIRED_NODE_FIELDS = {"id", "label", "file_type", "source_file"} REQUIRED_EDGE_FIELDS = {"source", "target", "relation", "confidence", "source_file"} diff --git a/pyproject.toml b/pyproject.toml index bae4849..e9ba9e5 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -5,6 +5,9 @@ build-backend = "setuptools.build_meta" [project] name = "graphify" version = "0.1.1" +description = "Turn any codebase, docs, or images into a queryable knowledge graph" +readme = "README.md" +license = { text = "MIT" } requires-python = ">=3.10" dependencies = [ "networkx", @@ -12,8 +15,25 @@ dependencies = [ "pyvis", "tree-sitter", "tree-sitter-python", + "tree-sitter-javascript", + "tree-sitter-typescript", + "tree-sitter-go", + "tree-sitter-rust", ] +[project.optional-dependencies] +mcp = ["mcp"] +neo4j = ["neo4j"] +pdf = ["pypdf", "html2text"] +watch = ["watchdog"] +all = ["mcp", "neo4j", "pypdf", "html2text", "watchdog"] + +[project.scripts] +graphify = "graphify.__main__:main" + [tool.setuptools.packages.find] where = ["."] include = ["graphify*"] + +[tool.setuptools.package-data] +graphify = ["skill.md"] diff --git a/skills/graphify/skill.md b/skills/graphify/skill.md index 25e259b..174dc5a 100644 --- a/skills/graphify/skill.md +++ b/skills/graphify/skill.md @@ -54,36 +54,41 @@ If no path was given, use `.` (current directory). Do not ask the user for a pat Follow these steps in order. Do not skip steps. -### Step 1 — Install dependencies (skip if already installed) +### Step 1 — Ensure graphify is installed ```bash -python3 -c "import graphify, networkx, graspologic, pyvis, tree_sitter" 2>/dev/null || { - pip install networkx graspologic pyvis tree-sitter tree-sitter-python -q --break-system-packages 2>&1 | tail -3 - pip install git+https://github.com/safishamsi/graphify.git -q --break-system-packages 2>&1 | tail -3 -} +python3 -c "import graphify" 2>/dev/null || pip install graphify -q --break-system-packages 2>&1 | tail -3 ``` -If all imports succeed, print nothing and move straight to Step 2. +If the import succeeds, print nothing and move straight to Step 2. ### Step 2 — Detect files ```bash python3 -c " -import sys, json +import json from graphify.detect import detect from pathlib import Path result = detect(Path('INPUT_PATH')) -print(json.dumps(result, indent=2)) +print(json.dumps(result)) " > .graphify_detect.json -cat .graphify_detect.json ``` -Replace INPUT_PATH with the actual path the user provided. +Replace INPUT_PATH with the actual path the user provided. Do NOT cat or print the JSON — read it silently and present a clean summary instead: -After detection: -- If `skipped_sensitive` is non-empty, tell the user which files were skipped — do not ask, just inform. -- If `total_files` is 0, stop: "No supported files found in [path]. Supported: .py .ts .js .go .rs .java .cpp .rb .md .txt .rst .pdf" -- If `total_words` > 2,000,000, tell the user the word count and suggest a subfolder, but **do not block** — proceed unless they say stop. +``` +Corpus: X files · ~Y words + code: N files (.py .ts .go ...) + docs: N files (.md .txt ...) + papers: N files (.pdf ...) + images: N files +``` + +Then act on it: +- If `total_files` is 0: stop with "No supported files found in [path]." +- If `skipped_sensitive` is non-empty: mention file count skipped, not the file names. +- If `total_words` > 2,000,000 OR `total_files` > 200: show the warning and the top 5 subdirectories by file count, then ask which subfolder to run on. Wait for the user's answer before proceeding. +- Otherwise: proceed directly to Step 3 — no need to ask anything. ### Step 3 — Extract entities and relationships @@ -131,35 +136,18 @@ Before dispatching any subagents, check which files already have cached extracti ```bash python3 -c " import json -from graphify.cache import load_cached +from graphify.cache import check_semantic_cache from pathlib import Path detect = json.loads(Path('.graphify_detect.json').read_text()) -all_files = [] -for ftype, files in detect['files'].items(): - all_files.extend(files) +all_files = [f for files in detect['files'].values() for f in files] -cached = [] -uncached = [] -for f in all_files: - result = load_cached(Path(f)) - if result is not None: - cached.append((f, result)) - else: - uncached.append(f) - -# Write cached results directly to a partial semantic file -if cached: - nodes, edges = [], [] - for f, r in cached: - nodes.extend(r.get('nodes', [])) - edges.extend(r.get('edges', [])) - import json - from pathlib import Path - Path('.graphify_cached.json').write_text(json.dumps({'nodes': nodes, 'edges': edges})) +cached_nodes, cached_edges, uncached = check_semantic_cache(all_files) +if cached_nodes or cached_edges: + Path('.graphify_cached.json').write_text(json.dumps({'nodes': cached_nodes, 'edges': cached_edges})) Path('.graphify_uncached.txt').write_text('\n'.join(uncached)) -print(f'Cache: {len(cached)} files hit, {len(uncached)} files need extraction') +print(f'Cache: {len(all_files)-len(uncached)} files hit, {len(uncached)} files need extraction') " ``` @@ -227,28 +215,13 @@ If more than half the chunks failed, stop and tell the user. Save new results to cache: ```bash python3 -c " -from graphify.cache import save_cached -from pathlib import Path import json +from graphify.cache import save_semantic_cache +from pathlib import Path -# Load the new semantic results (written by subagents as .graphify_semantic_new.json) -new = json.loads(Path('.graphify_semantic_new.json').read_text()) - -# Group nodes/edges back by source_file and cache each file's results -from collections import defaultdict -by_file = defaultdict(lambda: {'nodes': [], 'edges': []}) -for n in new.get('nodes', []): - if n.get('source_file'): - by_file[n['source_file']]['nodes'].append(n) -for e in new.get('edges', []): - if e.get('source_file'): - by_file[e['source_file']]['edges'].append(e) - -for fpath, result in by_file.items(): - p = Path(fpath) - if p.exists(): - save_cached(p, result) -print(f'Cached {len(by_file)} files') +new = json.loads(Path('.graphify_semantic_new.json').read_text()) if Path('.graphify_semantic_new.json').exists() else {'nodes':[],'edges':[]} +saved = save_semantic_cache(new.get('nodes', []), new.get('edges', [])) +print(f'Cached {saved} files') " ``` diff --git a/tests/test_serve.py b/tests/test_serve.py new file mode 100644 index 0000000..363e1c6 --- /dev/null +++ b/tests/test_serve.py @@ -0,0 +1,151 @@ +"""Tests for serve.py — MCP graph query helpers (no mcp package required).""" +import json +import pytest +import networkx as nx +from networkx.readwrite import json_graph + +from graphify.serve import ( + _communities_from_graph, + _score_nodes, + _bfs, + _dfs, + _subgraph_to_text, + _load_graph, +) + + +def _make_graph() -> nx.Graph: + G = nx.Graph() + G.add_node("n1", label="extract", source_file="extract.py", source_location="L10", community=0) + G.add_node("n2", label="cluster", source_file="cluster.py", source_location="L5", community=0) + G.add_node("n3", label="build", source_file="build.py", source_location="L1", community=1) + G.add_node("n4", label="report", source_file="report.py", source_location="L1", community=1) + G.add_node("n5", label="isolated", source_file="other.py", source_location="L1", community=2) + G.add_edge("n1", "n2", relation="calls", confidence="INFERRED") + G.add_edge("n2", "n3", relation="imports", confidence="EXTRACTED") + G.add_edge("n3", "n4", relation="uses", confidence="EXTRACTED") + return G + + +# --- _communities_from_graph --- + +def test_communities_from_graph_basic(): + G = _make_graph() + communities = _communities_from_graph(G) + assert 0 in communities + assert 1 in communities + assert "n1" in communities[0] + assert "n2" in communities[0] + assert "n3" in communities[1] + +def test_communities_from_graph_no_community_attr(): + G = nx.Graph() + G.add_node("a", label="foo") # no community attr + communities = _communities_from_graph(G) + assert communities == {} + +def test_communities_from_graph_isolated(): + G = _make_graph() + communities = _communities_from_graph(G) + assert 2 in communities + assert "n5" in communities[2] + + +# --- _score_nodes --- + +def test_score_nodes_exact_label_match(): + G = _make_graph() + scored = _score_nodes(G, ["extract"]) + nids = [nid for _, nid in scored] + assert "n1" in nids + assert scored[0][1] == "n1" # highest score first + +def test_score_nodes_no_match(): + G = _make_graph() + scored = _score_nodes(G, ["xyzzy"]) + assert scored == [] + +def test_score_nodes_source_file_partial(): + G = _make_graph() + # "cluster.py" contains "cluster" — should score 0.5 for source match + scored = _score_nodes(G, ["cluster"]) + nids = [nid for _, nid in scored] + assert "n2" in nids + + +# --- _bfs --- + +def test_bfs_depth_1(): + G = _make_graph() + visited, edges = _bfs(G, ["n1"], depth=1) + assert "n1" in visited + assert "n2" in visited # direct neighbor + assert "n3" not in visited # 2 hops away + +def test_bfs_depth_2(): + G = _make_graph() + visited, edges = _bfs(G, ["n1"], depth=2) + assert "n3" in visited # n1 -> n2 -> n3 + +def test_bfs_disconnected(): + G = _make_graph() + visited, edges = _bfs(G, ["n5"], depth=3) + assert visited == {"n5"} # isolated node + +def test_bfs_returns_edges(): + G = _make_graph() + visited, edges = _bfs(G, ["n1"], depth=1) + assert len(edges) >= 1 + assert any(u == "n1" or v == "n1" for u, v in edges) + + +# --- _dfs --- + +def test_dfs_depth_1(): + G = _make_graph() + visited, edges = _dfs(G, ["n1"], depth=1) + assert "n1" in visited + assert "n2" in visited + assert "n3" not in visited + +def test_dfs_full_chain(): + G = _make_graph() + visited, edges = _dfs(G, ["n1"], depth=5) + assert {"n1", "n2", "n3", "n4"}.issubset(visited) + + +# --- _subgraph_to_text --- + +def test_subgraph_to_text_contains_labels(): + G = _make_graph() + text = _subgraph_to_text(G, {"n1", "n2"}, [("n1", "n2")]) + assert "extract" in text + assert "cluster" in text + +def test_subgraph_to_text_truncates(): + G = _make_graph() + # Very small budget forces truncation + text = _subgraph_to_text(G, {"n1", "n2", "n3", "n4"}, [("n1", "n2")], token_budget=1) + assert "truncated" in text + +def test_subgraph_to_text_edge_included(): + G = _make_graph() + text = _subgraph_to_text(G, {"n1", "n2"}, [("n1", "n2")]) + assert "EDGE" in text + assert "calls" in text + + +# --- _load_graph --- + +def test_load_graph_roundtrip(tmp_path): + G = _make_graph() + data = json_graph.node_link_data(G, edges="links") + p = tmp_path / "graph.json" + p.write_text(json.dumps(data)) + G2 = _load_graph(str(p)) + assert G2.number_of_nodes() == G.number_of_nodes() + assert G2.number_of_edges() == G.number_of_edges() + +def test_load_graph_missing_file(tmp_path): + with pytest.raises(Exception): + _load_graph(str(tmp_path / "nonexistent.json")) diff --git a/tests/test_watch.py b/tests/test_watch.py new file mode 100644 index 0000000..5492c69 --- /dev/null +++ b/tests/test_watch.py @@ -0,0 +1,68 @@ +"""Tests for watch.py — file watcher helpers (no watchdog required).""" +import time +from pathlib import Path +import pytest + +from graphify.watch import _run_update, _WATCHED_EXTENSIONS + + +# --- _run_update --- + +def test_run_update_creates_flag(tmp_path): + _run_update(tmp_path) + flag = tmp_path / ".graphify" / "needs_update" + assert flag.exists() + assert flag.read_text() == "1" + +def test_run_update_creates_flag_dir(tmp_path): + # .graphify dir does not exist yet + assert not (tmp_path / ".graphify").exists() + _run_update(tmp_path) + assert (tmp_path / ".graphify").is_dir() + +def test_run_update_idempotent(tmp_path): + _run_update(tmp_path) + _run_update(tmp_path) + flag = tmp_path / ".graphify" / "needs_update" + assert flag.read_text() == "1" + + +# --- _WATCHED_EXTENSIONS --- + +def test_watched_extensions_includes_code(): + assert ".py" in _WATCHED_EXTENSIONS + assert ".ts" in _WATCHED_EXTENSIONS + assert ".go" in _WATCHED_EXTENSIONS + assert ".rs" in _WATCHED_EXTENSIONS + +def test_watched_extensions_includes_docs(): + assert ".md" in _WATCHED_EXTENSIONS + assert ".txt" in _WATCHED_EXTENSIONS + assert ".pdf" in _WATCHED_EXTENSIONS + +def test_watched_extensions_includes_images(): + assert ".png" in _WATCHED_EXTENSIONS + assert ".jpg" in _WATCHED_EXTENSIONS + +def test_watched_extensions_excludes_noise(): + assert ".json" not in _WATCHED_EXTENSIONS + assert ".pyc" not in _WATCHED_EXTENSIONS + assert ".log" not in _WATCHED_EXTENSIONS + + +# --- watch() import error without watchdog --- + +def test_watch_raises_without_watchdog(tmp_path, monkeypatch): + import builtins + real_import = builtins.__import__ + + def mock_import(name, *args, **kwargs): + if name == "watchdog.observers" or name == "watchdog.events": + raise ImportError("mocked missing watchdog") + return real_import(name, *args, **kwargs) + + monkeypatch.setattr(builtins, "__import__", mock_import) + + from graphify.watch import watch + with pytest.raises(ImportError, match="watchdog not installed"): + watch(tmp_path)