From e8b8e2039dc7aa45ea028e9ad699858323f8817f Mon Sep 17 00:00:00 2001 From: Safi Date: Sun, 5 Apr 2026 23:00:24 +0100 Subject: [PATCH] v2: fast --update for code-only changes, parallel AST+semantic, graphify claude install --- graphify/__main__.py | 76 +++++++++++++++++++++++++++++++ graphify/skill.md | 47 +++++++++++++++++-- pyproject.toml | 2 +- skills/graphify/skill.md | 47 +++++++++++++++++-- tests/test_claude_md.py | 97 ++++++++++++++++++++++++++++++++++++++++ 5 files changed, 262 insertions(+), 7 deletions(-) create mode 100644 tests/test_claude_md.py diff --git a/graphify/__main__.py b/graphify/__main__.py index d2fa063..8d9b856 100644 --- a/graphify/__main__.py +++ b/graphify/__main__.py @@ -1,6 +1,7 @@ """graphify CLI - `graphify install` sets up the Claude Code skill.""" from __future__ import annotations import json +import re import shutil import sys from pathlib import Path @@ -52,6 +53,70 @@ def install() -> None: print() +_CLAUDE_MD_SECTION = """\ +## graphify + +This project has a graphify knowledge graph at graphify-out/. + +Rules: +- Before answering architecture or codebase questions, read graphify-out/GRAPH_REPORT.md for god nodes and community structure +- If graphify-out/wiki/index.md exists, navigate it instead of reading raw files +- After modifying code files in this session, run `python3 -c "from graphify.watch import _rebuild_code; from pathlib import Path; _rebuild_code(Path('.'))"` to keep the graph current +""" + +_CLAUDE_MD_MARKER = "## graphify" + + +def claude_install(project_dir: Path | None = None) -> None: + """Write the graphify section to the local CLAUDE.md.""" + target = (project_dir or Path(".")) / "CLAUDE.md" + + if target.exists(): + content = target.read_text() + if _CLAUDE_MD_MARKER in content: + print("graphify already configured in CLAUDE.md") + return + new_content = content.rstrip() + "\n\n" + _CLAUDE_MD_SECTION + else: + new_content = _CLAUDE_MD_SECTION + + target.write_text(new_content) + print(f"graphify section written to {target.resolve()}") + print() + print("Claude Code will now check the knowledge graph before answering") + print("codebase questions and rebuild it after code changes.") + + +def claude_uninstall(project_dir: Path | None = None) -> None: + """Remove the graphify section from the local CLAUDE.md.""" + target = (project_dir or Path(".")) / "CLAUDE.md" + + if not target.exists(): + print("No CLAUDE.md found in current directory - nothing to do") + return + + content = target.read_text() + if _CLAUDE_MD_MARKER not in content: + print("graphify section not found in CLAUDE.md - nothing to do") + return + + # Remove the ## graphify section: from the marker to the next ## heading or EOF + cleaned = re.sub( + r"\n*## graphify\n.*?(?=\n## |\Z)", + "", + content, + flags=re.DOTALL, + ).rstrip() + if cleaned: + target.write_text(cleaned + "\n") + else: + target.unlink() + print(f"CLAUDE.md was empty after removal - deleted {target.resolve()}") + return + + print(f"graphify section removed from {target.resolve()}") + + def main() -> None: if len(sys.argv) < 2 or sys.argv[1] in ("-h", "--help"): print("Usage: graphify ") @@ -62,12 +127,23 @@ def main() -> None: print(" hook install install post-commit git hook (auto-rebuilds graph on commit)") print(" hook uninstall remove post-commit git hook") print(" hook status check if hook is installed") + print(" claude install write graphify section to local CLAUDE.md") + print(" claude uninstall remove graphify section from local CLAUDE.md") print() return cmd = sys.argv[1] if cmd == "install": install() + elif cmd == "claude": + subcmd = sys.argv[2] if len(sys.argv) > 2 else "" + if subcmd == "install": + claude_install() + elif subcmd == "uninstall": + claude_uninstall() + else: + print("Usage: graphify claude [install|uninstall]", file=sys.stderr) + sys.exit(1) elif cmd == "hook": from graphify.hooks import install as hook_install, uninstall as hook_uninstall, status as hook_status subcmd = sys.argv[2] if len(sys.argv) > 2 else "" diff --git a/graphify/skill.md b/graphify/skill.md index 1b67795..a2aed2d 100644 --- a/graphify/skill.md +++ b/graphify/skill.md @@ -95,11 +95,15 @@ Then act on it: **Before starting:** note whether `--mode deep` was given. You must pass `DEEP_MODE=true` to every subagent in Step B2 if it was. Track this from the original invocation - do not lose it. -This step has two parts: **structural extraction** (deterministic, free) then **semantic extraction** (Claude, costs tokens). +This step has two parts: **structural extraction** (deterministic, free) and **semantic extraction** (Claude, costs tokens). + +**Run Part A (AST) and Part B (semantic) in parallel. Dispatch all semantic subagents AND start AST extraction in the same message. Both can run simultaneously since they operate on different file types. Merge results in Part C as before.** + +Note: Parallelizing AST + semantic saves 5-15s on large corpora. AST is deterministic and fast; start it while subagents are processing docs/papers. #### Part A - Structural extraction for code files -For any code files detected, run AST extraction first: +For any code files detected, run AST extraction in parallel with Part B subagents: ```bash python3 -c " @@ -653,6 +657,7 @@ from pathlib import Path result = detect_incremental(Path('INPUT_PATH')) new_total = result.get('new_total', 0) print(json.dumps(result, indent=2)) +Path('.graphify_incremental.json').write_text(json.dumps(result)) if new_total == 0: print('No files changed since last run. Nothing to update.') raise SystemExit(0) @@ -660,7 +665,27 @@ print(f'{new_total} new/changed file(s) to re-extract.') " ``` -If new files exist, run **Steps 3A–3C** on `result['new_files']` only (not the full corpus). Then: +If new files exist, first check whether all changed files are code files: + +```bash +python3 -c " +import json +from pathlib import Path + +result = json.loads(open('.graphify_incremental.json').read()) if Path('.graphify_incremental.json').exists() else {} +code_exts = {'.py','.ts','.js','.go','.rs','.java','.cpp','.c','.rb','.swift','.kt','.cs','.scala','.php','.cc','.cxx','.hpp','.h','.kts'} +new_files = result.get('new_files', {}) +all_changed = [f for files in new_files.values() for f in files] +code_only = all(Path(f).suffix.lower() in code_exts for f in all_changed) +print('code_only:', code_only) +" +``` + +If `code_only` is True: print `[graphify update] Code-only changes detected - skipping semantic extraction (no LLM needed)`, run only Step 3A (AST) on the changed files, skip Step 3B entirely (no subagents), then go straight to merge and Steps 4–8. + +If `code_only` is False (any changed file is a doc/paper/image): run the full Steps 3A–3C pipeline as normal. + +Then: ```bash python3 -c " @@ -1135,6 +1160,22 @@ If a post-commit hook already exists, graphify appends to it rather than replaci --- +## For native CLAUDE.md integration + +Run once per project to make graphify always-on in Claude Code sessions: + +```bash +graphify claude install +``` + +This writes a `## graphify` section to the local `CLAUDE.md` that instructs Claude to check the graph before answering codebase questions and rebuild it after code changes. No manual `/graphify` needed in future sessions. + +```bash +graphify claude uninstall # remove the section +``` + +--- + ## Honesty Rules - Never invent an edge. If unsure, use AMBIGUOUS. diff --git a/pyproject.toml b/pyproject.toml index ea9fa16..6cbcfe6 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta" [project] name = "graphifyy" -version = "0.1.10" +version = "0.1.11" description = "Claude Code skill - turn any folder of code, docs, papers, images, or tweets into a queryable knowledge graph" readme = "README.md" license = { text = "MIT" } diff --git a/skills/graphify/skill.md b/skills/graphify/skill.md index 1b67795..a2aed2d 100644 --- a/skills/graphify/skill.md +++ b/skills/graphify/skill.md @@ -95,11 +95,15 @@ Then act on it: **Before starting:** note whether `--mode deep` was given. You must pass `DEEP_MODE=true` to every subagent in Step B2 if it was. Track this from the original invocation - do not lose it. -This step has two parts: **structural extraction** (deterministic, free) then **semantic extraction** (Claude, costs tokens). +This step has two parts: **structural extraction** (deterministic, free) and **semantic extraction** (Claude, costs tokens). + +**Run Part A (AST) and Part B (semantic) in parallel. Dispatch all semantic subagents AND start AST extraction in the same message. Both can run simultaneously since they operate on different file types. Merge results in Part C as before.** + +Note: Parallelizing AST + semantic saves 5-15s on large corpora. AST is deterministic and fast; start it while subagents are processing docs/papers. #### Part A - Structural extraction for code files -For any code files detected, run AST extraction first: +For any code files detected, run AST extraction in parallel with Part B subagents: ```bash python3 -c " @@ -653,6 +657,7 @@ from pathlib import Path result = detect_incremental(Path('INPUT_PATH')) new_total = result.get('new_total', 0) print(json.dumps(result, indent=2)) +Path('.graphify_incremental.json').write_text(json.dumps(result)) if new_total == 0: print('No files changed since last run. Nothing to update.') raise SystemExit(0) @@ -660,7 +665,27 @@ print(f'{new_total} new/changed file(s) to re-extract.') " ``` -If new files exist, run **Steps 3A–3C** on `result['new_files']` only (not the full corpus). Then: +If new files exist, first check whether all changed files are code files: + +```bash +python3 -c " +import json +from pathlib import Path + +result = json.loads(open('.graphify_incremental.json').read()) if Path('.graphify_incremental.json').exists() else {} +code_exts = {'.py','.ts','.js','.go','.rs','.java','.cpp','.c','.rb','.swift','.kt','.cs','.scala','.php','.cc','.cxx','.hpp','.h','.kts'} +new_files = result.get('new_files', {}) +all_changed = [f for files in new_files.values() for f in files] +code_only = all(Path(f).suffix.lower() in code_exts for f in all_changed) +print('code_only:', code_only) +" +``` + +If `code_only` is True: print `[graphify update] Code-only changes detected - skipping semantic extraction (no LLM needed)`, run only Step 3A (AST) on the changed files, skip Step 3B entirely (no subagents), then go straight to merge and Steps 4–8. + +If `code_only` is False (any changed file is a doc/paper/image): run the full Steps 3A–3C pipeline as normal. + +Then: ```bash python3 -c " @@ -1135,6 +1160,22 @@ If a post-commit hook already exists, graphify appends to it rather than replaci --- +## For native CLAUDE.md integration + +Run once per project to make graphify always-on in Claude Code sessions: + +```bash +graphify claude install +``` + +This writes a `## graphify` section to the local `CLAUDE.md` that instructs Claude to check the graph before answering codebase questions and rebuild it after code changes. No manual `/graphify` needed in future sessions. + +```bash +graphify claude uninstall # remove the section +``` + +--- + ## Honesty Rules - Never invent an edge. If unsure, use AMBIGUOUS. diff --git a/tests/test_claude_md.py b/tests/test_claude_md.py new file mode 100644 index 0000000..8ec6d3d --- /dev/null +++ b/tests/test_claude_md.py @@ -0,0 +1,97 @@ +"""Tests for graphify claude install / uninstall commands.""" +from pathlib import Path +import pytest +from graphify.__main__ import claude_install, claude_uninstall, _CLAUDE_MD_MARKER, _CLAUDE_MD_SECTION + + +# --------------------------------------------------------------------------- +# install +# --------------------------------------------------------------------------- + +def test_install_creates_claude_md(tmp_path): + """Creates CLAUDE.md when none exists.""" + claude_install(tmp_path) + target = tmp_path / "CLAUDE.md" + assert target.exists() + assert _CLAUDE_MD_MARKER in target.read_text() + + +def test_install_contains_expected_rules(tmp_path): + """Written section includes the three rules.""" + claude_install(tmp_path) + content = (tmp_path / "CLAUDE.md").read_text() + assert "GRAPH_REPORT.md" in content + assert "wiki/index.md" in content + assert "_rebuild_code" in content + + +def test_install_appends_to_existing_claude_md(tmp_path): + """Appends to an existing CLAUDE.md without clobbering it.""" + target = tmp_path / "CLAUDE.md" + target.write_text("# Existing content\n\nSome rules here.\n") + claude_install(tmp_path) + content = target.read_text() + assert "Existing content" in content + assert _CLAUDE_MD_MARKER in content + + +def test_install_is_idempotent(tmp_path, capsys): + """Running install twice does not duplicate the section.""" + claude_install(tmp_path) + claude_install(tmp_path) + content = (tmp_path / "CLAUDE.md").read_text() + assert content.count(_CLAUDE_MD_MARKER) == 1 + captured = capsys.readouterr() + assert "already configured" in captured.out + + +def test_install_idempotent_message(tmp_path, capsys): + """Second install prints the 'already configured' message.""" + claude_install(tmp_path) + capsys.readouterr() # clear first call output + claude_install(tmp_path) + out = capsys.readouterr().out + assert "already configured" in out + + +# --------------------------------------------------------------------------- +# uninstall +# --------------------------------------------------------------------------- + +def test_uninstall_removes_section(tmp_path): + """Removes the graphify section after it was installed.""" + claude_install(tmp_path) + claude_uninstall(tmp_path) + target = tmp_path / "CLAUDE.md" + # File may or may not exist depending on whether it was empty + if target.exists(): + assert _CLAUDE_MD_MARKER not in target.read_text() + + +def test_uninstall_preserves_other_content(tmp_path): + """Uninstall keeps pre-existing content outside the graphify section.""" + target = tmp_path / "CLAUDE.md" + target.write_text("# My Project\n\nSome rules.\n") + claude_install(tmp_path) + claude_uninstall(tmp_path) + assert target.exists() + content = target.read_text() + assert "My Project" in content + assert "Some rules" in content + assert _CLAUDE_MD_MARKER not in content + + +def test_uninstall_no_op_when_not_installed(tmp_path, capsys): + """Uninstall on a CLAUDE.md without graphify section prints a message and exits cleanly.""" + target = tmp_path / "CLAUDE.md" + target.write_text("# Other stuff\n") + claude_uninstall(tmp_path) + out = capsys.readouterr().out + assert "not found" in out or "nothing to do" in out + + +def test_uninstall_no_op_when_no_file(tmp_path, capsys): + """Uninstall when no CLAUDE.md exists prints a message and exits cleanly.""" + claude_uninstall(tmp_path) + out = capsys.readouterr().out + assert "No CLAUDE.md" in out or "nothing to do" in out