v2: fast --update for code-only changes, parallel AST+semantic, graphify claude install

This commit is contained in:
Safi
2026-04-05 23:00:24 +01:00
parent e4bc87a284
commit e8b8e2039d
5 changed files with 262 additions and 7 deletions
+76
View File
@@ -1,6 +1,7 @@
"""graphify CLI - `graphify install` sets up the Claude Code skill."""
from __future__ import annotations
import json
import re
import shutil
import sys
from pathlib import Path
@@ -52,6 +53,70 @@ def install() -> None:
print()
_CLAUDE_MD_SECTION = """\
## graphify
This project has a graphify knowledge graph at graphify-out/.
Rules:
- Before answering architecture or codebase questions, read graphify-out/GRAPH_REPORT.md for god nodes and community structure
- If graphify-out/wiki/index.md exists, navigate it instead of reading raw files
- After modifying code files in this session, run `python3 -c "from graphify.watch import _rebuild_code; from pathlib import Path; _rebuild_code(Path('.'))"` to keep the graph current
"""
_CLAUDE_MD_MARKER = "## graphify"
def claude_install(project_dir: Path | None = None) -> None:
"""Write the graphify section to the local CLAUDE.md."""
target = (project_dir or Path(".")) / "CLAUDE.md"
if target.exists():
content = target.read_text()
if _CLAUDE_MD_MARKER in content:
print("graphify already configured in CLAUDE.md")
return
new_content = content.rstrip() + "\n\n" + _CLAUDE_MD_SECTION
else:
new_content = _CLAUDE_MD_SECTION
target.write_text(new_content)
print(f"graphify section written to {target.resolve()}")
print()
print("Claude Code will now check the knowledge graph before answering")
print("codebase questions and rebuild it after code changes.")
def claude_uninstall(project_dir: Path | None = None) -> None:
"""Remove the graphify section from the local CLAUDE.md."""
target = (project_dir or Path(".")) / "CLAUDE.md"
if not target.exists():
print("No CLAUDE.md found in current directory - nothing to do")
return
content = target.read_text()
if _CLAUDE_MD_MARKER not in content:
print("graphify section not found in CLAUDE.md - nothing to do")
return
# Remove the ## graphify section: from the marker to the next ## heading or EOF
cleaned = re.sub(
r"\n*## graphify\n.*?(?=\n## |\Z)",
"",
content,
flags=re.DOTALL,
).rstrip()
if cleaned:
target.write_text(cleaned + "\n")
else:
target.unlink()
print(f"CLAUDE.md was empty after removal - deleted {target.resolve()}")
return
print(f"graphify section removed from {target.resolve()}")
def main() -> None:
if len(sys.argv) < 2 or sys.argv[1] in ("-h", "--help"):
print("Usage: graphify <command>")
@@ -62,12 +127,23 @@ def main() -> None:
print(" hook install install post-commit git hook (auto-rebuilds graph on commit)")
print(" hook uninstall remove post-commit git hook")
print(" hook status check if hook is installed")
print(" claude install write graphify section to local CLAUDE.md")
print(" claude uninstall remove graphify section from local CLAUDE.md")
print()
return
cmd = sys.argv[1]
if cmd == "install":
install()
elif cmd == "claude":
subcmd = sys.argv[2] if len(sys.argv) > 2 else ""
if subcmd == "install":
claude_install()
elif subcmd == "uninstall":
claude_uninstall()
else:
print("Usage: graphify claude [install|uninstall]", file=sys.stderr)
sys.exit(1)
elif cmd == "hook":
from graphify.hooks import install as hook_install, uninstall as hook_uninstall, status as hook_status
subcmd = sys.argv[2] if len(sys.argv) > 2 else ""
+44 -3
View File
@@ -95,11 +95,15 @@ Then act on it:
**Before starting:** note whether `--mode deep` was given. You must pass `DEEP_MODE=true` to every subagent in Step B2 if it was. Track this from the original invocation - do not lose it.
This step has two parts: **structural extraction** (deterministic, free) then **semantic extraction** (Claude, costs tokens).
This step has two parts: **structural extraction** (deterministic, free) and **semantic extraction** (Claude, costs tokens).
**Run Part A (AST) and Part B (semantic) in parallel. Dispatch all semantic subagents AND start AST extraction in the same message. Both can run simultaneously since they operate on different file types. Merge results in Part C as before.**
Note: Parallelizing AST + semantic saves 5-15s on large corpora. AST is deterministic and fast; start it while subagents are processing docs/papers.
#### Part A - Structural extraction for code files
For any code files detected, run AST extraction first:
For any code files detected, run AST extraction in parallel with Part B subagents:
```bash
python3 -c "
@@ -653,6 +657,7 @@ from pathlib import Path
result = detect_incremental(Path('INPUT_PATH'))
new_total = result.get('new_total', 0)
print(json.dumps(result, indent=2))
Path('.graphify_incremental.json').write_text(json.dumps(result))
if new_total == 0:
print('No files changed since last run. Nothing to update.')
raise SystemExit(0)
@@ -660,7 +665,27 @@ print(f'{new_total} new/changed file(s) to re-extract.')
"
```
If new files exist, run **Steps 3A–3C** on `result['new_files']` only (not the full corpus). Then:
If new files exist, first check whether all changed files are code files:
```bash
python3 -c "
import json
from pathlib import Path
result = json.loads(open('.graphify_incremental.json').read()) if Path('.graphify_incremental.json').exists() else {}
code_exts = {'.py','.ts','.js','.go','.rs','.java','.cpp','.c','.rb','.swift','.kt','.cs','.scala','.php','.cc','.cxx','.hpp','.h','.kts'}
new_files = result.get('new_files', {})
all_changed = [f for files in new_files.values() for f in files]
code_only = all(Path(f).suffix.lower() in code_exts for f in all_changed)
print('code_only:', code_only)
"
```
If `code_only` is True: print `[graphify update] Code-only changes detected - skipping semantic extraction (no LLM needed)`, run only Step 3A (AST) on the changed files, skip Step 3B entirely (no subagents), then go straight to merge and Steps 4–8.
If `code_only` is False (any changed file is a doc/paper/image): run the full Steps 3A–3C pipeline as normal.
Then:
```bash
python3 -c "
@@ -1135,6 +1160,22 @@ If a post-commit hook already exists, graphify appends to it rather than replaci
---
## For native CLAUDE.md integration
Run once per project to make graphify always-on in Claude Code sessions:
```bash
graphify claude install
```
This writes a `## graphify` section to the local `CLAUDE.md` that instructs Claude to check the graph before answering codebase questions and rebuild it after code changes. No manual `/graphify` needed in future sessions.
```bash
graphify claude uninstall # remove the section
```
---
## Honesty Rules
- Never invent an edge. If unsure, use AMBIGUOUS.
+1 -1
View File
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
[project]
name = "graphifyy"
version = "0.1.10"
version = "0.1.11"
description = "Claude Code skill - turn any folder of code, docs, papers, images, or tweets into a queryable knowledge graph"
readme = "README.md"
license = { text = "MIT" }
+44 -3
View File
@@ -95,11 +95,15 @@ Then act on it:
**Before starting:** note whether `--mode deep` was given. You must pass `DEEP_MODE=true` to every subagent in Step B2 if it was. Track this from the original invocation - do not lose it.
This step has two parts: **structural extraction** (deterministic, free) then **semantic extraction** (Claude, costs tokens).
This step has two parts: **structural extraction** (deterministic, free) and **semantic extraction** (Claude, costs tokens).
**Run Part A (AST) and Part B (semantic) in parallel. Dispatch all semantic subagents AND start AST extraction in the same message. Both can run simultaneously since they operate on different file types. Merge results in Part C as before.**
Note: Parallelizing AST + semantic saves 5-15s on large corpora. AST is deterministic and fast; start it while subagents are processing docs/papers.
#### Part A - Structural extraction for code files
For any code files detected, run AST extraction first:
For any code files detected, run AST extraction in parallel with Part B subagents:
```bash
python3 -c "
@@ -653,6 +657,7 @@ from pathlib import Path
result = detect_incremental(Path('INPUT_PATH'))
new_total = result.get('new_total', 0)
print(json.dumps(result, indent=2))
Path('.graphify_incremental.json').write_text(json.dumps(result))
if new_total == 0:
print('No files changed since last run. Nothing to update.')
raise SystemExit(0)
@@ -660,7 +665,27 @@ print(f'{new_total} new/changed file(s) to re-extract.')
"
```
If new files exist, run **Steps 3A–3C** on `result['new_files']` only (not the full corpus). Then:
If new files exist, first check whether all changed files are code files:
```bash
python3 -c "
import json
from pathlib import Path
result = json.loads(open('.graphify_incremental.json').read()) if Path('.graphify_incremental.json').exists() else {}
code_exts = {'.py','.ts','.js','.go','.rs','.java','.cpp','.c','.rb','.swift','.kt','.cs','.scala','.php','.cc','.cxx','.hpp','.h','.kts'}
new_files = result.get('new_files', {})
all_changed = [f for files in new_files.values() for f in files]
code_only = all(Path(f).suffix.lower() in code_exts for f in all_changed)
print('code_only:', code_only)
"
```
If `code_only` is True: print `[graphify update] Code-only changes detected - skipping semantic extraction (no LLM needed)`, run only Step 3A (AST) on the changed files, skip Step 3B entirely (no subagents), then go straight to merge and Steps 4–8.
If `code_only` is False (any changed file is a doc/paper/image): run the full Steps 3A–3C pipeline as normal.
Then:
```bash
python3 -c "
@@ -1135,6 +1160,22 @@ If a post-commit hook already exists, graphify appends to it rather than replaci
---
## For native CLAUDE.md integration
Run once per project to make graphify always-on in Claude Code sessions:
```bash
graphify claude install
```
This writes a `## graphify` section to the local `CLAUDE.md` that instructs Claude to check the graph before answering codebase questions and rebuild it after code changes. No manual `/graphify` needed in future sessions.
```bash
graphify claude uninstall # remove the section
```
---
## Honesty Rules
- Never invent an edge. If unsure, use AMBIGUOUS.
+97
View File
@@ -0,0 +1,97 @@
"""Tests for graphify claude install / uninstall commands."""
from pathlib import Path
import pytest
from graphify.__main__ import claude_install, claude_uninstall, _CLAUDE_MD_MARKER, _CLAUDE_MD_SECTION
# ---------------------------------------------------------------------------
# install
# ---------------------------------------------------------------------------
def test_install_creates_claude_md(tmp_path):
"""Creates CLAUDE.md when none exists."""
claude_install(tmp_path)
target = tmp_path / "CLAUDE.md"
assert target.exists()
assert _CLAUDE_MD_MARKER in target.read_text()
def test_install_contains_expected_rules(tmp_path):
"""Written section includes the three rules."""
claude_install(tmp_path)
content = (tmp_path / "CLAUDE.md").read_text()
assert "GRAPH_REPORT.md" in content
assert "wiki/index.md" in content
assert "_rebuild_code" in content
def test_install_appends_to_existing_claude_md(tmp_path):
"""Appends to an existing CLAUDE.md without clobbering it."""
target = tmp_path / "CLAUDE.md"
target.write_text("# Existing content\n\nSome rules here.\n")
claude_install(tmp_path)
content = target.read_text()
assert "Existing content" in content
assert _CLAUDE_MD_MARKER in content
def test_install_is_idempotent(tmp_path, capsys):
"""Running install twice does not duplicate the section."""
claude_install(tmp_path)
claude_install(tmp_path)
content = (tmp_path / "CLAUDE.md").read_text()
assert content.count(_CLAUDE_MD_MARKER) == 1
captured = capsys.readouterr()
assert "already configured" in captured.out
def test_install_idempotent_message(tmp_path, capsys):
"""Second install prints the 'already configured' message."""
claude_install(tmp_path)
capsys.readouterr() # clear first call output
claude_install(tmp_path)
out = capsys.readouterr().out
assert "already configured" in out
# ---------------------------------------------------------------------------
# uninstall
# ---------------------------------------------------------------------------
def test_uninstall_removes_section(tmp_path):
"""Removes the graphify section after it was installed."""
claude_install(tmp_path)
claude_uninstall(tmp_path)
target = tmp_path / "CLAUDE.md"
# File may or may not exist depending on whether it was empty
if target.exists():
assert _CLAUDE_MD_MARKER not in target.read_text()
def test_uninstall_preserves_other_content(tmp_path):
"""Uninstall keeps pre-existing content outside the graphify section."""
target = tmp_path / "CLAUDE.md"
target.write_text("# My Project\n\nSome rules.\n")
claude_install(tmp_path)
claude_uninstall(tmp_path)
assert target.exists()
content = target.read_text()
assert "My Project" in content
assert "Some rules" in content
assert _CLAUDE_MD_MARKER not in content
def test_uninstall_no_op_when_not_installed(tmp_path, capsys):
"""Uninstall on a CLAUDE.md without graphify section prints a message and exits cleanly."""
target = tmp_path / "CLAUDE.md"
target.write_text("# Other stuff\n")
claude_uninstall(tmp_path)
out = capsys.readouterr().out
assert "not found" in out or "nothing to do" in out
def test_uninstall_no_op_when_no_file(tmp_path, capsys):
"""Uninstall when no CLAUDE.md exists prints a message and exits cleanly."""
claude_uninstall(tmp_path)
out = capsys.readouterr().out
assert "No CLAUDE.md" in out or "nothing to do" in out