diff --git a/graphify/__main__.py b/graphify/__main__.py index e0b1cca..2b25d7c 100644 --- a/graphify/__main__.py +++ b/graphify/__main__.py @@ -367,6 +367,35 @@ _SETTINGS_HOOK = { ], } +_READ_SETTINGS_HOOK = { + # The Bash hook above never sees a file read through the native Read tool or a + # Glob, which is the most common way an agent skips the graph: answering a + # codebase question by Read-ing many source files one by one (issue #1114). + # Match Read|Glob, inspect the target path, and nudge (never block) only for a + # source/doc file outside graphify-out/ when a graph exists. The parser is + # python3 (already a graphify dependency), the shell is POSIX, and every branch + # fails open, so a legitimate read always goes through. Reading the graph's own + # report under graphify-out/ is suppressed so it never starts a feedback loop. + "matcher": "Read|Glob", + "hooks": [ + { + "type": "command", + "command": ( + "HIT=$(python3 -c \"" + "import json,sys;" + "d=json.load(sys.stdin);" + "t=d.get('tool_input',d);" + "s=(str(t.get('file_path') or '')+' '+str(t.get('pattern') or '')+' '+str(t.get('path') or '')).lower().replace(chr(92),'/');" + "exts=('.py','.js','.ts','.tsx','.jsx','.go','.rs','.java','.rb','.c','.h','.cpp','.hpp','.cc','.cs','.kt','.swift','.php','.scala','.lua','.sh','.md','.rst','.txt','.mdx');" + "sys.stdout.write('1' if 'graphify-out/' not in s and any(e in s for e in exts) else '')\" 2>/dev/null || true); " + "if [ \"$HIT\" = 1 ] && [ -f graphify-out/graph.json ]; then " + r"""echo '{"hookSpecificOutput":{"hookEventName":"PreToolUse","additionalContext":"graphify: knowledge graph at graphify-out/. For codebase questions, run `graphify query \"\"` (scoped subgraph, usually much smaller than reading files one by one), `graphify explain \"\"`, or `graphify path \"\" \"\"`, instead of reading source files to answer. Read raw files to modify or debug specific code, or when the graph lacks the detail."}}'; """ + "fi || true" + ), + } + ], +} + def _skill_registration(skill_path: str = "~/.claude/skills/graphify/SKILL.md") -> str: return ( "\n# graphify\n" @@ -1724,10 +1753,11 @@ def _install_claude_hook(project_dir: Path) -> None: hooks = settings.setdefault("hooks", {}) pre_tool = hooks.setdefault("PreToolUse", []) - hooks["PreToolUse"] = [h for h in pre_tool if not (h.get("matcher") in ("Glob|Grep", "Bash") and "graphify" in str(h))] + hooks["PreToolUse"] = [h for h in pre_tool if not (h.get("matcher") in ("Glob|Grep", "Bash", "Read|Glob") and "graphify" in str(h))] hooks["PreToolUse"].append(_SETTINGS_HOOK) + hooks["PreToolUse"].append(_READ_SETTINGS_HOOK) settings_path.write_text(json.dumps(settings, indent=2), encoding="utf-8") - print(f" .claude/settings.json -> PreToolUse hook registered") + print(f" .claude/settings.json -> PreToolUse hooks registered (Bash search + Read/Glob)") def _uninstall_claude_hook(project_dir: Path) -> None: @@ -1740,7 +1770,7 @@ def _uninstall_claude_hook(project_dir: Path) -> None: except json.JSONDecodeError: return pre_tool = settings.get("hooks", {}).get("PreToolUse", []) - filtered = [h for h in pre_tool if not (h.get("matcher") in ("Glob|Grep", "Bash") and "graphify" in str(h))] + filtered = [h for h in pre_tool if not (h.get("matcher") in ("Glob|Grep", "Bash", "Read|Glob") and "graphify" in str(h))] if len(filtered) == len(pre_tool): return settings["hooks"]["PreToolUse"] = filtered diff --git a/tests/test_install_strings.py b/tests/test_install_strings.py index 5e00370..fb8fd0d 100644 --- a/tests/test_install_strings.py +++ b/tests/test_install_strings.py @@ -13,6 +13,7 @@ import json from graphify.__main__ import ( _SETTINGS_HOOK, + _READ_SETTINGS_HOOK, _CLAUDE_MD_SECTION, _AGENTS_MD_SECTION, _GEMINI_MD_SECTION, @@ -31,6 +32,7 @@ from graphify.__main__ import ( # against the actual payload text the assistant will receive. _INSTALL_TEXTS: dict[str, str] = { "_SETTINGS_HOOK": json.dumps(_SETTINGS_HOOK), + "_READ_SETTINGS_HOOK": json.dumps(_READ_SETTINGS_HOOK), "_CLAUDE_MD_SECTION": _CLAUDE_MD_SECTION, "_AGENTS_MD_SECTION": _AGENTS_MD_SECTION, "_GEMINI_MD_SECTION": _GEMINI_MD_SECTION, diff --git a/tests/test_read_hook.py b/tests/test_read_hook.py new file mode 100644 index 0000000..00c901b --- /dev/null +++ b/tests/test_read_hook.py @@ -0,0 +1,80 @@ +"""The Read|Glob PreToolUse hook nudges toward the graph instead of raw reads. + +Closes the issue #1114 gap: the Bash search hook never sees a file read through +the native Read tool or a Glob. These tests run the hook command the way Claude +Code does - via `sh -c` with crafted stdin JSON - and assert it nudges only for +a source/doc file outside graphify-out/ when a graph exists, and otherwise stays +silent and fails open. +""" +import json +import subprocess + +from graphify.__main__ import _READ_SETTINGS_HOOK + +CMD = _READ_SETTINGS_HOOK["hooks"][0]["command"] + + +def _run(tool_input, cwd, *, graph: bool): + if graph: + (cwd / "graphify-out").mkdir(parents=True, exist_ok=True) + (cwd / "graphify-out" / "graph.json").write_text("{}", encoding="utf-8") + stdin = json.dumps({"tool_input": tool_input}) + return subprocess.run( + ["sh", "-c", CMD], input=stdin, capture_output=True, text=True, cwd=cwd + ) + + +def test_matcher_targets_read_and_glob(): + assert _READ_SETTINGS_HOOK["matcher"] == "Read|Glob" + + +def test_silent_without_graph(tmp_path): + out = _run({"file_path": "src/app.py"}, tmp_path, graph=False).stdout + assert out.strip() == "" + + +def test_nudges_on_source_read_with_graph(tmp_path): + out = _run({"file_path": "src/app.py"}, tmp_path, graph=True).stdout + assert "graphify query" in out + + +def test_nudge_payload_is_valid_pretooluse_json(tmp_path): + out = _run({"file_path": "pkg/mod.ts"}, tmp_path, graph=True).stdout + payload = json.loads(out) + assert payload["hookSpecificOutput"]["hookEventName"] == "PreToolUse" + assert "graphify query" in payload["hookSpecificOutput"]["additionalContext"] + + +def test_silent_on_graphify_out_targets(tmp_path): + """Reading the graph's own report must not start a go-read-the-graph loop.""" + out = _run({"file_path": "graphify-out/GRAPH_REPORT.md"}, tmp_path, graph=True).stdout + assert out.strip() == "" + + +def test_silent_on_non_source_files(tmp_path): + for path in ("uv.lock", "logo.png", "data.bin", ".gitignore"): + out = _run({"file_path": path}, tmp_path, graph=True).stdout + assert out.strip() == "", f"{path} should not nudge" + + +def test_glob_pattern_nudges(tmp_path): + out = _run({"pattern": "**/*.py", "path": "src"}, tmp_path, graph=True).stdout + assert "graphify query" in out + + +def test_fails_open_on_malformed_stdin(tmp_path): + (tmp_path / "graphify-out").mkdir() + (tmp_path / "graphify-out" / "graph.json").write_text("{}", encoding="utf-8") + r = subprocess.run( + ["sh", "-c", CMD], input="this is not json", capture_output=True, text=True, cwd=tmp_path + ) + assert r.returncode == 0 + assert r.stdout.strip() == "" + + +def test_never_blocks(tmp_path): + """A nudge is additionalContext only - the hook must exit 0, never deny.""" + r = _run({"file_path": "src/app.py"}, tmp_path, graph=True) + assert r.returncode == 0 + assert '"permissionDecision"' not in r.stdout + assert '"deny"' not in r.stdout