fix: OpenCode project path, hook loop guard, deterministic output, Amp platform, punctuation search, builtin god-node filter, .svh Verilog (#1040, #1018, #1037, #948, #994, #916, #1042)

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
Safi
2026-05-27 12:39:22 +01:00
co-authored by Claude Sonnet 4.6
parent a54a542b6c
commit 80301a06bf
9 changed files with 1390 additions and 22 deletions
+11
View File
@@ -2,6 +2,17 @@
Full release notes with details on each version: [GitHub Releases](https://github.com/safishamsi/graphify/releases)
## 0.8.21 (2026-05-27)
- Fix: `graphify update` (no `--changed` flag) no longer leaves ghost nodes from files deleted between runs — full re-extraction path now reconciles the existing graph against current disk state and evicts any node whose `source_file` no longer exists; `_norm_source_file` used on both sides to guarantee path format consistency (#1007)
- Fix: `graphify install --platform opencode --project` now writes `SKILL.md` to `.opencode/skills/graphify/SKILL.md` (discoverable by OpenCode) instead of the incorrect `.config/opencode/skills/` path; git-add hint updated accordingly (#1040)
- Fix: post-commit hook no longer triggers rebuild when only `graphify-out/` files were committed (avoids infinite dirty-tree loop when graph outputs are tracked in git); `GRAPHIFY_SKIP_HOOK=1` env var added for one-off skip; hook rebuild log now appends (`>>`) instead of overwriting (`>`) (#1018, #1037)
- Fix: graph output is now byte-for-byte deterministic across runs — edges sorted by `(source, target, relation)` in `build_from_json`; `PYTHONHASHSEED=0` exported in hook scripts to stabilize Louvain community ordering (#1010)
- Feat: Amp (ampcode.com) platform support — `graphify amp install/uninstall` installs the skill into `.amp/skills/graphify/SKILL.md` (#948)
- Fix: query punctuation no longer breaks node matching — `"what calls extract?"` correctly finds the `extract` node; `_search_tokens` helper strips punctuation from search terms in `_query_terms`, `_score_nodes`, and `_find_node` (#994, #978)
- Fix: language built-in globals (`String`, `Number`, `Boolean`, `Object`, `Array`, etc.) no longer accumulate spurious call edges — filtered at same-file and cross-file resolution in the AST extractor, eliminating god-node pollution from constructor-style calls (#916, #726)
- Feat: SystemVerilog header files (`.svh`) now extracted using the Verilog parser alongside `.v` and `.sv` (#1042)
## 0.8.20 (2026-05-26)
- Fix: stale nodes persist after `graphify update` when files are deleted on Windows — `deleted_paths` and `evict_sources` in `_rebuild_code` now use `.as_posix()` for consistent forward-slash paths; `_relativize_source_files` called on the existing graph before eviction (not after); `_relativize_source_files` itself now produces forward slashes (#1007)
+13 -3
View File
@@ -77,6 +77,11 @@ def _platform_skill_destination(platform_name: str, *, project: bool = False, pr
return Path.home() / ".agents" / "skills" / "graphify" / "SKILL.md"
return Path.home() / ".gemini" / "skills" / "graphify" / "SKILL.md"
if platform_name == "opencode":
if project:
return (project_dir or Path(".")) / ".opencode" / "skills" / "graphify" / "SKILL.md"
return Path.home() / ".config" / "opencode" / "skills" / "graphify" / "SKILL.md"
if platform_name == "devin":
if project:
return (project_dir or Path(".")) / ".devin" / "skills" / "graphify" / "SKILL.md"
@@ -289,6 +294,11 @@ _PLATFORM_CONFIG: dict[str, dict] = {
"skill_dst": Path(".kimi") / "skills" / "graphify" / "SKILL.md",
"claude_md": False,
},
"amp": {
"skill_file": "skill-amp.md",
"skill_dst": Path(".amp") / "skills" / "graphify" / "SKILL.md",
"claude_md": False,
},
"devin": {
"skill_file": "skill-devin.md",
# User scope: ~/.config/devin/skills/graphify/SKILL.md
@@ -1120,7 +1130,7 @@ def _project_install(platform_name: str, project_dir: Path | None = None) -> Non
elif platform_name == "kiro":
_kiro_install(project_dir)
_print_project_git_add_hint([project_dir / ".kiro"])
elif platform_name in ("aider", "codex", "opencode", "claw", "droid", "trae", "trae-cn", "hermes"):
elif platform_name in ("aider", "amp", "codex", "opencode", "claw", "droid", "trae", "trae-cn", "hermes"):
skill_dst = _copy_skill_file(platform_name, project=True, project_dir=project_dir)
_agents_install(project_dir, platform_name)
hint_paths = [_project_scope_root(skill_dst, project_dir), project_dir / "AGENTS.md"]
@@ -1153,7 +1163,7 @@ def _project_uninstall(platform_name: str, project_dir: Path | None = None) -> N
_cursor_uninstall(project_dir)
elif platform_name == "kiro":
_kiro_uninstall(project_dir)
elif platform_name in ("aider", "codex", "opencode", "claw", "droid", "trae", "trae-cn", "hermes"):
elif platform_name in ("aider", "amp", "codex", "opencode", "claw", "droid", "trae", "trae-cn", "hermes"):
_remove_skill_file(platform_name, project=True, project_dir=project_dir)
_agents_uninstall(project_dir, platform=platform_name)
if platform_name == "codex":
@@ -1729,7 +1739,7 @@ def main() -> None:
else:
print("Usage: graphify pi [install|uninstall]", file=sys.stderr)
sys.exit(1)
elif cmd in ("aider", "codex", "opencode", "claw", "droid", "trae", "trae-cn", "hermes"):
elif cmd in ("aider", "amp", "codex", "opencode", "claw", "droid", "trae", "trae-cn", "hermes"):
subcmd = sys.argv[2] if len(sys.argv) > 2 else ""
if subcmd == "install":
if "--project" in sys.argv[3:]:
+1 -1
View File
@@ -25,7 +25,7 @@ class FileType(str, Enum):
_MANIFEST_PATH = "graphify-out/manifest.json"
CODE_EXTENSIONS = {'.py', '.ts', '.tsx', '.js', '.jsx', '.mjs', '.ejs', '.ets', '.go', '.rs', '.java', '.groovy', '.gradle', '.cpp', '.cc', '.cxx', '.c', '.h', '.hpp', '.rb', '.swift', '.kt', '.kts', '.cs', '.scala', '.php', '.lua', '.luau', '.toc', '.zig', '.ps1', '.ex', '.exs', '.m', '.mm', '.jl', '.vue', '.svelte', '.astro', '.dart', '.v', '.sv', '.sql', '.r', '.f', '.F', '.f90', '.F90', '.f95', '.F95', '.f03', '.F03', '.f08', '.F08', '.pas', '.pp', '.dpr', '.dpk', '.lpr', '.inc', '.dfm', '.lfm', '.lpk', '.sh', '.bash', '.json', '.sln', '.csproj', '.fsproj', '.vbproj', '.razor', '.cshtml'}
CODE_EXTENSIONS = {'.py', '.ts', '.tsx', '.js', '.jsx', '.mjs', '.ejs', '.ets', '.go', '.rs', '.java', '.groovy', '.gradle', '.cpp', '.cc', '.cxx', '.c', '.h', '.hpp', '.rb', '.swift', '.kt', '.kts', '.cs', '.scala', '.php', '.lua', '.luau', '.toc', '.zig', '.ps1', '.ex', '.exs', '.m', '.mm', '.jl', '.vue', '.svelte', '.astro', '.dart', '.v', '.sv', '.svh', '.sql', '.r', '.f', '.F', '.f90', '.F90', '.f95', '.F95', '.f03', '.F03', '.f08', '.F08', '.pas', '.pp', '.dpr', '.dpk', '.lpr', '.inc', '.dfm', '.lfm', '.lpk', '.sh', '.bash', '.json', '.sln', '.csproj', '.fsproj', '.vbproj', '.razor', '.cshtml'}
DOC_EXTENSIONS = {'.md', '.mdx', '.qmd', '.txt', '.rst', '.html', '.yaml', '.yml'}
PAPER_EXTENSIONS = {'.pdf'}
IMAGE_EXTENSIONS = {'.png', '.jpg', '.jpeg', '.gif', '.webp', '.svg'}
+33 -4
View File
@@ -14,6 +14,32 @@ from .mcp_ingest import extract_mcp_config, is_mcp_config_path
_RECURSION_LIMIT = 10_000
# Language built-in globals that AST may classify as call targets when used as
# constructors or coercion functions (e.g. String(x), Number(x), Boolean(x)).
# Without this filter they become god-nodes accumulating spurious edges from
# every call site. Filter applied at same-file and cross-file resolution.
# See issue #726.
_LANGUAGE_BUILTIN_GLOBALS: frozenset[str] = frozenset({
# JavaScript / TypeScript ECMAScript built-ins
"String", "Number", "Boolean", "Object", "Array", "Symbol", "BigInt",
"Date", "RegExp", "Error", "TypeError", "RangeError", "SyntaxError",
"ReferenceError", "EvalError", "URIError",
"Promise", "Map", "Set", "WeakMap", "WeakSet", "JSON", "Math",
"Reflect", "Proxy", "Intl",
"parseInt", "parseFloat", "isNaN", "isFinite",
"encodeURIComponent", "decodeURIComponent", "encodeURI", "decodeURI",
# Browser / Node common globals
"URL", "URLSearchParams", "FormData", "Blob", "File",
"Headers", "Request", "Response", "AbortController", "AbortSignal",
"TextEncoder", "TextDecoder", "console",
# Python built-in callables
"str", "int", "float", "bool", "list", "dict", "set", "tuple", "bytes",
"len", "range", "enumerate", "zip", "map", "filter", "sum", "min", "max",
"print", "open", "isinstance", "type", "super", "sorted", "reversed",
"any", "all", "abs", "round", "next", "iter", "hash", "id", "repr",
"callable", "getattr", "setattr", "hasattr", "delattr", "vars", "dir",
})
def _raise_recursion_limit() -> None:
if sys.getrecursionlimit() < _RECURSION_LIMIT:
@@ -2285,7 +2311,7 @@ def _extract_generic(path: Path, config: LanguageConfig) -> dict:
# Try reading the node directly (e.g. Java name field is the callee)
callee_name = _read_text(func_node, source)
if callee_name:
if callee_name and callee_name not in _LANGUAGE_BUILTIN_GLOBALS:
tgt_nid = label_to_nid.get(callee_name)
if tgt_nid and tgt_nid != caller_nid:
pair = (caller_nid, tgt_nid)
@@ -4136,7 +4162,7 @@ def extract_go(path: Path) -> dict:
is_member_call = receiver_name not in go_imported_pkgs
if field:
callee_name = _read_text(field, source)
if callee_name:
if callee_name and callee_name not in _LANGUAGE_BUILTIN_GLOBALS:
tgt_nid = label_to_nid.get(callee_name)
if tgt_nid and tgt_nid != caller_nid:
pair = (caller_nid, tgt_nid)
@@ -4337,7 +4363,7 @@ def extract_rust(path: Path) -> dict:
name = func_node.child_by_field_name("name")
if name:
callee_name = _read_text(name, source)
if callee_name:
if callee_name and callee_name not in _LANGUAGE_BUILTIN_GLOBALS:
tgt_nid = label_to_nid.get(callee_name)
if tgt_nid and tgt_nid != caller_nid:
pair = (caller_nid, tgt_nid)
@@ -6456,7 +6482,7 @@ def extract_elixir(path: Path) -> dict:
if child.type == "identifier":
callee_name = source[child.start_byte:child.end_byte].decode("utf-8", errors="replace")
break
if callee_name:
if callee_name and callee_name not in _LANGUAGE_BUILTIN_GLOBALS:
tgt_nid = label_to_nid.get(callee_name)
if tgt_nid and tgt_nid != caller_nid:
pair = (caller_nid, tgt_nid)
@@ -8245,6 +8271,7 @@ _DISPATCH: dict[str, Any] = {
".dart": extract_dart,
".v": extract_verilog,
".sv": extract_verilog,
".svh": extract_verilog,
".sql": extract_sql,
".md": extract_markdown,
".mdx": extract_markdown,
@@ -8630,6 +8657,8 @@ def extract(
callee = rc.get("callee", "")
if not callee:
continue
if callee in _LANGUAGE_BUILTIN_GLOBALS:
continue
# Skip member-call callees: obj.log() → "log" has no import evidence
# and collides with any top-level function named "log" in the corpus.
if rc.get("is_member_call"):
+10 -2
View File
@@ -60,11 +60,19 @@ GIT_DIR=$(git rev-parse --git-dir 2>/dev/null)
[ -f "$GIT_DIR/MERGE_HEAD" ] && exit 0
[ -f "$GIT_DIR/CHERRY_PICK_HEAD" ] && exit 0
[ "${GRAPHIFY_SKIP_HOOK:-0}" = "1" ] && exit 0
CHANGED=$(git diff --name-only HEAD~1 HEAD 2>/dev/null || git diff --name-only HEAD 2>/dev/null)
if [ -z "$CHANGED" ]; then
exit 0
fi
# Skip when only graphify-out/ artifacts changed (avoids rebuild loop when graph outputs are tracked in git)
_NON_GRAPH=$(echo "$CHANGED" | grep -v '^graphify-out/' || true)
if [ -z "$_NON_GRAPH" ]; then
exit 0
fi
""" + _PYTHON_DETECT + """
export GRAPHIFY_CHANGED="$CHANGED"
@@ -100,7 +108,7 @@ except TimeoutError as exc:
except Exception as exc:
print(f'[graphify hook] Rebuild failed: {exc}')
sys.exit(1)
" > "$_GRAPHIFY_LOG" 2>&1 < /dev/null &
" >> "$_GRAPHIFY_LOG" 2>&1 < /dev/null &
disown 2>/dev/null || true
# graphify-hook-end
"""
@@ -162,7 +170,7 @@ except TimeoutError as exc:
except Exception as exc:
print(f'[graphify] Rebuild failed: {exc}')
sys.exit(1)
" > "$_GRAPHIFY_LOG" 2>&1 < /dev/null &
" >> "$_GRAPHIFY_LOG" 2>&1 < /dev/null &
disown 2>/dev/null || true
# graphify-checkout-hook-end
"""
+20 -10
View File
@@ -2,6 +2,7 @@
from __future__ import annotations
import json
import math
import re
import sys
from pathlib import Path
import networkx as nx
@@ -56,6 +57,11 @@ def _strip_diacritics(text: str) -> str:
return "".join(c for c in nfkd if not unicodedata.combining(c))
def _search_tokens(text: str) -> list[str]:
"""Split text into word tokens, stripping punctuation and diacritics."""
return re.findall(r"\w+", _strip_diacritics(str(text)).lower())
def _has_chinese(text: str) -> bool:
return any("一" <= ch <= "鿿" for ch in text)
@@ -82,14 +88,16 @@ def _query_terms(question: str) -> list[str]:
"""Split a query into searchable terms, segmenting Chinese text."""
terms: list[str] = []
for raw in question.split():
term = raw.lower().strip()
if not term:
continue
candidates = _segment_chinese(term) if _has_chinese(term) else [term]
for seg in candidates:
seg = seg.strip()
if seg and _is_searchable(seg):
terms.append(seg)
if _has_chinese(raw):
for seg in _segment_chinese(raw.lower().strip()):
seg = seg.strip()
if seg and _is_searchable(seg):
terms.append(seg)
else:
# Strip punctuation without touching Unicode characters (avoid NFKD mangling non-Latin scripts)
for tok in re.findall(r"\w+", raw.lower()):
if _is_searchable(tok):
terms.append(tok)
return terms
@@ -126,7 +134,7 @@ def _compute_idf(G: nx.Graph, terms: list[str]) -> dict[str, float]:
def _score_nodes(G: nx.Graph, terms: list[str]) -> list[tuple[float, str]]:
scored = []
norm_terms = [_strip_diacritics(t).lower() for t in terms]
norm_terms = [tok for t in terms for tok in _search_tokens(t)]
idf = _compute_idf(G, norm_terms)
for nid, data in G.nodes(data=True):
norm_label = data.get("norm_label") or _strip_diacritics(data.get("label") or "").lower()
@@ -415,7 +423,9 @@ def _find_node(G: nx.Graph, label: str) -> list[str]:
Results are ordered by three-tier precedence: exact match, then prefix match,
then substring match. Node-ID exact matches are grouped with label exact matches.
"""
term = _strip_diacritics(label).lower()
term = " ".join(_search_tokens(label))
if not term:
return []
exact: list[str] = []
prefix: list[str] = []
substring: list[str] = []
File diff suppressed because it is too large Load Diff
+2 -2
View File
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
[project]
name = "graphifyy"
version = "0.8.20"
version = "0.8.21"
description = "AI coding assistant skill (Claude Code, Codex, OpenCode, Cursor, Gemini CLI, Aider, OpenClaw, Factory Droid, Trae, Hermes, Kiro, Pi, Devin CLI, Google Antigravity) - turn any folder of code, docs, papers, images, or videos into a queryable knowledge graph"
readme = "README.md"
license = { file = "LICENSE" }
@@ -97,7 +97,7 @@ packages = ["graphify"]
include-package-data = false
[tool.setuptools.package-data]
graphify = ["skill.md", "skill-codex.md", "skill-opencode.md", "skill-aider.md", "skill-copilot.md", "skill-claw.md", "skill-windows.md", "skill-droid.md", "skill-trae.md", "skill-kiro.md", "skill-vscode.md", "skill-pi.md", "skill-devin.md"]
graphify = ["skill.md", "skill-codex.md", "skill-opencode.md", "skill-aider.md", "skill-amp.md", "skill-copilot.md", "skill-claw.md", "skill-windows.md", "skill-droid.md", "skill-trae.md", "skill-kiro.md", "skill-vscode.md", "skill-pi.md", "skill-devin.md"]
[tool.pytest.ini_options]
testpaths = ["tests"]
+16
View File
@@ -11,6 +11,7 @@ from graphify.serve import (
_pick_seeds,
_bfs,
_dfs,
_find_node,
_filter_graph_by_context,
_infer_context_filters,
_query_terms,
@@ -80,6 +81,21 @@ def test_score_nodes_source_file_partial():
assert "n2" in nids
def test_score_nodes_ignores_trailing_punctuation():
G = _make_graph()
scored = _score_nodes(G, ["extract?"])
assert scored[0][1] == "n1"
def test_find_node_ignores_trailing_punctuation():
G = _make_graph()
assert _find_node(G, "extract?") == ["n1"]
def test_query_terms_strips_search_punctuation():
assert _query_terms("what calls extract?") == ["what", "calls", "extract"]
def test_query_terms_filters_only_short_english_terms(monkeypatch):
import graphify.serve as serve_mod