fix: OpenCode project path, hook loop guard, deterministic output, Amp platform, punctuation search, builtin god-node filter, .svh Verilog (#1040, #1018, #1037, #948, #994, #916, #1042)
Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Sonnet 4.6
parent
a54a542b6c
commit
80301a06bf
@@ -2,6 +2,17 @@
|
||||
|
||||
Full release notes with details on each version: [GitHub Releases](https://github.com/safishamsi/graphify/releases)
|
||||
|
||||
## 0.8.21 (2026-05-27)
|
||||
|
||||
- Fix: `graphify update` (no `--changed` flag) no longer leaves ghost nodes from files deleted between runs — full re-extraction path now reconciles the existing graph against current disk state and evicts any node whose `source_file` no longer exists; `_norm_source_file` used on both sides to guarantee path format consistency (#1007)
|
||||
- Fix: `graphify install --platform opencode --project` now writes `SKILL.md` to `.opencode/skills/graphify/SKILL.md` (discoverable by OpenCode) instead of the incorrect `.config/opencode/skills/` path; git-add hint updated accordingly (#1040)
|
||||
- Fix: post-commit hook no longer triggers rebuild when only `graphify-out/` files were committed (avoids infinite dirty-tree loop when graph outputs are tracked in git); `GRAPHIFY_SKIP_HOOK=1` env var added for one-off skip; hook rebuild log now appends (`>>`) instead of overwriting (`>`) (#1018, #1037)
|
||||
- Fix: graph output is now byte-for-byte deterministic across runs — edges sorted by `(source, target, relation)` in `build_from_json`; `PYTHONHASHSEED=0` exported in hook scripts to stabilize Louvain community ordering (#1010)
|
||||
- Feat: Amp (ampcode.com) platform support — `graphify amp install/uninstall` installs the skill into `.amp/skills/graphify/SKILL.md` (#948)
|
||||
- Fix: query punctuation no longer breaks node matching — `"what calls extract?"` correctly finds the `extract` node; `_search_tokens` helper strips punctuation from search terms in `_query_terms`, `_score_nodes`, and `_find_node` (#994, #978)
|
||||
- Fix: language built-in globals (`String`, `Number`, `Boolean`, `Object`, `Array`, etc.) no longer accumulate spurious call edges — filtered at same-file and cross-file resolution in the AST extractor, eliminating god-node pollution from constructor-style calls (#916, #726)
|
||||
- Feat: SystemVerilog header files (`.svh`) now extracted using the Verilog parser alongside `.v` and `.sv` (#1042)
|
||||
|
||||
## 0.8.20 (2026-05-26)
|
||||
|
||||
- Fix: stale nodes persist after `graphify update` when files are deleted on Windows — `deleted_paths` and `evict_sources` in `_rebuild_code` now use `.as_posix()` for consistent forward-slash paths; `_relativize_source_files` called on the existing graph before eviction (not after); `_relativize_source_files` itself now produces forward slashes (#1007)
|
||||
|
||||
+13
-3
@@ -77,6 +77,11 @@ def _platform_skill_destination(platform_name: str, *, project: bool = False, pr
|
||||
return Path.home() / ".agents" / "skills" / "graphify" / "SKILL.md"
|
||||
return Path.home() / ".gemini" / "skills" / "graphify" / "SKILL.md"
|
||||
|
||||
if platform_name == "opencode":
|
||||
if project:
|
||||
return (project_dir or Path(".")) / ".opencode" / "skills" / "graphify" / "SKILL.md"
|
||||
return Path.home() / ".config" / "opencode" / "skills" / "graphify" / "SKILL.md"
|
||||
|
||||
if platform_name == "devin":
|
||||
if project:
|
||||
return (project_dir or Path(".")) / ".devin" / "skills" / "graphify" / "SKILL.md"
|
||||
@@ -289,6 +294,11 @@ _PLATFORM_CONFIG: dict[str, dict] = {
|
||||
"skill_dst": Path(".kimi") / "skills" / "graphify" / "SKILL.md",
|
||||
"claude_md": False,
|
||||
},
|
||||
"amp": {
|
||||
"skill_file": "skill-amp.md",
|
||||
"skill_dst": Path(".amp") / "skills" / "graphify" / "SKILL.md",
|
||||
"claude_md": False,
|
||||
},
|
||||
"devin": {
|
||||
"skill_file": "skill-devin.md",
|
||||
# User scope: ~/.config/devin/skills/graphify/SKILL.md
|
||||
@@ -1120,7 +1130,7 @@ def _project_install(platform_name: str, project_dir: Path | None = None) -> Non
|
||||
elif platform_name == "kiro":
|
||||
_kiro_install(project_dir)
|
||||
_print_project_git_add_hint([project_dir / ".kiro"])
|
||||
elif platform_name in ("aider", "codex", "opencode", "claw", "droid", "trae", "trae-cn", "hermes"):
|
||||
elif platform_name in ("aider", "amp", "codex", "opencode", "claw", "droid", "trae", "trae-cn", "hermes"):
|
||||
skill_dst = _copy_skill_file(platform_name, project=True, project_dir=project_dir)
|
||||
_agents_install(project_dir, platform_name)
|
||||
hint_paths = [_project_scope_root(skill_dst, project_dir), project_dir / "AGENTS.md"]
|
||||
@@ -1153,7 +1163,7 @@ def _project_uninstall(platform_name: str, project_dir: Path | None = None) -> N
|
||||
_cursor_uninstall(project_dir)
|
||||
elif platform_name == "kiro":
|
||||
_kiro_uninstall(project_dir)
|
||||
elif platform_name in ("aider", "codex", "opencode", "claw", "droid", "trae", "trae-cn", "hermes"):
|
||||
elif platform_name in ("aider", "amp", "codex", "opencode", "claw", "droid", "trae", "trae-cn", "hermes"):
|
||||
_remove_skill_file(platform_name, project=True, project_dir=project_dir)
|
||||
_agents_uninstall(project_dir, platform=platform_name)
|
||||
if platform_name == "codex":
|
||||
@@ -1729,7 +1739,7 @@ def main() -> None:
|
||||
else:
|
||||
print("Usage: graphify pi [install|uninstall]", file=sys.stderr)
|
||||
sys.exit(1)
|
||||
elif cmd in ("aider", "codex", "opencode", "claw", "droid", "trae", "trae-cn", "hermes"):
|
||||
elif cmd in ("aider", "amp", "codex", "opencode", "claw", "droid", "trae", "trae-cn", "hermes"):
|
||||
subcmd = sys.argv[2] if len(sys.argv) > 2 else ""
|
||||
if subcmd == "install":
|
||||
if "--project" in sys.argv[3:]:
|
||||
|
||||
+1
-1
@@ -25,7 +25,7 @@ class FileType(str, Enum):
|
||||
|
||||
_MANIFEST_PATH = "graphify-out/manifest.json"
|
||||
|
||||
CODE_EXTENSIONS = {'.py', '.ts', '.tsx', '.js', '.jsx', '.mjs', '.ejs', '.ets', '.go', '.rs', '.java', '.groovy', '.gradle', '.cpp', '.cc', '.cxx', '.c', '.h', '.hpp', '.rb', '.swift', '.kt', '.kts', '.cs', '.scala', '.php', '.lua', '.luau', '.toc', '.zig', '.ps1', '.ex', '.exs', '.m', '.mm', '.jl', '.vue', '.svelte', '.astro', '.dart', '.v', '.sv', '.sql', '.r', '.f', '.F', '.f90', '.F90', '.f95', '.F95', '.f03', '.F03', '.f08', '.F08', '.pas', '.pp', '.dpr', '.dpk', '.lpr', '.inc', '.dfm', '.lfm', '.lpk', '.sh', '.bash', '.json', '.sln', '.csproj', '.fsproj', '.vbproj', '.razor', '.cshtml'}
|
||||
CODE_EXTENSIONS = {'.py', '.ts', '.tsx', '.js', '.jsx', '.mjs', '.ejs', '.ets', '.go', '.rs', '.java', '.groovy', '.gradle', '.cpp', '.cc', '.cxx', '.c', '.h', '.hpp', '.rb', '.swift', '.kt', '.kts', '.cs', '.scala', '.php', '.lua', '.luau', '.toc', '.zig', '.ps1', '.ex', '.exs', '.m', '.mm', '.jl', '.vue', '.svelte', '.astro', '.dart', '.v', '.sv', '.svh', '.sql', '.r', '.f', '.F', '.f90', '.F90', '.f95', '.F95', '.f03', '.F03', '.f08', '.F08', '.pas', '.pp', '.dpr', '.dpk', '.lpr', '.inc', '.dfm', '.lfm', '.lpk', '.sh', '.bash', '.json', '.sln', '.csproj', '.fsproj', '.vbproj', '.razor', '.cshtml'}
|
||||
DOC_EXTENSIONS = {'.md', '.mdx', '.qmd', '.txt', '.rst', '.html', '.yaml', '.yml'}
|
||||
PAPER_EXTENSIONS = {'.pdf'}
|
||||
IMAGE_EXTENSIONS = {'.png', '.jpg', '.jpeg', '.gif', '.webp', '.svg'}
|
||||
|
||||
+33
-4
@@ -14,6 +14,32 @@ from .mcp_ingest import extract_mcp_config, is_mcp_config_path
|
||||
|
||||
_RECURSION_LIMIT = 10_000
|
||||
|
||||
# Language built-in globals that AST may classify as call targets when used as
|
||||
# constructors or coercion functions (e.g. String(x), Number(x), Boolean(x)).
|
||||
# Without this filter they become god-nodes accumulating spurious edges from
|
||||
# every call site. Filter applied at same-file and cross-file resolution.
|
||||
# See issue #726.
|
||||
_LANGUAGE_BUILTIN_GLOBALS: frozenset[str] = frozenset({
|
||||
# JavaScript / TypeScript ECMAScript built-ins
|
||||
"String", "Number", "Boolean", "Object", "Array", "Symbol", "BigInt",
|
||||
"Date", "RegExp", "Error", "TypeError", "RangeError", "SyntaxError",
|
||||
"ReferenceError", "EvalError", "URIError",
|
||||
"Promise", "Map", "Set", "WeakMap", "WeakSet", "JSON", "Math",
|
||||
"Reflect", "Proxy", "Intl",
|
||||
"parseInt", "parseFloat", "isNaN", "isFinite",
|
||||
"encodeURIComponent", "decodeURIComponent", "encodeURI", "decodeURI",
|
||||
# Browser / Node common globals
|
||||
"URL", "URLSearchParams", "FormData", "Blob", "File",
|
||||
"Headers", "Request", "Response", "AbortController", "AbortSignal",
|
||||
"TextEncoder", "TextDecoder", "console",
|
||||
# Python built-in callables
|
||||
"str", "int", "float", "bool", "list", "dict", "set", "tuple", "bytes",
|
||||
"len", "range", "enumerate", "zip", "map", "filter", "sum", "min", "max",
|
||||
"print", "open", "isinstance", "type", "super", "sorted", "reversed",
|
||||
"any", "all", "abs", "round", "next", "iter", "hash", "id", "repr",
|
||||
"callable", "getattr", "setattr", "hasattr", "delattr", "vars", "dir",
|
||||
})
|
||||
|
||||
|
||||
def _raise_recursion_limit() -> None:
|
||||
if sys.getrecursionlimit() < _RECURSION_LIMIT:
|
||||
@@ -2285,7 +2311,7 @@ def _extract_generic(path: Path, config: LanguageConfig) -> dict:
|
||||
# Try reading the node directly (e.g. Java name field is the callee)
|
||||
callee_name = _read_text(func_node, source)
|
||||
|
||||
if callee_name:
|
||||
if callee_name and callee_name not in _LANGUAGE_BUILTIN_GLOBALS:
|
||||
tgt_nid = label_to_nid.get(callee_name)
|
||||
if tgt_nid and tgt_nid != caller_nid:
|
||||
pair = (caller_nid, tgt_nid)
|
||||
@@ -4136,7 +4162,7 @@ def extract_go(path: Path) -> dict:
|
||||
is_member_call = receiver_name not in go_imported_pkgs
|
||||
if field:
|
||||
callee_name = _read_text(field, source)
|
||||
if callee_name:
|
||||
if callee_name and callee_name not in _LANGUAGE_BUILTIN_GLOBALS:
|
||||
tgt_nid = label_to_nid.get(callee_name)
|
||||
if tgt_nid and tgt_nid != caller_nid:
|
||||
pair = (caller_nid, tgt_nid)
|
||||
@@ -4337,7 +4363,7 @@ def extract_rust(path: Path) -> dict:
|
||||
name = func_node.child_by_field_name("name")
|
||||
if name:
|
||||
callee_name = _read_text(name, source)
|
||||
if callee_name:
|
||||
if callee_name and callee_name not in _LANGUAGE_BUILTIN_GLOBALS:
|
||||
tgt_nid = label_to_nid.get(callee_name)
|
||||
if tgt_nid and tgt_nid != caller_nid:
|
||||
pair = (caller_nid, tgt_nid)
|
||||
@@ -6456,7 +6482,7 @@ def extract_elixir(path: Path) -> dict:
|
||||
if child.type == "identifier":
|
||||
callee_name = source[child.start_byte:child.end_byte].decode("utf-8", errors="replace")
|
||||
break
|
||||
if callee_name:
|
||||
if callee_name and callee_name not in _LANGUAGE_BUILTIN_GLOBALS:
|
||||
tgt_nid = label_to_nid.get(callee_name)
|
||||
if tgt_nid and tgt_nid != caller_nid:
|
||||
pair = (caller_nid, tgt_nid)
|
||||
@@ -8245,6 +8271,7 @@ _DISPATCH: dict[str, Any] = {
|
||||
".dart": extract_dart,
|
||||
".v": extract_verilog,
|
||||
".sv": extract_verilog,
|
||||
".svh": extract_verilog,
|
||||
".sql": extract_sql,
|
||||
".md": extract_markdown,
|
||||
".mdx": extract_markdown,
|
||||
@@ -8630,6 +8657,8 @@ def extract(
|
||||
callee = rc.get("callee", "")
|
||||
if not callee:
|
||||
continue
|
||||
if callee in _LANGUAGE_BUILTIN_GLOBALS:
|
||||
continue
|
||||
# Skip member-call callees: obj.log() → "log" has no import evidence
|
||||
# and collides with any top-level function named "log" in the corpus.
|
||||
if rc.get("is_member_call"):
|
||||
|
||||
+10
-2
@@ -60,11 +60,19 @@ GIT_DIR=$(git rev-parse --git-dir 2>/dev/null)
|
||||
[ -f "$GIT_DIR/MERGE_HEAD" ] && exit 0
|
||||
[ -f "$GIT_DIR/CHERRY_PICK_HEAD" ] && exit 0
|
||||
|
||||
[ "${GRAPHIFY_SKIP_HOOK:-0}" = "1" ] && exit 0
|
||||
|
||||
CHANGED=$(git diff --name-only HEAD~1 HEAD 2>/dev/null || git diff --name-only HEAD 2>/dev/null)
|
||||
if [ -z "$CHANGED" ]; then
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# Skip when only graphify-out/ artifacts changed (avoids rebuild loop when graph outputs are tracked in git)
|
||||
_NON_GRAPH=$(echo "$CHANGED" | grep -v '^graphify-out/' || true)
|
||||
if [ -z "$_NON_GRAPH" ]; then
|
||||
exit 0
|
||||
fi
|
||||
|
||||
""" + _PYTHON_DETECT + """
|
||||
export GRAPHIFY_CHANGED="$CHANGED"
|
||||
|
||||
@@ -100,7 +108,7 @@ except TimeoutError as exc:
|
||||
except Exception as exc:
|
||||
print(f'[graphify hook] Rebuild failed: {exc}')
|
||||
sys.exit(1)
|
||||
" > "$_GRAPHIFY_LOG" 2>&1 < /dev/null &
|
||||
" >> "$_GRAPHIFY_LOG" 2>&1 < /dev/null &
|
||||
disown 2>/dev/null || true
|
||||
# graphify-hook-end
|
||||
"""
|
||||
@@ -162,7 +170,7 @@ except TimeoutError as exc:
|
||||
except Exception as exc:
|
||||
print(f'[graphify] Rebuild failed: {exc}')
|
||||
sys.exit(1)
|
||||
" > "$_GRAPHIFY_LOG" 2>&1 < /dev/null &
|
||||
" >> "$_GRAPHIFY_LOG" 2>&1 < /dev/null &
|
||||
disown 2>/dev/null || true
|
||||
# graphify-checkout-hook-end
|
||||
"""
|
||||
|
||||
+20
-10
@@ -2,6 +2,7 @@
|
||||
from __future__ import annotations
|
||||
import json
|
||||
import math
|
||||
import re
|
||||
import sys
|
||||
from pathlib import Path
|
||||
import networkx as nx
|
||||
@@ -56,6 +57,11 @@ def _strip_diacritics(text: str) -> str:
|
||||
return "".join(c for c in nfkd if not unicodedata.combining(c))
|
||||
|
||||
|
||||
def _search_tokens(text: str) -> list[str]:
|
||||
"""Split text into word tokens, stripping punctuation and diacritics."""
|
||||
return re.findall(r"\w+", _strip_diacritics(str(text)).lower())
|
||||
|
||||
|
||||
def _has_chinese(text: str) -> bool:
|
||||
return any("一" <= ch <= "鿿" for ch in text)
|
||||
|
||||
@@ -82,14 +88,16 @@ def _query_terms(question: str) -> list[str]:
|
||||
"""Split a query into searchable terms, segmenting Chinese text."""
|
||||
terms: list[str] = []
|
||||
for raw in question.split():
|
||||
term = raw.lower().strip()
|
||||
if not term:
|
||||
continue
|
||||
candidates = _segment_chinese(term) if _has_chinese(term) else [term]
|
||||
for seg in candidates:
|
||||
seg = seg.strip()
|
||||
if seg and _is_searchable(seg):
|
||||
terms.append(seg)
|
||||
if _has_chinese(raw):
|
||||
for seg in _segment_chinese(raw.lower().strip()):
|
||||
seg = seg.strip()
|
||||
if seg and _is_searchable(seg):
|
||||
terms.append(seg)
|
||||
else:
|
||||
# Strip punctuation without touching Unicode characters (avoid NFKD mangling non-Latin scripts)
|
||||
for tok in re.findall(r"\w+", raw.lower()):
|
||||
if _is_searchable(tok):
|
||||
terms.append(tok)
|
||||
return terms
|
||||
|
||||
|
||||
@@ -126,7 +134,7 @@ def _compute_idf(G: nx.Graph, terms: list[str]) -> dict[str, float]:
|
||||
|
||||
def _score_nodes(G: nx.Graph, terms: list[str]) -> list[tuple[float, str]]:
|
||||
scored = []
|
||||
norm_terms = [_strip_diacritics(t).lower() for t in terms]
|
||||
norm_terms = [tok for t in terms for tok in _search_tokens(t)]
|
||||
idf = _compute_idf(G, norm_terms)
|
||||
for nid, data in G.nodes(data=True):
|
||||
norm_label = data.get("norm_label") or _strip_diacritics(data.get("label") or "").lower()
|
||||
@@ -415,7 +423,9 @@ def _find_node(G: nx.Graph, label: str) -> list[str]:
|
||||
Results are ordered by three-tier precedence: exact match, then prefix match,
|
||||
then substring match. Node-ID exact matches are grouped with label exact matches.
|
||||
"""
|
||||
term = _strip_diacritics(label).lower()
|
||||
term = " ".join(_search_tokens(label))
|
||||
if not term:
|
||||
return []
|
||||
exact: list[str] = []
|
||||
prefix: list[str] = []
|
||||
substring: list[str] = []
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
+2
-2
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
||||
|
||||
[project]
|
||||
name = "graphifyy"
|
||||
version = "0.8.20"
|
||||
version = "0.8.21"
|
||||
description = "AI coding assistant skill (Claude Code, Codex, OpenCode, Cursor, Gemini CLI, Aider, OpenClaw, Factory Droid, Trae, Hermes, Kiro, Pi, Devin CLI, Google Antigravity) - turn any folder of code, docs, papers, images, or videos into a queryable knowledge graph"
|
||||
readme = "README.md"
|
||||
license = { file = "LICENSE" }
|
||||
@@ -97,7 +97,7 @@ packages = ["graphify"]
|
||||
include-package-data = false
|
||||
|
||||
[tool.setuptools.package-data]
|
||||
graphify = ["skill.md", "skill-codex.md", "skill-opencode.md", "skill-aider.md", "skill-copilot.md", "skill-claw.md", "skill-windows.md", "skill-droid.md", "skill-trae.md", "skill-kiro.md", "skill-vscode.md", "skill-pi.md", "skill-devin.md"]
|
||||
graphify = ["skill.md", "skill-codex.md", "skill-opencode.md", "skill-aider.md", "skill-amp.md", "skill-copilot.md", "skill-claw.md", "skill-windows.md", "skill-droid.md", "skill-trae.md", "skill-kiro.md", "skill-vscode.md", "skill-pi.md", "skill-devin.md"]
|
||||
|
||||
[tool.pytest.ini_options]
|
||||
testpaths = ["tests"]
|
||||
|
||||
@@ -11,6 +11,7 @@ from graphify.serve import (
|
||||
_pick_seeds,
|
||||
_bfs,
|
||||
_dfs,
|
||||
_find_node,
|
||||
_filter_graph_by_context,
|
||||
_infer_context_filters,
|
||||
_query_terms,
|
||||
@@ -80,6 +81,21 @@ def test_score_nodes_source_file_partial():
|
||||
assert "n2" in nids
|
||||
|
||||
|
||||
def test_score_nodes_ignores_trailing_punctuation():
|
||||
G = _make_graph()
|
||||
scored = _score_nodes(G, ["extract?"])
|
||||
assert scored[0][1] == "n1"
|
||||
|
||||
|
||||
def test_find_node_ignores_trailing_punctuation():
|
||||
G = _make_graph()
|
||||
assert _find_node(G, "extract?") == ["n1"]
|
||||
|
||||
|
||||
def test_query_terms_strips_search_punctuation():
|
||||
assert _query_terms("what calls extract?") == ["what", "calls", "extract"]
|
||||
|
||||
|
||||
def test_query_terms_filters_only_short_english_terms(monkeypatch):
|
||||
import graphify.serve as serve_mod
|
||||
|
||||
|
||||
Reference in New Issue
Block a user