fix 7 bugs: cluster-only --graph, _is_sensitive boundaries, max_tokens 16384, prune message clarity, svelte stub source_file, svelte static imports, manifest on full rebuild + pi skill YAML fix

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
Safi
2026-05-05 16:00:08 +01:00
co-authored by Claude Sonnet 4.6
parent ee85bbfbfd
commit 005a36608a
7 changed files with 130 additions and 13 deletions
+26 -2
View File
@@ -1075,6 +1075,7 @@ def main() -> None:
print(" (also: GRAPHIFY_FORCE=1 env var; use after refactors that delete code)")
print(" cluster-only <path> rerun clustering on an existing graph.json and regenerate report")
print(" --no-viz skip graph.html generation (useful for >5000 node graphs / CI)")
print(" --graph <path> path to graph.json (default <path>/graphify-out/graph.json)")
print(" query \"<question>\" BFS traversal of graph.json for a question")
print(" --dfs use depth-first instead of breadth-first")
print(" --context C explicit edge-context filter (repeatable)")
@@ -1494,11 +1495,30 @@ def main() -> None:
sys.exit(1)
elif cmd == "cluster-only":
watch_path = Path(sys.argv[2]) if len(sys.argv) > 2 else Path(".")
# Mirror the tree/export arg-parsing pattern: walk argv so flags and
# the optional positional path can appear in any order (#724).
no_viz = "--no-viz" in sys.argv
_min_cs_arg = next((a for a in sys.argv if a.startswith("--min-community-size=")), None)
min_community_size = int(_min_cs_arg.split("=")[1]) if _min_cs_arg else 3
graph_json = watch_path / "graphify-out" / "graph.json"
args = sys.argv[2:]
watch_path: Path | None = None
graph_override: Path | None = None
i_arg = 0
while i_arg < len(args):
a = args[i_arg]
if a == "--graph" and i_arg + 1 < len(args):
graph_override = Path(args[i_arg + 1]); i_arg += 2
elif a == "--no-viz" or a.startswith("--min-community-size="):
i_arg += 1
elif a.startswith("--"):
i_arg += 1
elif watch_path is None:
watch_path = Path(a); i_arg += 1
else:
i_arg += 1
if watch_path is None:
watch_path = Path(".")
graph_json = graph_override if graph_override is not None else watch_path / "graphify-out" / "graph.json"
if not graph_json.exists():
print(f"error: no graph found at {graph_json} — run /graphify first", file=sys.stderr)
sys.exit(1)
@@ -2122,6 +2142,10 @@ def main() -> None:
f"{merged['output_tokens']:,} out, "
f"est. cost: ${cost:.4f}"
)
try:
_save_manifest(files_by_type, manifest_path=str(manifest_path))
except Exception as exc:
print(f"[graphify extract] warning: could not write manifest: {exc}", file=sys.stderr)
sys.exit(0)
# Build graph + cluster + score + write.
+13 -2
View File
@@ -245,8 +245,19 @@ def build_merge(
if d.get("source_file") in prune_sources
]
G.remove_nodes_from(to_remove)
if to_remove:
print(f"[graphify] Pruned {len(to_remove)} node(s) from deleted sources.", file=sys.stderr)
n_files = len(prune_sources)
n_nodes = len(to_remove)
if n_nodes:
print(
f"[graphify] Pruned {n_nodes} node(s) from {n_files} deleted source file(s).",
file=sys.stderr,
)
else:
print(
f"[graphify] {n_files} source file(s) deleted since last run — "
f"no matching nodes in graph, already clean.",
file=sys.stderr,
)
# Safety check: refuse to shrink the graph silently (#479)
# Skip when dedup or prune_sources is active — shrinkage is intentional there.
+1 -1
View File
@@ -33,7 +33,7 @@ FILE_COUNT_UPPER = 200 # files - above this, warn about token cost
_SENSITIVE_PATTERNS = [
re.compile(r'(^|[\\/])\.(env|envrc)(\.|$)', re.IGNORECASE),
re.compile(r'\.(pem|key|p12|pfx|cert|crt|der|p8)$', re.IGNORECASE),
re.compile(r'(credential|secret|passwd|password|token|private_key)', re.IGNORECASE),
re.compile(r'\b(credential|secret|passwd|password|token|private_key)s?\b', re.IGNORECASE),
re.compile(r'(id_rsa|id_dsa|id_ecdsa|id_ed25519)(\.pub)?$'),
re.compile(r'(\.netrc|\.pgpass|\.htpasswd)$', re.IGNORECASE),
re.compile(r'(aws_credentials|gcloud_credentials|service.account)', re.IGNORECASE),
+63 -1
View File
@@ -1767,6 +1767,7 @@ def extract_svelte(path: Path) -> dict:
elif resolved.suffix == ".jsx":
resolved = resolved.with_suffix(".tsx")
node_id = _make_id(str(resolved))
stub_source_file = str(resolved)
else:
# Check tsconfig.json path aliases (e.g. "$lib/" -> "src/lib/", "@/" -> "src/")
# before treating as external. Mirrors _import_js logic so SvelteKit alias
@@ -1779,6 +1780,7 @@ def extract_svelte(path: Path) -> dict:
break
if resolved_alias is not None:
node_id = _make_id(str(resolved_alias))
stub_source_file = str(resolved_alias)
else:
# Bare/scoped import (node_modules) - use last segment;
# build_from_json drops as external if no matching node exists.
@@ -1786,6 +1788,7 @@ def extract_svelte(path: Path) -> dict:
if not module_name:
continue
node_id = _make_id(module_name)
stub_source_file = raw
if node_id in existing_ids:
# Edge target already a real node - just add the edge, don't add a node.
result.setdefault("edges", []).append({
@@ -1796,7 +1799,7 @@ def extract_svelte(path: Path) -> dict:
continue
result.setdefault("nodes", []).append({
"id": node_id, "label": raw,
"file_type": "code", "source_file": str(path),
"file_type": "code", "source_file": stub_source_file,
"confidence": "EXTRACTED",
})
result.setdefault("edges", []).append({
@@ -1805,6 +1808,65 @@ def extract_svelte(path: Path) -> dict:
"source_file": str(path),
})
existing_ids.add(node_id)
# Static imports inside <script> blocks. The JS tree-sitter parser fed
# the full .svelte file produces a top-level ERROR node (HTML markup
# is not valid JS), so import_statement nodes are never reached and
# static imports are silently dropped (#713). Regex over each script
# body recovers them.
script_re = _re.compile(
r"<script\b[^>]*>([\s\S]*?)</script\s*>", _re.IGNORECASE
)
static_import_re = _re.compile(
r"""import\s+(?:[^'"`;]+?\s+from\s+)?['"]([^'"]+)['"]"""
)
for script_match in script_re.finditer(src):
script_body = script_match.group(1)
for m in static_import_re.finditer(script_body):
raw = m.group(1)
if not raw:
continue
if raw.startswith("."):
resolved = Path(os.path.normpath(path.parent / raw))
if resolved.suffix == ".js":
resolved = resolved.with_suffix(".ts")
elif resolved.suffix == ".jsx":
resolved = resolved.with_suffix(".tsx")
node_id = _make_id(str(resolved))
stub_source_file = str(resolved)
else:
resolved_alias = None
for alias_prefix, alias_base in aliases.items():
if raw == alias_prefix or raw.startswith(alias_prefix + "/"):
rest = raw[len(alias_prefix):].lstrip("/")
resolved_alias = Path(os.path.normpath(Path(alias_base) / rest))
break
if resolved_alias is not None:
node_id = _make_id(str(resolved_alias))
stub_source_file = str(resolved_alias)
else:
module_name = raw.split("/")[-1]
if not module_name:
continue
node_id = _make_id(module_name)
stub_source_file = raw
if node_id in existing_ids:
result.setdefault("edges", []).append({
"source": file_node_id, "target": node_id,
"relation": "imports_from", "confidence": "EXTRACTED",
"source_file": str(path),
})
continue
result.setdefault("nodes", []).append({
"id": node_id, "label": raw,
"file_type": "code", "source_file": stub_source_file,
"confidence": "EXTRACTED",
})
result.setdefault("edges", []).append({
"source": file_node_id, "target": node_id,
"relation": "imports_from", "confidence": "EXTRACTED",
"source_file": str(path),
})
existing_ids.add(node_id)
except Exception:
pass
return result
+22 -5
View File
@@ -50,6 +50,7 @@ BACKENDS: dict[str, dict] = {
"env_key": "ANTHROPIC_API_KEY",
"pricing": {"input": 3.0, "output": 15.0}, # USD per 1M tokens
"temperature": 0,
"max_tokens": 16384,
},
"kimi": {
"base_url": "https://api.moonshot.ai/v1",
@@ -57,9 +58,23 @@ BACKENDS: dict[str, dict] = {
"env_key": "MOONSHOT_API_KEY",
"pricing": {"input": 0.74, "output": 4.66}, # USD per 1M tokens
"temperature": None, # kimi-k2.6 enforces its own fixed temperature; sending any value raises 400
"max_tokens": 16384,
},
}
def _resolve_max_tokens(default: int) -> int:
"""Honour GRAPHIFY_MAX_OUTPUT_TOKENS env var override, else use backend default."""
raw = os.environ.get("GRAPHIFY_MAX_OUTPUT_TOKENS", "").strip()
if raw:
try:
v = int(raw)
if v > 0:
return v
except ValueError:
pass
return default
_EXTRACTION_SYSTEM = """\
You are a graphify semantic extraction agent. Extract a knowledge graph fragment from the files provided.
Output ONLY valid JSON — no explanation, no markdown fences, no preamble.
@@ -113,6 +128,7 @@ def _call_openai_compat(
model: str,
user_message: str,
temperature: float | None = 0,
max_tokens: int = 8192,
) -> dict:
"""Call any OpenAI-compatible API (Kimi, OpenAI, etc.) and return parsed JSON."""
try:
@@ -130,7 +146,7 @@ def _call_openai_compat(
{"role": "system", "content": _EXTRACTION_SYSTEM},
{"role": "user", "content": user_message},
],
"max_completion_tokens": 8192,
"max_completion_tokens": max_tokens,
}
if temperature is not None:
kwargs["temperature"] = temperature
@@ -149,7 +165,7 @@ def _call_openai_compat(
return result
def _call_claude(api_key: str, model: str, user_message: str) -> dict:
def _call_claude(api_key: str, model: str, user_message: str, max_tokens: int = 8192) -> dict:
"""Call Anthropic Claude directly (not via OpenAI compat layer)."""
try:
import anthropic
@@ -162,7 +178,7 @@ def _call_claude(api_key: str, model: str, user_message: str) -> dict:
client = anthropic.Anthropic(api_key=api_key)
resp = client.messages.create(
model=model,
max_tokens=8192,
max_tokens=max_tokens,
system=_EXTRACTION_SYSTEM,
messages=[{"role": "user", "content": user_message}],
)
@@ -201,11 +217,12 @@ def extract_files_direct(
)
mdl = model or cfg["default_model"]
user_msg = _read_files(files, root)
max_out = _resolve_max_tokens(cfg.get("max_tokens", 8192))
if backend == "claude":
return _call_claude(key, mdl, user_msg)
return _call_claude(key, mdl, user_msg, max_tokens=max_out)
else:
return _call_openai_compat(cfg["base_url"], key, mdl, user_msg, temperature=cfg.get("temperature", 0))
return _call_openai_compat(cfg["base_url"], key, mdl, user_msg, temperature=cfg.get("temperature", 0), max_tokens=max_out)
def _estimate_file_tokens(path: Path) -> int:
+4 -1
View File
@@ -843,7 +843,10 @@ deleted = set(incremental.get('deleted_files', []))
if deleted:
to_remove = [n for n, d in G_existing.nodes(data=True) if d.get('source_file') in deleted]
G_existing.remove_nodes_from(to_remove)
print(f'Pruned {len(to_remove)} ghost nodes from {len(deleted)} deleted file(s)')
if to_remove:
print(f'Pruned {len(to_remove)} ghost node(s) from {len(deleted)} deleted file(s).')
else:
print(f'{len(deleted)} file(s) deleted since last run — no ghost nodes in graph, already clean.')
# Merge: new nodes/edges into existing graph
G_existing.update(G_new)
+1 -1
View File
@@ -1,6 +1,6 @@
---
name: graphify
description: Turn any folder of files (code, docs, papers, images, video) into a queryable knowledge graph with community detection, an honest audit trail, and three outputs: interactive HTML, GraphRAG-ready JSON, and a plain-language GRAPH_REPORT.md. Use when asked to analyze a codebase, understand architecture, map dependencies, or build a knowledge graph.
description: any input (code, docs, papers, images, video) → knowledge graph → clustered communities → HTML + JSON + GRAPH_REPORT.md
---
# /graphify