- extract/_make_id + build/_normalize_id: use NFKC normalization and casefold so composed/decomposed Unicode forms produce the same ID; collapse consecutive underscores; both functions are now byte-for-byte equivalent (#811) - dedup: use explicit key-presence check instead of `or` for source/from fallback; pop stale from/to keys so they don't leak into graph.json attrs (#803) - skill --update: use build_merge() to avoid NetworkX round-trip direction flip; fix dict merge ordering so explicit source/target win; pull hyperedges from G.graph (merged) not new_extraction only (#801) - skill subagents: inject absolute CHUNK_PATH so Write tool doesn't lose chunk files to undefined cwd (#808) - __main__: skip skill version check during hook-check (runs on every editor tool use, must be silent); move warning to stderr Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Sonnet 4.6
parent
4cec58e072
commit
95e2c5eb32
@@ -33,7 +33,7 @@ def _check_skill_version(skill_dst: Path) -> None:
|
||||
return
|
||||
installed = version_file.read_text(encoding="utf-8").strip()
|
||||
if installed != __version__:
|
||||
print(f" warning: skill is from graphify {installed}, package is {__version__}. Run 'graphify install' to update.")
|
||||
print(f" warning: skill is from graphify {installed}, package is {__version__}. Run 'graphify install' to update.", file=sys.stderr)
|
||||
|
||||
|
||||
def _refresh_all_version_stamps() -> None:
|
||||
@@ -1115,8 +1115,10 @@ def _clone_repo(url: str, branch: str | None = None, out_dir: Path | None = None
|
||||
def main() -> None:
|
||||
# Check all known skill install locations for a stale version stamp.
|
||||
# Skip during install/uninstall (hook writes trigger a fresh check anyway).
|
||||
# Skip during hook-check — it runs on every editor tool use and must be silent.
|
||||
# Deduplicate paths so platforms sharing the same install dir don't warn twice.
|
||||
if not any(arg in ("install", "uninstall") for arg in sys.argv):
|
||||
_silent_cmds = {"install", "uninstall", "hook-check"}
|
||||
if not any(arg in _silent_cmds for arg in sys.argv):
|
||||
for skill_dst in {Path.home() / cfg["skill_dst"] for cfg in _PLATFORM_CONFIG.values()}:
|
||||
_check_skill_version(skill_dst)
|
||||
|
||||
|
||||
+9
-4
@@ -24,19 +24,24 @@ from __future__ import annotations
|
||||
import json
|
||||
import re
|
||||
import sys
|
||||
import unicodedata
|
||||
from pathlib import Path
|
||||
import networkx as nx
|
||||
from .validate import validate_extraction
|
||||
|
||||
|
||||
def _normalize_id(s: str) -> str:
|
||||
"""Normalize an ID string the same way extract._make_id does.
|
||||
r"""Normalize an ID string the same way extract._make_id does.
|
||||
|
||||
Used to reconcile edge endpoints when the LLM generates IDs with slightly
|
||||
different punctuation or casing than the AST extractor.
|
||||
different punctuation or casing than the AST extractor. Must stay in sync
|
||||
with extract._make_id — NFKC normalization, \w with re.UNICODE, underscore
|
||||
collapse, and casefold must all match (#811).
|
||||
"""
|
||||
cleaned = re.sub(r"[^a-zA-Z0-9]+", "_", s)
|
||||
return cleaned.strip("_").lower()
|
||||
s = unicodedata.normalize("NFKC", s)
|
||||
cleaned = re.sub(r"[^\w]+", "_", s, flags=re.UNICODE)
|
||||
cleaned = re.sub(r"_+", "_", cleaned)
|
||||
return cleaned.strip("_").casefold()
|
||||
|
||||
|
||||
def _norm_source_file(p: str | None) -> str | None:
|
||||
|
||||
+14
-2
@@ -233,8 +233,20 @@ def deduplicate_entities(
|
||||
deduped_edges = []
|
||||
for edge in edges:
|
||||
e = dict(edge)
|
||||
e["source"] = remap.get(e["source"], e["source"])
|
||||
e["target"] = remap.get(e["target"], e["target"])
|
||||
# Tolerate "from"/"to" keys from LLM backends that don't follow the
|
||||
# schema exactly — build_from_json normalises later but dedup runs
|
||||
# first so bracket access would KeyError here (#803).
|
||||
# Use explicit key presence check (not `or`) so empty-string src/tgt
|
||||
# aren't silently replaced by the fallback key.
|
||||
src = e["source"] if "source" in e else e.get("from")
|
||||
tgt = e["target"] if "target" in e else e.get("to")
|
||||
if src is None or tgt is None:
|
||||
continue
|
||||
e["source"] = remap.get(src, src)
|
||||
e["target"] = remap.get(tgt, tgt)
|
||||
# Remove legacy keys so they don't leak into edge attrs in graph.json.
|
||||
e.pop("from", None)
|
||||
e.pop("to", None)
|
||||
if e["source"] != e["target"]:
|
||||
deduped_edges.append(e)
|
||||
|
||||
|
||||
+13
-3
@@ -5,6 +5,7 @@ import json
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
import unicodedata
|
||||
from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
from typing import Callable, Any
|
||||
@@ -30,10 +31,19 @@ def _safe_extract(extractor: Callable, path: Path) -> dict:
|
||||
|
||||
|
||||
def _make_id(*parts: str) -> str:
|
||||
"""Build a stable node ID from one or more name parts."""
|
||||
r"""Build a stable node ID from one or more name parts.
|
||||
|
||||
Preserves Unicode letters/digits (CJK, Cyrillic, Arabic, accented Latin,
|
||||
etc.) so non-ASCII identifiers produce distinct IDs and don't collapse to
|
||||
a single per-file node (#811). NFKC normalization ensures composed and
|
||||
decomposed forms of the same character (e.g. é vs e+combining-acute)
|
||||
produce the same ID. Must stay in sync with build._normalize_id.
|
||||
"""
|
||||
combined = "_".join(p.strip("_.") for p in parts if p)
|
||||
cleaned = re.sub(r"[^a-zA-Z0-9]+", "_", combined)
|
||||
return cleaned.strip("_").lower()
|
||||
combined = unicodedata.normalize("NFKC", combined)
|
||||
cleaned = re.sub(r"[^\w]+", "_", combined, flags=re.UNICODE)
|
||||
cleaned = re.sub(r"_+", "_", cleaned)
|
||||
return cleaned.strip("_").casefold()
|
||||
|
||||
|
||||
def _file_stem(path: Path) -> str:
|
||||
|
||||
+40
-35
@@ -282,7 +282,15 @@ Concrete example for 3 chunks:
|
||||
```
|
||||
All three in one message. Not three separate messages.
|
||||
|
||||
Each subagent receives this exact prompt (substitute FILE_LIST, CHUNK_NUM, TOTAL_CHUNKS, and DEEP_MODE):
|
||||
Each subagent receives this exact prompt (substitute FILE_LIST, CHUNK_NUM, TOTAL_CHUNKS, DEEP_MODE, and CHUNK_PATH).
|
||||
|
||||
CHUNK_PATH must be an **absolute** path — derive it before dispatching:
|
||||
```bash
|
||||
PROJECT_ROOT=$(cat graphify-out/.graphify_root)
|
||||
# Then for chunk N: CHUNK_PATH="${PROJECT_ROOT}/graphify-out/.graphify_chunk_0N.json"
|
||||
```
|
||||
|
||||
Subagent prompt template:
|
||||
|
||||
```
|
||||
You are a graphify extraction subagent. Read the files listed and extract a knowledge graph fragment.
|
||||
@@ -342,8 +350,11 @@ confidence_score is REQUIRED on every edge - never omit it, never use 0.5 as a d
|
||||
|
||||
Node ID format: lowercase, only `[a-z0-9_]`, no dots or slashes. Format: `{stem}_{entity}` where stem is the filename without extension and entity is the symbol name, both normalized (lowercase, non-alphanumeric chars replaced with `_`). Example: `src/auth/session.py` + `ValidateToken` → `session_validatetoken`. This must match the ID the AST extractor generates so cross-references between code and semantic nodes connect correctly. CRITICAL: never append chunk numbers, sequence numbers, or any suffix to an ID (no `_c1`, `_c2`, `_chunk2`, etc.). IDs must be deterministic from the label alone — the same entity must always produce the same ID regardless of which chunk processes it.
|
||||
|
||||
Output exactly this JSON (no other text):
|
||||
Generate the extraction JSON matching this schema exactly:
|
||||
{"nodes":[{"id":"session_validatetoken","label":"Human Readable Name","file_type":"code|document|paper|image|rationale","source_file":"relative/path","source_location":null,"source_url":null,"captured_at":null,"author":null,"contributor":null}],"edges":[{"source":"node_id","target":"node_id","relation":"calls|implements|references|cites|conceptually_related_to|shares_data_with|semantically_similar_to|rationale_for","confidence":"EXTRACTED|INFERRED|AMBIGUOUS","confidence_score":1.0,"source_file":"relative/path","source_location":null,"weight":1.0}],"hyperedges":[{"id":"snake_case_id","label":"Human Readable Label","nodes":["node_id1","node_id2","node_id3"],"relation":"participate_in|implement|form","confidence":"EXTRACTED|INFERRED","confidence_score":0.75,"source_file":"relative/path"}],"input_tokens":0,"output_tokens":0}
|
||||
|
||||
Then write the JSON to disk using the Write tool at this exact absolute path (no relative paths — Write resolves relative paths against an undefined cwd and the file will be silently lost):
|
||||
CHUNK_PATH
|
||||
```
|
||||
|
||||
**Step B3 - Collect, cache, and merge**
|
||||
@@ -776,55 +787,49 @@ Then:
|
||||
|
||||
```bash
|
||||
$(cat graphify-out/.graphify_python) -c "
|
||||
import sys, json
|
||||
from graphify.build import build_from_json
|
||||
from graphify.export import to_json
|
||||
from networkx.readwrite import json_graph
|
||||
import networkx as nx
|
||||
import json
|
||||
from pathlib import Path
|
||||
from graphify.build import build_merge
|
||||
from graphify.detect import save_manifest
|
||||
|
||||
# Load existing graph
|
||||
existing_data = json.loads(Path('graphify-out/graph.json').read_text())
|
||||
G_existing = json_graph.node_link_graph(existing_data, edges='links')
|
||||
|
||||
# Load new extraction
|
||||
# Load new extraction and incremental state
|
||||
new_extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text())
|
||||
G_new = build_from_json(new_extraction)
|
||||
|
||||
# Prune nodes from deleted files
|
||||
incremental = json.loads(Path('graphify-out/.graphify_incremental.json').read_text())
|
||||
deleted = set(incremental.get('deleted_files', []))
|
||||
if deleted:
|
||||
to_remove = [n for n, d in G_existing.nodes(data=True) if d.get('source_file') in deleted]
|
||||
G_existing.remove_nodes_from(to_remove)
|
||||
if to_remove:
|
||||
print(f'Pruned {len(to_remove)} ghost node(s) from {len(deleted)} deleted file(s) — drift detected and corrected.')
|
||||
else:
|
||||
print(f'{len(deleted)} file(s) deleted since last run, but no ghost nodes were present in the graph — no drift.')
|
||||
deleted = list(incremental.get('deleted_files', []))
|
||||
|
||||
# Merge: new nodes/edges into existing graph
|
||||
G_existing.update(G_new)
|
||||
print(f'Merged: {G_existing.number_of_nodes()} nodes, {G_existing.number_of_edges()} edges')
|
||||
# Use build_merge() — reads graph.json directly without NetworkX round-trip
|
||||
# so edge direction (calls, implements, imports) is always preserved (#801).
|
||||
G = build_merge(
|
||||
[new_extraction],
|
||||
graph_path='graphify-out/graph.json',
|
||||
prune_sources=deleted or None,
|
||||
)
|
||||
print(f'[graphify update] Merged: {G.number_of_nodes()} nodes, {G.number_of_edges()} edges')
|
||||
|
||||
# Write merged result back to .graphify_extract.json so Step 4 sees the full graph
|
||||
merged_out = {
|
||||
'nodes': [{'id': n, **d} for n, d in G_existing.nodes(data=True)],
|
||||
'edges': [{'source': u, 'target': v, **d} for u, v, d in G_existing.edges(data=True)],
|
||||
'hyperedges': new_extraction.get('hyperedges', []),
|
||||
'nodes': [{'id': n, **d} for n, d in G.nodes(data=True)],
|
||||
'edges': [
|
||||
# Explicit source/target last so they win over any stale attrs in d.
|
||||
{**{k: val for k, val in d.items() if k not in ('_src', '_tgt', 'source', 'target')},
|
||||
'source': d.get('_src', u), 'target': d.get('_tgt', v)}
|
||||
for u, v, d in G.edges(data=True)
|
||||
],
|
||||
# G.graph["hyperedges"] holds hyperedges from both existing graph.json
|
||||
# and new_extraction (build_merge combines them). Falling back to
|
||||
# new_extraction only would silently drop prior-run hyperedges (#801).
|
||||
'hyperedges': list(G.graph.get('hyperedges', [])),
|
||||
'input_tokens': new_extraction.get('input_tokens', 0),
|
||||
'output_tokens': new_extraction.get('output_tokens', 0),
|
||||
}
|
||||
Path('graphify-out/.graphify_extract.json').write_text(json.dumps(merged_out))
|
||||
print(f'[graphify update] Merged extraction written ({len(merged_out[\"nodes\"])} nodes, {len(merged_out[\"edges\"])} edges)')
|
||||
|
||||
# Save manifest with the CURRENT full file list so the next --update
|
||||
# diffs against today's filesystem state, not the prior --update's
|
||||
# baseline. Without this, deleted files get reported as ghosts again
|
||||
# on every subsequent --update until a full rebuild runs.
|
||||
from graphify.detect import save_manifest
|
||||
# Save manifest so next --update diffs against today's state, not the
|
||||
# prior run's baseline (prevents ghost-node reports on subsequent updates).
|
||||
save_manifest(incremental['files'])
|
||||
print('[graphify update] Manifest saved.')
|
||||
"
|
||||
"
|
||||
```
|
||||
|
||||
Then run Steps 4–8 on the merged graph as normal.
|
||||
|
||||
Reference in New Issue
Block a user