fix #811 #803 #801 #808: Unicode IDs, dedup edge keys, direction flip, chunk paths

- extract/_make_id + build/_normalize_id: use NFKC normalization and casefold
  so composed/decomposed Unicode forms produce the same ID; collapse consecutive
  underscores; both functions are now byte-for-byte equivalent (#811)
- dedup: use explicit key-presence check instead of `or` for source/from
  fallback; pop stale from/to keys so they don't leak into graph.json attrs (#803)
- skill --update: use build_merge() to avoid NetworkX round-trip direction flip;
  fix dict merge ordering so explicit source/target win; pull hyperedges from
  G.graph (merged) not new_extraction only (#801)
- skill subagents: inject absolute CHUNK_PATH so Write tool doesn't lose chunk
  files to undefined cwd (#808)
- __main__: skip skill version check during hook-check (runs on every editor
  tool use, must be silent); move warning to stderr

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
Safi
2026-05-11 18:46:12 +01:00
co-authored by Claude Sonnet 4.6
parent 4cec58e072
commit 95e2c5eb32
5 changed files with 80 additions and 46 deletions
+4 -2
View File
@@ -33,7 +33,7 @@ def _check_skill_version(skill_dst: Path) -> None:
return
installed = version_file.read_text(encoding="utf-8").strip()
if installed != __version__:
print(f" warning: skill is from graphify {installed}, package is {__version__}. Run 'graphify install' to update.")
print(f" warning: skill is from graphify {installed}, package is {__version__}. Run 'graphify install' to update.", file=sys.stderr)
def _refresh_all_version_stamps() -> None:
@@ -1115,8 +1115,10 @@ def _clone_repo(url: str, branch: str | None = None, out_dir: Path | None = None
def main() -> None:
# Check all known skill install locations for a stale version stamp.
# Skip during install/uninstall (hook writes trigger a fresh check anyway).
# Skip during hook-check — it runs on every editor tool use and must be silent.
# Deduplicate paths so platforms sharing the same install dir don't warn twice.
if not any(arg in ("install", "uninstall") for arg in sys.argv):
_silent_cmds = {"install", "uninstall", "hook-check"}
if not any(arg in _silent_cmds for arg in sys.argv):
for skill_dst in {Path.home() / cfg["skill_dst"] for cfg in _PLATFORM_CONFIG.values()}:
_check_skill_version(skill_dst)
+9 -4
View File
@@ -24,19 +24,24 @@ from __future__ import annotations
import json
import re
import sys
import unicodedata
from pathlib import Path
import networkx as nx
from .validate import validate_extraction
def _normalize_id(s: str) -> str:
"""Normalize an ID string the same way extract._make_id does.
r"""Normalize an ID string the same way extract._make_id does.
Used to reconcile edge endpoints when the LLM generates IDs with slightly
different punctuation or casing than the AST extractor.
different punctuation or casing than the AST extractor. Must stay in sync
with extract._make_id — NFKC normalization, \w with re.UNICODE, underscore
collapse, and casefold must all match (#811).
"""
cleaned = re.sub(r"[^a-zA-Z0-9]+", "_", s)
return cleaned.strip("_").lower()
s = unicodedata.normalize("NFKC", s)
cleaned = re.sub(r"[^\w]+", "_", s, flags=re.UNICODE)
cleaned = re.sub(r"_+", "_", cleaned)
return cleaned.strip("_").casefold()
def _norm_source_file(p: str | None) -> str | None:
+14 -2
View File
@@ -233,8 +233,20 @@ def deduplicate_entities(
deduped_edges = []
for edge in edges:
e = dict(edge)
e["source"] = remap.get(e["source"], e["source"])
e["target"] = remap.get(e["target"], e["target"])
# Tolerate "from"/"to" keys from LLM backends that don't follow the
# schema exactly — build_from_json normalises later but dedup runs
# first so bracket access would KeyError here (#803).
# Use explicit key presence check (not `or`) so empty-string src/tgt
# aren't silently replaced by the fallback key.
src = e["source"] if "source" in e else e.get("from")
tgt = e["target"] if "target" in e else e.get("to")
if src is None or tgt is None:
continue
e["source"] = remap.get(src, src)
e["target"] = remap.get(tgt, tgt)
# Remove legacy keys so they don't leak into edge attrs in graph.json.
e.pop("from", None)
e.pop("to", None)
if e["source"] != e["target"]:
deduped_edges.append(e)
+13 -3
View File
@@ -5,6 +5,7 @@ import json
import os
import re
import sys
import unicodedata
from dataclasses import dataclass, field
from pathlib import Path
from typing import Callable, Any
@@ -30,10 +31,19 @@ def _safe_extract(extractor: Callable, path: Path) -> dict:
def _make_id(*parts: str) -> str:
"""Build a stable node ID from one or more name parts."""
r"""Build a stable node ID from one or more name parts.
Preserves Unicode letters/digits (CJK, Cyrillic, Arabic, accented Latin,
etc.) so non-ASCII identifiers produce distinct IDs and don't collapse to
a single per-file node (#811). NFKC normalization ensures composed and
decomposed forms of the same character (e.g. é vs e+combining-acute)
produce the same ID. Must stay in sync with build._normalize_id.
"""
combined = "_".join(p.strip("_.") for p in parts if p)
cleaned = re.sub(r"[^a-zA-Z0-9]+", "_", combined)
return cleaned.strip("_").lower()
combined = unicodedata.normalize("NFKC", combined)
cleaned = re.sub(r"[^\w]+", "_", combined, flags=re.UNICODE)
cleaned = re.sub(r"_+", "_", cleaned)
return cleaned.strip("_").casefold()
def _file_stem(path: Path) -> str:
+40 -35
View File
@@ -282,7 +282,15 @@ Concrete example for 3 chunks:
```
All three in one message. Not three separate messages.
Each subagent receives this exact prompt (substitute FILE_LIST, CHUNK_NUM, TOTAL_CHUNKS, and DEEP_MODE):
Each subagent receives this exact prompt (substitute FILE_LIST, CHUNK_NUM, TOTAL_CHUNKS, DEEP_MODE, and CHUNK_PATH).
CHUNK_PATH must be an **absolute** path — derive it before dispatching:
```bash
PROJECT_ROOT=$(cat graphify-out/.graphify_root)
# Then for chunk N: CHUNK_PATH="${PROJECT_ROOT}/graphify-out/.graphify_chunk_0N.json"
```
Subagent prompt template:
```
You are a graphify extraction subagent. Read the files listed and extract a knowledge graph fragment.
@@ -342,8 +350,11 @@ confidence_score is REQUIRED on every edge - never omit it, never use 0.5 as a d
Node ID format: lowercase, only `[a-z0-9_]`, no dots or slashes. Format: `{stem}_{entity}` where stem is the filename without extension and entity is the symbol name, both normalized (lowercase, non-alphanumeric chars replaced with `_`). Example: `src/auth/session.py` + `ValidateToken` → `session_validatetoken`. This must match the ID the AST extractor generates so cross-references between code and semantic nodes connect correctly. CRITICAL: never append chunk numbers, sequence numbers, or any suffix to an ID (no `_c1`, `_c2`, `_chunk2`, etc.). IDs must be deterministic from the label alone — the same entity must always produce the same ID regardless of which chunk processes it.
Output exactly this JSON (no other text):
Generate the extraction JSON matching this schema exactly:
{"nodes":[{"id":"session_validatetoken","label":"Human Readable Name","file_type":"code|document|paper|image|rationale","source_file":"relative/path","source_location":null,"source_url":null,"captured_at":null,"author":null,"contributor":null}],"edges":[{"source":"node_id","target":"node_id","relation":"calls|implements|references|cites|conceptually_related_to|shares_data_with|semantically_similar_to|rationale_for","confidence":"EXTRACTED|INFERRED|AMBIGUOUS","confidence_score":1.0,"source_file":"relative/path","source_location":null,"weight":1.0}],"hyperedges":[{"id":"snake_case_id","label":"Human Readable Label","nodes":["node_id1","node_id2","node_id3"],"relation":"participate_in|implement|form","confidence":"EXTRACTED|INFERRED","confidence_score":0.75,"source_file":"relative/path"}],"input_tokens":0,"output_tokens":0}
Then write the JSON to disk using the Write tool at this exact absolute path (no relative paths — Write resolves relative paths against an undefined cwd and the file will be silently lost):
CHUNK_PATH
```
**Step B3 - Collect, cache, and merge**
@@ -776,55 +787,49 @@ Then:
```bash
$(cat graphify-out/.graphify_python) -c "
import sys, json
from graphify.build import build_from_json
from graphify.export import to_json
from networkx.readwrite import json_graph
import networkx as nx
import json
from pathlib import Path
from graphify.build import build_merge
from graphify.detect import save_manifest
# Load existing graph
existing_data = json.loads(Path('graphify-out/graph.json').read_text())
G_existing = json_graph.node_link_graph(existing_data, edges='links')
# Load new extraction
# Load new extraction and incremental state
new_extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text())
G_new = build_from_json(new_extraction)
# Prune nodes from deleted files
incremental = json.loads(Path('graphify-out/.graphify_incremental.json').read_text())
deleted = set(incremental.get('deleted_files', []))
if deleted:
to_remove = [n for n, d in G_existing.nodes(data=True) if d.get('source_file') in deleted]
G_existing.remove_nodes_from(to_remove)
if to_remove:
print(f'Pruned {len(to_remove)} ghost node(s) from {len(deleted)} deleted file(s) — drift detected and corrected.')
else:
print(f'{len(deleted)} file(s) deleted since last run, but no ghost nodes were present in the graph — no drift.')
deleted = list(incremental.get('deleted_files', []))
# Merge: new nodes/edges into existing graph
G_existing.update(G_new)
print(f'Merged: {G_existing.number_of_nodes()} nodes, {G_existing.number_of_edges()} edges')
# Use build_merge() — reads graph.json directly without NetworkX round-trip
# so edge direction (calls, implements, imports) is always preserved (#801).
G = build_merge(
[new_extraction],
graph_path='graphify-out/graph.json',
prune_sources=deleted or None,
)
print(f'[graphify update] Merged: {G.number_of_nodes()} nodes, {G.number_of_edges()} edges')
# Write merged result back to .graphify_extract.json so Step 4 sees the full graph
merged_out = {
'nodes': [{'id': n, **d} for n, d in G_existing.nodes(data=True)],
'edges': [{'source': u, 'target': v, **d} for u, v, d in G_existing.edges(data=True)],
'hyperedges': new_extraction.get('hyperedges', []),
'nodes': [{'id': n, **d} for n, d in G.nodes(data=True)],
'edges': [
# Explicit source/target last so they win over any stale attrs in d.
{**{k: val for k, val in d.items() if k not in ('_src', '_tgt', 'source', 'target')},
'source': d.get('_src', u), 'target': d.get('_tgt', v)}
for u, v, d in G.edges(data=True)
],
# G.graph["hyperedges"] holds hyperedges from both existing graph.json
# and new_extraction (build_merge combines them). Falling back to
# new_extraction only would silently drop prior-run hyperedges (#801).
'hyperedges': list(G.graph.get('hyperedges', [])),
'input_tokens': new_extraction.get('input_tokens', 0),
'output_tokens': new_extraction.get('output_tokens', 0),
}
Path('graphify-out/.graphify_extract.json').write_text(json.dumps(merged_out))
print(f'[graphify update] Merged extraction written ({len(merged_out[\"nodes\"])} nodes, {len(merged_out[\"edges\"])} edges)')
# Save manifest with the CURRENT full file list so the next --update
# diffs against today's filesystem state, not the prior --update's
# baseline. Without this, deleted files get reported as ghosts again
# on every subsequent --update until a full rebuild runs.
from graphify.detect import save_manifest
# Save manifest so next --update diffs against today's state, not the
# prior run's baseline (prevents ghost-node reports on subsequent updates).
save_manifest(incremental['files'])
print('[graphify update] Manifest saved.')
"
"
```
Then run Steps 4–8 on the merged graph as normal.