cherry-pick PRs #212, #220, #204, #221 into 0.4.2

- build/validate: accept NetworkX <=3.1 "links" key alongside "edges" (#212)
- __main__: skip version check during install/uninstall, deduplicate paths (#220)
- all file IO: explicit encoding="utf-8" to prevent crashes on Windows CJK locales (#204)
- hooks: add newline="\n" on write to prevent CRLF shebang breakage on Windows (#204)
- export: strip trailing .md from safe_name so "CLAUDE.md" doesn't become "CLAUDE.md.md" (#221)
- report: add Community Hubs navigation block so Obsidian vault stays connected (#221)

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
Safi
2026-04-11 21:07:39 +01:00
co-authored by Claude Sonnet 4.6
parent 1cb882a156
commit 8c17230586
11 changed files with 61 additions and 30 deletions
+6 -4
View File
@@ -628,10 +628,12 @@ def claude_uninstall(project_dir: Path | None = None) -> None:
def main() -> None: def main() -> None:
# Check all known skill install locations for a stale version stamp # Check all known skill install locations for a stale version stamp.
for cfg in _PLATFORM_CONFIG.values(): # Skip during install/uninstall (hook writes trigger a fresh check anyway).
skill_dst = Path.home() / cfg["skill_dst"] # Deduplicate paths so platforms sharing the same install dir don't warn twice.
_check_skill_version(skill_dst) if not any(arg in ("install", "uninstall") for arg in sys.argv):
for skill_dst in {Path.home() / cfg["skill_dst"] for cfg in _PLATFORM_CONFIG.values()}:
_check_skill_version(skill_dst)
if len(sys.argv) < 2 or sys.argv[1] in ("-h", "--help"): if len(sys.argv) < 2 or sys.argv[1] in ("-h", "--help"):
print("Usage: graphify <command>") print("Usage: graphify <command>")
+1 -1
View File
@@ -75,7 +75,7 @@ def run_benchmark(
Returns dict with: corpus_tokens, avg_query_tokens, reduction_ratio, per_question Returns dict with: corpus_tokens, avg_query_tokens, reduction_ratio, per_question
""" """
data = json.loads(Path(graph_path).read_text()) data = json.loads(Path(graph_path).read_text(encoding="utf-8"))
try: try:
G = json_graph.node_link_graph(data, edges="links") G = json_graph.node_link_graph(data, edges="links")
except TypeError: except TypeError:
+3
View File
@@ -32,6 +32,9 @@ def build_from_json(extraction: dict, *, directed: bool = False) -> nx.Graph:
directed=True produces a DiGraph that preserves edge direction (source→target). directed=True produces a DiGraph that preserves edge direction (source→target).
directed=False (default) produces an undirected Graph for backward compatibility. directed=False (default) produces an undirected Graph for backward compatibility.
""" """
# NetworkX <= 3.1 serialised edges as "links"; remap to "edges" for compatibility.
if "edges" not in extraction and "links" in extraction:
extraction = dict(extraction, edges=extraction["links"])
errors = validate_extraction(extraction) errors = validate_extraction(extraction)
# Dangling edges (stdlib/external imports) are expected - only warn about real schema errors. # Dangling edges (stdlib/external imports) are expected - only warn about real schema errors.
real_errors = [e for e in errors if "does not match any node id" not in e] real_errors = [e for e in errors if "does not match any node id" not in e]
+2 -2
View File
@@ -55,7 +55,7 @@ def load_cached(path: Path, root: Path = Path(".")) -> dict | None:
if not entry.exists(): if not entry.exists():
return None return None
try: try:
return json.loads(entry.read_text()) return json.loads(entry.read_text(encoding="utf-8"))
except (json.JSONDecodeError, OSError): except (json.JSONDecodeError, OSError):
return None return None
@@ -70,7 +70,7 @@ def save_cached(path: Path, result: dict, root: Path = Path(".")) -> None:
entry = cache_dir(root) / f"{h}.json" entry = cache_dir(root) / f"{h}.json"
tmp = entry.with_suffix(".tmp") tmp = entry.with_suffix(".tmp")
try: try:
tmp.write_text(json.dumps(result)) tmp.write_text(json.dumps(result), encoding="utf-8")
os.replace(tmp, entry) os.replace(tmp, entry)
except Exception: except Exception:
tmp.unlink(missing_ok=True) tmp.unlink(missing_ok=True)
+5 -5
View File
@@ -69,7 +69,7 @@ def _looks_like_paper(path: Path) -> bool:
"""Heuristic: does this text file read like an academic paper?""" """Heuristic: does this text file read like an academic paper?"""
try: try:
# Only scan first 3000 chars for speed # Only scan first 3000 chars for speed
text = path.read_text(errors="ignore")[:3000] text = path.read_text(encoding="utf-8", errors="ignore")[:3000]
hits = sum(1 for pattern in _PAPER_SIGNALS if pattern.search(text)) hits = sum(1 for pattern in _PAPER_SIGNALS if pattern.search(text))
return hits >= _PAPER_SIGNAL_THRESHOLD return hits >= _PAPER_SIGNAL_THRESHOLD
except Exception: except Exception:
@@ -226,7 +226,7 @@ def count_words(path: Path) -> int:
return len(docx_to_markdown(path).split()) return len(docx_to_markdown(path).split())
if ext == ".xlsx": if ext == ".xlsx":
return len(xlsx_to_markdown(path).split()) return len(xlsx_to_markdown(path).split())
return len(path.read_text(errors="ignore").split()) return len(path.read_text(encoding="utf-8", errors="ignore").split())
except Exception: except Exception:
return 0 return 0
@@ -271,7 +271,7 @@ def _load_graphifyignore(root: Path) -> list[str]:
while True: while True:
ignore_file = current / ".graphifyignore" ignore_file = current / ".graphifyignore"
if ignore_file.exists(): if ignore_file.exists():
for line in ignore_file.read_text(errors="ignore").splitlines(): for line in ignore_file.read_text(encoding="utf-8", errors="ignore").splitlines():
line = line.strip() line = line.strip()
if line and not line.startswith("#"): if line and not line.startswith("#"):
patterns.append(line) patterns.append(line)
@@ -427,7 +427,7 @@ def detect(root: Path, *, follow_symlinks: bool = False) -> dict:
def load_manifest(manifest_path: str = _MANIFEST_PATH) -> dict[str, float]: def load_manifest(manifest_path: str = _MANIFEST_PATH) -> dict[str, float]:
"""Load the file modification time manifest from a previous run.""" """Load the file modification time manifest from a previous run."""
try: try:
return json.loads(Path(manifest_path).read_text()) return json.loads(Path(manifest_path).read_text(encoding="utf-8"))
except Exception: except Exception:
return {} return {}
@@ -442,7 +442,7 @@ def save_manifest(files: dict[str, list[str]], manifest_path: str = _MANIFEST_PA
except OSError: except OSError:
pass # file deleted between detect() and manifest write - skip it pass # file deleted between detect() and manifest write - skip it
Path(manifest_path).parent.mkdir(parents=True, exist_ok=True) Path(manifest_path).parent.mkdir(parents=True, exist_ok=True)
Path(manifest_path).write_text(json.dumps(manifest, indent=2)) Path(manifest_path).write_text(json.dumps(manifest, indent=2), encoding="utf-8")
def detect_incremental(root: Path, manifest_path: str = _MANIFEST_PATH) -> dict: def detect_incremental(root: Path, manifest_path: str = _MANIFEST_PATH) -> dict:
+10 -5
View File
@@ -295,7 +295,7 @@ def to_json(G: nx.Graph, communities: dict[int, list[str]], output_path: str) ->
conf = link.get("confidence", "EXTRACTED") conf = link.get("confidence", "EXTRACTED")
link["confidence_score"] = _CONFIDENCE_SCORE_DEFAULTS.get(conf, 1.0) link["confidence_score"] = _CONFIDENCE_SCORE_DEFAULTS.get(conf, 1.0)
data["hyperedges"] = getattr(G, "graph", {}).get("hyperedges", []) data["hyperedges"] = getattr(G, "graph", {}).get("hyperedges", [])
with open(output_path, "w") as f: with open(output_path, "w", encoding="utf-8") as f:
json.dump(data, f, indent=2) json.dump(data, f, indent=2)
@@ -322,7 +322,7 @@ def to_cypher(G: nx.Graph, output_path: str) -> None:
f"MATCH (a {{id: '{u_esc}'}}), (b {{id: '{v_esc}'}}) " f"MATCH (a {{id: '{u_esc}'}}), (b {{id: '{v_esc}'}}) "
f"MERGE (a)-[:{rel} {{confidence: '{conf}'}}]->(b);" f"MERGE (a)-[:{rel} {{confidence: '{conf}'}}]->(b);"
) )
with open(output_path, "w") as f: with open(output_path, "w", encoding="utf-8") as f:
f.write("\n".join(lines)) f.write("\n".join(lines))
@@ -467,7 +467,10 @@ def to_obsidian(
# Map node_id → safe filename so wikilinks stay consistent. # Map node_id → safe filename so wikilinks stay consistent.
# Deduplicate: if two nodes produce the same filename, append a numeric suffix. # Deduplicate: if two nodes produce the same filename, append a numeric suffix.
def safe_name(label: str) -> str: def safe_name(label: str) -> str:
return re.sub(r'[\\/*?:"<>|#^[\]]', "", label.replace("\r\n", " ").replace("\r", " ").replace("\n", " ")).strip() or "unnamed" cleaned = re.sub(r'[\\/*?:"<>|#^[\]]', "", label.replace("\r\n", " ").replace("\r", " ").replace("\n", " ")).strip()
# Strip trailing .md/.mdx/.markdown so "CLAUDE.md" doesn't become "CLAUDE.md.md"
cleaned = re.sub(r"\.(md|mdx|markdown)$", "", cleaned, flags=re.IGNORECASE)
return cleaned or "unnamed"
node_filename: dict[str, str] = {} node_filename: dict[str, str] = {}
seen_names: dict[str, int] = {} seen_names: dict[str, int] = {}
@@ -681,7 +684,7 @@ def to_obsidian(
for cid, label in sorted((community_labels or {}).items()) for cid, label in sorted((community_labels or {}).items())
] ]
} }
(obsidian_dir / "graph.json").write_text(json.dumps(graph_config, indent=2)) (obsidian_dir / "graph.json").write_text(json.dumps(graph_config, indent=2), encoding="utf-8")
return G.number_of_nodes() + community_notes_written return G.number_of_nodes() + community_notes_written
@@ -703,7 +706,9 @@ def to_canvas(
CANVAS_COLORS = ["1", "2", "3", "4", "5", "6"] # red, orange, yellow, green, cyan, purple CANVAS_COLORS = ["1", "2", "3", "4", "5", "6"] # red, orange, yellow, green, cyan, purple
def safe_name(label: str) -> str: def safe_name(label: str) -> str:
return re.sub(r'[\\/*?:"<>|#^[\]]', "", label.replace("\r\n", " ").replace("\r", " ").replace("\n", " ")).strip() or "unnamed" cleaned = re.sub(r'[\\/*?:"<>|#^[\]]', "", label.replace("\r\n", " ").replace("\r", " ").replace("\n", " ")).strip()
cleaned = re.sub(r"\.(md|mdx|markdown)$", "", cleaned, flags=re.IGNORECASE)
return cleaned or "unnamed"
# Build node_filenames if not provided (same dedup logic as to_obsidian) # Build node_filenames if not provided (same dedup logic as to_obsidian)
if node_filenames is None: if node_filenames is None:
+6 -6
View File
@@ -113,12 +113,12 @@ def _install_hook(hooks_dir: Path, name: str, script: str, marker: str) -> str:
"""Install a single git hook, appending if an existing hook is present.""" """Install a single git hook, appending if an existing hook is present."""
hook_path = hooks_dir / name hook_path = hooks_dir / name
if hook_path.exists(): if hook_path.exists():
content = hook_path.read_text() content = hook_path.read_text(encoding="utf-8")
if marker in content: if marker in content:
return f"already installed at {hook_path}" return f"already installed at {hook_path}"
hook_path.write_text(content.rstrip() + "\n\n" + script) hook_path.write_text(content.rstrip() + "\n\n" + script, encoding="utf-8", newline="\n")
return f"appended to existing {name} hook at {hook_path}" return f"appended to existing {name} hook at {hook_path}"
hook_path.write_text("#!/bin/sh\n" + script) hook_path.write_text("#!/bin/sh\n" + script, encoding="utf-8", newline="\n")
hook_path.chmod(0o755) hook_path.chmod(0o755)
return f"installed at {hook_path}" return f"installed at {hook_path}"
@@ -128,7 +128,7 @@ def _uninstall_hook(hooks_dir: Path, name: str, marker: str, marker_end: str) ->
hook_path = hooks_dir / name hook_path = hooks_dir / name
if not hook_path.exists(): if not hook_path.exists():
return f"no {name} hook found - nothing to remove." return f"no {name} hook found - nothing to remove."
content = hook_path.read_text() content = hook_path.read_text(encoding="utf-8")
if marker not in content: if marker not in content:
return f"graphify hook not found in {name} - nothing to remove." return f"graphify hook not found in {name} - nothing to remove."
new_content = re.sub( new_content = re.sub(
@@ -140,7 +140,7 @@ def _uninstall_hook(hooks_dir: Path, name: str, marker: str, marker_end: str) ->
if not new_content or new_content in ("#!/bin/bash", "#!/bin/sh"): if not new_content or new_content in ("#!/bin/bash", "#!/bin/sh"):
hook_path.unlink() hook_path.unlink()
return f"removed {name} hook at {hook_path}" return f"removed {name} hook at {hook_path}"
hook_path.write_text(new_content + "\n") hook_path.write_text(new_content + "\n", encoding="utf-8", newline="\n")
return f"graphify removed from {name} at {hook_path} (other hook content preserved)" return f"graphify removed from {name} at {hook_path} (other hook content preserved)"
@@ -183,7 +183,7 @@ def status(path: Path = Path(".")) -> str:
p = hooks_dir / name p = hooks_dir / name
if not p.exists(): if not p.exists():
return "not installed" return "not installed"
return "installed" if marker in p.read_text() else "not installed (hook exists but graphify not found)" return "installed" if marker in p.read_text(encoding="utf-8") else "not installed (hook exists but graphify not found)"
commit = _check("post-commit", _HOOK_MARKER) commit = _check("post-commit", _HOOK_MARKER)
checkout = _check("post-checkout", _CHECKOUT_MARKER) checkout = _check("post-checkout", _CHECKOUT_MARKER)
+20
View File
@@ -1,9 +1,17 @@
# generate GRAPH_REPORT.md - the human-readable audit trail # generate GRAPH_REPORT.md - the human-readable audit trail
from __future__ import annotations from __future__ import annotations
import re
from datetime import date from datetime import date
import networkx as nx import networkx as nx
def _safe_community_name(label: str) -> str:
"""Mirrors export.safe_name so community hub filenames and report wikilinks always agree."""
cleaned = re.sub(r'[\\/*?:"<>|#^[\]]', "", label.replace("\r\n", " ").replace("\r", " ").replace("\n", " ")).strip()
cleaned = re.sub(r"\.(md|mdx|markdown)$", "", cleaned, flags=re.IGNORECASE)
return cleaned or "unnamed"
def generate( def generate(
G: nx.Graph, G: nx.Graph,
communities: dict[int, list[str]], communities: dict[int, list[str]],
@@ -48,6 +56,18 @@ def generate(
f"- Extraction: {ext_pct}% EXTRACTED · {inf_pct}% INFERRED · {amb_pct}% AMBIGUOUS" f"- Extraction: {ext_pct}% EXTRACTED · {inf_pct}% INFERRED · {amb_pct}% AMBIGUOUS"
+ (f" · INFERRED: {len(inf_edges)} edges (avg confidence: {inf_avg})" if inf_avg is not None else ""), + (f" · INFERRED: {len(inf_edges)} edges (avg confidence: {inf_avg})" if inf_avg is not None else ""),
f"- Token cost: {token_cost.get('input', 0):,} input · {token_cost.get('output', 0):,} output", f"- Token cost: {token_cost.get('input', 0):,} input · {token_cost.get('output', 0):,} output",
]
# Community hub navigation - links to _COMMUNITY_*.md files in the Obsidian vault.
# Without these, GRAPH_REPORT.md is a dead-end and the vault splits into disconnected components.
if communities:
lines += ["", "## Community Hubs (Navigation)"]
for cid in communities:
label = community_labels.get(cid, f"Community {cid}")
safe = _safe_community_name(label)
lines.append(f"- [[_COMMUNITY_{safe}|{label}]]")
lines += [
"", "",
"## God Nodes (most connected - your core abstractions)", "## God Nodes (most connected - your core abstractions)",
] ]
+1 -1
View File
@@ -16,7 +16,7 @@ def _load_graph(graph_path: str) -> nx.Graph:
if not resolved.exists(): if not resolved.exists():
raise FileNotFoundError(f"Graph file not found: {resolved}") raise FileNotFoundError(f"Graph file not found: {resolved}")
safe = resolved safe = resolved
data = json.loads(safe.read_text()) data = json.loads(safe.read_text(encoding="utf-8"))
try: try:
return json_graph.node_link_graph(data, edges="links") return json_graph.node_link_graph(data, edges="links")
except TypeError: except TypeError:
+5 -4
View File
@@ -36,14 +36,15 @@ def validate_extraction(data: dict) -> list[str]:
f"'{node['file_type']}' - must be one of {sorted(VALID_FILE_TYPES)}" f"'{node['file_type']}' - must be one of {sorted(VALID_FILE_TYPES)}"
) )
# Edges # Edges - accept "links" (NetworkX <= 3.1) as fallback for "edges"
if "edges" not in data: edge_list = data.get("edges") if "edges" in data else data.get("links")
if edge_list is None:
errors.append("Missing required key 'edges'") errors.append("Missing required key 'edges'")
elif not isinstance(data["edges"], list): elif not isinstance(edge_list, list):
errors.append("'edges' must be a list") errors.append("'edges' must be a list")
else: else:
node_ids = {n["id"] for n in data.get("nodes", []) if isinstance(n, dict) and "id" in n} node_ids = {n["id"] for n in data.get("nodes", []) if isinstance(n, dict) and "id" in n}
for i, edge in enumerate(data["edges"]): for i, edge in enumerate(edge_list):
if not isinstance(edge, dict): if not isinstance(edge, dict):
errors.append(f"Edge {i} must be an object") errors.append(f"Edge {i} must be an object")
continue continue
+2 -2
View File
@@ -53,7 +53,7 @@ def _rebuild_code(watch_path: Path, *, follow_symlinks: bool = False) -> bool:
report = generate(G, communities, cohesion, labels, gods, surprises, detection, report = generate(G, communities, cohesion, labels, gods, surprises, detection,
{"input": 0, "output": 0}, str(watch_path), suggested_questions=questions) {"input": 0, "output": 0}, str(watch_path), suggested_questions=questions)
(out / "GRAPH_REPORT.md").write_text(report) (out / "GRAPH_REPORT.md").write_text(report, encoding="utf-8")
to_json(G, communities, str(out / "graph.json")) to_json(G, communities, str(out / "graph.json"))
# clear stale needs_update flag if present # clear stale needs_update flag if present
@@ -75,7 +75,7 @@ def _notify_only(watch_path: Path) -> None:
"""Write a flag file and print a notification (fallback for non-code-only corpora).""" """Write a flag file and print a notification (fallback for non-code-only corpora)."""
flag = watch_path / "graphify-out" / "needs_update" flag = watch_path / "graphify-out" / "needs_update"
flag.parent.mkdir(parents=True, exist_ok=True) flag.parent.mkdir(parents=True, exist_ok=True)
flag.write_text("1") flag.write_text("1", encoding="utf-8")
print(f"\n[graphify watch] New or changed files detected in {watch_path}") print(f"\n[graphify watch] New or changed files detected in {watch_path}")
print("[graphify watch] Non-code files changed - semantic re-extraction requires LLM.") print("[graphify watch] Non-code files changed - semantic re-extraction requires LLM.")
print("[graphify watch] Run `/graphify --update` in Claude Code to update the graph.") print("[graphify watch] Run `/graphify --update` in Claude Code to update the graph.")