- build/validate: accept NetworkX <=3.1 "links" key alongside "edges" (#212) - __main__: skip version check during install/uninstall, deduplicate paths (#220) - all file IO: explicit encoding="utf-8" to prevent crashes on Windows CJK locales (#204) - hooks: add newline="\n" on write to prevent CRLF shebang breakage on Windows (#204) - export: strip trailing .md from safe_name so "CLAUDE.md" doesn't become "CLAUDE.md.md" (#221) - report: add Community Hubs navigation block so Obsidian vault stays connected (#221) Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Sonnet 4.6
parent
1cb882a156
commit
8c17230586
@@ -628,10 +628,12 @@ def claude_uninstall(project_dir: Path | None = None) -> None:
|
||||
|
||||
|
||||
def main() -> None:
|
||||
# Check all known skill install locations for a stale version stamp
|
||||
for cfg in _PLATFORM_CONFIG.values():
|
||||
skill_dst = Path.home() / cfg["skill_dst"]
|
||||
_check_skill_version(skill_dst)
|
||||
# Check all known skill install locations for a stale version stamp.
|
||||
# Skip during install/uninstall (hook writes trigger a fresh check anyway).
|
||||
# Deduplicate paths so platforms sharing the same install dir don't warn twice.
|
||||
if not any(arg in ("install", "uninstall") for arg in sys.argv):
|
||||
for skill_dst in {Path.home() / cfg["skill_dst"] for cfg in _PLATFORM_CONFIG.values()}:
|
||||
_check_skill_version(skill_dst)
|
||||
|
||||
if len(sys.argv) < 2 or sys.argv[1] in ("-h", "--help"):
|
||||
print("Usage: graphify <command>")
|
||||
|
||||
@@ -75,7 +75,7 @@ def run_benchmark(
|
||||
|
||||
Returns dict with: corpus_tokens, avg_query_tokens, reduction_ratio, per_question
|
||||
"""
|
||||
data = json.loads(Path(graph_path).read_text())
|
||||
data = json.loads(Path(graph_path).read_text(encoding="utf-8"))
|
||||
try:
|
||||
G = json_graph.node_link_graph(data, edges="links")
|
||||
except TypeError:
|
||||
|
||||
@@ -32,6 +32,9 @@ def build_from_json(extraction: dict, *, directed: bool = False) -> nx.Graph:
|
||||
directed=True produces a DiGraph that preserves edge direction (source→target).
|
||||
directed=False (default) produces an undirected Graph for backward compatibility.
|
||||
"""
|
||||
# NetworkX <= 3.1 serialised edges as "links"; remap to "edges" for compatibility.
|
||||
if "edges" not in extraction and "links" in extraction:
|
||||
extraction = dict(extraction, edges=extraction["links"])
|
||||
errors = validate_extraction(extraction)
|
||||
# Dangling edges (stdlib/external imports) are expected - only warn about real schema errors.
|
||||
real_errors = [e for e in errors if "does not match any node id" not in e]
|
||||
|
||||
+2
-2
@@ -55,7 +55,7 @@ def load_cached(path: Path, root: Path = Path(".")) -> dict | None:
|
||||
if not entry.exists():
|
||||
return None
|
||||
try:
|
||||
return json.loads(entry.read_text())
|
||||
return json.loads(entry.read_text(encoding="utf-8"))
|
||||
except (json.JSONDecodeError, OSError):
|
||||
return None
|
||||
|
||||
@@ -70,7 +70,7 @@ def save_cached(path: Path, result: dict, root: Path = Path(".")) -> None:
|
||||
entry = cache_dir(root) / f"{h}.json"
|
||||
tmp = entry.with_suffix(".tmp")
|
||||
try:
|
||||
tmp.write_text(json.dumps(result))
|
||||
tmp.write_text(json.dumps(result), encoding="utf-8")
|
||||
os.replace(tmp, entry)
|
||||
except Exception:
|
||||
tmp.unlink(missing_ok=True)
|
||||
|
||||
+5
-5
@@ -69,7 +69,7 @@ def _looks_like_paper(path: Path) -> bool:
|
||||
"""Heuristic: does this text file read like an academic paper?"""
|
||||
try:
|
||||
# Only scan first 3000 chars for speed
|
||||
text = path.read_text(errors="ignore")[:3000]
|
||||
text = path.read_text(encoding="utf-8", errors="ignore")[:3000]
|
||||
hits = sum(1 for pattern in _PAPER_SIGNALS if pattern.search(text))
|
||||
return hits >= _PAPER_SIGNAL_THRESHOLD
|
||||
except Exception:
|
||||
@@ -226,7 +226,7 @@ def count_words(path: Path) -> int:
|
||||
return len(docx_to_markdown(path).split())
|
||||
if ext == ".xlsx":
|
||||
return len(xlsx_to_markdown(path).split())
|
||||
return len(path.read_text(errors="ignore").split())
|
||||
return len(path.read_text(encoding="utf-8", errors="ignore").split())
|
||||
except Exception:
|
||||
return 0
|
||||
|
||||
@@ -271,7 +271,7 @@ def _load_graphifyignore(root: Path) -> list[str]:
|
||||
while True:
|
||||
ignore_file = current / ".graphifyignore"
|
||||
if ignore_file.exists():
|
||||
for line in ignore_file.read_text(errors="ignore").splitlines():
|
||||
for line in ignore_file.read_text(encoding="utf-8", errors="ignore").splitlines():
|
||||
line = line.strip()
|
||||
if line and not line.startswith("#"):
|
||||
patterns.append(line)
|
||||
@@ -427,7 +427,7 @@ def detect(root: Path, *, follow_symlinks: bool = False) -> dict:
|
||||
def load_manifest(manifest_path: str = _MANIFEST_PATH) -> dict[str, float]:
|
||||
"""Load the file modification time manifest from a previous run."""
|
||||
try:
|
||||
return json.loads(Path(manifest_path).read_text())
|
||||
return json.loads(Path(manifest_path).read_text(encoding="utf-8"))
|
||||
except Exception:
|
||||
return {}
|
||||
|
||||
@@ -442,7 +442,7 @@ def save_manifest(files: dict[str, list[str]], manifest_path: str = _MANIFEST_PA
|
||||
except OSError:
|
||||
pass # file deleted between detect() and manifest write - skip it
|
||||
Path(manifest_path).parent.mkdir(parents=True, exist_ok=True)
|
||||
Path(manifest_path).write_text(json.dumps(manifest, indent=2))
|
||||
Path(manifest_path).write_text(json.dumps(manifest, indent=2), encoding="utf-8")
|
||||
|
||||
|
||||
def detect_incremental(root: Path, manifest_path: str = _MANIFEST_PATH) -> dict:
|
||||
|
||||
+10
-5
@@ -295,7 +295,7 @@ def to_json(G: nx.Graph, communities: dict[int, list[str]], output_path: str) ->
|
||||
conf = link.get("confidence", "EXTRACTED")
|
||||
link["confidence_score"] = _CONFIDENCE_SCORE_DEFAULTS.get(conf, 1.0)
|
||||
data["hyperedges"] = getattr(G, "graph", {}).get("hyperedges", [])
|
||||
with open(output_path, "w") as f:
|
||||
with open(output_path, "w", encoding="utf-8") as f:
|
||||
json.dump(data, f, indent=2)
|
||||
|
||||
|
||||
@@ -322,7 +322,7 @@ def to_cypher(G: nx.Graph, output_path: str) -> None:
|
||||
f"MATCH (a {{id: '{u_esc}'}}), (b {{id: '{v_esc}'}}) "
|
||||
f"MERGE (a)-[:{rel} {{confidence: '{conf}'}}]->(b);"
|
||||
)
|
||||
with open(output_path, "w") as f:
|
||||
with open(output_path, "w", encoding="utf-8") as f:
|
||||
f.write("\n".join(lines))
|
||||
|
||||
|
||||
@@ -467,7 +467,10 @@ def to_obsidian(
|
||||
# Map node_id → safe filename so wikilinks stay consistent.
|
||||
# Deduplicate: if two nodes produce the same filename, append a numeric suffix.
|
||||
def safe_name(label: str) -> str:
|
||||
return re.sub(r'[\\/*?:"<>|#^[\]]', "", label.replace("\r\n", " ").replace("\r", " ").replace("\n", " ")).strip() or "unnamed"
|
||||
cleaned = re.sub(r'[\\/*?:"<>|#^[\]]', "", label.replace("\r\n", " ").replace("\r", " ").replace("\n", " ")).strip()
|
||||
# Strip trailing .md/.mdx/.markdown so "CLAUDE.md" doesn't become "CLAUDE.md.md"
|
||||
cleaned = re.sub(r"\.(md|mdx|markdown)$", "", cleaned, flags=re.IGNORECASE)
|
||||
return cleaned or "unnamed"
|
||||
|
||||
node_filename: dict[str, str] = {}
|
||||
seen_names: dict[str, int] = {}
|
||||
@@ -681,7 +684,7 @@ def to_obsidian(
|
||||
for cid, label in sorted((community_labels or {}).items())
|
||||
]
|
||||
}
|
||||
(obsidian_dir / "graph.json").write_text(json.dumps(graph_config, indent=2))
|
||||
(obsidian_dir / "graph.json").write_text(json.dumps(graph_config, indent=2), encoding="utf-8")
|
||||
|
||||
return G.number_of_nodes() + community_notes_written
|
||||
|
||||
@@ -703,7 +706,9 @@ def to_canvas(
|
||||
CANVAS_COLORS = ["1", "2", "3", "4", "5", "6"] # red, orange, yellow, green, cyan, purple
|
||||
|
||||
def safe_name(label: str) -> str:
|
||||
return re.sub(r'[\\/*?:"<>|#^[\]]', "", label.replace("\r\n", " ").replace("\r", " ").replace("\n", " ")).strip() or "unnamed"
|
||||
cleaned = re.sub(r'[\\/*?:"<>|#^[\]]', "", label.replace("\r\n", " ").replace("\r", " ").replace("\n", " ")).strip()
|
||||
cleaned = re.sub(r"\.(md|mdx|markdown)$", "", cleaned, flags=re.IGNORECASE)
|
||||
return cleaned or "unnamed"
|
||||
|
||||
# Build node_filenames if not provided (same dedup logic as to_obsidian)
|
||||
if node_filenames is None:
|
||||
|
||||
+6
-6
@@ -113,12 +113,12 @@ def _install_hook(hooks_dir: Path, name: str, script: str, marker: str) -> str:
|
||||
"""Install a single git hook, appending if an existing hook is present."""
|
||||
hook_path = hooks_dir / name
|
||||
if hook_path.exists():
|
||||
content = hook_path.read_text()
|
||||
content = hook_path.read_text(encoding="utf-8")
|
||||
if marker in content:
|
||||
return f"already installed at {hook_path}"
|
||||
hook_path.write_text(content.rstrip() + "\n\n" + script)
|
||||
hook_path.write_text(content.rstrip() + "\n\n" + script, encoding="utf-8", newline="\n")
|
||||
return f"appended to existing {name} hook at {hook_path}"
|
||||
hook_path.write_text("#!/bin/sh\n" + script)
|
||||
hook_path.write_text("#!/bin/sh\n" + script, encoding="utf-8", newline="\n")
|
||||
hook_path.chmod(0o755)
|
||||
return f"installed at {hook_path}"
|
||||
|
||||
@@ -128,7 +128,7 @@ def _uninstall_hook(hooks_dir: Path, name: str, marker: str, marker_end: str) ->
|
||||
hook_path = hooks_dir / name
|
||||
if not hook_path.exists():
|
||||
return f"no {name} hook found - nothing to remove."
|
||||
content = hook_path.read_text()
|
||||
content = hook_path.read_text(encoding="utf-8")
|
||||
if marker not in content:
|
||||
return f"graphify hook not found in {name} - nothing to remove."
|
||||
new_content = re.sub(
|
||||
@@ -140,7 +140,7 @@ def _uninstall_hook(hooks_dir: Path, name: str, marker: str, marker_end: str) ->
|
||||
if not new_content or new_content in ("#!/bin/bash", "#!/bin/sh"):
|
||||
hook_path.unlink()
|
||||
return f"removed {name} hook at {hook_path}"
|
||||
hook_path.write_text(new_content + "\n")
|
||||
hook_path.write_text(new_content + "\n", encoding="utf-8", newline="\n")
|
||||
return f"graphify removed from {name} at {hook_path} (other hook content preserved)"
|
||||
|
||||
|
||||
@@ -183,7 +183,7 @@ def status(path: Path = Path(".")) -> str:
|
||||
p = hooks_dir / name
|
||||
if not p.exists():
|
||||
return "not installed"
|
||||
return "installed" if marker in p.read_text() else "not installed (hook exists but graphify not found)"
|
||||
return "installed" if marker in p.read_text(encoding="utf-8") else "not installed (hook exists but graphify not found)"
|
||||
|
||||
commit = _check("post-commit", _HOOK_MARKER)
|
||||
checkout = _check("post-checkout", _CHECKOUT_MARKER)
|
||||
|
||||
@@ -1,9 +1,17 @@
|
||||
# generate GRAPH_REPORT.md - the human-readable audit trail
|
||||
from __future__ import annotations
|
||||
import re
|
||||
from datetime import date
|
||||
import networkx as nx
|
||||
|
||||
|
||||
def _safe_community_name(label: str) -> str:
|
||||
"""Mirrors export.safe_name so community hub filenames and report wikilinks always agree."""
|
||||
cleaned = re.sub(r'[\\/*?:"<>|#^[\]]', "", label.replace("\r\n", " ").replace("\r", " ").replace("\n", " ")).strip()
|
||||
cleaned = re.sub(r"\.(md|mdx|markdown)$", "", cleaned, flags=re.IGNORECASE)
|
||||
return cleaned or "unnamed"
|
||||
|
||||
|
||||
def generate(
|
||||
G: nx.Graph,
|
||||
communities: dict[int, list[str]],
|
||||
@@ -48,6 +56,18 @@ def generate(
|
||||
f"- Extraction: {ext_pct}% EXTRACTED · {inf_pct}% INFERRED · {amb_pct}% AMBIGUOUS"
|
||||
+ (f" · INFERRED: {len(inf_edges)} edges (avg confidence: {inf_avg})" if inf_avg is not None else ""),
|
||||
f"- Token cost: {token_cost.get('input', 0):,} input · {token_cost.get('output', 0):,} output",
|
||||
]
|
||||
|
||||
# Community hub navigation - links to _COMMUNITY_*.md files in the Obsidian vault.
|
||||
# Without these, GRAPH_REPORT.md is a dead-end and the vault splits into disconnected components.
|
||||
if communities:
|
||||
lines += ["", "## Community Hubs (Navigation)"]
|
||||
for cid in communities:
|
||||
label = community_labels.get(cid, f"Community {cid}")
|
||||
safe = _safe_community_name(label)
|
||||
lines.append(f"- [[_COMMUNITY_{safe}|{label}]]")
|
||||
|
||||
lines += [
|
||||
"",
|
||||
"## God Nodes (most connected - your core abstractions)",
|
||||
]
|
||||
|
||||
+1
-1
@@ -16,7 +16,7 @@ def _load_graph(graph_path: str) -> nx.Graph:
|
||||
if not resolved.exists():
|
||||
raise FileNotFoundError(f"Graph file not found: {resolved}")
|
||||
safe = resolved
|
||||
data = json.loads(safe.read_text())
|
||||
data = json.loads(safe.read_text(encoding="utf-8"))
|
||||
try:
|
||||
return json_graph.node_link_graph(data, edges="links")
|
||||
except TypeError:
|
||||
|
||||
@@ -36,14 +36,15 @@ def validate_extraction(data: dict) -> list[str]:
|
||||
f"'{node['file_type']}' - must be one of {sorted(VALID_FILE_TYPES)}"
|
||||
)
|
||||
|
||||
# Edges
|
||||
if "edges" not in data:
|
||||
# Edges - accept "links" (NetworkX <= 3.1) as fallback for "edges"
|
||||
edge_list = data.get("edges") if "edges" in data else data.get("links")
|
||||
if edge_list is None:
|
||||
errors.append("Missing required key 'edges'")
|
||||
elif not isinstance(data["edges"], list):
|
||||
elif not isinstance(edge_list, list):
|
||||
errors.append("'edges' must be a list")
|
||||
else:
|
||||
node_ids = {n["id"] for n in data.get("nodes", []) if isinstance(n, dict) and "id" in n}
|
||||
for i, edge in enumerate(data["edges"]):
|
||||
for i, edge in enumerate(edge_list):
|
||||
if not isinstance(edge, dict):
|
||||
errors.append(f"Edge {i} must be an object")
|
||||
continue
|
||||
|
||||
+2
-2
@@ -53,7 +53,7 @@ def _rebuild_code(watch_path: Path, *, follow_symlinks: bool = False) -> bool:
|
||||
|
||||
report = generate(G, communities, cohesion, labels, gods, surprises, detection,
|
||||
{"input": 0, "output": 0}, str(watch_path), suggested_questions=questions)
|
||||
(out / "GRAPH_REPORT.md").write_text(report)
|
||||
(out / "GRAPH_REPORT.md").write_text(report, encoding="utf-8")
|
||||
to_json(G, communities, str(out / "graph.json"))
|
||||
|
||||
# clear stale needs_update flag if present
|
||||
@@ -75,7 +75,7 @@ def _notify_only(watch_path: Path) -> None:
|
||||
"""Write a flag file and print a notification (fallback for non-code-only corpora)."""
|
||||
flag = watch_path / "graphify-out" / "needs_update"
|
||||
flag.parent.mkdir(parents=True, exist_ok=True)
|
||||
flag.write_text("1")
|
||||
flag.write_text("1", encoding="utf-8")
|
||||
print(f"\n[graphify watch] New or changed files detected in {watch_path}")
|
||||
print("[graphify watch] Non-code files changed - semantic re-extraction requires LLM.")
|
||||
print("[graphify watch] Run `/graphify --update` in Claude Code to update the graph.")
|
||||
|
||||
Reference in New Issue
Block a user