From bd24ddb1d6df10af9a411e3f7f043f266e02d26f Mon Sep 17 00:00:00 2001 From: Safi Date: Wed, 8 Apr 2026 19:20:48 +0100 Subject: [PATCH] fix: hook JSON format, Go pkg scoping, xcassets PDF, cross-file guard, skill file paths (#83, #85, #52, #81) --- CHANGELOG.md | 8 ++ graphify/__main__.py | 4 +- graphify/detect.py | 6 ++ graphify/extract.py | 18 +++-- graphify/skill.md | 175 ++++++++++++++++++++-------------------- pyproject.toml | 2 +- tests/test_detect.py | 9 +++ tests/test_languages.py | 24 +++++- 8 files changed, 149 insertions(+), 97 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index d76310f..d6d21da 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,14 @@ Full release notes with details on each version: [GitHub Releases](https://github.com/safishamsi/graphify/releases) +## 0.3.13 (2026-04-08) + +- Fix: PreToolUse hook now outputs `additionalContext` JSON so Claude actually sees the graph reminder before Glob/Grep calls (#83) +- Fix: Go AST method receivers and type declarations now use package directory scope, eliminating disconnected duplicate type nodes across files in the same package (#85) +- Fix: PDFs inside Xcode asset catalogs (`.imageset`, `.xcassets`) are no longer misclassified as academic papers (#52) +- Fix: `_resolve_cross_file_imports` is now guarded with `if py_paths` and wrapped in try/except so a Python parser crash can't abort extraction for non-Python files (#52) +- Fix: Skill intermediate files (`.graphify_*.json`) now live in `graphify-out/` instead of project root, preventing git pollution (#81) + ## 0.3.12 (2026-04-07) - Fix: `sanitize_label` was double-encoding HTML entities in the interactive graph (`&lt;` instead of `<`) — removed `html.escape()` from `sanitize_label`; callers that inject directly into HTML now call `html.escape()` themselves (#66) diff --git a/graphify/__main__.py b/graphify/__main__.py index 2b3f3b5..c54d566 100644 --- a/graphify/__main__.py +++ b/graphify/__main__.py @@ -30,8 +30,8 @@ _SETTINGS_HOOK = { "type": "command", "command": ( "[ -f graphify-out/graph.json ] && " - "echo 'graphify: Knowledge graph exists. Read graphify-out/GRAPH_REPORT.md " - "for god nodes and community structure before searching raw files.' || true" + r"""echo '{"hookSpecificOutput":{"hookEventName":"PreToolUse","additionalContext":"graphify: Knowledge graph exists. Read graphify-out/GRAPH_REPORT.md for god nodes and community structure before searching raw files."}}' """ + "|| true" ), } ], diff --git a/graphify/detect.py b/graphify/detect.py index 5306d92..9779a8e 100644 --- a/graphify/detect.py +++ b/graphify/detect.py @@ -74,11 +74,17 @@ def _looks_like_paper(path: Path) -> bool: return False +_ASSET_DIR_MARKERS = {".imageset", ".xcassets", ".appiconset", ".colorset", ".launchimage"} + + def classify_file(path: Path) -> FileType | None: ext = path.suffix.lower() if ext in CODE_EXTENSIONS: return FileType.CODE if ext in PAPER_EXTENSIONS: + # PDFs inside Xcode asset catalogs are vector icons, not papers + if any(part.endswith(tuple(_ASSET_DIR_MARKERS)) for part in path.parts): + return None return FileType.PAPER if ext in IMAGE_EXTENSIONS: return FileType.IMAGE diff --git a/graphify/extract.py b/graphify/extract.py index c2d858d..f20fc5d 100644 --- a/graphify/extract.py +++ b/graphify/extract.py @@ -1171,6 +1171,9 @@ def extract_go(path: Path) -> dict: return {"nodes": [], "edges": [], "error": str(e)} stem = path.stem + # Use directory name as package scope so methods on the same type across + # multiple files in a package share one canonical type node. + pkg_scope = path.parent.name or stem str_path = str(path) nodes: list[dict] = [] edges: list[dict] = [] @@ -1235,7 +1238,7 @@ def extract_go(path: Path) -> dict: method_name = _read_text(name_node, source) line = node.start_point[0] + 1 if receiver_type: - parent_nid = _make_id(stem, receiver_type) + parent_nid = _make_id(pkg_scope, receiver_type) add_node(parent_nid, receiver_type, line) method_nid = _make_id(parent_nid, method_name) add_node(method_nid, f".{method_name}()", line) @@ -1256,7 +1259,7 @@ def extract_go(path: Path) -> dict: if name_node: type_name = _read_text(name_node, source) line = child.start_point[0] + 1 - type_nid = _make_id(stem, type_name) + type_nid = _make_id(pkg_scope, type_name) add_node(type_nid, type_name, line) add_edge(file_nid, type_nid, "contains", line) return @@ -2408,9 +2411,14 @@ def extract(paths: list[Path]) -> dict: # Add cross-file class-level edges (Python only - uses Python parser internally) py_paths = [p for p in paths if p.suffix == ".py"] - py_results = [r for r, p in zip(per_file, paths) if p.suffix == ".py"] - cross_file_edges = _resolve_cross_file_imports(py_results, py_paths) - all_edges.extend(cross_file_edges) + if py_paths: + py_results = [r for r, p in zip(per_file, paths) if p.suffix == ".py"] + try: + cross_file_edges = _resolve_cross_file_imports(py_results, py_paths) + all_edges.extend(cross_file_edges) + except Exception as exc: + import logging + logging.getLogger(__name__).warning("Cross-file import resolution failed, skipping: %s", exc) return { "nodes": all_nodes, diff --git a/graphify/skill.md b/graphify/skill.md index 3a6f51b..853f3ac 100644 --- a/graphify/skill.md +++ b/graphify/skill.md @@ -69,23 +69,24 @@ else fi $PYTHON -c "import graphify" 2>/dev/null || pip install graphifyy -q --break-system-packages 2>&1 | tail -3 # Write interpreter path for all subsequent steps -$PYTHON -c "import sys; open('.graphify_python', 'w').write(sys.executable)" +$PYTHON -c "import sys; open('graphify-out/.graphify_python', 'w').write(sys.executable)" +mkdir -p graphify-out ``` If the import succeeds, print nothing and move straight to Step 2. -**In every subsequent bash block, replace `python3` with `$(cat .graphify_python)` to use the correct interpreter.** +**In every subsequent bash block, replace `python3` with `$(cat graphify-out/.graphify_python)` to use the correct interpreter.** ### Step 2 - Detect files ```bash -$(cat .graphify_python) -c " +$(cat graphify-out/.graphify_python) -c " import json from graphify.detect import detect from pathlib import Path result = detect(Path('INPUT_PATH')) print(json.dumps(result)) -" > .graphify_detect.json +" > graphify-out/.graphify_detect.json ``` Replace INPUT_PATH with the actual path the user provided. Do NOT cat or print the JSON - read it silently and present a clean summary instead: @@ -119,23 +120,23 @@ Note: Parallelizing AST + semantic saves 5-15s on large corpora. AST is determin For any code files detected, run AST extraction in parallel with Part B subagents: ```bash -$(cat .graphify_python) -c " +$(cat graphify-out/.graphify_python) -c " import sys, json from graphify.extract import collect_files, extract from pathlib import Path import json code_files = [] -detect = json.loads(Path('.graphify_detect.json').read_text()) +detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text()) for f in detect.get('files', {}).get('code', []): code_files.extend(collect_files(Path(f)) if Path(f).is_dir() else [Path(f)]) if code_files: result = extract(code_files) - Path('.graphify_ast.json').write_text(json.dumps(result, indent=2)) + Path('graphify-out/.graphify_ast.json').write_text(json.dumps(result, indent=2)) print(f'AST: {len(result[\"nodes\"])} nodes, {len(result[\"edges\"])} edges') else: - Path('.graphify_ast.json').write_text(json.dumps({'nodes':[],'edges':[],'input_tokens':0,'output_tokens':0})) + Path('graphify-out/.graphify_ast.json').write_text(json.dumps({'nodes':[],'edges':[],'input_tokens':0,'output_tokens':0})) print('No code files - skipping AST extraction') " ``` @@ -147,7 +148,7 @@ else: **MANDATORY: You MUST use the Agent tool here. Reading files yourself one-by-one is forbidden - it is 5-10x slower. If you do not use the Agent tool you are doing this wrong.** Before dispatching subagents, print a timing estimate: -- Load `total_words` and file counts from `.graphify_detect.json` +- Load `total_words` and file counts from `graphify-out/.graphify_detect.json` - Estimate agents needed: `ceil(uncached_non_code_files / 22)` (chunk size is 20-25) - Estimate time: ~45s per agent batch (they run in parallel, so total ≈ 45s × ceil(agents/parallel_limit)) - Print: "Semantic extraction: ~N files → X agents, estimated ~Ys" @@ -157,28 +158,28 @@ Before dispatching subagents, print a timing estimate: Before dispatching any subagents, check which files already have cached extraction results: ```bash -$(cat .graphify_python) -c " +$(cat graphify-out/.graphify_python) -c " import json from graphify.cache import check_semantic_cache from pathlib import Path -detect = json.loads(Path('.graphify_detect.json').read_text()) +detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text()) all_files = [f for files in detect['files'].values() for f in files] cached_nodes, cached_edges, cached_hyperedges, uncached = check_semantic_cache(all_files) if cached_nodes or cached_edges or cached_hyperedges: - Path('.graphify_cached.json').write_text(json.dumps({'nodes': cached_nodes, 'edges': cached_edges, 'hyperedges': cached_hyperedges})) -Path('.graphify_uncached.txt').write_text('\n'.join(uncached)) + Path('graphify-out/.graphify_cached.json').write_text(json.dumps({'nodes': cached_nodes, 'edges': cached_edges, 'hyperedges': cached_hyperedges})) +Path('graphify-out/.graphify_uncached.txt').write_text('\n'.join(uncached)) print(f'Cache: {len(all_files)-len(uncached)} files hit, {len(uncached)} files need extraction') " ``` -Only dispatch subagents for files listed in `.graphify_uncached.txt`. If all files are cached, skip to Part C directly. +Only dispatch subagents for files listed in `graphify-out/.graphify_uncached.txt`. If all files are cached, skip to Part C directly. **Step B1 - Split into chunks** -Load files from `.graphify_uncached.txt`. Split into chunks of 20-25 files each. Each image gets its own chunk (vision needs separate context). +Load files from `graphify-out/.graphify_uncached.txt`. Split into chunks of 20-25 files each. Each image gets its own chunk (vision needs separate context). **Step B2 - Dispatch ALL subagents in a single message** @@ -257,25 +258,25 @@ If more than half the chunks failed, stop and tell the user. Save new results to cache: ```bash -$(cat .graphify_python) -c " +$(cat graphify-out/.graphify_python) -c " import json from graphify.cache import save_semantic_cache from pathlib import Path -new = json.loads(Path('.graphify_semantic_new.json').read_text()) if Path('.graphify_semantic_new.json').exists() else {'nodes':[],'edges':[],'hyperedges':[]} +new = json.loads(Path('graphify-out/.graphify_semantic_new.json').read_text()) if Path('graphify-out/.graphify_semantic_new.json').exists() else {'nodes':[],'edges':[],'hyperedges':[]} saved = save_semantic_cache(new.get('nodes', []), new.get('edges', []), new.get('hyperedges', [])) print(f'Cached {saved} files') " ``` -Merge cached + new results into `.graphify_semantic.json`: +Merge cached + new results into `graphify-out/.graphify_semantic.json`: ```bash -$(cat .graphify_python) -c " +$(cat graphify-out/.graphify_python) -c " import json from pathlib import Path -cached = json.loads(Path('.graphify_cached.json').read_text()) if Path('.graphify_cached.json').exists() else {'nodes':[],'edges':[],'hyperedges':[]} -new = json.loads(Path('.graphify_semantic_new.json').read_text()) if Path('.graphify_semantic_new.json').exists() else {'nodes':[],'edges':[],'hyperedges':[]} +cached = json.loads(Path('graphify-out/.graphify_cached.json').read_text()) if Path('graphify-out/.graphify_cached.json').exists() else {'nodes':[],'edges':[],'hyperedges':[]} +new = json.loads(Path('graphify-out/.graphify_semantic_new.json').read_text()) if Path('graphify-out/.graphify_semantic_new.json').exists() else {'nodes':[],'edges':[],'hyperedges':[]} all_nodes = cached['nodes'] + new.get('nodes', []) all_edges = cached['edges'] + new.get('edges', []) @@ -294,21 +295,21 @@ merged = { 'input_tokens': new.get('input_tokens', 0), 'output_tokens': new.get('output_tokens', 0), } -Path('.graphify_semantic.json').write_text(json.dumps(merged, indent=2)) +Path('graphify-out/.graphify_semantic.json').write_text(json.dumps(merged, indent=2)) print(f'Extraction complete - {len(deduped)} nodes, {len(all_edges)} edges ({len(cached[\"nodes\"])} from cache, {len(new.get(\"nodes\",[]))} new)') " ``` -Clean up temp files: `rm -f .graphify_cached.json .graphify_uncached.txt .graphify_semantic_new.json` +Clean up temp files: `rm -f graphify-out/.graphify_cached.json graphify-out/.graphify_uncached.txt graphify-out/.graphify_semantic_new.json` #### Part C - Merge AST + semantic into final extraction ```bash -$(cat .graphify_python) -c " +$(cat graphify-out/.graphify_python) -c " import sys, json from pathlib import Path -ast = json.loads(Path('.graphify_ast.json').read_text()) -sem = json.loads(Path('.graphify_semantic.json').read_text()) +ast = json.loads(Path('graphify-out/.graphify_ast.json').read_text()) +sem = json.loads(Path('graphify-out/.graphify_semantic.json').read_text()) # Merge: AST nodes first, semantic nodes deduplicated by id seen = {n['id'] for n in ast['nodes']} @@ -327,7 +328,7 @@ merged = { 'input_tokens': sem.get('input_tokens', 0), 'output_tokens': sem.get('output_tokens', 0), } -Path('.graphify_extract.json').write_text(json.dumps(merged, indent=2)) +Path('graphify-out/.graphify_extract.json').write_text(json.dumps(merged, indent=2)) total = len(merged_nodes) edges = len(merged_edges) print(f'Merged: {total} nodes, {edges} edges ({len(ast[\"nodes\"])} AST + {len(sem[\"nodes\"])} semantic)') @@ -338,7 +339,7 @@ print(f'Merged: {total} nodes, {edges} edges ({len(ast[\"nodes\"])} AST + {len(s ```bash mkdir -p graphify-out -$(cat .graphify_python) -c " +$(cat graphify-out/.graphify_python) -c " import sys, json from graphify.build import build_from_json from graphify.cluster import cluster, score_all @@ -347,8 +348,8 @@ from graphify.report import generate from graphify.export import to_json from pathlib import Path -extraction = json.loads(Path('.graphify_extract.json').read_text()) -detection = json.loads(Path('.graphify_detect.json').read_text()) +extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text()) +detection = json.loads(Path('graphify-out/.graphify_detect.json').read_text()) G = build_from_json(extraction) communities = cluster(G) @@ -371,7 +372,7 @@ analysis = { 'surprises': surprises, 'questions': questions, } -Path('.graphify_analysis.json').write_text(json.dumps(analysis, indent=2)) +Path('graphify-out/.graphify_analysis.json').write_text(json.dumps(analysis, indent=2)) if G.number_of_nodes() == 0: print('ERROR: Graph is empty - extraction produced no nodes.') print('Possible causes: all files were skipped, binary-only corpus, or extraction failed.') @@ -386,12 +387,12 @@ Replace INPUT_PATH with the actual path. ### Step 5 - Label communities -Read `.graphify_analysis.json`. For each community key, look at its node labels and write a 2-5 word plain-language name (e.g. "Attention Mechanism", "Training Pipeline", "Data Loading"). +Read `graphify-out/.graphify_analysis.json`. For each community key, look at its node labels and write a 2-5 word plain-language name (e.g. "Attention Mechanism", "Training Pipeline", "Data Loading"). Then regenerate the report and save the labels for the visualizer: ```bash -$(cat .graphify_python) -c " +$(cat graphify-out/.graphify_python) -c " import sys, json from graphify.build import build_from_json from graphify.cluster import score_all @@ -399,9 +400,9 @@ from graphify.analyze import god_nodes, surprising_connections, suggest_question from graphify.report import generate from pathlib import Path -extraction = json.loads(Path('.graphify_extract.json').read_text()) -detection = json.loads(Path('.graphify_detect.json').read_text()) -analysis = json.loads(Path('.graphify_analysis.json').read_text()) +extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text()) +detection = json.loads(Path('graphify-out/.graphify_detect.json').read_text()) +analysis = json.loads(Path('graphify-out/.graphify_analysis.json').read_text()) G = build_from_json(extraction) communities = {int(k): v for k, v in analysis['communities'].items()} @@ -416,7 +417,7 @@ questions = suggest_questions(G, communities, labels) report = generate(G, communities, cohesion, labels, analysis['gods'], analysis['surprises'], detection, tokens, 'INPUT_PATH', suggested_questions=questions) Path('graphify-out/GRAPH_REPORT.md').write_text(report) -Path('.graphify_labels.json').write_text(json.dumps({str(k): v for k, v in labels.items()})) +Path('graphify-out/.graphify_labels.json').write_text(json.dumps({str(k): v for k, v in labels.items()})) print('Report updated with community labels') " ``` @@ -433,15 +434,15 @@ If `--obsidian` was given: - If `--obsidian-dir ` was also given, use that path as the vault directory. Otherwise default to `graphify-out/obsidian`. ```bash -$(cat .graphify_python) -c " +$(cat graphify-out/.graphify_python) -c " import sys, json from graphify.build import build_from_json from graphify.export import to_obsidian, to_canvas from pathlib import Path -extraction = json.loads(Path('.graphify_extract.json').read_text()) -analysis = json.loads(Path('.graphify_analysis.json').read_text()) -labels_raw = json.loads(Path('.graphify_labels.json').read_text()) if Path('.graphify_labels.json').exists() else {} +extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text()) +analysis = json.loads(Path('graphify-out/.graphify_analysis.json').read_text()) +labels_raw = json.loads(Path('graphify-out/.graphify_labels.json').read_text()) if Path('graphify-out/.graphify_labels.json').exists() else {} G = build_from_json(extraction) communities = {int(k): v for k, v in analysis['communities'].items()} @@ -466,15 +467,15 @@ print(' _COMMUNITY_* - overview notes with cohesion scores and dataview queries Generate the HTML graph (always, unless `--no-viz`): ```bash -$(cat .graphify_python) -c " +$(cat graphify-out/.graphify_python) -c " import sys, json from graphify.build import build_from_json from graphify.export import to_html from pathlib import Path -extraction = json.loads(Path('.graphify_extract.json').read_text()) -analysis = json.loads(Path('.graphify_analysis.json').read_text()) -labels_raw = json.loads(Path('.graphify_labels.json').read_text()) if Path('.graphify_labels.json').exists() else {} +extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text()) +analysis = json.loads(Path('graphify-out/.graphify_analysis.json').read_text()) +labels_raw = json.loads(Path('graphify-out/.graphify_labels.json').read_text()) if Path('graphify-out/.graphify_labels.json').exists() else {} G = build_from_json(extraction) communities = {int(k): v for k, v in analysis['communities'].items()} @@ -493,13 +494,13 @@ else: **If `--neo4j`** - generate a Cypher file for manual import: ```bash -$(cat .graphify_python) -c " +$(cat graphify-out/.graphify_python) -c " import sys, json from graphify.build import build_from_json from graphify.export import to_cypher from pathlib import Path -G = build_from_json(json.loads(Path('.graphify_extract.json').read_text())) +G = build_from_json(json.loads(Path('graphify-out/.graphify_extract.json').read_text())) to_cypher(G, 'graphify-out/cypher.txt') print('cypher.txt written - import with: cypher-shell < graphify-out/cypher.txt') " @@ -508,15 +509,15 @@ print('cypher.txt written - import with: cypher-shell < graphify-out/cypher.txt' **If `--neo4j-push `** - push directly to a running Neo4j instance. Ask the user for credentials if not provided: ```bash -$(cat .graphify_python) -c " +$(cat graphify-out/.graphify_python) -c " import sys, json from graphify.build import build_from_json from graphify.cluster import cluster from graphify.export import push_to_neo4j from pathlib import Path -extraction = json.loads(Path('.graphify_extract.json').read_text()) -analysis = json.loads(Path('.graphify_analysis.json').read_text()) +extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text()) +analysis = json.loads(Path('graphify-out/.graphify_analysis.json').read_text()) G = build_from_json(extraction) communities = {int(k): v for k, v in analysis['communities'].items()} @@ -530,15 +531,15 @@ Replace `NEO4J_URI`, `NEO4J_USER`, `NEO4J_PASSWORD` with actual values. Default ### Step 7b - SVG export (only if --svg flag) ```bash -$(cat .graphify_python) -c " +$(cat graphify-out/.graphify_python) -c " import sys, json from graphify.build import build_from_json from graphify.export import to_svg from pathlib import Path -extraction = json.loads(Path('.graphify_extract.json').read_text()) -analysis = json.loads(Path('.graphify_analysis.json').read_text()) -labels_raw = json.loads(Path('.graphify_labels.json').read_text()) if Path('.graphify_labels.json').exists() else {} +extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text()) +analysis = json.loads(Path('graphify-out/.graphify_analysis.json').read_text()) +labels_raw = json.loads(Path('graphify-out/.graphify_labels.json').read_text()) if Path('graphify-out/.graphify_labels.json').exists() else {} G = build_from_json(extraction) communities = {int(k): v for k, v in analysis['communities'].items()} @@ -552,14 +553,14 @@ print('graph.svg written - embeds in Obsidian, Notion, GitHub READMEs') ### Step 7c - GraphML export (only if --graphml flag) ```bash -$(cat .graphify_python) -c " +$(cat graphify-out/.graphify_python) -c " import json from graphify.build import build_from_json from graphify.export import to_graphml from pathlib import Path -extraction = json.loads(Path('.graphify_extract.json').read_text()) -analysis = json.loads(Path('.graphify_analysis.json').read_text()) +extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text()) +analysis = json.loads(Path('graphify-out/.graphify_analysis.json').read_text()) G = build_from_json(extraction) communities = {int(k): v for k, v in analysis['communities'].items()} @@ -591,15 +592,15 @@ To configure in Claude Desktop, add to `claude_desktop_config.json`: ### Step 8 - Token reduction benchmark (only if total_words > 5000) -If `total_words` from `.graphify_detect.json` is greater than 5,000, run: +If `total_words` from `graphify-out/.graphify_detect.json` is greater than 5,000, run: ```bash -$(cat .graphify_python) -c " +$(cat graphify-out/.graphify_python) -c " import json from graphify.benchmark import run_benchmark, print_benchmark from pathlib import Path -detection = json.loads(Path('.graphify_detect.json').read_text()) +detection = json.loads(Path('graphify-out/.graphify_detect.json').read_text()) result = run_benchmark('graphify-out/graph.json', corpus_words=detection['total_words']) print_benchmark(result) " @@ -612,18 +613,18 @@ Print the output directly in chat. If `total_words <= 5000`, skip silently - the ### Step 9 - Save manifest, update cost tracker, clean up, and report ```bash -$(cat .graphify_python) -c " +$(cat graphify-out/.graphify_python) -c " import json from pathlib import Path from datetime import datetime, timezone from graphify.detect import save_manifest # Save manifest for --update -detect = json.loads(Path('.graphify_detect.json').read_text()) +detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text()) save_manifest(detect['files']) # Update cumulative cost tracker -extract = json.loads(Path('.graphify_extract.json').read_text()) +extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text()) input_tok = extract.get('input_tokens', 0) output_tok = extract.get('output_tokens', 0) @@ -646,7 +647,7 @@ cost_path.write_text(json.dumps(cost, indent=2)) print(f'This run: {input_tok:,} input tokens, {output_tok:,} output tokens') print(f'All time: {cost[\"total_input_tokens\"]:,} input, {cost[\"total_output_tokens\"]:,} output ({len(cost[\"runs\"])} runs)') " -rm -f .graphify_detect.json .graphify_extract.json .graphify_ast.json .graphify_semantic.json .graphify_analysis.json .graphify_labels.json .graphify_python +rm -f graphify-out/.graphify_detect.json graphify-out/.graphify_extract.json graphify-out/.graphify_ast.json graphify-out/.graphify_semantic.json graphify-out/.graphify_analysis.json graphify-out/.graphify_labels.json graphify-out/.graphify_python rm -f graphify-out/.needs_update 2>/dev/null || true ``` @@ -684,7 +685,7 @@ The graph is the map. Your job after the pipeline is to be the guide. Use when you've added or modified files since the last run. Only re-extracts changed files - saves tokens and time. ```bash -$(cat .graphify_python) -c " +$(cat graphify-out/.graphify_python) -c " import sys, json from graphify.detect import detect_incremental, save_manifest from pathlib import Path @@ -692,7 +693,7 @@ from pathlib import Path result = detect_incremental(Path('INPUT_PATH')) new_total = result.get('new_total', 0) print(json.dumps(result, indent=2)) -Path('.graphify_incremental.json').write_text(json.dumps(result)) +Path('graphify-out/.graphify_incremental.json').write_text(json.dumps(result)) if new_total == 0: print('No files changed since last run. Nothing to update.') raise SystemExit(0) @@ -703,11 +704,11 @@ print(f'{new_total} new/changed file(s) to re-extract.') If new files exist, first check whether all changed files are code files: ```bash -$(cat .graphify_python) -c " +$(cat graphify-out/.graphify_python) -c " import json from pathlib import Path -result = json.loads(open('.graphify_incremental.json').read()) if Path('.graphify_incremental.json').exists() else {} +result = json.loads(open('graphify-out/.graphify_incremental.json').read()) if Path('graphify-out/.graphify_incremental.json').exists() else {} code_exts = {'.py','.ts','.js','.go','.rs','.java','.cpp','.c','.rb','.swift','.kt','.cs','.scala','.php','.cc','.cxx','.hpp','.h','.kts','.lua','.toc'} new_files = result.get('new_files', {}) all_changed = [f for files in new_files.values() for f in files] @@ -723,7 +724,7 @@ If `code_only` is False (any changed file is a doc/paper/image): run the full St Then: ```bash -$(cat .graphify_python) -c " +$(cat graphify-out/.graphify_python) -c " import sys, json from graphify.build import build_from_json from graphify.export import to_json @@ -736,7 +737,7 @@ existing_data = json.loads(Path('graphify-out/graph.json').read_text()) G_existing = json_graph.node_link_graph(existing_data, edges='links') # Load new extraction -new_extraction = json.loads(Path('.graphify_extract.json').read_text()) +new_extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text()) G_new = build_from_json(new_extraction) # Merge: new nodes/edges into existing graph @@ -750,7 +751,7 @@ Then run Steps 4–8 on the merged graph as normal. After Step 4, show the graph diff: ```bash -$(cat .graphify_python) -c " +$(cat graphify-out/.graphify_python) -c " import json from graphify.analyze import graph_diff from graphify.build import build_from_json @@ -759,8 +760,8 @@ import networkx as nx from pathlib import Path # Load old graph (before update) from backup written before merge -old_data = json.loads(Path('.graphify_old.json').read_text()) if Path('.graphify_old.json').exists() else None -new_extract = json.loads(Path('.graphify_extract.json').read_text()) +old_data = json.loads(Path('graphify-out/.graphify_old.json').read_text()) if Path('graphify-out/.graphify_old.json').exists() else None +new_extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text()) G_new = build_from_json(new_extract) if old_data: @@ -774,8 +775,8 @@ if old_data: " ``` -Before the merge step, save the old graph: `cp graphify-out/graph.json .graphify_old.json` -Clean up after: `rm -f .graphify_old.json` +Before the merge step, save the old graph: `cp graphify-out/graph.json graphify-out/.graphify_old.json` +Clean up after: `rm -f graphify-out/.graphify_old.json` --- @@ -784,7 +785,7 @@ Clean up after: `rm -f .graphify_old.json` Skip Steps 1–3. Load the existing graph from `graphify-out/graph.json` and re-run clustering: ```bash -$(cat .graphify_python) -c " +$(cat graphify-out/.graphify_python) -c " import sys, json from graphify.cluster import cluster, score_all from graphify.analyze import god_nodes, surprising_connections @@ -817,7 +818,7 @@ analysis = { 'gods': gods, 'surprises': surprises, } -Path('.graphify_analysis.json').write_text(json.dumps(analysis, indent=2)) +Path('graphify-out/.graphify_analysis.json').write_text(json.dumps(analysis, indent=2)) print(f'Re-clustered: {len(communities)} communities') " ``` @@ -837,7 +838,7 @@ Two traversal modes - choose based on the question: First check the graph exists: ```bash -$(cat .graphify_python) -c " +$(cat graphify-out/.graphify_python) -c " from pathlib import Path if not Path('graphify-out/graph.json').exists(): print('ERROR: No graph found. Run /graphify first to build the graph.') @@ -855,7 +856,7 @@ Load `graphify-out/graph.json`, then: 5. If the graph lacks enough information, say so - do not hallucinate edges. ```bash -$(cat .graphify_python) -c " +$(cat graphify-out/.graphify_python) -c " import sys, json from networkx.readwrite import json_graph import networkx as nx @@ -946,7 +947,7 @@ Replace `QUESTION` with the user's actual question, `MODE` with `bfs` or `dfs`, After writing the answer, save it back into the graph so it improves future queries: ```bash -$(cat .graphify_python) -c " +$(cat graphify-out/.graphify_python) -c " from graphify.ingest import save_query_result from pathlib import Path save_query_result( @@ -970,7 +971,7 @@ Find the shortest path between two named concepts in the graph. First check the graph exists: ```bash -$(cat .graphify_python) -c " +$(cat graphify-out/.graphify_python) -c " from pathlib import Path if not Path('graphify-out/graph.json').exists(): print('ERROR: No graph found. Run /graphify first to build the graph.') @@ -980,7 +981,7 @@ if not Path('graphify-out/graph.json').exists(): If it fails, stop and tell the user to run `/graphify ` first. ```bash -$(cat .graphify_python) -c " +$(cat graphify-out/.graphify_python) -c " import json, sys import networkx as nx from networkx.readwrite import json_graph @@ -1032,7 +1033,7 @@ Replace `NODE_A` and `NODE_B` with the actual concept names from the user. Then After writing the explanation, save it back: ```bash -$(cat .graphify_python) -c " +$(cat graphify-out/.graphify_python) -c " from graphify.ingest import save_query_result from pathlib import Path save_query_result( @@ -1054,7 +1055,7 @@ Give a plain-language explanation of a single node - everything connected to it. First check the graph exists: ```bash -$(cat .graphify_python) -c " +$(cat graphify-out/.graphify_python) -c " from pathlib import Path if not Path('graphify-out/graph.json').exists(): print('ERROR: No graph found. Run /graphify first to build the graph.') @@ -1064,7 +1065,7 @@ if not Path('graphify-out/graph.json').exists(): If it fails, stop and tell the user to run `/graphify ` first. ```bash -$(cat .graphify_python) -c " +$(cat graphify-out/.graphify_python) -c " import json, sys import networkx as nx from networkx.readwrite import json_graph @@ -1109,7 +1110,7 @@ Replace `NODE_NAME` with the concept the user asked about. Then write a 3-5 sent After writing the explanation, save it back: ```bash -$(cat .graphify_python) -c " +$(cat graphify-out/.graphify_python) -c " from graphify.ingest import save_query_result from pathlib import Path save_query_result( @@ -1130,7 +1131,7 @@ print('Explanation saved to graphify-out/memory/') Fetch a URL and add it to the corpus, then update the graph. ```bash -$(cat .graphify_python) -c " +$(cat graphify-out/.graphify_python) -c " import sys from graphify.ingest import ingest from pathlib import Path diff --git a/pyproject.toml b/pyproject.toml index 28b292a..56ef195 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta" [project] name = "graphifyy" -version = "0.3.12" +version = "0.3.13" description = "AI coding assistant skill (Claude Code, Codex, OpenCode, OpenClaw) - turn any folder of code, docs, papers, or images into a queryable knowledge graph" readme = "README.md" license = { file = "LICENSE" } diff --git a/tests/test_detect.py b/tests/test_detect.py index aabe8a6..520d02d 100644 --- a/tests/test_detect.py +++ b/tests/test_detect.py @@ -15,6 +15,15 @@ def test_classify_markdown(): def test_classify_pdf(): assert classify_file(Path("paper.pdf")) == FileType.PAPER +def test_classify_pdf_in_xcassets_skipped(): + # PDFs inside Xcode asset catalogs are vector icons, not papers + asset_pdf = Path("MyApp/Images.xcassets/icon.imageset/icon.pdf") + assert classify_file(asset_pdf) is None + +def test_classify_pdf_in_xcassets_root_skipped(): + asset_pdf = Path("Pods/HXPHPicker/Assets.xcassets/photo.pdf") + assert classify_file(asset_pdf) is None + def test_classify_unknown_returns_none(): assert classify_file(Path("archive.zip")) is None diff --git a/tests/test_languages.py b/tests/test_languages.py index 1391007..551062c 100644 --- a/tests/test_languages.py +++ b/tests/test_languages.py @@ -1,11 +1,11 @@ -"""Tests for language extractors: Java, C, C++, Ruby, C#, Kotlin, Scala, PHP, Swift.""" +"""Tests for language extractors: Java, C, C++, Ruby, C#, Kotlin, Scala, PHP, Swift, Go.""" from __future__ import annotations from pathlib import Path import pytest from graphify.extract import ( extract_java, extract_c, extract_cpp, extract_ruby, extract_csharp, extract_kotlin, extract_scala, extract_php, - extract_swift, + extract_swift, extract_go, ) FIXTURES = Path(__file__).parent / "fixtures" @@ -427,3 +427,23 @@ def test_objc_no_dangling_edges(): node_ids = {n["id"] for n in r["nodes"]} for e in r["edges"]: assert e["source"] in node_ids, f"Dangling source: {e}" + + +# --------------------------------------------------------------------------- +# Go +# --------------------------------------------------------------------------- + +def test_go_receiver_methods_share_type_node(): + """Methods on the same receiver type must share one canonical type node.""" + r = extract_go(FIXTURES / "sample.go") + server_nodes = [n for n in r["nodes"] if n["label"] == "Server"] + # Both Start() and Stop() are on *Server — should produce exactly one Server node + assert len(server_nodes) == 1 + +def test_go_receiver_uses_pkg_scope(): + """Type node id should be scoped to directory, not file stem.""" + r = extract_go(FIXTURES / "sample.go") + server_nodes = [n for n in r["nodes"] if n["label"] == "Server"] + assert server_nodes + # Should NOT contain the file stem "sample" in the type node id + assert "sample" not in server_nodes[0]["id"].split(":")[0]