From 5d053721aba875156cf2a6ddd6953d8beee98147 Mon Sep 17 00:00:00 2001 From: Safi Date: Fri, 19 Jun 2026 16:17:49 +0100 Subject: [PATCH] Apply the #1392 runbook fixes to the Aider and Devin monolith skills The monoliths are hand-maintained single files frozen against a pinned pristine-v8 blob by the round-trip guard, so they were excluded from the 0.8.44 #1392 batch. Evolve the guard from a positional zip (line-count-exact, single-line-class allowlist) to a multiset diff that classifies every added/removed line against documented sanctioned change-classes, so the multi-line fixes can land while any unsanctioned drift still fails. Add predicates for the four fix classes and broaden the enum/chunk-cleanup predicates to match both the v8 and rewritten forms. Both monoliths now: thread directed=IS_DIRECTED through every build_from_json call (a --directed run no longer collapses reciprocal edges), scope semantic extraction to document/paper/image, unlink a stale .graphify_cached.json on a cache miss, and run Step 4's zero-node guard before any write with the report/analysis gated on to_json persisting the graph (#1392). Co-Authored-By: Claude Opus 4.8 (1M context) --- CHANGELOG.md | 2 + graphify/skill-aider.md | 49 +++-- graphify/skill-devin.md | 51 ++++-- tests/test_skillgen.py | 89 +++++----- .../expected/graphify__skill-aider.md | 49 +++-- .../expected/graphify__skill-devin.md | 51 ++++-- tools/skillgen/fragments/core/aider.md | 49 +++-- tools/skillgen/fragments/core/devin.md | 51 ++++-- tools/skillgen/gen.py | 168 ++++++++++++++---- 9 files changed, 386 insertions(+), 173 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 1b3859a..a4af2f3 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,8 @@ Full release notes with details on each version: [GitHub Releases](https://githu ## Unreleased +- Fix: the Aider and Devin monolith skills now carry the #1392 runbook fixes that the split skill got in 0.8.44. These single-file skills are hand-maintained and frozen against a pinned pristine-v8 blob by a round-trip guard, so they had been excluded. The guard is now a multiset diff that classifies every added/removed line against documented sanctioned change-classes (rather than a positional zip that forbade any line-count change), which lets the multi-line fixes land while still failing on any unsanctioned drift. Both monoliths now propagate `directed=IS_DIRECTED` into every `build_from_json` call (a `--directed` run no longer collapses reciprocal edges), scope semantic extraction to document/paper/image (code is covered by the AST pass), delete `.graphify_cached.json` on a cache miss, and run Step 4's zero-node guard before any write with the report/analysis gated on `to_json` actually persisting the graph (#1392). + ## 0.8.44 (2026-06-19) - Fix: generated Claude/agent skill, crash & data-loss bugs in the runbooks (#1392). (1) Semantic chunk files were written under the **scanned dir** (`.graphify_root`) but the merge globs **cwd** `graphify-out/`, so a non-cwd scan produced "no nodes"; chunk paths are now derived from cwd. (2) Code-only corpora skipped Part B but Part C reads `.graphify_semantic.json` unconditionally, raising `FileNotFoundError`; the fast path now writes an empty semantic file first. (3) `--cluster-only` told the agent to re-run Steps 5-9, which read intermediate files a prior cleanup deleted (`FileNotFoundError`); it now relies on the self-contained `graphify cluster-only` CLI. (4) Step 4's zero-node guard ran *after* `GRAPH_REPORT.md`/`graph.json`/analysis were written, and `GRAPH_REPORT.md` was written before `to_json`'s #479 shrink-guard; the guard now runs before any write and the report/analysis are written only when `to_json` actually persisted the graph. diff --git a/graphify/skill-aider.md b/graphify/skill-aider.md index 6f37a5d..7cea232 100644 --- a/graphify/skill-aider.md +++ b/graphify/skill-aider.md @@ -226,12 +226,19 @@ from graphify.cache import check_semantic_cache from pathlib import Path detect = json.loads(Path('.graphify_detect.json').read_text()) -all_files = [f for files in detect['files'].values() for f in files] +# Only content files go to semantic extraction. Code is already covered +# structurally by the AST pass; flattening every category here makes the +# extraction step re-read every source file (#1392). +all_files = [f for cat in ('document', 'paper', 'image') for f in detect['files'].get(cat, [])] cached_nodes, cached_edges, cached_hyperedges, uncached = check_semantic_cache(all_files) +# Always (re)write the cache file: write hits, else DELETE any leftover from a +# prior run so Part C never merges a stale .graphify_cached.json (#1392). if cached_nodes or cached_edges or cached_hyperedges: Path('.graphify_cached.json').write_text(json.dumps({'nodes': cached_nodes, 'edges': cached_edges, 'hyperedges': cached_hyperedges})) +else: + Path('.graphify_cached.json').unlink(missing_ok=True) Path('.graphify_uncached.txt').write_text('\n'.join(uncached)) print(f'Cache: {len(all_files)-len(uncached)} files hit, {len(uncached)} files need extraction') " @@ -377,6 +384,8 @@ print(f'Merged: {total} nodes, {edges} edges ({len(ast[\"nodes\"])} AST + {len(s ### Step 4 - Build graph, cluster, analyze, generate outputs +**Before starting:** the code blocks below pass `directed=IS_DIRECTED` to `build_from_json()`. Replace `IS_DIRECTED` with `True` if `--directed` was given (builds a `DiGraph` preserving edge direction source->target), otherwise `False` (the default undirected `Graph`). Substitute it everywhere it appears, the same way you substitute `INPUT_PATH` - do not leave the literal `IS_DIRECTED` in the code. + ```bash mkdir -p graphify-out $(cat graphify-out/.graphify_python) -c " @@ -391,7 +400,13 @@ from pathlib import Path extraction = json.loads(Path('.graphify_extract.json').read_text()) detection = json.loads(Path('.graphify_detect.json').read_text()) -G = build_from_json(extraction) +G = build_from_json(extraction, directed=IS_DIRECTED) +# Guard BEFORE any write: an empty extraction must not clobber a good graph.json / +# GRAPH_REPORT.md / analysis sidecar. Check immediately after build (#1392). +if G.number_of_nodes() == 0: + print('ERROR: Graph is empty - extraction produced no nodes.') + print('Possible causes: all files were skipped, binary-only corpus, or extraction failed.') + raise SystemExit(1) communities = cluster(G) cohesion = score_all(G, communities) tokens = {'input': extraction.get('input_tokens', 0), 'output': extraction.get('output_tokens', 0)} @@ -401,9 +416,15 @@ labels = {cid: 'Community ' + str(cid) for cid in communities} # Placeholder questions - regenerated with real labels in Step 5 questions = suggest_questions(G, communities, labels) +# Persist the graph first and only write the report/analysis if it actually +# persisted - to_json refuses to shrink an existing graph.json (#479), and a +# report describing a graph we did not write would be a lie (#1392). +wrote = to_json(G, communities, 'graphify-out/graph.json') +if not wrote: + print('ERROR: refused to shrink graphify-out/graph.json (fewer nodes than the existing graph). Run a full rebuild to be safe.') + raise SystemExit(1) report = generate(G, communities, cohesion, labels, gods, surprises, detection, tokens, 'INPUT_PATH', suggested_questions=questions) Path('graphify-out/GRAPH_REPORT.md').write_text(report) -to_json(G, communities, 'graphify-out/graph.json') analysis = { 'communities': {str(k): v for k, v in communities.items()}, @@ -413,10 +434,6 @@ analysis = { 'questions': questions, } Path('.graphify_analysis.json').write_text(json.dumps(analysis, indent=2)) -if G.number_of_nodes() == 0: - print('ERROR: Graph is empty - extraction produced no nodes.') - print('Possible causes: all files were skipped, binary-only corpus, or extraction failed.') - raise SystemExit(1) print(f'Graph: {G.number_of_nodes()} nodes, {G.number_of_edges()} edges, {len(communities)} communities') " ``` @@ -444,7 +461,7 @@ extraction = json.loads(Path('.graphify_extract.json').read_text()) detection = json.loads(Path('.graphify_detect.json').read_text()) analysis = json.loads(Path('.graphify_analysis.json').read_text()) -G = build_from_json(extraction) +G = build_from_json(extraction, directed=IS_DIRECTED) communities = {int(k): v for k, v in analysis['communities'].items()} cohesion = {int(k): v for k, v in analysis['cohesion'].items()} tokens = {'input': extraction.get('input_tokens', 0), 'output': extraction.get('output_tokens', 0)} @@ -482,7 +499,7 @@ extraction = json.loads(Path('.graphify_extract.json').read_text()) analysis = json.loads(Path('.graphify_analysis.json').read_text()) labels_raw = json.loads(Path('.graphify_labels.json').read_text()) if Path('.graphify_labels.json').exists() else {} -G = build_from_json(extraction) +G = build_from_json(extraction, directed=IS_DIRECTED) communities = {int(k): v for k, v in analysis['communities'].items()} cohesion = {int(k): v for k, v in analysis['cohesion'].items()} labels = {int(k): v for k, v in labels_raw.items()} @@ -513,7 +530,7 @@ extraction = json.loads(Path('.graphify_extract.json').read_text()) analysis = json.loads(Path('.graphify_analysis.json').read_text()) labels_raw = json.loads(Path('.graphify_labels.json').read_text()) if Path('.graphify_labels.json').exists() else {} -G = build_from_json(extraction) +G = build_from_json(extraction, directed=IS_DIRECTED) communities = {int(k): v for k, v in analysis['communities'].items()} labels = {int(k): v for k, v in labels_raw.items()} @@ -536,7 +553,7 @@ from graphify.build import build_from_json from graphify.export import to_cypher from pathlib import Path -G = build_from_json(json.loads(Path('.graphify_extract.json').read_text())) +G = build_from_json(json.loads(Path('.graphify_extract.json').read_text()), directed=IS_DIRECTED) to_cypher(G, 'graphify-out/cypher.txt') print('cypher.txt written - import with: cypher-shell < graphify-out/cypher.txt') " @@ -554,7 +571,7 @@ from pathlib import Path extraction = json.loads(Path('.graphify_extract.json').read_text()) analysis = json.loads(Path('.graphify_analysis.json').read_text()) -G = build_from_json(extraction) +G = build_from_json(extraction, directed=IS_DIRECTED) communities = {int(k): v for k, v in analysis['communities'].items()} result = push_to_neo4j(G, uri='NEO4J_URI', user='NEO4J_USER', password='NEO4J_PASSWORD', communities=communities) @@ -577,7 +594,7 @@ extraction = json.loads(Path('.graphify_extract.json').read_text()) analysis = json.loads(Path('.graphify_analysis.json').read_text()) labels_raw = json.loads(Path('.graphify_labels.json').read_text()) if Path('.graphify_labels.json').exists() else {} -G = build_from_json(extraction) +G = build_from_json(extraction, directed=IS_DIRECTED) communities = {int(k): v for k, v in analysis['communities'].items()} labels = {int(k): v for k, v in labels_raw.items()} @@ -598,7 +615,7 @@ from pathlib import Path extraction = json.loads(Path('.graphify_extract.json').read_text()) analysis = json.loads(Path('.graphify_analysis.json').read_text()) -G = build_from_json(extraction) +G = build_from_json(extraction, directed=IS_DIRECTED) communities = {int(k): v for k, v in analysis['communities'].items()} to_graphml(G, communities, 'graphify-out/graph.graphml') @@ -795,7 +812,7 @@ G_existing = json_graph.node_link_graph(existing_data, edges='links') # Load new extraction new_extraction = json.loads(Path('.graphify_extract.json').read_text()) -G_new = build_from_json(new_extraction) +G_new = build_from_json(new_extraction, directed=IS_DIRECTED) # Merge: new nodes/edges into existing graph G_existing.update(G_new) @@ -819,7 +836,7 @@ from pathlib import Path # Load old graph (before update) from backup written before merge old_data = json.loads(Path('.graphify_old.json').read_text()) if Path('.graphify_old.json').exists() else None new_extract = json.loads(Path('.graphify_extract.json').read_text()) -G_new = build_from_json(new_extract) +G_new = build_from_json(new_extract, directed=IS_DIRECTED) if old_data: G_old = json_graph.node_link_graph(old_data, edges='links') diff --git a/graphify/skill-devin.md b/graphify/skill-devin.md index 6e46612..6cbb55d 100644 --- a/graphify/skill-devin.md +++ b/graphify/skill-devin.md @@ -243,12 +243,19 @@ from graphify.cache import check_semantic_cache from pathlib import Path detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text()) -all_files = [f for files in detect['files'].values() for f in files] +# Only content files go to semantic extraction. Code is already covered +# structurally by the AST pass; flattening every category here makes the +# extraction step re-read every source file (#1392). +all_files = [f for cat in ('document', 'paper', 'image') for f in detect['files'].get(cat, [])] cached_nodes, cached_edges, cached_hyperedges, uncached = check_semantic_cache(all_files) +# Always (re)write the cache file: write hits, else DELETE any leftover from a +# prior run so Part C never merges a stale .graphify_cached.json (#1392). if cached_nodes or cached_edges or cached_hyperedges: Path('graphify-out/.graphify_cached.json').write_text(json.dumps({'nodes': cached_nodes, 'edges': cached_edges, 'hyperedges': cached_hyperedges})) +else: + Path('graphify-out/.graphify_cached.json').unlink(missing_ok=True) Path('graphify-out/.graphify_uncached.txt').write_text('\n'.join(uncached)) print(f'Cache: {len(all_files)-len(uncached)} files hit, {len(uncached)} files need extraction') " @@ -442,6 +449,8 @@ print(f'Merged: {total} nodes, {edges} edges ({len(ast[\"nodes\"])} AST + {len(s ### Step 4 - Build graph, cluster, analyze, generate outputs +**Before starting:** the code blocks below pass `directed=IS_DIRECTED` to `build_from_json()`. Replace `IS_DIRECTED` with `True` if `--directed` was given (builds a `DiGraph` preserving edge direction source->target), otherwise `False` (the default undirected `Graph`). Substitute it everywhere it appears, the same way you substitute `INPUT_PATH` - do not leave the literal `IS_DIRECTED` in the code. + ```bash mkdir -p graphify-out $(cat graphify-out/.graphify_python) -c " @@ -456,7 +465,13 @@ from pathlib import Path extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text()) detection = json.loads(Path('graphify-out/.graphify_detect.json').read_text()) -G = build_from_json(extraction) +G = build_from_json(extraction, directed=IS_DIRECTED) +# Guard BEFORE any write: an empty extraction must not clobber a good graph.json / +# GRAPH_REPORT.md / analysis sidecar. Check immediately after build (#1392). +if G.number_of_nodes() == 0: + print('ERROR: Graph is empty - extraction produced no nodes.') + print('Possible causes: all files were skipped, binary-only corpus, or extraction failed.') + raise SystemExit(1) communities = cluster(G) cohesion = score_all(G, communities) tokens = {'input': extraction.get('input_tokens', 0), 'output': extraction.get('output_tokens', 0)} @@ -466,9 +481,15 @@ labels = {cid: 'Community ' + str(cid) for cid in communities} # Placeholder questions - regenerated with real labels in Step 5 questions = suggest_questions(G, communities, labels) +# Persist the graph first and only write the report/analysis if it actually +# persisted - to_json refuses to shrink an existing graph.json (#479), and a +# report describing a graph we did not write would be a lie (#1392). +wrote = to_json(G, communities, 'graphify-out/graph.json') +if not wrote: + print('ERROR: refused to shrink graphify-out/graph.json (fewer nodes than the existing graph). Run a full rebuild to be safe.') + raise SystemExit(1) report = generate(G, communities, cohesion, labels, gods, surprises, detection, tokens, 'INPUT_PATH', suggested_questions=questions) Path('graphify-out/GRAPH_REPORT.md').write_text(report) -to_json(G, communities, 'graphify-out/graph.json') analysis = { 'communities': {str(k): v for k, v in communities.items()}, @@ -478,10 +499,6 @@ analysis = { 'questions': questions, } Path('graphify-out/.graphify_analysis.json').write_text(json.dumps(analysis, indent=2)) -if G.number_of_nodes() == 0: - print('ERROR: Graph is empty - extraction produced no nodes.') - print('Possible causes: all files were skipped, binary-only corpus, or extraction failed.') - raise SystemExit(1) print(f'Graph: {G.number_of_nodes()} nodes, {G.number_of_edges()} edges, {len(communities)} communities') " ``` @@ -509,7 +526,7 @@ extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text()) detection = json.loads(Path('graphify-out/.graphify_detect.json').read_text()) analysis = json.loads(Path('graphify-out/.graphify_analysis.json').read_text()) -G = build_from_json(extraction) +G = build_from_json(extraction, directed=IS_DIRECTED) communities = {int(k): v for k, v in analysis['communities'].items()} cohesion = {int(k): v for k, v in analysis['cohesion'].items()} tokens = {'input': extraction.get('input_tokens', 0), 'output': extraction.get('output_tokens', 0)} @@ -547,7 +564,7 @@ extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text()) analysis = json.loads(Path('graphify-out/.graphify_analysis.json').read_text()) labels_raw = json.loads(Path('graphify-out/.graphify_labels.json').read_text()) if Path('graphify-out/.graphify_labels.json').exists() else {} -G = build_from_json(extraction) +G = build_from_json(extraction, directed=IS_DIRECTED) communities = {int(k): v for k, v in analysis['communities'].items()} cohesion = {int(k): v for k, v in analysis['cohesion'].items()} labels = {int(k): v for k, v in labels_raw.items()} @@ -578,7 +595,7 @@ extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text()) analysis = json.loads(Path('graphify-out/.graphify_analysis.json').read_text()) labels_raw = json.loads(Path('graphify-out/.graphify_labels.json').read_text()) if Path('graphify-out/.graphify_labels.json').exists() else {} -G = build_from_json(extraction) +G = build_from_json(extraction, directed=IS_DIRECTED) communities = {int(k): v for k, v in analysis['communities'].items()} labels = {int(k): v for k, v in labels_raw.items()} @@ -632,7 +649,7 @@ extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text()) analysis = json.loads(Path('graphify-out/.graphify_analysis.json').read_text()) labels_raw = json.loads(Path('graphify-out/.graphify_labels.json').read_text()) if Path('graphify-out/.graphify_labels.json').exists() else {} -G = build_from_json(extraction) +G = build_from_json(extraction, directed=IS_DIRECTED) communities = {int(k): v for k, v in analysis['communities'].items()} cohesion = {int(k): v for k, v in analysis['cohesion'].items()} labels = {int(k): v for k, v in labels_raw.items()} @@ -655,7 +672,7 @@ from graphify.build import build_from_json from graphify.export import to_cypher from pathlib import Path -G = build_from_json(json.loads(Path('graphify-out/.graphify_extract.json').read_text())) +G = build_from_json(json.loads(Path('graphify-out/.graphify_extract.json').read_text()), directed=IS_DIRECTED) to_cypher(G, 'graphify-out/cypher.txt') print('cypher.txt written - import with: cypher-shell < graphify-out/cypher.txt') " @@ -672,7 +689,7 @@ from pathlib import Path extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text()) analysis = json.loads(Path('graphify-out/.graphify_analysis.json').read_text()) -G = build_from_json(extraction) +G = build_from_json(extraction, directed=IS_DIRECTED) communities = {int(k): v for k, v in analysis['communities'].items()} result = push_to_neo4j(G, uri='NEO4J_URI', user='NEO4J_USER', password='NEO4J_PASSWORD', communities=communities) @@ -695,7 +712,7 @@ extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text()) analysis = json.loads(Path('graphify-out/.graphify_analysis.json').read_text()) labels_raw = json.loads(Path('graphify-out/.graphify_labels.json').read_text()) if Path('graphify-out/.graphify_labels.json').exists() else {} -G = build_from_json(extraction) +G = build_from_json(extraction, directed=IS_DIRECTED) communities = {int(k): v for k, v in analysis['communities'].items()} labels = {int(k): v for k, v in labels_raw.items()} @@ -716,7 +733,7 @@ from pathlib import Path extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text()) analysis = json.loads(Path('graphify-out/.graphify_analysis.json').read_text()) -G = build_from_json(extraction) +G = build_from_json(extraction, directed=IS_DIRECTED) communities = {int(k): v for k, v in analysis['communities'].items()} to_graphml(G, communities, 'graphify-out/graph.graphml') @@ -932,7 +949,7 @@ G_existing = json_graph.node_link_graph(existing_data, edges='links') # Load new extraction new_extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text()) -G_new = build_from_json(new_extraction) +G_new = build_from_json(new_extraction, directed=IS_DIRECTED) # Merge: new nodes/edges into existing graph G_existing.update(G_new) @@ -955,7 +972,7 @@ from pathlib import Path old_data = json.loads(Path('graphify-out/.graphify_old.json').read_text()) if Path('graphify-out/.graphify_old.json').exists() else None new_extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text()) -G_new = build_from_json(new_extract) +G_new = build_from_json(new_extract, directed=IS_DIRECTED) if old_data: G_old = json_graph.node_link_graph(old_data, edges='links') diff --git a/tests/test_skillgen.py b/tests/test_skillgen.py index da6e28f..08a875c 100644 --- a/tests/test_skillgen.py +++ b/tests/test_skillgen.py @@ -450,54 +450,57 @@ def test_monolith_roundtrip_passes_for_aider_and_devin(): assert problems == [], f"[{key}]\n" + "\n".join(problems) -def test_monoliths_change_only_the_enum_description_and_chunk_cleanup(): - """The rendered monolith differs from v8 on exactly the allowed lines. +def test_monoliths_change_only_sanctioned_lines(): + """Every line that differs from pristine v8 is a sanctioned change-class. - Three changes are now in play for the monoliths: the file_type enum unified to - the six-value superset (the prose guidance line + the schema line), the - frontmatter description unified across all platforms, and the shell-agnostic - chunk-cleanup rewrite (#1172). Nothing else may differ. + The round-trip (multiset diff vs the pinned v8 blob) must come back clean: + each added/removed line matches one of the documented sanctioned predicates + in gen — the enum unification, the unified description, the chunk-cleanup + rewrite (#1172), and the four #1392 runbook fixes. Anything else is drift. """ platforms = gen.load_platforms() for key in ("aider", "devin"): - rendered = gen.render(platforms[key])[0].content.splitlines() - # Strip trigger: lines from the reference — their removal (#1180) is a - # permitted diff alongside enum, description, and chunk-cleanup changes. - original = [ - l for l in gen._normalise(gen._git_show(platforms[key].roundtrip_ref)).splitlines() - if not gen._is_trigger_line(l) - ] - assert len(rendered) == len(original), f"[{key}] line count changed" - diff_idx = [i for i, (r, o) in enumerate(zip(rendered, original)) if r != o] - # Four lines change: the prose enum guidance, the schema line, - # the frontmatter description, and the chunk-cleanup rewrite. - assert len(diff_idx) == 4, f"[{key}] expected 4 changed lines, got {len(diff_idx)}" - enum_changes = 0 - desc_changes = 0 - cleanup_changes = 0 - for i in diff_idx: - line = rendered[i] - if gen.ENUM_VALUES in line or gen.ENUM_PROSE in line: - enum_changes += 1 - elif line.lstrip().startswith("description:"): - desc_changes += 1 - assert UNIFIED_DESCRIPTION in line, ( - f"[{key}] description line is not the unified text: {line!r}" - ) - elif gen._is_chunk_cleanup_line(line): - cleanup_changes += 1 - # The unmatched-glob abort is fixed: the rm no longer carries the - # bare chunk glob, and a find ... -delete sweeps the chunks. - assert ".graphify_chunk_*.json" not in line.split("find", 1)[0] - else: - raise AssertionError( - f"[{key}] changed line {i} is none of enum/description/cleanup: {line!r}" - ) - assert enum_changes == 2, f"[{key}] expected 2 enum line changes, got {enum_changes}" - assert desc_changes == 1, f"[{key}] expected 1 description change, got {desc_changes}" - assert cleanup_changes == 1, f"[{key}] expected 1 cleanup change, got {cleanup_changes}" + assert gen.monolith_roundtrip(platforms[key]) == [] # The six-value superset replaced the five-value enum in both files. - assert any(gen.ENUM_VALUES in line for line in rendered) + rendered = gen.render(platforms[key])[0].content + assert gen.ENUM_VALUES in rendered + assert UNIFIED_DESCRIPTION in rendered + + +def test_monoliths_carry_the_1392_runbook_fixes(): + """The four #1392 data-loss/correctness fixes are present in both monoliths. + + The round-trip allows these change-classes; this test asserts they are + actually applied, so a regression that drops a fix fails here even though the + round-trip (which only forbids *unsanctioned* drift) would still pass. + """ + platforms = gen.load_platforms() + for key in ("aider", "devin"): + body = gen.render(platforms[key])[0].content + + # #6/#7 directed propagation: no bare build_from_json call survives, and + # the IS_DIRECTED substitution instruction is present. + assert "directed=IS_DIRECTED" in body + assert "build_from_json(extraction)" not in body + assert "Substitute it everywhere it appears" in body + + # #10 content-only semantic scope: code is no longer flattened in. + assert "for cat in ('document', 'paper', 'image')" in body + assert "detect['files'].values()" not in body + + # #12 stale-cache unlink on a miss. + assert ".graphify_cached.json').unlink(missing_ok=True)" in body + + # #18/#20 zero-node guard before any write, report/analysis gated on + # to_json's return. + lines = body.splitlines() + build_i = next(i for i, l in enumerate(lines) if "G = build_from_json(extraction, directed=IS_DIRECTED)" in l) + guard_i = next(i for i, l in enumerate(lines[build_i:], build_i) if "number_of_nodes() == 0" in l) + report_i = next(i for i, l in enumerate(lines[build_i:], build_i) if "GRAPH_REPORT.md').write_text(report)" in l) + wrote_i = next(i for i, l in enumerate(lines[build_i:], build_i) if l.strip().startswith("wrote = to_json(")) + # guard fires right after the build, before the graph/report are written. + assert build_i < guard_i < wrote_i < report_i, f"[{key}] Step 4 ordering not fixed" + assert "if not wrote:" in body def test_devin_keeps_its_multi_field_frontmatter(): diff --git a/tools/skillgen/expected/graphify__skill-aider.md b/tools/skillgen/expected/graphify__skill-aider.md index 6f37a5d..7cea232 100644 --- a/tools/skillgen/expected/graphify__skill-aider.md +++ b/tools/skillgen/expected/graphify__skill-aider.md @@ -226,12 +226,19 @@ from graphify.cache import check_semantic_cache from pathlib import Path detect = json.loads(Path('.graphify_detect.json').read_text()) -all_files = [f for files in detect['files'].values() for f in files] +# Only content files go to semantic extraction. Code is already covered +# structurally by the AST pass; flattening every category here makes the +# extraction step re-read every source file (#1392). +all_files = [f for cat in ('document', 'paper', 'image') for f in detect['files'].get(cat, [])] cached_nodes, cached_edges, cached_hyperedges, uncached = check_semantic_cache(all_files) +# Always (re)write the cache file: write hits, else DELETE any leftover from a +# prior run so Part C never merges a stale .graphify_cached.json (#1392). if cached_nodes or cached_edges or cached_hyperedges: Path('.graphify_cached.json').write_text(json.dumps({'nodes': cached_nodes, 'edges': cached_edges, 'hyperedges': cached_hyperedges})) +else: + Path('.graphify_cached.json').unlink(missing_ok=True) Path('.graphify_uncached.txt').write_text('\n'.join(uncached)) print(f'Cache: {len(all_files)-len(uncached)} files hit, {len(uncached)} files need extraction') " @@ -377,6 +384,8 @@ print(f'Merged: {total} nodes, {edges} edges ({len(ast[\"nodes\"])} AST + {len(s ### Step 4 - Build graph, cluster, analyze, generate outputs +**Before starting:** the code blocks below pass `directed=IS_DIRECTED` to `build_from_json()`. Replace `IS_DIRECTED` with `True` if `--directed` was given (builds a `DiGraph` preserving edge direction source->target), otherwise `False` (the default undirected `Graph`). Substitute it everywhere it appears, the same way you substitute `INPUT_PATH` - do not leave the literal `IS_DIRECTED` in the code. + ```bash mkdir -p graphify-out $(cat graphify-out/.graphify_python) -c " @@ -391,7 +400,13 @@ from pathlib import Path extraction = json.loads(Path('.graphify_extract.json').read_text()) detection = json.loads(Path('.graphify_detect.json').read_text()) -G = build_from_json(extraction) +G = build_from_json(extraction, directed=IS_DIRECTED) +# Guard BEFORE any write: an empty extraction must not clobber a good graph.json / +# GRAPH_REPORT.md / analysis sidecar. Check immediately after build (#1392). +if G.number_of_nodes() == 0: + print('ERROR: Graph is empty - extraction produced no nodes.') + print('Possible causes: all files were skipped, binary-only corpus, or extraction failed.') + raise SystemExit(1) communities = cluster(G) cohesion = score_all(G, communities) tokens = {'input': extraction.get('input_tokens', 0), 'output': extraction.get('output_tokens', 0)} @@ -401,9 +416,15 @@ labels = {cid: 'Community ' + str(cid) for cid in communities} # Placeholder questions - regenerated with real labels in Step 5 questions = suggest_questions(G, communities, labels) +# Persist the graph first and only write the report/analysis if it actually +# persisted - to_json refuses to shrink an existing graph.json (#479), and a +# report describing a graph we did not write would be a lie (#1392). +wrote = to_json(G, communities, 'graphify-out/graph.json') +if not wrote: + print('ERROR: refused to shrink graphify-out/graph.json (fewer nodes than the existing graph). Run a full rebuild to be safe.') + raise SystemExit(1) report = generate(G, communities, cohesion, labels, gods, surprises, detection, tokens, 'INPUT_PATH', suggested_questions=questions) Path('graphify-out/GRAPH_REPORT.md').write_text(report) -to_json(G, communities, 'graphify-out/graph.json') analysis = { 'communities': {str(k): v for k, v in communities.items()}, @@ -413,10 +434,6 @@ analysis = { 'questions': questions, } Path('.graphify_analysis.json').write_text(json.dumps(analysis, indent=2)) -if G.number_of_nodes() == 0: - print('ERROR: Graph is empty - extraction produced no nodes.') - print('Possible causes: all files were skipped, binary-only corpus, or extraction failed.') - raise SystemExit(1) print(f'Graph: {G.number_of_nodes()} nodes, {G.number_of_edges()} edges, {len(communities)} communities') " ``` @@ -444,7 +461,7 @@ extraction = json.loads(Path('.graphify_extract.json').read_text()) detection = json.loads(Path('.graphify_detect.json').read_text()) analysis = json.loads(Path('.graphify_analysis.json').read_text()) -G = build_from_json(extraction) +G = build_from_json(extraction, directed=IS_DIRECTED) communities = {int(k): v for k, v in analysis['communities'].items()} cohesion = {int(k): v for k, v in analysis['cohesion'].items()} tokens = {'input': extraction.get('input_tokens', 0), 'output': extraction.get('output_tokens', 0)} @@ -482,7 +499,7 @@ extraction = json.loads(Path('.graphify_extract.json').read_text()) analysis = json.loads(Path('.graphify_analysis.json').read_text()) labels_raw = json.loads(Path('.graphify_labels.json').read_text()) if Path('.graphify_labels.json').exists() else {} -G = build_from_json(extraction) +G = build_from_json(extraction, directed=IS_DIRECTED) communities = {int(k): v for k, v in analysis['communities'].items()} cohesion = {int(k): v for k, v in analysis['cohesion'].items()} labels = {int(k): v for k, v in labels_raw.items()} @@ -513,7 +530,7 @@ extraction = json.loads(Path('.graphify_extract.json').read_text()) analysis = json.loads(Path('.graphify_analysis.json').read_text()) labels_raw = json.loads(Path('.graphify_labels.json').read_text()) if Path('.graphify_labels.json').exists() else {} -G = build_from_json(extraction) +G = build_from_json(extraction, directed=IS_DIRECTED) communities = {int(k): v for k, v in analysis['communities'].items()} labels = {int(k): v for k, v in labels_raw.items()} @@ -536,7 +553,7 @@ from graphify.build import build_from_json from graphify.export import to_cypher from pathlib import Path -G = build_from_json(json.loads(Path('.graphify_extract.json').read_text())) +G = build_from_json(json.loads(Path('.graphify_extract.json').read_text()), directed=IS_DIRECTED) to_cypher(G, 'graphify-out/cypher.txt') print('cypher.txt written - import with: cypher-shell < graphify-out/cypher.txt') " @@ -554,7 +571,7 @@ from pathlib import Path extraction = json.loads(Path('.graphify_extract.json').read_text()) analysis = json.loads(Path('.graphify_analysis.json').read_text()) -G = build_from_json(extraction) +G = build_from_json(extraction, directed=IS_DIRECTED) communities = {int(k): v for k, v in analysis['communities'].items()} result = push_to_neo4j(G, uri='NEO4J_URI', user='NEO4J_USER', password='NEO4J_PASSWORD', communities=communities) @@ -577,7 +594,7 @@ extraction = json.loads(Path('.graphify_extract.json').read_text()) analysis = json.loads(Path('.graphify_analysis.json').read_text()) labels_raw = json.loads(Path('.graphify_labels.json').read_text()) if Path('.graphify_labels.json').exists() else {} -G = build_from_json(extraction) +G = build_from_json(extraction, directed=IS_DIRECTED) communities = {int(k): v for k, v in analysis['communities'].items()} labels = {int(k): v for k, v in labels_raw.items()} @@ -598,7 +615,7 @@ from pathlib import Path extraction = json.loads(Path('.graphify_extract.json').read_text()) analysis = json.loads(Path('.graphify_analysis.json').read_text()) -G = build_from_json(extraction) +G = build_from_json(extraction, directed=IS_DIRECTED) communities = {int(k): v for k, v in analysis['communities'].items()} to_graphml(G, communities, 'graphify-out/graph.graphml') @@ -795,7 +812,7 @@ G_existing = json_graph.node_link_graph(existing_data, edges='links') # Load new extraction new_extraction = json.loads(Path('.graphify_extract.json').read_text()) -G_new = build_from_json(new_extraction) +G_new = build_from_json(new_extraction, directed=IS_DIRECTED) # Merge: new nodes/edges into existing graph G_existing.update(G_new) @@ -819,7 +836,7 @@ from pathlib import Path # Load old graph (before update) from backup written before merge old_data = json.loads(Path('.graphify_old.json').read_text()) if Path('.graphify_old.json').exists() else None new_extract = json.loads(Path('.graphify_extract.json').read_text()) -G_new = build_from_json(new_extract) +G_new = build_from_json(new_extract, directed=IS_DIRECTED) if old_data: G_old = json_graph.node_link_graph(old_data, edges='links') diff --git a/tools/skillgen/expected/graphify__skill-devin.md b/tools/skillgen/expected/graphify__skill-devin.md index 6e46612..6cbb55d 100644 --- a/tools/skillgen/expected/graphify__skill-devin.md +++ b/tools/skillgen/expected/graphify__skill-devin.md @@ -243,12 +243,19 @@ from graphify.cache import check_semantic_cache from pathlib import Path detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text()) -all_files = [f for files in detect['files'].values() for f in files] +# Only content files go to semantic extraction. Code is already covered +# structurally by the AST pass; flattening every category here makes the +# extraction step re-read every source file (#1392). +all_files = [f for cat in ('document', 'paper', 'image') for f in detect['files'].get(cat, [])] cached_nodes, cached_edges, cached_hyperedges, uncached = check_semantic_cache(all_files) +# Always (re)write the cache file: write hits, else DELETE any leftover from a +# prior run so Part C never merges a stale .graphify_cached.json (#1392). if cached_nodes or cached_edges or cached_hyperedges: Path('graphify-out/.graphify_cached.json').write_text(json.dumps({'nodes': cached_nodes, 'edges': cached_edges, 'hyperedges': cached_hyperedges})) +else: + Path('graphify-out/.graphify_cached.json').unlink(missing_ok=True) Path('graphify-out/.graphify_uncached.txt').write_text('\n'.join(uncached)) print(f'Cache: {len(all_files)-len(uncached)} files hit, {len(uncached)} files need extraction') " @@ -442,6 +449,8 @@ print(f'Merged: {total} nodes, {edges} edges ({len(ast[\"nodes\"])} AST + {len(s ### Step 4 - Build graph, cluster, analyze, generate outputs +**Before starting:** the code blocks below pass `directed=IS_DIRECTED` to `build_from_json()`. Replace `IS_DIRECTED` with `True` if `--directed` was given (builds a `DiGraph` preserving edge direction source->target), otherwise `False` (the default undirected `Graph`). Substitute it everywhere it appears, the same way you substitute `INPUT_PATH` - do not leave the literal `IS_DIRECTED` in the code. + ```bash mkdir -p graphify-out $(cat graphify-out/.graphify_python) -c " @@ -456,7 +465,13 @@ from pathlib import Path extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text()) detection = json.loads(Path('graphify-out/.graphify_detect.json').read_text()) -G = build_from_json(extraction) +G = build_from_json(extraction, directed=IS_DIRECTED) +# Guard BEFORE any write: an empty extraction must not clobber a good graph.json / +# GRAPH_REPORT.md / analysis sidecar. Check immediately after build (#1392). +if G.number_of_nodes() == 0: + print('ERROR: Graph is empty - extraction produced no nodes.') + print('Possible causes: all files were skipped, binary-only corpus, or extraction failed.') + raise SystemExit(1) communities = cluster(G) cohesion = score_all(G, communities) tokens = {'input': extraction.get('input_tokens', 0), 'output': extraction.get('output_tokens', 0)} @@ -466,9 +481,15 @@ labels = {cid: 'Community ' + str(cid) for cid in communities} # Placeholder questions - regenerated with real labels in Step 5 questions = suggest_questions(G, communities, labels) +# Persist the graph first and only write the report/analysis if it actually +# persisted - to_json refuses to shrink an existing graph.json (#479), and a +# report describing a graph we did not write would be a lie (#1392). +wrote = to_json(G, communities, 'graphify-out/graph.json') +if not wrote: + print('ERROR: refused to shrink graphify-out/graph.json (fewer nodes than the existing graph). Run a full rebuild to be safe.') + raise SystemExit(1) report = generate(G, communities, cohesion, labels, gods, surprises, detection, tokens, 'INPUT_PATH', suggested_questions=questions) Path('graphify-out/GRAPH_REPORT.md').write_text(report) -to_json(G, communities, 'graphify-out/graph.json') analysis = { 'communities': {str(k): v for k, v in communities.items()}, @@ -478,10 +499,6 @@ analysis = { 'questions': questions, } Path('graphify-out/.graphify_analysis.json').write_text(json.dumps(analysis, indent=2)) -if G.number_of_nodes() == 0: - print('ERROR: Graph is empty - extraction produced no nodes.') - print('Possible causes: all files were skipped, binary-only corpus, or extraction failed.') - raise SystemExit(1) print(f'Graph: {G.number_of_nodes()} nodes, {G.number_of_edges()} edges, {len(communities)} communities') " ``` @@ -509,7 +526,7 @@ extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text()) detection = json.loads(Path('graphify-out/.graphify_detect.json').read_text()) analysis = json.loads(Path('graphify-out/.graphify_analysis.json').read_text()) -G = build_from_json(extraction) +G = build_from_json(extraction, directed=IS_DIRECTED) communities = {int(k): v for k, v in analysis['communities'].items()} cohesion = {int(k): v for k, v in analysis['cohesion'].items()} tokens = {'input': extraction.get('input_tokens', 0), 'output': extraction.get('output_tokens', 0)} @@ -547,7 +564,7 @@ extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text()) analysis = json.loads(Path('graphify-out/.graphify_analysis.json').read_text()) labels_raw = json.loads(Path('graphify-out/.graphify_labels.json').read_text()) if Path('graphify-out/.graphify_labels.json').exists() else {} -G = build_from_json(extraction) +G = build_from_json(extraction, directed=IS_DIRECTED) communities = {int(k): v for k, v in analysis['communities'].items()} cohesion = {int(k): v for k, v in analysis['cohesion'].items()} labels = {int(k): v for k, v in labels_raw.items()} @@ -578,7 +595,7 @@ extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text()) analysis = json.loads(Path('graphify-out/.graphify_analysis.json').read_text()) labels_raw = json.loads(Path('graphify-out/.graphify_labels.json').read_text()) if Path('graphify-out/.graphify_labels.json').exists() else {} -G = build_from_json(extraction) +G = build_from_json(extraction, directed=IS_DIRECTED) communities = {int(k): v for k, v in analysis['communities'].items()} labels = {int(k): v for k, v in labels_raw.items()} @@ -632,7 +649,7 @@ extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text()) analysis = json.loads(Path('graphify-out/.graphify_analysis.json').read_text()) labels_raw = json.loads(Path('graphify-out/.graphify_labels.json').read_text()) if Path('graphify-out/.graphify_labels.json').exists() else {} -G = build_from_json(extraction) +G = build_from_json(extraction, directed=IS_DIRECTED) communities = {int(k): v for k, v in analysis['communities'].items()} cohesion = {int(k): v for k, v in analysis['cohesion'].items()} labels = {int(k): v for k, v in labels_raw.items()} @@ -655,7 +672,7 @@ from graphify.build import build_from_json from graphify.export import to_cypher from pathlib import Path -G = build_from_json(json.loads(Path('graphify-out/.graphify_extract.json').read_text())) +G = build_from_json(json.loads(Path('graphify-out/.graphify_extract.json').read_text()), directed=IS_DIRECTED) to_cypher(G, 'graphify-out/cypher.txt') print('cypher.txt written - import with: cypher-shell < graphify-out/cypher.txt') " @@ -672,7 +689,7 @@ from pathlib import Path extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text()) analysis = json.loads(Path('graphify-out/.graphify_analysis.json').read_text()) -G = build_from_json(extraction) +G = build_from_json(extraction, directed=IS_DIRECTED) communities = {int(k): v for k, v in analysis['communities'].items()} result = push_to_neo4j(G, uri='NEO4J_URI', user='NEO4J_USER', password='NEO4J_PASSWORD', communities=communities) @@ -695,7 +712,7 @@ extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text()) analysis = json.loads(Path('graphify-out/.graphify_analysis.json').read_text()) labels_raw = json.loads(Path('graphify-out/.graphify_labels.json').read_text()) if Path('graphify-out/.graphify_labels.json').exists() else {} -G = build_from_json(extraction) +G = build_from_json(extraction, directed=IS_DIRECTED) communities = {int(k): v for k, v in analysis['communities'].items()} labels = {int(k): v for k, v in labels_raw.items()} @@ -716,7 +733,7 @@ from pathlib import Path extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text()) analysis = json.loads(Path('graphify-out/.graphify_analysis.json').read_text()) -G = build_from_json(extraction) +G = build_from_json(extraction, directed=IS_DIRECTED) communities = {int(k): v for k, v in analysis['communities'].items()} to_graphml(G, communities, 'graphify-out/graph.graphml') @@ -932,7 +949,7 @@ G_existing = json_graph.node_link_graph(existing_data, edges='links') # Load new extraction new_extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text()) -G_new = build_from_json(new_extraction) +G_new = build_from_json(new_extraction, directed=IS_DIRECTED) # Merge: new nodes/edges into existing graph G_existing.update(G_new) @@ -955,7 +972,7 @@ from pathlib import Path old_data = json.loads(Path('graphify-out/.graphify_old.json').read_text()) if Path('graphify-out/.graphify_old.json').exists() else None new_extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text()) -G_new = build_from_json(new_extract) +G_new = build_from_json(new_extract, directed=IS_DIRECTED) if old_data: G_old = json_graph.node_link_graph(old_data, edges='links') diff --git a/tools/skillgen/fragments/core/aider.md b/tools/skillgen/fragments/core/aider.md index 6f37a5d..7cea232 100644 --- a/tools/skillgen/fragments/core/aider.md +++ b/tools/skillgen/fragments/core/aider.md @@ -226,12 +226,19 @@ from graphify.cache import check_semantic_cache from pathlib import Path detect = json.loads(Path('.graphify_detect.json').read_text()) -all_files = [f for files in detect['files'].values() for f in files] +# Only content files go to semantic extraction. Code is already covered +# structurally by the AST pass; flattening every category here makes the +# extraction step re-read every source file (#1392). +all_files = [f for cat in ('document', 'paper', 'image') for f in detect['files'].get(cat, [])] cached_nodes, cached_edges, cached_hyperedges, uncached = check_semantic_cache(all_files) +# Always (re)write the cache file: write hits, else DELETE any leftover from a +# prior run so Part C never merges a stale .graphify_cached.json (#1392). if cached_nodes or cached_edges or cached_hyperedges: Path('.graphify_cached.json').write_text(json.dumps({'nodes': cached_nodes, 'edges': cached_edges, 'hyperedges': cached_hyperedges})) +else: + Path('.graphify_cached.json').unlink(missing_ok=True) Path('.graphify_uncached.txt').write_text('\n'.join(uncached)) print(f'Cache: {len(all_files)-len(uncached)} files hit, {len(uncached)} files need extraction') " @@ -377,6 +384,8 @@ print(f'Merged: {total} nodes, {edges} edges ({len(ast[\"nodes\"])} AST + {len(s ### Step 4 - Build graph, cluster, analyze, generate outputs +**Before starting:** the code blocks below pass `directed=IS_DIRECTED` to `build_from_json()`. Replace `IS_DIRECTED` with `True` if `--directed` was given (builds a `DiGraph` preserving edge direction source->target), otherwise `False` (the default undirected `Graph`). Substitute it everywhere it appears, the same way you substitute `INPUT_PATH` - do not leave the literal `IS_DIRECTED` in the code. + ```bash mkdir -p graphify-out $(cat graphify-out/.graphify_python) -c " @@ -391,7 +400,13 @@ from pathlib import Path extraction = json.loads(Path('.graphify_extract.json').read_text()) detection = json.loads(Path('.graphify_detect.json').read_text()) -G = build_from_json(extraction) +G = build_from_json(extraction, directed=IS_DIRECTED) +# Guard BEFORE any write: an empty extraction must not clobber a good graph.json / +# GRAPH_REPORT.md / analysis sidecar. Check immediately after build (#1392). +if G.number_of_nodes() == 0: + print('ERROR: Graph is empty - extraction produced no nodes.') + print('Possible causes: all files were skipped, binary-only corpus, or extraction failed.') + raise SystemExit(1) communities = cluster(G) cohesion = score_all(G, communities) tokens = {'input': extraction.get('input_tokens', 0), 'output': extraction.get('output_tokens', 0)} @@ -401,9 +416,15 @@ labels = {cid: 'Community ' + str(cid) for cid in communities} # Placeholder questions - regenerated with real labels in Step 5 questions = suggest_questions(G, communities, labels) +# Persist the graph first and only write the report/analysis if it actually +# persisted - to_json refuses to shrink an existing graph.json (#479), and a +# report describing a graph we did not write would be a lie (#1392). +wrote = to_json(G, communities, 'graphify-out/graph.json') +if not wrote: + print('ERROR: refused to shrink graphify-out/graph.json (fewer nodes than the existing graph). Run a full rebuild to be safe.') + raise SystemExit(1) report = generate(G, communities, cohesion, labels, gods, surprises, detection, tokens, 'INPUT_PATH', suggested_questions=questions) Path('graphify-out/GRAPH_REPORT.md').write_text(report) -to_json(G, communities, 'graphify-out/graph.json') analysis = { 'communities': {str(k): v for k, v in communities.items()}, @@ -413,10 +434,6 @@ analysis = { 'questions': questions, } Path('.graphify_analysis.json').write_text(json.dumps(analysis, indent=2)) -if G.number_of_nodes() == 0: - print('ERROR: Graph is empty - extraction produced no nodes.') - print('Possible causes: all files were skipped, binary-only corpus, or extraction failed.') - raise SystemExit(1) print(f'Graph: {G.number_of_nodes()} nodes, {G.number_of_edges()} edges, {len(communities)} communities') " ``` @@ -444,7 +461,7 @@ extraction = json.loads(Path('.graphify_extract.json').read_text()) detection = json.loads(Path('.graphify_detect.json').read_text()) analysis = json.loads(Path('.graphify_analysis.json').read_text()) -G = build_from_json(extraction) +G = build_from_json(extraction, directed=IS_DIRECTED) communities = {int(k): v for k, v in analysis['communities'].items()} cohesion = {int(k): v for k, v in analysis['cohesion'].items()} tokens = {'input': extraction.get('input_tokens', 0), 'output': extraction.get('output_tokens', 0)} @@ -482,7 +499,7 @@ extraction = json.loads(Path('.graphify_extract.json').read_text()) analysis = json.loads(Path('.graphify_analysis.json').read_text()) labels_raw = json.loads(Path('.graphify_labels.json').read_text()) if Path('.graphify_labels.json').exists() else {} -G = build_from_json(extraction) +G = build_from_json(extraction, directed=IS_DIRECTED) communities = {int(k): v for k, v in analysis['communities'].items()} cohesion = {int(k): v for k, v in analysis['cohesion'].items()} labels = {int(k): v for k, v in labels_raw.items()} @@ -513,7 +530,7 @@ extraction = json.loads(Path('.graphify_extract.json').read_text()) analysis = json.loads(Path('.graphify_analysis.json').read_text()) labels_raw = json.loads(Path('.graphify_labels.json').read_text()) if Path('.graphify_labels.json').exists() else {} -G = build_from_json(extraction) +G = build_from_json(extraction, directed=IS_DIRECTED) communities = {int(k): v for k, v in analysis['communities'].items()} labels = {int(k): v for k, v in labels_raw.items()} @@ -536,7 +553,7 @@ from graphify.build import build_from_json from graphify.export import to_cypher from pathlib import Path -G = build_from_json(json.loads(Path('.graphify_extract.json').read_text())) +G = build_from_json(json.loads(Path('.graphify_extract.json').read_text()), directed=IS_DIRECTED) to_cypher(G, 'graphify-out/cypher.txt') print('cypher.txt written - import with: cypher-shell < graphify-out/cypher.txt') " @@ -554,7 +571,7 @@ from pathlib import Path extraction = json.loads(Path('.graphify_extract.json').read_text()) analysis = json.loads(Path('.graphify_analysis.json').read_text()) -G = build_from_json(extraction) +G = build_from_json(extraction, directed=IS_DIRECTED) communities = {int(k): v for k, v in analysis['communities'].items()} result = push_to_neo4j(G, uri='NEO4J_URI', user='NEO4J_USER', password='NEO4J_PASSWORD', communities=communities) @@ -577,7 +594,7 @@ extraction = json.loads(Path('.graphify_extract.json').read_text()) analysis = json.loads(Path('.graphify_analysis.json').read_text()) labels_raw = json.loads(Path('.graphify_labels.json').read_text()) if Path('.graphify_labels.json').exists() else {} -G = build_from_json(extraction) +G = build_from_json(extraction, directed=IS_DIRECTED) communities = {int(k): v for k, v in analysis['communities'].items()} labels = {int(k): v for k, v in labels_raw.items()} @@ -598,7 +615,7 @@ from pathlib import Path extraction = json.loads(Path('.graphify_extract.json').read_text()) analysis = json.loads(Path('.graphify_analysis.json').read_text()) -G = build_from_json(extraction) +G = build_from_json(extraction, directed=IS_DIRECTED) communities = {int(k): v for k, v in analysis['communities'].items()} to_graphml(G, communities, 'graphify-out/graph.graphml') @@ -795,7 +812,7 @@ G_existing = json_graph.node_link_graph(existing_data, edges='links') # Load new extraction new_extraction = json.loads(Path('.graphify_extract.json').read_text()) -G_new = build_from_json(new_extraction) +G_new = build_from_json(new_extraction, directed=IS_DIRECTED) # Merge: new nodes/edges into existing graph G_existing.update(G_new) @@ -819,7 +836,7 @@ from pathlib import Path # Load old graph (before update) from backup written before merge old_data = json.loads(Path('.graphify_old.json').read_text()) if Path('.graphify_old.json').exists() else None new_extract = json.loads(Path('.graphify_extract.json').read_text()) -G_new = build_from_json(new_extract) +G_new = build_from_json(new_extract, directed=IS_DIRECTED) if old_data: G_old = json_graph.node_link_graph(old_data, edges='links') diff --git a/tools/skillgen/fragments/core/devin.md b/tools/skillgen/fragments/core/devin.md index 6e46612..6cbb55d 100644 --- a/tools/skillgen/fragments/core/devin.md +++ b/tools/skillgen/fragments/core/devin.md @@ -243,12 +243,19 @@ from graphify.cache import check_semantic_cache from pathlib import Path detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text()) -all_files = [f for files in detect['files'].values() for f in files] +# Only content files go to semantic extraction. Code is already covered +# structurally by the AST pass; flattening every category here makes the +# extraction step re-read every source file (#1392). +all_files = [f for cat in ('document', 'paper', 'image') for f in detect['files'].get(cat, [])] cached_nodes, cached_edges, cached_hyperedges, uncached = check_semantic_cache(all_files) +# Always (re)write the cache file: write hits, else DELETE any leftover from a +# prior run so Part C never merges a stale .graphify_cached.json (#1392). if cached_nodes or cached_edges or cached_hyperedges: Path('graphify-out/.graphify_cached.json').write_text(json.dumps({'nodes': cached_nodes, 'edges': cached_edges, 'hyperedges': cached_hyperedges})) +else: + Path('graphify-out/.graphify_cached.json').unlink(missing_ok=True) Path('graphify-out/.graphify_uncached.txt').write_text('\n'.join(uncached)) print(f'Cache: {len(all_files)-len(uncached)} files hit, {len(uncached)} files need extraction') " @@ -442,6 +449,8 @@ print(f'Merged: {total} nodes, {edges} edges ({len(ast[\"nodes\"])} AST + {len(s ### Step 4 - Build graph, cluster, analyze, generate outputs +**Before starting:** the code blocks below pass `directed=IS_DIRECTED` to `build_from_json()`. Replace `IS_DIRECTED` with `True` if `--directed` was given (builds a `DiGraph` preserving edge direction source->target), otherwise `False` (the default undirected `Graph`). Substitute it everywhere it appears, the same way you substitute `INPUT_PATH` - do not leave the literal `IS_DIRECTED` in the code. + ```bash mkdir -p graphify-out $(cat graphify-out/.graphify_python) -c " @@ -456,7 +465,13 @@ from pathlib import Path extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text()) detection = json.loads(Path('graphify-out/.graphify_detect.json').read_text()) -G = build_from_json(extraction) +G = build_from_json(extraction, directed=IS_DIRECTED) +# Guard BEFORE any write: an empty extraction must not clobber a good graph.json / +# GRAPH_REPORT.md / analysis sidecar. Check immediately after build (#1392). +if G.number_of_nodes() == 0: + print('ERROR: Graph is empty - extraction produced no nodes.') + print('Possible causes: all files were skipped, binary-only corpus, or extraction failed.') + raise SystemExit(1) communities = cluster(G) cohesion = score_all(G, communities) tokens = {'input': extraction.get('input_tokens', 0), 'output': extraction.get('output_tokens', 0)} @@ -466,9 +481,15 @@ labels = {cid: 'Community ' + str(cid) for cid in communities} # Placeholder questions - regenerated with real labels in Step 5 questions = suggest_questions(G, communities, labels) +# Persist the graph first and only write the report/analysis if it actually +# persisted - to_json refuses to shrink an existing graph.json (#479), and a +# report describing a graph we did not write would be a lie (#1392). +wrote = to_json(G, communities, 'graphify-out/graph.json') +if not wrote: + print('ERROR: refused to shrink graphify-out/graph.json (fewer nodes than the existing graph). Run a full rebuild to be safe.') + raise SystemExit(1) report = generate(G, communities, cohesion, labels, gods, surprises, detection, tokens, 'INPUT_PATH', suggested_questions=questions) Path('graphify-out/GRAPH_REPORT.md').write_text(report) -to_json(G, communities, 'graphify-out/graph.json') analysis = { 'communities': {str(k): v for k, v in communities.items()}, @@ -478,10 +499,6 @@ analysis = { 'questions': questions, } Path('graphify-out/.graphify_analysis.json').write_text(json.dumps(analysis, indent=2)) -if G.number_of_nodes() == 0: - print('ERROR: Graph is empty - extraction produced no nodes.') - print('Possible causes: all files were skipped, binary-only corpus, or extraction failed.') - raise SystemExit(1) print(f'Graph: {G.number_of_nodes()} nodes, {G.number_of_edges()} edges, {len(communities)} communities') " ``` @@ -509,7 +526,7 @@ extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text()) detection = json.loads(Path('graphify-out/.graphify_detect.json').read_text()) analysis = json.loads(Path('graphify-out/.graphify_analysis.json').read_text()) -G = build_from_json(extraction) +G = build_from_json(extraction, directed=IS_DIRECTED) communities = {int(k): v for k, v in analysis['communities'].items()} cohesion = {int(k): v for k, v in analysis['cohesion'].items()} tokens = {'input': extraction.get('input_tokens', 0), 'output': extraction.get('output_tokens', 0)} @@ -547,7 +564,7 @@ extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text()) analysis = json.loads(Path('graphify-out/.graphify_analysis.json').read_text()) labels_raw = json.loads(Path('graphify-out/.graphify_labels.json').read_text()) if Path('graphify-out/.graphify_labels.json').exists() else {} -G = build_from_json(extraction) +G = build_from_json(extraction, directed=IS_DIRECTED) communities = {int(k): v for k, v in analysis['communities'].items()} cohesion = {int(k): v for k, v in analysis['cohesion'].items()} labels = {int(k): v for k, v in labels_raw.items()} @@ -578,7 +595,7 @@ extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text()) analysis = json.loads(Path('graphify-out/.graphify_analysis.json').read_text()) labels_raw = json.loads(Path('graphify-out/.graphify_labels.json').read_text()) if Path('graphify-out/.graphify_labels.json').exists() else {} -G = build_from_json(extraction) +G = build_from_json(extraction, directed=IS_DIRECTED) communities = {int(k): v for k, v in analysis['communities'].items()} labels = {int(k): v for k, v in labels_raw.items()} @@ -632,7 +649,7 @@ extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text()) analysis = json.loads(Path('graphify-out/.graphify_analysis.json').read_text()) labels_raw = json.loads(Path('graphify-out/.graphify_labels.json').read_text()) if Path('graphify-out/.graphify_labels.json').exists() else {} -G = build_from_json(extraction) +G = build_from_json(extraction, directed=IS_DIRECTED) communities = {int(k): v for k, v in analysis['communities'].items()} cohesion = {int(k): v for k, v in analysis['cohesion'].items()} labels = {int(k): v for k, v in labels_raw.items()} @@ -655,7 +672,7 @@ from graphify.build import build_from_json from graphify.export import to_cypher from pathlib import Path -G = build_from_json(json.loads(Path('graphify-out/.graphify_extract.json').read_text())) +G = build_from_json(json.loads(Path('graphify-out/.graphify_extract.json').read_text()), directed=IS_DIRECTED) to_cypher(G, 'graphify-out/cypher.txt') print('cypher.txt written - import with: cypher-shell < graphify-out/cypher.txt') " @@ -672,7 +689,7 @@ from pathlib import Path extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text()) analysis = json.loads(Path('graphify-out/.graphify_analysis.json').read_text()) -G = build_from_json(extraction) +G = build_from_json(extraction, directed=IS_DIRECTED) communities = {int(k): v for k, v in analysis['communities'].items()} result = push_to_neo4j(G, uri='NEO4J_URI', user='NEO4J_USER', password='NEO4J_PASSWORD', communities=communities) @@ -695,7 +712,7 @@ extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text()) analysis = json.loads(Path('graphify-out/.graphify_analysis.json').read_text()) labels_raw = json.loads(Path('graphify-out/.graphify_labels.json').read_text()) if Path('graphify-out/.graphify_labels.json').exists() else {} -G = build_from_json(extraction) +G = build_from_json(extraction, directed=IS_DIRECTED) communities = {int(k): v for k, v in analysis['communities'].items()} labels = {int(k): v for k, v in labels_raw.items()} @@ -716,7 +733,7 @@ from pathlib import Path extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text()) analysis = json.loads(Path('graphify-out/.graphify_analysis.json').read_text()) -G = build_from_json(extraction) +G = build_from_json(extraction, directed=IS_DIRECTED) communities = {int(k): v for k, v in analysis['communities'].items()} to_graphml(G, communities, 'graphify-out/graph.graphml') @@ -932,7 +949,7 @@ G_existing = json_graph.node_link_graph(existing_data, edges='links') # Load new extraction new_extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text()) -G_new = build_from_json(new_extraction) +G_new = build_from_json(new_extraction, directed=IS_DIRECTED) # Merge: new nodes/edges into existing graph G_existing.update(G_new) @@ -955,7 +972,7 @@ from pathlib import Path old_data = json.loads(Path('graphify-out/.graphify_old.json').read_text()) if Path('graphify-out/.graphify_old.json').exists() else None new_extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text()) -G_new = build_from_json(new_extract) +G_new = build_from_json(new_extract, directed=IS_DIRECTED) if old_data: G_old = json_graph.node_link_graph(old_data, edges='links') diff --git a/tools/skillgen/gen.py b/tools/skillgen/gen.py index 32e0c6e..4b6ab2d 100644 --- a/tools/skillgen/gen.py +++ b/tools/skillgen/gen.py @@ -27,6 +27,7 @@ import argparse import re import subprocess import sys +from collections import Counter try: import tomllib # Python 3.11+ stdlib except ModuleNotFoundError: # Python 3.10 - graphify supports >=3.10 @@ -672,8 +673,20 @@ def schema_singleton(platforms: dict[str, Platform]) -> list[str]: def _is_enum_line(line: str) -> bool: - """Whether a rendered line carries the unified six-value file_type enum.""" - return ENUM_VALUES in line or ENUM_PROSE in line + """Whether a line carries the file_type enum (its v8 or unified form). + + The unified six-value enum (``ENUM_VALUES``/``ENUM_PROSE``) and the v8 + five-value form it replaced both match here, so the round-trip's multiset + diff classifies the removed-v8 line as well as the added-rendered line. The + five-value schema string is a prefix of the six-value one, and both prose + forms open with the same ``file_type:"rationale"`` guidance clause. + """ + return ( + ENUM_VALUES in line + or ENUM_PROSE in line + or "code|document|paper|image|rationale" in line + or 'file_type:"rationale"` for concept-like nodes' in line + ) def _is_frontmatter_description_line(line: str) -> bool: @@ -695,8 +708,14 @@ def _is_chunk_cleanup_line(line: str) -> bool: ``rm`` and deletes the chunk files with ``find ... -delete`` instead. That rewrite touches the single cleanup line in place (no line added or removed), so it joins the enum and description unifications as an allowed monolith diff. + Both the v8 form (bare ``.graphify_chunk_*.json`` glob, removed) and the fixed + form (``find ... -delete``, added) match here so the multiset diff classifies + each side of the change. """ - return line.lstrip().startswith("rm -f") and "find " in line and "-name '.graphify_chunk_" in line + s = line.lstrip() + if not s.startswith("rm -f"): + return False + return ".graphify_chunk_*.json" in line or ("find " in line and "-name '.graphify_chunk_" in line) def _is_trigger_line(line: str) -> bool: @@ -708,48 +727,135 @@ def _is_trigger_line(line: str) -> bool: return line.strip().startswith("trigger:") +def _is_directed_fix_line(line: str) -> bool: + """Whether a line is part of the ``--directed`` propagation fix (#1392). + + The monolith runbooks built every graph undirected, so a ``--directed`` run + silently collapsed reciprocal A<->B edges. Every ``build_from_json(...)`` call + now threads ``directed=IS_DIRECTED`` and a prose line tells the agent to + substitute it like ``INPUT_PATH``. Both the old bare call (removed) and the + new threaded call (added) match here, plus the substitution instruction. + """ + return ( + "build_from_json(" in line and "import" not in line + ) or "directed=IS_DIRECTED" in line or ( + "IS_DIRECTED" in line and "Substitute it everywhere" in line + ) + + +def _is_content_scope_fix_line(line: str) -> bool: + """Whether a line is part of the content-only semantic scope fix (#1392). + + Flattening every detect category fed code files (already covered by the AST + pass) back to the semantic step. The fix scopes to document/paper/image. + """ + return ( + "detect['files'].values()" in line + or "for cat in ('document', 'paper', 'image')" in line + or "Only content files go to semantic extraction" in line + or "structurally by the AST pass" in line + or "extraction step re-read every source file" in line + ) + + +def _is_cache_unlink_fix_line(line: str) -> bool: + """Whether a line is part of the stale-cache unlink fix (#1392). + + The cache file was written only on a hit, so a miss left a prior run's + ``.graphify_cached.json`` for Part C to merge. The miss branch now deletes it. + """ + return ( + ".graphify_cached.json').unlink(missing_ok=True)" in line + or line.strip() == "else:" + or "Always (re)write the cache file" in line + or "stale .graphify_cached.json" in line + ) + + +def _is_zero_node_guard_fix_line(line: str) -> bool: + """Whether a line is part of the zero-node / shrink-guard ordering fix (#1392). + + Step 4 wrote GRAPH_REPORT.md, graph.json and the analysis sidecar *before* + the zero-node guard, so an empty extraction clobbered a good graph; and the + report was written even when ``to_json`` refused to shrink (#479). The guard + now runs before any write and the report/analysis are gated on ``to_json``. + Both the old (removed) and new (added) forms of these lines match here. + """ + s = line.strip() + return ( + "number_of_nodes() == 0" in line + or "Graph is empty - extraction produced no nodes" in line + or s.startswith("print('Possible causes:") + or s == "raise SystemExit(1)" + or "to_json(G, communities," in line + or s == "if not wrote:" + or "refused to shrink graphify-out/graph.json" in line + or "Guard BEFORE any write" in line + or "GRAPH_REPORT.md / analysis sidecar" in line + or "Persist the graph first" in line + or "to_json refuses to shrink an existing graph.json" in line + or "report describing a graph we did not write" in line + ) + + +# Every line that may differ between a rendered monolith and its pristine v8 +# baseline. Each predicate documents one sanctioned change-class; a blank line is +# allowed because the multi-line fix blocks insert spacing. Anything else failing +# all of these is an unsanctioned drift the round-trip must catch. +_SANCTIONED_MONOLITH_DIFFS = ( + _is_enum_line, + _is_frontmatter_description_line, + _is_chunk_cleanup_line, + _is_directed_fix_line, + _is_content_scope_fix_line, + _is_cache_unlink_fix_line, + _is_zero_node_guard_fix_line, +) + + +def _is_sanctioned_monolith_diff(line: str) -> bool: + """Whether a single added/removed monolith line is an allowed change.""" + return not line.strip() or any(pred(line) for pred in _SANCTIONED_MONOLITH_DIFFS) + + def monolith_roundtrip(platform: Platform) -> list[str]: """Assert a monolith renders diff-clean vs its v8 blob modulo allowed changes. - Two classes of line are allowed to differ between the rendered monolith and - the v8 source: the file_type enum lines (unified to the six-value superset) - and the frontmatter ``description`` line (unified across all platforms for - discovery). Every other line must match byte for byte. + The monolith bodies are hand-maintained single files frozen against a pinned + pristine v8 blob (``roundtrip_ref``); this is the guard that stops an + arbitrary edit (even a blessed one) from drifting them. Sanctioned changes are + enumerated as predicates in ``_SANCTIONED_MONOLITH_DIFFS``: the file_type enum + unification, the unified frontmatter description, the chunk-cleanup rewrite + (#1172), and the four #1392 runbook fixes (directed propagation, content-only + semantic scope, stale-cache unlink, and the zero-node/shrink-guard ordering). + + The comparison is a multiset diff, not a positional zip: a line whose text is + unchanged but merely *moved* (the report-write line shifted below ``to_json`` + in the ordering fix) cancels out and is not flagged. Only lines whose content + is genuinely added or removed are checked, and each must be sanctioned. """ if platform.bucket != "monolith": return [] if platform.roundtrip_ref is None: return [f"[{platform.key}] monolith is missing roundtrip_ref"] - rendered = render(platform)[0].content - original = _normalise(_git_show(platform.roundtrip_ref)) + rendered_lines = render(platform)[0].content.splitlines() + # Strip trigger lines from the original — they are non-spec and their removal + # (#1180) is a permitted diff. + original_lines = [ + l for l in _normalise(_git_show(platform.roundtrip_ref)).splitlines() + if not _is_trigger_line(l) + ] - rendered_lines = rendered.splitlines() - # Strip trigger lines from the original before comparing — they are non-spec - # and their removal (#1180) is a permitted diff. Filter here so the line-count - # check and the per-line zip both operate on the same reduced set. - original_lines = [l for l in original.splitlines() if not _is_trigger_line(l)] + added = Counter(rendered_lines) - Counter(original_lines) + removed = Counter(original_lines) - Counter(rendered_lines) problems: list[str] = [] - if len(rendered_lines) != len(original_lines): - problems.append( - f"[{platform.key}] line count differs: rendered {len(rendered_lines)} vs v8 {len(original_lines)} " - "(the only allowed changes are the enum line(s), the description line, " - "the chunk-cleanup rewrite, and trigger: removal — none must add or remove other lines)" - ) - return problems - - for i, (r, o) in enumerate(zip(rendered_lines, original_lines), start=1): - if r == o: - continue - # The permitted diffs are the enum unification, the unified description, - # the shell-agnostic chunk-cleanup rewrite (#1172), and trigger removal (#1180). - if _is_enum_line(r) or _is_frontmatter_description_line(r) or _is_chunk_cleanup_line(r): + for line in list(added.elements()) + list(removed.elements()): + if _is_sanctioned_monolith_diff(line): continue problems.append( - f"[{platform.key}] line {i} differs and is not an enum or description unification:\n" - f" v8: {o!r}\n" - f" rendered: {r!r}" + f"[{platform.key}] unsanctioned monolith change vs pristine v8: {line!r}" ) return problems