From 3d19463484ebcf773b399ddad9fd3363b2ab3bff Mon Sep 17 00:00:00 2001 From: safishamsi Date: Fri, 7 Aug 2026 22:49:25 +0100 Subject: [PATCH] fix(skill): make the Windows skill variant runnable on PowerShell; bump to 0.9.36 (#2528) The Windows skill variant had a PowerShell Step 1 but its later steps came from the shared core fragment as bash-only shell (cat-piped interpreter invocations, rm -f, find -delete). The skillgen renderer now translates the composed core to PowerShell for powershell-shell platforms (here-string interpreter invocations, Remove-Item cleanup); POSIX skills are unchanged and step / #2490 parity is enforced by a generator check. Not a regression from 0.9.35 (skill files were unchanged between 0.9.34 and 0.9.35). Thanks @tannermosher2015-debug. Co-Authored-By: Claude Opus 4.8 (1M context) --- CHANGELOG.md | 9 +- graphify/skill-agents.md | 7 +- graphify/skill-amp.md | 7 +- graphify/skill-claw.md | 7 +- graphify/skill-codex.md | 7 +- graphify/skill-copilot.md | 7 +- graphify/skill-droid.md | 7 +- graphify/skill-kilo.md | 7 +- graphify/skill-kiro.md | 7 +- graphify/skill-opencode.md | 7 +- graphify/skill-pi.md | 7 +- graphify/skill-trae.md | 7 +- graphify/skill-vscode.md | 7 +- graphify/skill-windows.md | 187 +++++++++--------- graphify/skill.md | 7 +- pyproject.toml | 2 +- tests/test_skillgen.py | 116 +++++++++++ .../expected/graphify__skill-agents.md | 7 +- .../skillgen/expected/graphify__skill-amp.md | 7 +- .../skillgen/expected/graphify__skill-claw.md | 7 +- .../expected/graphify__skill-codex.md | 7 +- .../expected/graphify__skill-copilot.md | 7 +- .../expected/graphify__skill-droid.md | 7 +- .../skillgen/expected/graphify__skill-kilo.md | 7 +- .../skillgen/expected/graphify__skill-kiro.md | 7 +- .../expected/graphify__skill-opencode.md | 7 +- tools/skillgen/expected/graphify__skill-pi.md | 7 +- .../skillgen/expected/graphify__skill-trae.md | 7 +- .../expected/graphify__skill-vscode.md | 7 +- .../expected/graphify__skill-windows.md | 187 +++++++++--------- tools/skillgen/expected/graphify__skill.md | 7 +- tools/skillgen/fragments/core/core.md | 21 +- .../shell/interpreter-guard-posix.md | 13 ++ .../shell/interpreter-guard-powershell.md | 15 ++ tools/skillgen/gen.py | 165 +++++++++++++++- 35 files changed, 645 insertions(+), 252 deletions(-) create mode 100644 tools/skillgen/fragments/shell/interpreter-guard-posix.md create mode 100644 tools/skillgen/fragments/shell/interpreter-guard-powershell.md diff --git a/CHANGELOG.md b/CHANGELOG.md index 077647f..d78fe39 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,7 +2,14 @@ Full release notes with details on each version: [GitHub Releases](https://github.com/safishamsi/graphify/releases) -## 0.9.35 (unreleased) +## 0.9.36 (unreleased) + +- Fix: four commands that failed silently while exiting 0 now surface the problem (#2534, thanks @elecnix). `cluster-only` warns when `--backend`/`--model`/`--batch-size` are ignored because saved labels are being reused; the community-label prompt no longer collides with the discard sentinel (a model echoing the key back is no longer silently dropped); `tree --root` exits non-zero when the root matches no source file instead of silently flattening the tree; and `cluster-only` stamps `built_at_commit` from the analysed graph rather than the shell's working directory. Also folds in the `cluster-only` refused-write guard from #2522 (thanks @aniJani). +- Fix: a Swift `extension Foo` in a different file from `Foo` no longer drops static and singleton call edges into the type (#2538, thanks @pawelo446). The extension node id is now remapped consistently so the extension merges onto its base type before call resolution, and the merge is gated so it never absorbs a same-named type from another language. +- Fix: node-id collision resolution is now deterministic and prefers active over archived paths (#2532, thanks @michaelxer for the active/archived idea in #2540). Two files that mint the same id (for example `plans/_done/x.md` vs `plans/in-progress/x.md`) are ranked by a lifecycle penalty computed on the root-relative path and a reversed-segment tie-break, so the winner no longer depends on ASCII filename order, absolute-vs-relative path form, or the checkout directory name. +- Fix: the Windows skill now runs on PowerShell (#2528, thanks @tannermosher2015-debug). The Windows skill variant's steps were bash-only (`$(cat ...)`, `rm -f`, `find -delete`); they are now emitted as PowerShell (here-string interpreter invocations and `Remove-Item` cleanup), with POSIX skills unchanged and step parity enforced by a generator check. + +## 0.9.35 (2026-08-06) - Fix: the `build_merge` #479 shrink guard is no longer effectively dead (#2497, thanks @sortakool). It read the post-replace node count, so a broken partial re-extract could silently destroy nodes without tripping the guard, and the guard was skipped entirely under `prune_sources`. The guard now diffs the on-disk baseline by node identity and refuses any loss from a source that was neither re-extracted nor pruned this run (active even under `prune_sources`, skipped only under `dedup`), and reports how many nodes a re-extract replaced. - Fix: `build_merge`/`merge_raw_extraction` `prune_sources` now prunes correctly when given absolute paths under a non-standard layout, deriving the scan root by suffix-matching stored source paths, and warns (instead of reporting "already clean") when a prune matches nothing (#2446, thanks @AI-invest). diff --git a/graphify/skill-agents.md b/graphify/skill-agents.md index b174e6d..190827d 100644 --- a/graphify/skill-agents.md +++ b/graphify/skill-agents.md @@ -112,8 +112,11 @@ import json from graphify.detect import detect from pathlib import Path result = detect(Path('INPUT_PATH')) -print(json.dumps(result, ensure_ascii=False)) -" > graphify-out/.graphify_detect.json +# Write the sidecar from Python, not a shell redirect, so the same block renders +# on PowerShell hosts without console-encoding drift (#2528). +Path('graphify-out/.graphify_detect.json').write_text(json.dumps(result, ensure_ascii=False), encoding=\"utf-8\") +print(f'Detected {result[\"total_files\"]} files') +" ``` Replace INPUT_PATH with the actual path the user provided. Do NOT cat or print the JSON - read it silently and present a clean summary instead: diff --git a/graphify/skill-amp.md b/graphify/skill-amp.md index b174e6d..190827d 100644 --- a/graphify/skill-amp.md +++ b/graphify/skill-amp.md @@ -112,8 +112,11 @@ import json from graphify.detect import detect from pathlib import Path result = detect(Path('INPUT_PATH')) -print(json.dumps(result, ensure_ascii=False)) -" > graphify-out/.graphify_detect.json +# Write the sidecar from Python, not a shell redirect, so the same block renders +# on PowerShell hosts without console-encoding drift (#2528). +Path('graphify-out/.graphify_detect.json').write_text(json.dumps(result, ensure_ascii=False), encoding=\"utf-8\") +print(f'Detected {result[\"total_files\"]} files') +" ``` Replace INPUT_PATH with the actual path the user provided. Do NOT cat or print the JSON - read it silently and present a clean summary instead: diff --git a/graphify/skill-claw.md b/graphify/skill-claw.md index cef4d5c..abd2811 100644 --- a/graphify/skill-claw.md +++ b/graphify/skill-claw.md @@ -112,8 +112,11 @@ import json from graphify.detect import detect from pathlib import Path result = detect(Path('INPUT_PATH')) -print(json.dumps(result, ensure_ascii=False)) -" > graphify-out/.graphify_detect.json +# Write the sidecar from Python, not a shell redirect, so the same block renders +# on PowerShell hosts without console-encoding drift (#2528). +Path('graphify-out/.graphify_detect.json').write_text(json.dumps(result, ensure_ascii=False), encoding=\"utf-8\") +print(f'Detected {result[\"total_files\"]} files') +" ``` Replace INPUT_PATH with the actual path the user provided. Do NOT cat or print the JSON - read it silently and present a clean summary instead: diff --git a/graphify/skill-codex.md b/graphify/skill-codex.md index bd8a020..af3f723 100644 --- a/graphify/skill-codex.md +++ b/graphify/skill-codex.md @@ -112,8 +112,11 @@ import json from graphify.detect import detect from pathlib import Path result = detect(Path('INPUT_PATH')) -print(json.dumps(result, ensure_ascii=False)) -" > graphify-out/.graphify_detect.json +# Write the sidecar from Python, not a shell redirect, so the same block renders +# on PowerShell hosts without console-encoding drift (#2528). +Path('graphify-out/.graphify_detect.json').write_text(json.dumps(result, ensure_ascii=False), encoding=\"utf-8\") +print(f'Detected {result[\"total_files\"]} files') +" ``` Replace INPUT_PATH with the actual path the user provided. Do NOT cat or print the JSON - read it silently and present a clean summary instead: diff --git a/graphify/skill-copilot.md b/graphify/skill-copilot.md index cef4d5c..abd2811 100644 --- a/graphify/skill-copilot.md +++ b/graphify/skill-copilot.md @@ -112,8 +112,11 @@ import json from graphify.detect import detect from pathlib import Path result = detect(Path('INPUT_PATH')) -print(json.dumps(result, ensure_ascii=False)) -" > graphify-out/.graphify_detect.json +# Write the sidecar from Python, not a shell redirect, so the same block renders +# on PowerShell hosts without console-encoding drift (#2528). +Path('graphify-out/.graphify_detect.json').write_text(json.dumps(result, ensure_ascii=False), encoding=\"utf-8\") +print(f'Detected {result[\"total_files\"]} files') +" ``` Replace INPUT_PATH with the actual path the user provided. Do NOT cat or print the JSON - read it silently and present a clean summary instead: diff --git a/graphify/skill-droid.md b/graphify/skill-droid.md index 370e0d6..fd148d4 100644 --- a/graphify/skill-droid.md +++ b/graphify/skill-droid.md @@ -112,8 +112,11 @@ import json from graphify.detect import detect from pathlib import Path result = detect(Path('INPUT_PATH')) -print(json.dumps(result, ensure_ascii=False)) -" > graphify-out/.graphify_detect.json +# Write the sidecar from Python, not a shell redirect, so the same block renders +# on PowerShell hosts without console-encoding drift (#2528). +Path('graphify-out/.graphify_detect.json').write_text(json.dumps(result, ensure_ascii=False), encoding=\"utf-8\") +print(f'Detected {result[\"total_files\"]} files') +" ``` Replace INPUT_PATH with the actual path the user provided. Do NOT cat or print the JSON - read it silently and present a clean summary instead: diff --git a/graphify/skill-kilo.md b/graphify/skill-kilo.md index f83cb06..3e70b05 100644 --- a/graphify/skill-kilo.md +++ b/graphify/skill-kilo.md @@ -112,8 +112,11 @@ import json from graphify.detect import detect from pathlib import Path result = detect(Path('INPUT_PATH')) -print(json.dumps(result, ensure_ascii=False)) -" > graphify-out/.graphify_detect.json +# Write the sidecar from Python, not a shell redirect, so the same block renders +# on PowerShell hosts without console-encoding drift (#2528). +Path('graphify-out/.graphify_detect.json').write_text(json.dumps(result, ensure_ascii=False), encoding=\"utf-8\") +print(f'Detected {result[\"total_files\"]} files') +" ``` Replace INPUT_PATH with the actual path the user provided. Do NOT cat or print the JSON - read it silently and present a clean summary instead: diff --git a/graphify/skill-kiro.md b/graphify/skill-kiro.md index cef4d5c..abd2811 100644 --- a/graphify/skill-kiro.md +++ b/graphify/skill-kiro.md @@ -112,8 +112,11 @@ import json from graphify.detect import detect from pathlib import Path result = detect(Path('INPUT_PATH')) -print(json.dumps(result, ensure_ascii=False)) -" > graphify-out/.graphify_detect.json +# Write the sidecar from Python, not a shell redirect, so the same block renders +# on PowerShell hosts without console-encoding drift (#2528). +Path('graphify-out/.graphify_detect.json').write_text(json.dumps(result, ensure_ascii=False), encoding=\"utf-8\") +print(f'Detected {result[\"total_files\"]} files') +" ``` Replace INPUT_PATH with the actual path the user provided. Do NOT cat or print the JSON - read it silently and present a clean summary instead: diff --git a/graphify/skill-opencode.md b/graphify/skill-opencode.md index c121617..91ced60 100644 --- a/graphify/skill-opencode.md +++ b/graphify/skill-opencode.md @@ -112,8 +112,11 @@ import json from graphify.detect import detect from pathlib import Path result = detect(Path('INPUT_PATH')) -print(json.dumps(result, ensure_ascii=False)) -" > graphify-out/.graphify_detect.json +# Write the sidecar from Python, not a shell redirect, so the same block renders +# on PowerShell hosts without console-encoding drift (#2528). +Path('graphify-out/.graphify_detect.json').write_text(json.dumps(result, ensure_ascii=False), encoding=\"utf-8\") +print(f'Detected {result[\"total_files\"]} files') +" ``` Replace INPUT_PATH with the actual path the user provided. Do NOT cat or print the JSON - read it silently and present a clean summary instead: diff --git a/graphify/skill-pi.md b/graphify/skill-pi.md index cef4d5c..abd2811 100644 --- a/graphify/skill-pi.md +++ b/graphify/skill-pi.md @@ -112,8 +112,11 @@ import json from graphify.detect import detect from pathlib import Path result = detect(Path('INPUT_PATH')) -print(json.dumps(result, ensure_ascii=False)) -" > graphify-out/.graphify_detect.json +# Write the sidecar from Python, not a shell redirect, so the same block renders +# on PowerShell hosts without console-encoding drift (#2528). +Path('graphify-out/.graphify_detect.json').write_text(json.dumps(result, ensure_ascii=False), encoding=\"utf-8\") +print(f'Detected {result[\"total_files\"]} files') +" ``` Replace INPUT_PATH with the actual path the user provided. Do NOT cat or print the JSON - read it silently and present a clean summary instead: diff --git a/graphify/skill-trae.md b/graphify/skill-trae.md index 98deab2..050667b 100644 --- a/graphify/skill-trae.md +++ b/graphify/skill-trae.md @@ -112,8 +112,11 @@ import json from graphify.detect import detect from pathlib import Path result = detect(Path('INPUT_PATH')) -print(json.dumps(result, ensure_ascii=False)) -" > graphify-out/.graphify_detect.json +# Write the sidecar from Python, not a shell redirect, so the same block renders +# on PowerShell hosts without console-encoding drift (#2528). +Path('graphify-out/.graphify_detect.json').write_text(json.dumps(result, ensure_ascii=False), encoding=\"utf-8\") +print(f'Detected {result[\"total_files\"]} files') +" ``` Replace INPUT_PATH with the actual path the user provided. Do NOT cat or print the JSON - read it silently and present a clean summary instead: diff --git a/graphify/skill-vscode.md b/graphify/skill-vscode.md index efb2401..20c7c08 100644 --- a/graphify/skill-vscode.md +++ b/graphify/skill-vscode.md @@ -112,8 +112,11 @@ import json from graphify.detect import detect from pathlib import Path result = detect(Path('INPUT_PATH')) -print(json.dumps(result, ensure_ascii=False)) -" > graphify-out/.graphify_detect.json +# Write the sidecar from Python, not a shell redirect, so the same block renders +# on PowerShell hosts without console-encoding drift (#2528). +Path('graphify-out/.graphify_detect.json').write_text(json.dumps(result, ensure_ascii=False), encoding=\"utf-8\") +print(f'Detected {result[\"total_files\"]} files') +" ``` Replace INPUT_PATH with the actual path the user provided. Do NOT cat or print the JSON - read it silently and present a clean summary instead: diff --git a/graphify/skill-windows.md b/graphify/skill-windows.md index c48d874..d631821 100644 --- a/graphify/skill-windows.md +++ b/graphify/skill-windows.md @@ -128,14 +128,17 @@ If the import succeeds, print nothing and move straight to Step 2. ### Step 2 - Detect files -```bash -$(cat graphify-out/.graphify_python) -c " +```powershell +@' import json from graphify.detect import detect from pathlib import Path result = detect(Path('INPUT_PATH')) -print(json.dumps(result, ensure_ascii=False)) -" > graphify-out/.graphify_detect.json +# Write the sidecar from Python, not a shell redirect, so the same block renders +# on PowerShell hosts without console-encoding drift (#2528). +Path('graphify-out/.graphify_detect.json').write_text(json.dumps(result, ensure_ascii=False), encoding="utf-8") +print(f'Detected {result["total_files"]} files') +'@ | & (Get-Content graphify-out\.graphify_python) - ``` Replace INPUT_PATH with the actual path the user provided. Do NOT cat or print the JSON - read it silently and present a clean summary instead: @@ -190,38 +193,38 @@ Note: Parallelizing AST + semantic saves 5-15s on large corpora. AST is determin For any code files detected, run AST extraction in parallel with Part B subagents: -```bash -$(cat graphify-out/.graphify_python) -c " +```powershell +@' import sys, json from graphify.extract import collect_files, extract from pathlib import Path import json code_files = [] -detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\")) +detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding="utf-8")) for f in detect.get('files', {}).get('code', []): code_files.extend(collect_files(Path(f)) if Path(f).is_dir() else [Path(f)]) if code_files: result = extract(code_files, cache_root=Path('INPUT_PATH')) - Path('graphify-out/.graphify_ast.json').write_text(json.dumps(result, indent=2, ensure_ascii=False), encoding=\"utf-8\") - print(f'AST: {len(result[\"nodes\"])} nodes, {len(result[\"edges\"])} edges') + Path('graphify-out/.graphify_ast.json').write_text(json.dumps(result, indent=2, ensure_ascii=False), encoding="utf-8") + print(f'AST: {len(result["nodes"])} nodes, {len(result["edges"])} edges') else: - Path('graphify-out/.graphify_ast.json').write_text(json.dumps({'nodes':[],'edges':[],'input_tokens':0,'output_tokens':0}, ensure_ascii=False), encoding=\"utf-8\") + Path('graphify-out/.graphify_ast.json').write_text(json.dumps({'nodes':[],'edges':[],'input_tokens':0,'output_tokens':0}, ensure_ascii=False), encoding="utf-8") print('No code files - skipping AST extraction') -" +'@ | & (Get-Content graphify-out\.graphify_python) - ``` #### Part B - Semantic extraction (parallel subagents) **Fast path:** If detection found zero docs, papers, and images (code-only corpus), skip Part B entirely and go straight to Part C. AST handles code - there is nothing for semantic subagents to do. **First write an empty semantic file** so Part C's merge has its input (it reads `.graphify_semantic.json` unconditionally; without this a code-only run hits `FileNotFoundError`): -```bash -$(cat graphify-out/.graphify_python) -c " +```powershell +@' import json from pathlib import Path Path('graphify-out/.graphify_semantic.json').write_text(json.dumps({'nodes':[],'edges':[],'hyperedges':[],'input_tokens':0,'output_tokens':0}), encoding='utf-8') -" +'@ | & (Get-Content graphify-out\.graphify_python) - ``` **MANDATORY: You MUST use the Agent tool here. Reading files yourself one-by-one is forbidden - it is 5-10x slower. If you do not use the Agent tool you are doing this wrong.** @@ -238,13 +241,13 @@ Before dispatching any subagents, check which files already have cached extracti SPEC_PATH below is the **absolute** path of the `references/extraction-spec.md` that ships beside this SKILL.md — the same file Step B2 loads and hands to every subagent. It is the extraction prompt, so cache entries are attributed to it: when a graphify upgrade changes the prompt, entries produced by the old one are re-extracted instead of replayed, and unchanged prompts keep their entries (#1939). Substitute the real path in both Step B0 and Step B3 — pass the same one to each, and do not drop the argument. -```bash -$(cat graphify-out/.graphify_python) -c " +```powershell +@' import json from graphify.cache import check_semantic_cache from pathlib import Path -detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\")) +detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding="utf-8")) # Only content files go to semantic extraction. Code is already covered structurally # by the AST pass (Part A); flattening every category here makes subagents re-read # every source file (#1392). Video is transcribed to a document in Step 2.5 first. @@ -255,12 +258,12 @@ cached_nodes, cached_edges, cached_hyperedges, uncached = check_semantic_cache(a # Always (re)write the cache file: write hits, else DELETE any leftover from a prior # run so Part C never merges a stale .graphify_cached.json (#1392). if cached_nodes or cached_edges or cached_hyperedges: - Path('graphify-out/.graphify_cached.json').write_text(json.dumps({'nodes': cached_nodes, 'edges': cached_edges, 'hyperedges': cached_hyperedges}, ensure_ascii=False), encoding=\"utf-8\") + Path('graphify-out/.graphify_cached.json').write_text(json.dumps({'nodes': cached_nodes, 'edges': cached_edges, 'hyperedges': cached_hyperedges}, ensure_ascii=False), encoding="utf-8") else: Path('graphify-out/.graphify_cached.json').unlink(missing_ok=True) -Path('graphify-out/.graphify_uncached.txt').write_text('\n'.join(uncached), encoding=\"utf-8\") +Path('graphify-out/.graphify_uncached.txt').write_text('\n'.join(uncached), encoding="utf-8") print(f'Cache: {len(all_files)-len(uncached)} files hit, {len(uncached)} files need extraction') -" +'@ | & (Get-Content graphify-out\.graphify_python) - ``` Only dispatch subagents for files listed in `graphify-out/.graphify_uncached.txt`. If all files are cached, skip to Part C directly. @@ -306,8 +309,8 @@ Wait for all subagents. For each result: If more than half the chunks failed or are missing, stop and tell the user to re-run and ensure `subagent_type="general-purpose"` is used. Merge all chunk files into `.graphify_semantic_new.json`. **After each Agent call completes, read the real token counts from the Agent tool result's `usage` field and write them back into the chunk JSON before merging** — the chunk JSON itself always has placeholder zeros. Then run: -```bash -$(cat graphify-out/.graphify_python) -c " +```powershell +@' import json, glob from pathlib import Path @@ -315,7 +318,7 @@ chunks = sorted(glob.glob('graphify-out/.graphify_chunk_*.json')) all_nodes, all_edges, all_hyperedges = [], [], [] total_in, total_out = 0, 0 for c in chunks: - d = json.loads(Path(c).read_text(encoding=\"utf-8\")) + d = json.loads(Path(c).read_text(encoding="utf-8")) all_nodes += d.get('nodes', []) all_edges += d.get('edges', []) all_hyperedges += d.get('hyperedges', []) @@ -324,33 +327,33 @@ for c in chunks: Path('graphify-out/.graphify_semantic_new.json').write_text(json.dumps({ 'nodes': all_nodes, 'edges': all_edges, 'hyperedges': all_hyperedges, 'input_tokens': total_in, 'output_tokens': total_out, -}, indent=2, ensure_ascii=False), encoding=\"utf-8\") +}, indent=2, ensure_ascii=False), encoding="utf-8") print(f'Merged {len(chunks)} chunks: {total_in:,} in / {total_out:,} out tokens') -" +'@ | & (Get-Content graphify-out\.graphify_python) - ``` Save new results to cache. Pass the same SPEC_PATH as Step B0 — it stamps each entry with the prompt that produced it, and a write under a different prompt than the read lands where the next run won't look (#1939): -```bash -$(cat graphify-out/.graphify_python) -c " +```powershell +@' import json from graphify.cache import save_semantic_cache from pathlib import Path -new = json.loads(Path('graphify-out/.graphify_semantic_new.json').read_text(encoding=\"utf-8\")) if Path('graphify-out/.graphify_semantic_new.json').exists() else {'nodes':[],'edges':[],'hyperedges':[]} -uncached = [line for line in Path('graphify-out/.graphify_uncached.txt').read_text(encoding=\"utf-8\").splitlines() if line] +new = json.loads(Path('graphify-out/.graphify_semantic_new.json').read_text(encoding="utf-8")) if Path('graphify-out/.graphify_semantic_new.json').exists() else {'nodes':[],'edges':[],'hyperedges':[]} +uncached = [line for line in Path('graphify-out/.graphify_uncached.txt').read_text(encoding="utf-8").splitlines() if line] saved = save_semantic_cache(new.get('nodes', []), new.get('edges', []), new.get('hyperedges', []), root='INPUT_PATH', allowed_source_files=uncached, prompt_file='SPEC_PATH') print(f'Cached {saved} files') -" +'@ | & (Get-Content graphify-out\.graphify_python) - ``` Merge cached + new results into `graphify-out/.graphify_semantic.json`: -```bash -$(cat graphify-out/.graphify_python) -c " +```powershell +@' import json from pathlib import Path -cached = json.loads(Path('graphify-out/.graphify_cached.json').read_text(encoding=\"utf-8\")) if Path('graphify-out/.graphify_cached.json').exists() else {'nodes':[],'edges':[],'hyperedges':[]} -new = json.loads(Path('graphify-out/.graphify_semantic_new.json').read_text(encoding=\"utf-8\")) if Path('graphify-out/.graphify_semantic_new.json').exists() else {'nodes':[],'edges':[],'hyperedges':[]} +cached = json.loads(Path('graphify-out/.graphify_cached.json').read_text(encoding="utf-8")) if Path('graphify-out/.graphify_cached.json').exists() else {'nodes':[],'edges':[],'hyperedges':[]} +new = json.loads(Path('graphify-out/.graphify_semantic_new.json').read_text(encoding="utf-8")) if Path('graphify-out/.graphify_semantic_new.json').exists() else {'nodes':[],'edges':[],'hyperedges':[]} all_nodes = cached['nodes'] + new.get('nodes', []) all_edges = cached['edges'] + new.get('edges', []) @@ -369,21 +372,21 @@ merged = { 'input_tokens': new.get('input_tokens', 0), 'output_tokens': new.get('output_tokens', 0), } -Path('graphify-out/.graphify_semantic.json').write_text(json.dumps(merged, indent=2, ensure_ascii=False), encoding=\"utf-8\") -print(f'Extraction complete - {len(deduped)} nodes, {len(all_edges)} edges ({len(cached[\"nodes\"])} from cache, {len(new.get(\"nodes\",[]))} new)') -" +Path('graphify-out/.graphify_semantic.json').write_text(json.dumps(merged, indent=2, ensure_ascii=False), encoding="utf-8") +print(f'Extraction complete - {len(deduped)} nodes, {len(all_edges)} edges ({len(cached["nodes"])} from cache, {len(new.get("nodes",[]))} new)') +'@ | & (Get-Content graphify-out\.graphify_python) - ``` -Clean up temp files: `rm -f graphify-out/.graphify_cached.json graphify-out/.graphify_uncached.txt graphify-out/.graphify_semantic_new.json` +Clean up temp files: `Remove-Item -Force -ErrorAction SilentlyContinue graphify-out\.graphify_cached.json, graphify-out\.graphify_uncached.txt, graphify-out\.graphify_semantic_new.json` #### Part C - Merge AST + semantic into final extraction -```bash -$(cat graphify-out/.graphify_python) -c " +```powershell +@' import sys, json from pathlib import Path -ast = json.loads(Path('graphify-out/.graphify_ast.json').read_text(encoding=\"utf-8\")) -sem = json.loads(Path('graphify-out/.graphify_semantic.json').read_text(encoding=\"utf-8\")) +ast = json.loads(Path('graphify-out/.graphify_ast.json').read_text(encoding="utf-8")) +sem = json.loads(Path('graphify-out/.graphify_semantic.json').read_text(encoding="utf-8")) # Merge: AST nodes first, semantic nodes deduplicated by id seen = {n['id'] for n in ast['nodes']} @@ -402,20 +405,20 @@ merged = { 'input_tokens': sem.get('input_tokens', 0), 'output_tokens': sem.get('output_tokens', 0), } -Path('graphify-out/.graphify_extract.json').write_text(json.dumps(merged, indent=2, ensure_ascii=False), encoding=\"utf-8\") +Path('graphify-out/.graphify_extract.json').write_text(json.dumps(merged, indent=2, ensure_ascii=False), encoding="utf-8") total = len(merged_nodes) edges = len(merged_edges) -print(f'Merged: {total} nodes, {edges} edges ({len(ast[\"nodes\"])} AST + {len(sem[\"nodes\"])} semantic)') -" +print(f'Merged: {total} nodes, {edges} edges ({len(ast["nodes"])} AST + {len(sem["nodes"])} semantic)') +'@ | & (Get-Content graphify-out\.graphify_python) - ``` ### Step 4 - Build graph, cluster, analyze, generate outputs **Before starting:** the code blocks below pass `directed=IS_DIRECTED` to `build_from_json()`. Replace `IS_DIRECTED` with `True` if `--directed` was given (builds a `DiGraph` preserving edge direction source→target), otherwise `False` (the default undirected `Graph`). Substitute it the same way you substitute `INPUT_PATH` — do not leave the literal `IS_DIRECTED` in the code. -```bash -mkdir -p graphify-out -$(cat graphify-out/.graphify_python) -c " +```powershell +New-Item -ItemType Directory -Force -Path graphify-out | Out-Null +@' import sys, json from graphify.build import build_from_json from graphify.cluster import cluster, score_all @@ -424,8 +427,8 @@ from graphify.report import generate from graphify.export import to_json from pathlib import Path -extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) -detection = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\")) +extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding="utf-8")) +detection = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding="utf-8")) # root= mirrors the --update runbook (#1361): relativize source_file to the same # base so the full build and incremental --update never drift apart on re-extract. @@ -455,7 +458,7 @@ if not wrote: print('If this shrink is intentional (you deleted files), re-run a full build with --force.') raise SystemExit(1) report = generate(G, communities, cohesion, labels, gods, surprises, detection, tokens, 'INPUT_PATH', suggested_questions=questions) -Path('graphify-out/GRAPH_REPORT.md').write_text(report, encoding=\"utf-8\") +Path('graphify-out/GRAPH_REPORT.md').write_text(report, encoding="utf-8") analysis = { 'communities': {str(k): v for k, v in communities.items()}, 'cohesion': {str(k): v for k, v in cohesion.items()}, @@ -463,9 +466,9 @@ analysis = { 'surprises': surprises, 'questions': questions, } -Path('graphify-out/.graphify_analysis.json').write_text(json.dumps(analysis, indent=2, ensure_ascii=False), encoding=\"utf-8\") +Path('graphify-out/.graphify_analysis.json').write_text(json.dumps(analysis, indent=2, ensure_ascii=False), encoding="utf-8") print(f'Graph: {G.number_of_nodes()} nodes, {G.number_of_edges()} edges, {len(communities)} communities') -" +'@ | & (Get-Content graphify-out\.graphify_python) - ``` If this step prints `ERROR: Graph is empty`, stop and tell the user what happened - do not proceed to labeling or visualization. @@ -476,13 +479,13 @@ Replace INPUT_PATH with the actual path. A non-destructive diagnostic on the extraction, before labeling. It surfaces edge collapse, dangling/missing endpoints, and self-loops — the silent-corruption modes of incremental updates and AST/LLM id mismatches. Read-only; never aborts. -```bash -$(cat graphify-out/.graphify_python) -c " +```powershell +@' import json from pathlib import Path from graphify.diagnostics import diagnose_extraction, format_diagnostic_report -extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) +extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding="utf-8")) summary = diagnose_extraction(extraction, directed=IS_DIRECTED, root='INPUT_PATH') print(format_diagnostic_report(summary)) flags = [f'{summary[k]} {label}' for k, label in ( @@ -493,7 +496,7 @@ flags = [f'{summary[k]} {label}' for k, label in ( ('undirected_same_endpoint_collapsed_edges', 'collapsed (undirected) edges'), ) if summary.get(k, 0)] print('GRAPH HEALTH WARNING: ' + '; '.join(flags) + ' - graph may be incomplete/corrupt.' if flags else 'Graph health: OK (no dangling/missing/collapsed edges).') -" +'@ | & (Get-Content graphify-out\.graphify_python) - ``` Substitute `IS_DIRECTED` and `INPUT_PATH` as in Step 4. If a `GRAPH HEALTH WARNING` prints, surface it in the final summary (do not abort — the graph is still usable, but the integrity issue must be visible, per the Honesty Rules). @@ -504,8 +507,8 @@ Read `graphify-out/.graphify_analysis.json`. For each community key, look at its Then regenerate the report and save the labels for the visualizer: -```bash -$(cat graphify-out/.graphify_python) -c " +```powershell +@' import sys, json from graphify.build import build_from_json from graphify.cluster import score_all @@ -514,9 +517,9 @@ from graphify.report import generate from graphify.export import to_json from pathlib import Path -extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) -detection = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\")) -analysis = json.loads(Path('graphify-out/.graphify_analysis.json').read_text(encoding=\"utf-8\")) +extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding="utf-8")) +detection = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding="utf-8")) +analysis = json.loads(Path('graphify-out/.graphify_analysis.json').read_text(encoding="utf-8")) # root= as in Step 4 / the --update runbook (#1361) — same base for node-key parity. G = build_from_json(extraction, root='INPUT_PATH', directed=IS_DIRECTED) @@ -531,8 +534,8 @@ labels = LABELS_DICT questions = suggest_questions(G, communities, labels) report = generate(G, communities, cohesion, labels, analysis['gods'], analysis['surprises'], detection, tokens, 'INPUT_PATH', suggested_questions=questions) -Path('graphify-out/GRAPH_REPORT.md').write_text(report, encoding=\"utf-8\") -Path('graphify-out/.graphify_labels.json').write_text(json.dumps({str(k): v for k, v in labels.items()}, ensure_ascii=False), encoding=\"utf-8\") +Path('graphify-out/GRAPH_REPORT.md').write_text(report, encoding="utf-8") +Path('graphify-out/.graphify_labels.json').write_text(json.dumps({str(k): v for k, v in labels.items()}, ensure_ascii=False), encoding="utf-8") # Re-export so graph.json nodes carry the curated community_name (#2490). # Same extraction as Step 4, so the #479 shrink-guard passes on node count; # if it still refuses, surface the guard message - do not force past it. @@ -541,7 +544,7 @@ if not wrote: print('ERROR: refused to shrink graphify-out/graph.json (existing graph has more nodes; #479).') print('If this shrink is intentional (you deleted files), re-run a full build with --force.') print('Report updated with community labels') -" +'@ | & (Get-Content graphify-out\.graphify_python) - ``` Replace `LABELS_DICT` with the actual dict you constructed (e.g. `{0: "Attention Mechanism", 1: "Training Pipeline"}`). @@ -555,14 +558,14 @@ If `--obsidian` was given: - If `--obsidian-dir ` was also given, pass it via `--dir`. Otherwise defaults to `graphify-out/obsidian`. -```bash +```powershell graphify export obsidian # or with custom dir: graphify export obsidian --dir ~/vaults/my-project ``` Generate the HTML graph (always, unless `--no-viz`): -```bash +```powershell graphify export html # auto-aggregates to community view if graph > 5000 nodes # or: graphify export html --no-viz ``` @@ -575,16 +578,16 @@ These run only when their flag is present (`--wiki`, `--neo4j`/`--neo4j-push`, ` ### Step 9 - Save manifest, update cost tracker, clean up, and report -```bash -$(cat graphify-out/.graphify_python) -c " +```powershell +@' import json from pathlib import Path from datetime import datetime, timezone from graphify.detect import save_manifest # Save manifest for --update -detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\")) -extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) +detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding="utf-8")) +extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding="utf-8")) # In --update mode, 'all_files' carries the full corpus; 'files' is the changed # subset. Full-rebuild mode populates only 'files', so the fallback handles that. # root= relativizes the manifest keys to the scan root (same base as the build), @@ -620,7 +623,7 @@ output_tok = extract.get('output_tokens', 0) cost_path = Path('graphify-out/cost.json') if cost_path.exists(): - cost = json.loads(cost_path.read_text(encoding=\"utf-8\")) + cost = json.loads(cost_path.read_text(encoding="utf-8")) else: cost = {'runs': [], 'total_input_tokens': 0, 'total_output_tokens': 0} @@ -632,14 +635,14 @@ cost['runs'].append({ }) cost['total_input_tokens'] += input_tok cost['total_output_tokens'] += output_tok -cost_path.write_text(json.dumps(cost, indent=2, ensure_ascii=False), encoding=\"utf-8\") +cost_path.write_text(json.dumps(cost, indent=2, ensure_ascii=False), encoding="utf-8") print(f'This run: {input_tok:,} input tokens, {output_tok:,} output tokens') -print(f'All time: {cost[\"total_input_tokens\"]:,} input, {cost[\"total_output_tokens\"]:,} output ({len(cost[\"runs\"])} runs)') -" -rm -f graphify-out/.graphify_detect.json graphify-out/.graphify_extract.json graphify-out/.graphify_ast.json graphify-out/.graphify_semantic.json graphify-out/.graphify_analysis.json -find graphify-out -maxdepth 1 -name '.graphify_chunk_*.json' -delete 2>/dev/null -rm -f graphify-out/.needs_update 2>/dev/null || true +print(f'All time: {cost["total_input_tokens"]:,} input, {cost["total_output_tokens"]:,} output ({len(cost["runs"])} runs)') +'@ | & (Get-Content graphify-out\.graphify_python) - +Remove-Item -Force -ErrorAction SilentlyContinue graphify-out\.graphify_detect.json, graphify-out\.graphify_extract.json, graphify-out\.graphify_ast.json, graphify-out\.graphify_semantic.json, graphify-out\.graphify_analysis.json +Get-ChildItem graphify-out -Filter '.graphify_chunk_*.json' -File -ErrorAction SilentlyContinue | Remove-Item -Force +Remove-Item -Force -ErrorAction SilentlyContinue graphify-out\.needs_update ``` Replace INPUT_PATH with the actual path (same value used in Steps 4-5) so the manifest is relativized to the scan root. @@ -679,18 +682,20 @@ The graph is the map. Your job after the pipeline is to be the guide. Before running any subcommand below (`--update`, `--cluster-only`, `query`, `path`, `explain`, `add`), check that `.graphify_python` exists. If it's missing (e.g. user deleted `graphify-out/`), re-resolve the interpreter first: -```bash -if [ ! -f graphify-out/.graphify_python ]; then - GRAPHIFY_BIN=$(which graphify 2>/dev/null) - if [ -n "$GRAPHIFY_BIN" ]; then - PYTHON=$(head -1 "$GRAPHIFY_BIN" | tr -d '#!') - case "$PYTHON" in *[!a-zA-Z0-9/_.@-]*) PYTHON="python3" ;; esac - else - PYTHON="python3" - fi - mkdir -p graphify-out - "$PYTHON" -c "import sys; open('graphify-out/.graphify_python', 'w', encoding='utf-8').write(sys.executable)" -fi +```powershell +if (-not (Test-Path graphify-out\.graphify_python)) { + $GRAPHIFY_PYTHON = $null + $graphifyCmd = Get-Command graphify -ErrorAction SilentlyContinue + if ($graphifyCmd) { + # The interpreter that owns the graphify entry point sits next to it + # (\Scripts\python.exe for uv tool, pipx, and venv installs). + $py = Join-Path (Split-Path $graphifyCmd.Source) "python.exe" + if (Test-Path $py) { $GRAPHIFY_PYTHON = $py } + } + if (-not $GRAPHIFY_PYTHON) { $GRAPHIFY_PYTHON = "python" } + New-Item -ItemType Directory -Force -Path graphify-out | Out-Null + & $GRAPHIFY_PYTHON -c "import sys; open('graphify-out/.graphify_python', 'w', encoding='utf-8').write(sys.executable)" +} ``` ## For --update and --cluster-only @@ -703,7 +708,7 @@ Both are non-default subcommands. `--update` re-extracts only new or changed fil When `graphify-out/graph.json` already exists and the user asks a question about the corpus, answer from the graph rather than rebuilding it: -```bash +```powershell graphify query "" ``` diff --git a/graphify/skill.md b/graphify/skill.md index cef4d5c..abd2811 100644 --- a/graphify/skill.md +++ b/graphify/skill.md @@ -112,8 +112,11 @@ import json from graphify.detect import detect from pathlib import Path result = detect(Path('INPUT_PATH')) -print(json.dumps(result, ensure_ascii=False)) -" > graphify-out/.graphify_detect.json +# Write the sidecar from Python, not a shell redirect, so the same block renders +# on PowerShell hosts without console-encoding drift (#2528). +Path('graphify-out/.graphify_detect.json').write_text(json.dumps(result, ensure_ascii=False), encoding=\"utf-8\") +print(f'Detected {result[\"total_files\"]} files') +" ``` Replace INPUT_PATH with the actual path the user provided. Do NOT cat or print the JSON - read it silently and present a clean summary instead: diff --git a/pyproject.toml b/pyproject.toml index 6288b9c..f99f81e 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta" [project] name = "graphifyy" -version = "0.9.35" +version = "0.9.36" description = "AI coding assistant skill (Claude Code, CodeBuddy, Codex, OpenCode, Kilo Code, Cursor, Gemini CLI, Aider, OpenClaw, Factory Droid, Trae, Hermes, Kiro, Pi, Devin CLI, Google Antigravity) - turn any folder of code, docs, papers, images, or videos into a queryable knowledge graph" readme = "README.md" license = "Apache-2.0" diff --git a/tests/test_skillgen.py b/tests/test_skillgen.py index 0c09e60..cf11686 100644 --- a/tests/test_skillgen.py +++ b/tests/test_skillgen.py @@ -353,6 +353,122 @@ def test_every_platform_query_has_expansion_and_fallback(): assert "## For /graphify explain" in q +# --- cross-shell parity for powershell hosts (#2528) --------------------------- + +# Bash-only tokens that must never reach a strict-PowerShell host's skill body. +_BASH_ONLY_TOKENS = ("$(cat ", "rm -f ", "2>/dev/null", "```bash") + +# The default-pipeline step headings that must exist on BOTH shells (parity). +_STEP_HEADINGS = ( + "### Step 1 - Ensure graphify is installed", + "### Step 2 - Detect files", + "### Step 3 - Extract entities and relationships", + "#### Part A - Structural extraction for code files", + "#### Part B - Semantic extraction (parallel subagents)", + "#### Part C - Merge AST + semantic into final extraction", + "### Step 4 - Build graph, cluster, analyze, generate outputs", + "### Step 4.5 - Graph health check (read-only integrity gate)", + "### Step 5 - Label communities", + "### Step 6 - Generate Obsidian vault (opt-in) + HTML", + "### Step 9 - Save manifest, update cost tracker, clean up, and report", + "## Interpreter guard for subcommands", + "## Honesty Rules", +) + + +def _powershell_platform_keys(): + """Every platform that renders for a strict-PowerShell host (windows today, + plus any future powershell-shell platform automatically).""" + platforms = gen.load_platforms() + keys = [k for k, p in platforms.items() if p.shell == "powershell"] + assert "windows" in keys, "the windows platform must declare shell = powershell" + return keys + + +def test_powershell_hosts_carry_no_bash_only_shell(): + """#2528: the Windows variant had a PowerShell Step 1 but bash for Steps 2+ + (``$(cat ...) -c``, ``rm -f``, ``find ... -delete``, ``2>/dev/null``), so a + strict-PowerShell host failed at Step 2. Every powershell-shell render must + now be free of bash-only tokens and carry the here-string stdin invocation.""" + import re + + platforms = gen.load_platforms() + for key in _powershell_platform_keys(): + core = gen.render(platforms[key])[0].content + for token in _BASH_ONLY_TOKENS: + assert token not in core, f"[{key}] bash-only token survived: {token!r}" + assert re.search(r"\bfind\b[^\n]*-delete", core) is None, f"[{key}] find -delete survived" + # The PowerShell invocation pattern replaces $(cat ...) -c "..." everywhere. + assert "```powershell" in core + assert "'@ | & (Get-Content graphify-out\\.graphify_python) -" in core, ( + f"[{key}] missing the here-string stdin python invocation" + ) + # Cleanup went through Remove-Item / Get-ChildItem, not rm/find. + assert "Remove-Item -Force -ErrorAction SilentlyContinue" in core + assert "Get-ChildItem graphify-out -Filter '.graphify_chunk_*.json'" in core + + +def test_windows_and_posix_cores_have_step_and_2490_parity(): + """Both shells run the same pipeline: every step heading and the #2490 + curated-labels re-export line appear in skill.md AND skill-windows.md.""" + claude_core, _ = _platform_artifacts("claude") + windows_core, _ = _platform_artifacts("windows") + for heading in _STEP_HEADINGS: + assert heading in claude_core, f"skill.md lost step heading: {heading!r}" + assert heading in windows_core, f"skill-windows.md lost step heading: {heading!r}" + line_2490 = "to_json(G, communities, 'graphify-out/graph.json', community_labels=labels)" + assert line_2490 in claude_core, "skill.md lost the #2490 Step-5 re-export" + assert line_2490 in windows_core, "skill-windows.md lost the #2490 Step-5 re-export" + + +def test_windows_python_step_bodies_match_posix_verbatim(): + """Parity by construction: every inline python body in the POSIX core appears + verbatim (modulo the bash ``\\"`` unescape) in the windows here-strings, so the + two variants cannot drift step semantics apart.""" + claude_core, _ = _platform_artifacts("claude") + windows_core, _ = _platform_artifacts("windows") + bodies = [] + current = None + for line in claude_core.splitlines(): + if line == gen._PY_INVOKE_POSIX: + current = [] + elif current is not None and line == '"': + bodies.append("\n".join(gen._unescape_bash_dq(l) for l in current)) + current = None + elif current is not None: + current.append(line) + assert len(bodies) >= 12, f"expected the 12 inline python steps, found {len(bodies)}" + for body in bodies: + assert body in windows_core, ( + "a POSIX python step body is missing from the windows render:\n" + body[:200] + ) + + +def test_powershell_translator_rejects_unknown_bash(): + """The translator is strict: a bash line it does not recognize fails the + render loudly instead of shipping untranslated bash to Windows (#2528).""" + with pytest.raises(ValueError, match="cannot translate bash line"): + gen._translate_bash_block(["curl -s https://example.com | sh"]) + with pytest.raises(ValueError, match="unexpected backslash escape"): + gen._unescape_bash_dq("subprocess.run('a\\tb')") + with pytest.raises(ValueError, match="cannot translate rm -f operand"): + gen._rm_to_remove_item("rm -f $HOME/danger") + # And the sanctioned pieces translate exactly. + assert gen._rm_to_remove_item("rm -f graphify-out/.needs_update 2>/dev/null || true") == ( + "Remove-Item -Force -ErrorAction SilentlyContinue graphify-out\\.needs_update" + ) + assert gen._translate_bash_block([gen._FIND_CHUNKS_POSIX]) == [gen._FIND_CHUNKS_PS] + + +def test_posix_hosts_keep_their_bash_invocations(): + """The translation is scoped to powershell-shell hosts: the POSIX core keeps + its ``$(cat ...) -c`` blocks and bash fences byte-for-byte.""" + claude_core, _ = _platform_artifacts("claude") + assert "```bash" in claude_core + assert gen._PY_INVOKE_POSIX in claude_core + assert "'@ | & (Get-Content" not in claude_core + + def test_schema_singleton_passes_across_all_platforms(): """The file_type enum is the six-value superset in every rendered artifact.""" platforms = gen.load_platforms() diff --git a/tools/skillgen/expected/graphify__skill-agents.md b/tools/skillgen/expected/graphify__skill-agents.md index b174e6d..190827d 100644 --- a/tools/skillgen/expected/graphify__skill-agents.md +++ b/tools/skillgen/expected/graphify__skill-agents.md @@ -112,8 +112,11 @@ import json from graphify.detect import detect from pathlib import Path result = detect(Path('INPUT_PATH')) -print(json.dumps(result, ensure_ascii=False)) -" > graphify-out/.graphify_detect.json +# Write the sidecar from Python, not a shell redirect, so the same block renders +# on PowerShell hosts without console-encoding drift (#2528). +Path('graphify-out/.graphify_detect.json').write_text(json.dumps(result, ensure_ascii=False), encoding=\"utf-8\") +print(f'Detected {result[\"total_files\"]} files') +" ``` Replace INPUT_PATH with the actual path the user provided. Do NOT cat or print the JSON - read it silently and present a clean summary instead: diff --git a/tools/skillgen/expected/graphify__skill-amp.md b/tools/skillgen/expected/graphify__skill-amp.md index b174e6d..190827d 100644 --- a/tools/skillgen/expected/graphify__skill-amp.md +++ b/tools/skillgen/expected/graphify__skill-amp.md @@ -112,8 +112,11 @@ import json from graphify.detect import detect from pathlib import Path result = detect(Path('INPUT_PATH')) -print(json.dumps(result, ensure_ascii=False)) -" > graphify-out/.graphify_detect.json +# Write the sidecar from Python, not a shell redirect, so the same block renders +# on PowerShell hosts without console-encoding drift (#2528). +Path('graphify-out/.graphify_detect.json').write_text(json.dumps(result, ensure_ascii=False), encoding=\"utf-8\") +print(f'Detected {result[\"total_files\"]} files') +" ``` Replace INPUT_PATH with the actual path the user provided. Do NOT cat or print the JSON - read it silently and present a clean summary instead: diff --git a/tools/skillgen/expected/graphify__skill-claw.md b/tools/skillgen/expected/graphify__skill-claw.md index cef4d5c..abd2811 100644 --- a/tools/skillgen/expected/graphify__skill-claw.md +++ b/tools/skillgen/expected/graphify__skill-claw.md @@ -112,8 +112,11 @@ import json from graphify.detect import detect from pathlib import Path result = detect(Path('INPUT_PATH')) -print(json.dumps(result, ensure_ascii=False)) -" > graphify-out/.graphify_detect.json +# Write the sidecar from Python, not a shell redirect, so the same block renders +# on PowerShell hosts without console-encoding drift (#2528). +Path('graphify-out/.graphify_detect.json').write_text(json.dumps(result, ensure_ascii=False), encoding=\"utf-8\") +print(f'Detected {result[\"total_files\"]} files') +" ``` Replace INPUT_PATH with the actual path the user provided. Do NOT cat or print the JSON - read it silently and present a clean summary instead: diff --git a/tools/skillgen/expected/graphify__skill-codex.md b/tools/skillgen/expected/graphify__skill-codex.md index bd8a020..af3f723 100644 --- a/tools/skillgen/expected/graphify__skill-codex.md +++ b/tools/skillgen/expected/graphify__skill-codex.md @@ -112,8 +112,11 @@ import json from graphify.detect import detect from pathlib import Path result = detect(Path('INPUT_PATH')) -print(json.dumps(result, ensure_ascii=False)) -" > graphify-out/.graphify_detect.json +# Write the sidecar from Python, not a shell redirect, so the same block renders +# on PowerShell hosts without console-encoding drift (#2528). +Path('graphify-out/.graphify_detect.json').write_text(json.dumps(result, ensure_ascii=False), encoding=\"utf-8\") +print(f'Detected {result[\"total_files\"]} files') +" ``` Replace INPUT_PATH with the actual path the user provided. Do NOT cat or print the JSON - read it silently and present a clean summary instead: diff --git a/tools/skillgen/expected/graphify__skill-copilot.md b/tools/skillgen/expected/graphify__skill-copilot.md index cef4d5c..abd2811 100644 --- a/tools/skillgen/expected/graphify__skill-copilot.md +++ b/tools/skillgen/expected/graphify__skill-copilot.md @@ -112,8 +112,11 @@ import json from graphify.detect import detect from pathlib import Path result = detect(Path('INPUT_PATH')) -print(json.dumps(result, ensure_ascii=False)) -" > graphify-out/.graphify_detect.json +# Write the sidecar from Python, not a shell redirect, so the same block renders +# on PowerShell hosts without console-encoding drift (#2528). +Path('graphify-out/.graphify_detect.json').write_text(json.dumps(result, ensure_ascii=False), encoding=\"utf-8\") +print(f'Detected {result[\"total_files\"]} files') +" ``` Replace INPUT_PATH with the actual path the user provided. Do NOT cat or print the JSON - read it silently and present a clean summary instead: diff --git a/tools/skillgen/expected/graphify__skill-droid.md b/tools/skillgen/expected/graphify__skill-droid.md index 370e0d6..fd148d4 100644 --- a/tools/skillgen/expected/graphify__skill-droid.md +++ b/tools/skillgen/expected/graphify__skill-droid.md @@ -112,8 +112,11 @@ import json from graphify.detect import detect from pathlib import Path result = detect(Path('INPUT_PATH')) -print(json.dumps(result, ensure_ascii=False)) -" > graphify-out/.graphify_detect.json +# Write the sidecar from Python, not a shell redirect, so the same block renders +# on PowerShell hosts without console-encoding drift (#2528). +Path('graphify-out/.graphify_detect.json').write_text(json.dumps(result, ensure_ascii=False), encoding=\"utf-8\") +print(f'Detected {result[\"total_files\"]} files') +" ``` Replace INPUT_PATH with the actual path the user provided. Do NOT cat or print the JSON - read it silently and present a clean summary instead: diff --git a/tools/skillgen/expected/graphify__skill-kilo.md b/tools/skillgen/expected/graphify__skill-kilo.md index f83cb06..3e70b05 100644 --- a/tools/skillgen/expected/graphify__skill-kilo.md +++ b/tools/skillgen/expected/graphify__skill-kilo.md @@ -112,8 +112,11 @@ import json from graphify.detect import detect from pathlib import Path result = detect(Path('INPUT_PATH')) -print(json.dumps(result, ensure_ascii=False)) -" > graphify-out/.graphify_detect.json +# Write the sidecar from Python, not a shell redirect, so the same block renders +# on PowerShell hosts without console-encoding drift (#2528). +Path('graphify-out/.graphify_detect.json').write_text(json.dumps(result, ensure_ascii=False), encoding=\"utf-8\") +print(f'Detected {result[\"total_files\"]} files') +" ``` Replace INPUT_PATH with the actual path the user provided. Do NOT cat or print the JSON - read it silently and present a clean summary instead: diff --git a/tools/skillgen/expected/graphify__skill-kiro.md b/tools/skillgen/expected/graphify__skill-kiro.md index cef4d5c..abd2811 100644 --- a/tools/skillgen/expected/graphify__skill-kiro.md +++ b/tools/skillgen/expected/graphify__skill-kiro.md @@ -112,8 +112,11 @@ import json from graphify.detect import detect from pathlib import Path result = detect(Path('INPUT_PATH')) -print(json.dumps(result, ensure_ascii=False)) -" > graphify-out/.graphify_detect.json +# Write the sidecar from Python, not a shell redirect, so the same block renders +# on PowerShell hosts without console-encoding drift (#2528). +Path('graphify-out/.graphify_detect.json').write_text(json.dumps(result, ensure_ascii=False), encoding=\"utf-8\") +print(f'Detected {result[\"total_files\"]} files') +" ``` Replace INPUT_PATH with the actual path the user provided. Do NOT cat or print the JSON - read it silently and present a clean summary instead: diff --git a/tools/skillgen/expected/graphify__skill-opencode.md b/tools/skillgen/expected/graphify__skill-opencode.md index c121617..91ced60 100644 --- a/tools/skillgen/expected/graphify__skill-opencode.md +++ b/tools/skillgen/expected/graphify__skill-opencode.md @@ -112,8 +112,11 @@ import json from graphify.detect import detect from pathlib import Path result = detect(Path('INPUT_PATH')) -print(json.dumps(result, ensure_ascii=False)) -" > graphify-out/.graphify_detect.json +# Write the sidecar from Python, not a shell redirect, so the same block renders +# on PowerShell hosts without console-encoding drift (#2528). +Path('graphify-out/.graphify_detect.json').write_text(json.dumps(result, ensure_ascii=False), encoding=\"utf-8\") +print(f'Detected {result[\"total_files\"]} files') +" ``` Replace INPUT_PATH with the actual path the user provided. Do NOT cat or print the JSON - read it silently and present a clean summary instead: diff --git a/tools/skillgen/expected/graphify__skill-pi.md b/tools/skillgen/expected/graphify__skill-pi.md index cef4d5c..abd2811 100644 --- a/tools/skillgen/expected/graphify__skill-pi.md +++ b/tools/skillgen/expected/graphify__skill-pi.md @@ -112,8 +112,11 @@ import json from graphify.detect import detect from pathlib import Path result = detect(Path('INPUT_PATH')) -print(json.dumps(result, ensure_ascii=False)) -" > graphify-out/.graphify_detect.json +# Write the sidecar from Python, not a shell redirect, so the same block renders +# on PowerShell hosts without console-encoding drift (#2528). +Path('graphify-out/.graphify_detect.json').write_text(json.dumps(result, ensure_ascii=False), encoding=\"utf-8\") +print(f'Detected {result[\"total_files\"]} files') +" ``` Replace INPUT_PATH with the actual path the user provided. Do NOT cat or print the JSON - read it silently and present a clean summary instead: diff --git a/tools/skillgen/expected/graphify__skill-trae.md b/tools/skillgen/expected/graphify__skill-trae.md index 98deab2..050667b 100644 --- a/tools/skillgen/expected/graphify__skill-trae.md +++ b/tools/skillgen/expected/graphify__skill-trae.md @@ -112,8 +112,11 @@ import json from graphify.detect import detect from pathlib import Path result = detect(Path('INPUT_PATH')) -print(json.dumps(result, ensure_ascii=False)) -" > graphify-out/.graphify_detect.json +# Write the sidecar from Python, not a shell redirect, so the same block renders +# on PowerShell hosts without console-encoding drift (#2528). +Path('graphify-out/.graphify_detect.json').write_text(json.dumps(result, ensure_ascii=False), encoding=\"utf-8\") +print(f'Detected {result[\"total_files\"]} files') +" ``` Replace INPUT_PATH with the actual path the user provided. Do NOT cat or print the JSON - read it silently and present a clean summary instead: diff --git a/tools/skillgen/expected/graphify__skill-vscode.md b/tools/skillgen/expected/graphify__skill-vscode.md index efb2401..20c7c08 100644 --- a/tools/skillgen/expected/graphify__skill-vscode.md +++ b/tools/skillgen/expected/graphify__skill-vscode.md @@ -112,8 +112,11 @@ import json from graphify.detect import detect from pathlib import Path result = detect(Path('INPUT_PATH')) -print(json.dumps(result, ensure_ascii=False)) -" > graphify-out/.graphify_detect.json +# Write the sidecar from Python, not a shell redirect, so the same block renders +# on PowerShell hosts without console-encoding drift (#2528). +Path('graphify-out/.graphify_detect.json').write_text(json.dumps(result, ensure_ascii=False), encoding=\"utf-8\") +print(f'Detected {result[\"total_files\"]} files') +" ``` Replace INPUT_PATH with the actual path the user provided. Do NOT cat or print the JSON - read it silently and present a clean summary instead: diff --git a/tools/skillgen/expected/graphify__skill-windows.md b/tools/skillgen/expected/graphify__skill-windows.md index c48d874..d631821 100644 --- a/tools/skillgen/expected/graphify__skill-windows.md +++ b/tools/skillgen/expected/graphify__skill-windows.md @@ -128,14 +128,17 @@ If the import succeeds, print nothing and move straight to Step 2. ### Step 2 - Detect files -```bash -$(cat graphify-out/.graphify_python) -c " +```powershell +@' import json from graphify.detect import detect from pathlib import Path result = detect(Path('INPUT_PATH')) -print(json.dumps(result, ensure_ascii=False)) -" > graphify-out/.graphify_detect.json +# Write the sidecar from Python, not a shell redirect, so the same block renders +# on PowerShell hosts without console-encoding drift (#2528). +Path('graphify-out/.graphify_detect.json').write_text(json.dumps(result, ensure_ascii=False), encoding="utf-8") +print(f'Detected {result["total_files"]} files') +'@ | & (Get-Content graphify-out\.graphify_python) - ``` Replace INPUT_PATH with the actual path the user provided. Do NOT cat or print the JSON - read it silently and present a clean summary instead: @@ -190,38 +193,38 @@ Note: Parallelizing AST + semantic saves 5-15s on large corpora. AST is determin For any code files detected, run AST extraction in parallel with Part B subagents: -```bash -$(cat graphify-out/.graphify_python) -c " +```powershell +@' import sys, json from graphify.extract import collect_files, extract from pathlib import Path import json code_files = [] -detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\")) +detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding="utf-8")) for f in detect.get('files', {}).get('code', []): code_files.extend(collect_files(Path(f)) if Path(f).is_dir() else [Path(f)]) if code_files: result = extract(code_files, cache_root=Path('INPUT_PATH')) - Path('graphify-out/.graphify_ast.json').write_text(json.dumps(result, indent=2, ensure_ascii=False), encoding=\"utf-8\") - print(f'AST: {len(result[\"nodes\"])} nodes, {len(result[\"edges\"])} edges') + Path('graphify-out/.graphify_ast.json').write_text(json.dumps(result, indent=2, ensure_ascii=False), encoding="utf-8") + print(f'AST: {len(result["nodes"])} nodes, {len(result["edges"])} edges') else: - Path('graphify-out/.graphify_ast.json').write_text(json.dumps({'nodes':[],'edges':[],'input_tokens':0,'output_tokens':0}, ensure_ascii=False), encoding=\"utf-8\") + Path('graphify-out/.graphify_ast.json').write_text(json.dumps({'nodes':[],'edges':[],'input_tokens':0,'output_tokens':0}, ensure_ascii=False), encoding="utf-8") print('No code files - skipping AST extraction') -" +'@ | & (Get-Content graphify-out\.graphify_python) - ``` #### Part B - Semantic extraction (parallel subagents) **Fast path:** If detection found zero docs, papers, and images (code-only corpus), skip Part B entirely and go straight to Part C. AST handles code - there is nothing for semantic subagents to do. **First write an empty semantic file** so Part C's merge has its input (it reads `.graphify_semantic.json` unconditionally; without this a code-only run hits `FileNotFoundError`): -```bash -$(cat graphify-out/.graphify_python) -c " +```powershell +@' import json from pathlib import Path Path('graphify-out/.graphify_semantic.json').write_text(json.dumps({'nodes':[],'edges':[],'hyperedges':[],'input_tokens':0,'output_tokens':0}), encoding='utf-8') -" +'@ | & (Get-Content graphify-out\.graphify_python) - ``` **MANDATORY: You MUST use the Agent tool here. Reading files yourself one-by-one is forbidden - it is 5-10x slower. If you do not use the Agent tool you are doing this wrong.** @@ -238,13 +241,13 @@ Before dispatching any subagents, check which files already have cached extracti SPEC_PATH below is the **absolute** path of the `references/extraction-spec.md` that ships beside this SKILL.md — the same file Step B2 loads and hands to every subagent. It is the extraction prompt, so cache entries are attributed to it: when a graphify upgrade changes the prompt, entries produced by the old one are re-extracted instead of replayed, and unchanged prompts keep their entries (#1939). Substitute the real path in both Step B0 and Step B3 — pass the same one to each, and do not drop the argument. -```bash -$(cat graphify-out/.graphify_python) -c " +```powershell +@' import json from graphify.cache import check_semantic_cache from pathlib import Path -detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\")) +detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding="utf-8")) # Only content files go to semantic extraction. Code is already covered structurally # by the AST pass (Part A); flattening every category here makes subagents re-read # every source file (#1392). Video is transcribed to a document in Step 2.5 first. @@ -255,12 +258,12 @@ cached_nodes, cached_edges, cached_hyperedges, uncached = check_semantic_cache(a # Always (re)write the cache file: write hits, else DELETE any leftover from a prior # run so Part C never merges a stale .graphify_cached.json (#1392). if cached_nodes or cached_edges or cached_hyperedges: - Path('graphify-out/.graphify_cached.json').write_text(json.dumps({'nodes': cached_nodes, 'edges': cached_edges, 'hyperedges': cached_hyperedges}, ensure_ascii=False), encoding=\"utf-8\") + Path('graphify-out/.graphify_cached.json').write_text(json.dumps({'nodes': cached_nodes, 'edges': cached_edges, 'hyperedges': cached_hyperedges}, ensure_ascii=False), encoding="utf-8") else: Path('graphify-out/.graphify_cached.json').unlink(missing_ok=True) -Path('graphify-out/.graphify_uncached.txt').write_text('\n'.join(uncached), encoding=\"utf-8\") +Path('graphify-out/.graphify_uncached.txt').write_text('\n'.join(uncached), encoding="utf-8") print(f'Cache: {len(all_files)-len(uncached)} files hit, {len(uncached)} files need extraction') -" +'@ | & (Get-Content graphify-out\.graphify_python) - ``` Only dispatch subagents for files listed in `graphify-out/.graphify_uncached.txt`. If all files are cached, skip to Part C directly. @@ -306,8 +309,8 @@ Wait for all subagents. For each result: If more than half the chunks failed or are missing, stop and tell the user to re-run and ensure `subagent_type="general-purpose"` is used. Merge all chunk files into `.graphify_semantic_new.json`. **After each Agent call completes, read the real token counts from the Agent tool result's `usage` field and write them back into the chunk JSON before merging** — the chunk JSON itself always has placeholder zeros. Then run: -```bash -$(cat graphify-out/.graphify_python) -c " +```powershell +@' import json, glob from pathlib import Path @@ -315,7 +318,7 @@ chunks = sorted(glob.glob('graphify-out/.graphify_chunk_*.json')) all_nodes, all_edges, all_hyperedges = [], [], [] total_in, total_out = 0, 0 for c in chunks: - d = json.loads(Path(c).read_text(encoding=\"utf-8\")) + d = json.loads(Path(c).read_text(encoding="utf-8")) all_nodes += d.get('nodes', []) all_edges += d.get('edges', []) all_hyperedges += d.get('hyperedges', []) @@ -324,33 +327,33 @@ for c in chunks: Path('graphify-out/.graphify_semantic_new.json').write_text(json.dumps({ 'nodes': all_nodes, 'edges': all_edges, 'hyperedges': all_hyperedges, 'input_tokens': total_in, 'output_tokens': total_out, -}, indent=2, ensure_ascii=False), encoding=\"utf-8\") +}, indent=2, ensure_ascii=False), encoding="utf-8") print(f'Merged {len(chunks)} chunks: {total_in:,} in / {total_out:,} out tokens') -" +'@ | & (Get-Content graphify-out\.graphify_python) - ``` Save new results to cache. Pass the same SPEC_PATH as Step B0 — it stamps each entry with the prompt that produced it, and a write under a different prompt than the read lands where the next run won't look (#1939): -```bash -$(cat graphify-out/.graphify_python) -c " +```powershell +@' import json from graphify.cache import save_semantic_cache from pathlib import Path -new = json.loads(Path('graphify-out/.graphify_semantic_new.json').read_text(encoding=\"utf-8\")) if Path('graphify-out/.graphify_semantic_new.json').exists() else {'nodes':[],'edges':[],'hyperedges':[]} -uncached = [line for line in Path('graphify-out/.graphify_uncached.txt').read_text(encoding=\"utf-8\").splitlines() if line] +new = json.loads(Path('graphify-out/.graphify_semantic_new.json').read_text(encoding="utf-8")) if Path('graphify-out/.graphify_semantic_new.json').exists() else {'nodes':[],'edges':[],'hyperedges':[]} +uncached = [line for line in Path('graphify-out/.graphify_uncached.txt').read_text(encoding="utf-8").splitlines() if line] saved = save_semantic_cache(new.get('nodes', []), new.get('edges', []), new.get('hyperedges', []), root='INPUT_PATH', allowed_source_files=uncached, prompt_file='SPEC_PATH') print(f'Cached {saved} files') -" +'@ | & (Get-Content graphify-out\.graphify_python) - ``` Merge cached + new results into `graphify-out/.graphify_semantic.json`: -```bash -$(cat graphify-out/.graphify_python) -c " +```powershell +@' import json from pathlib import Path -cached = json.loads(Path('graphify-out/.graphify_cached.json').read_text(encoding=\"utf-8\")) if Path('graphify-out/.graphify_cached.json').exists() else {'nodes':[],'edges':[],'hyperedges':[]} -new = json.loads(Path('graphify-out/.graphify_semantic_new.json').read_text(encoding=\"utf-8\")) if Path('graphify-out/.graphify_semantic_new.json').exists() else {'nodes':[],'edges':[],'hyperedges':[]} +cached = json.loads(Path('graphify-out/.graphify_cached.json').read_text(encoding="utf-8")) if Path('graphify-out/.graphify_cached.json').exists() else {'nodes':[],'edges':[],'hyperedges':[]} +new = json.loads(Path('graphify-out/.graphify_semantic_new.json').read_text(encoding="utf-8")) if Path('graphify-out/.graphify_semantic_new.json').exists() else {'nodes':[],'edges':[],'hyperedges':[]} all_nodes = cached['nodes'] + new.get('nodes', []) all_edges = cached['edges'] + new.get('edges', []) @@ -369,21 +372,21 @@ merged = { 'input_tokens': new.get('input_tokens', 0), 'output_tokens': new.get('output_tokens', 0), } -Path('graphify-out/.graphify_semantic.json').write_text(json.dumps(merged, indent=2, ensure_ascii=False), encoding=\"utf-8\") -print(f'Extraction complete - {len(deduped)} nodes, {len(all_edges)} edges ({len(cached[\"nodes\"])} from cache, {len(new.get(\"nodes\",[]))} new)') -" +Path('graphify-out/.graphify_semantic.json').write_text(json.dumps(merged, indent=2, ensure_ascii=False), encoding="utf-8") +print(f'Extraction complete - {len(deduped)} nodes, {len(all_edges)} edges ({len(cached["nodes"])} from cache, {len(new.get("nodes",[]))} new)') +'@ | & (Get-Content graphify-out\.graphify_python) - ``` -Clean up temp files: `rm -f graphify-out/.graphify_cached.json graphify-out/.graphify_uncached.txt graphify-out/.graphify_semantic_new.json` +Clean up temp files: `Remove-Item -Force -ErrorAction SilentlyContinue graphify-out\.graphify_cached.json, graphify-out\.graphify_uncached.txt, graphify-out\.graphify_semantic_new.json` #### Part C - Merge AST + semantic into final extraction -```bash -$(cat graphify-out/.graphify_python) -c " +```powershell +@' import sys, json from pathlib import Path -ast = json.loads(Path('graphify-out/.graphify_ast.json').read_text(encoding=\"utf-8\")) -sem = json.loads(Path('graphify-out/.graphify_semantic.json').read_text(encoding=\"utf-8\")) +ast = json.loads(Path('graphify-out/.graphify_ast.json').read_text(encoding="utf-8")) +sem = json.loads(Path('graphify-out/.graphify_semantic.json').read_text(encoding="utf-8")) # Merge: AST nodes first, semantic nodes deduplicated by id seen = {n['id'] for n in ast['nodes']} @@ -402,20 +405,20 @@ merged = { 'input_tokens': sem.get('input_tokens', 0), 'output_tokens': sem.get('output_tokens', 0), } -Path('graphify-out/.graphify_extract.json').write_text(json.dumps(merged, indent=2, ensure_ascii=False), encoding=\"utf-8\") +Path('graphify-out/.graphify_extract.json').write_text(json.dumps(merged, indent=2, ensure_ascii=False), encoding="utf-8") total = len(merged_nodes) edges = len(merged_edges) -print(f'Merged: {total} nodes, {edges} edges ({len(ast[\"nodes\"])} AST + {len(sem[\"nodes\"])} semantic)') -" +print(f'Merged: {total} nodes, {edges} edges ({len(ast["nodes"])} AST + {len(sem["nodes"])} semantic)') +'@ | & (Get-Content graphify-out\.graphify_python) - ``` ### Step 4 - Build graph, cluster, analyze, generate outputs **Before starting:** the code blocks below pass `directed=IS_DIRECTED` to `build_from_json()`. Replace `IS_DIRECTED` with `True` if `--directed` was given (builds a `DiGraph` preserving edge direction source→target), otherwise `False` (the default undirected `Graph`). Substitute it the same way you substitute `INPUT_PATH` — do not leave the literal `IS_DIRECTED` in the code. -```bash -mkdir -p graphify-out -$(cat graphify-out/.graphify_python) -c " +```powershell +New-Item -ItemType Directory -Force -Path graphify-out | Out-Null +@' import sys, json from graphify.build import build_from_json from graphify.cluster import cluster, score_all @@ -424,8 +427,8 @@ from graphify.report import generate from graphify.export import to_json from pathlib import Path -extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) -detection = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\")) +extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding="utf-8")) +detection = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding="utf-8")) # root= mirrors the --update runbook (#1361): relativize source_file to the same # base so the full build and incremental --update never drift apart on re-extract. @@ -455,7 +458,7 @@ if not wrote: print('If this shrink is intentional (you deleted files), re-run a full build with --force.') raise SystemExit(1) report = generate(G, communities, cohesion, labels, gods, surprises, detection, tokens, 'INPUT_PATH', suggested_questions=questions) -Path('graphify-out/GRAPH_REPORT.md').write_text(report, encoding=\"utf-8\") +Path('graphify-out/GRAPH_REPORT.md').write_text(report, encoding="utf-8") analysis = { 'communities': {str(k): v for k, v in communities.items()}, 'cohesion': {str(k): v for k, v in cohesion.items()}, @@ -463,9 +466,9 @@ analysis = { 'surprises': surprises, 'questions': questions, } -Path('graphify-out/.graphify_analysis.json').write_text(json.dumps(analysis, indent=2, ensure_ascii=False), encoding=\"utf-8\") +Path('graphify-out/.graphify_analysis.json').write_text(json.dumps(analysis, indent=2, ensure_ascii=False), encoding="utf-8") print(f'Graph: {G.number_of_nodes()} nodes, {G.number_of_edges()} edges, {len(communities)} communities') -" +'@ | & (Get-Content graphify-out\.graphify_python) - ``` If this step prints `ERROR: Graph is empty`, stop and tell the user what happened - do not proceed to labeling or visualization. @@ -476,13 +479,13 @@ Replace INPUT_PATH with the actual path. A non-destructive diagnostic on the extraction, before labeling. It surfaces edge collapse, dangling/missing endpoints, and self-loops — the silent-corruption modes of incremental updates and AST/LLM id mismatches. Read-only; never aborts. -```bash -$(cat graphify-out/.graphify_python) -c " +```powershell +@' import json from pathlib import Path from graphify.diagnostics import diagnose_extraction, format_diagnostic_report -extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) +extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding="utf-8")) summary = diagnose_extraction(extraction, directed=IS_DIRECTED, root='INPUT_PATH') print(format_diagnostic_report(summary)) flags = [f'{summary[k]} {label}' for k, label in ( @@ -493,7 +496,7 @@ flags = [f'{summary[k]} {label}' for k, label in ( ('undirected_same_endpoint_collapsed_edges', 'collapsed (undirected) edges'), ) if summary.get(k, 0)] print('GRAPH HEALTH WARNING: ' + '; '.join(flags) + ' - graph may be incomplete/corrupt.' if flags else 'Graph health: OK (no dangling/missing/collapsed edges).') -" +'@ | & (Get-Content graphify-out\.graphify_python) - ``` Substitute `IS_DIRECTED` and `INPUT_PATH` as in Step 4. If a `GRAPH HEALTH WARNING` prints, surface it in the final summary (do not abort — the graph is still usable, but the integrity issue must be visible, per the Honesty Rules). @@ -504,8 +507,8 @@ Read `graphify-out/.graphify_analysis.json`. For each community key, look at its Then regenerate the report and save the labels for the visualizer: -```bash -$(cat graphify-out/.graphify_python) -c " +```powershell +@' import sys, json from graphify.build import build_from_json from graphify.cluster import score_all @@ -514,9 +517,9 @@ from graphify.report import generate from graphify.export import to_json from pathlib import Path -extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) -detection = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\")) -analysis = json.loads(Path('graphify-out/.graphify_analysis.json').read_text(encoding=\"utf-8\")) +extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding="utf-8")) +detection = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding="utf-8")) +analysis = json.loads(Path('graphify-out/.graphify_analysis.json').read_text(encoding="utf-8")) # root= as in Step 4 / the --update runbook (#1361) — same base for node-key parity. G = build_from_json(extraction, root='INPUT_PATH', directed=IS_DIRECTED) @@ -531,8 +534,8 @@ labels = LABELS_DICT questions = suggest_questions(G, communities, labels) report = generate(G, communities, cohesion, labels, analysis['gods'], analysis['surprises'], detection, tokens, 'INPUT_PATH', suggested_questions=questions) -Path('graphify-out/GRAPH_REPORT.md').write_text(report, encoding=\"utf-8\") -Path('graphify-out/.graphify_labels.json').write_text(json.dumps({str(k): v for k, v in labels.items()}, ensure_ascii=False), encoding=\"utf-8\") +Path('graphify-out/GRAPH_REPORT.md').write_text(report, encoding="utf-8") +Path('graphify-out/.graphify_labels.json').write_text(json.dumps({str(k): v for k, v in labels.items()}, ensure_ascii=False), encoding="utf-8") # Re-export so graph.json nodes carry the curated community_name (#2490). # Same extraction as Step 4, so the #479 shrink-guard passes on node count; # if it still refuses, surface the guard message - do not force past it. @@ -541,7 +544,7 @@ if not wrote: print('ERROR: refused to shrink graphify-out/graph.json (existing graph has more nodes; #479).') print('If this shrink is intentional (you deleted files), re-run a full build with --force.') print('Report updated with community labels') -" +'@ | & (Get-Content graphify-out\.graphify_python) - ``` Replace `LABELS_DICT` with the actual dict you constructed (e.g. `{0: "Attention Mechanism", 1: "Training Pipeline"}`). @@ -555,14 +558,14 @@ If `--obsidian` was given: - If `--obsidian-dir ` was also given, pass it via `--dir`. Otherwise defaults to `graphify-out/obsidian`. -```bash +```powershell graphify export obsidian # or with custom dir: graphify export obsidian --dir ~/vaults/my-project ``` Generate the HTML graph (always, unless `--no-viz`): -```bash +```powershell graphify export html # auto-aggregates to community view if graph > 5000 nodes # or: graphify export html --no-viz ``` @@ -575,16 +578,16 @@ These run only when their flag is present (`--wiki`, `--neo4j`/`--neo4j-push`, ` ### Step 9 - Save manifest, update cost tracker, clean up, and report -```bash -$(cat graphify-out/.graphify_python) -c " +```powershell +@' import json from pathlib import Path from datetime import datetime, timezone from graphify.detect import save_manifest # Save manifest for --update -detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\")) -extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) +detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding="utf-8")) +extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding="utf-8")) # In --update mode, 'all_files' carries the full corpus; 'files' is the changed # subset. Full-rebuild mode populates only 'files', so the fallback handles that. # root= relativizes the manifest keys to the scan root (same base as the build), @@ -620,7 +623,7 @@ output_tok = extract.get('output_tokens', 0) cost_path = Path('graphify-out/cost.json') if cost_path.exists(): - cost = json.loads(cost_path.read_text(encoding=\"utf-8\")) + cost = json.loads(cost_path.read_text(encoding="utf-8")) else: cost = {'runs': [], 'total_input_tokens': 0, 'total_output_tokens': 0} @@ -632,14 +635,14 @@ cost['runs'].append({ }) cost['total_input_tokens'] += input_tok cost['total_output_tokens'] += output_tok -cost_path.write_text(json.dumps(cost, indent=2, ensure_ascii=False), encoding=\"utf-8\") +cost_path.write_text(json.dumps(cost, indent=2, ensure_ascii=False), encoding="utf-8") print(f'This run: {input_tok:,} input tokens, {output_tok:,} output tokens') -print(f'All time: {cost[\"total_input_tokens\"]:,} input, {cost[\"total_output_tokens\"]:,} output ({len(cost[\"runs\"])} runs)') -" -rm -f graphify-out/.graphify_detect.json graphify-out/.graphify_extract.json graphify-out/.graphify_ast.json graphify-out/.graphify_semantic.json graphify-out/.graphify_analysis.json -find graphify-out -maxdepth 1 -name '.graphify_chunk_*.json' -delete 2>/dev/null -rm -f graphify-out/.needs_update 2>/dev/null || true +print(f'All time: {cost["total_input_tokens"]:,} input, {cost["total_output_tokens"]:,} output ({len(cost["runs"])} runs)') +'@ | & (Get-Content graphify-out\.graphify_python) - +Remove-Item -Force -ErrorAction SilentlyContinue graphify-out\.graphify_detect.json, graphify-out\.graphify_extract.json, graphify-out\.graphify_ast.json, graphify-out\.graphify_semantic.json, graphify-out\.graphify_analysis.json +Get-ChildItem graphify-out -Filter '.graphify_chunk_*.json' -File -ErrorAction SilentlyContinue | Remove-Item -Force +Remove-Item -Force -ErrorAction SilentlyContinue graphify-out\.needs_update ``` Replace INPUT_PATH with the actual path (same value used in Steps 4-5) so the manifest is relativized to the scan root. @@ -679,18 +682,20 @@ The graph is the map. Your job after the pipeline is to be the guide. Before running any subcommand below (`--update`, `--cluster-only`, `query`, `path`, `explain`, `add`), check that `.graphify_python` exists. If it's missing (e.g. user deleted `graphify-out/`), re-resolve the interpreter first: -```bash -if [ ! -f graphify-out/.graphify_python ]; then - GRAPHIFY_BIN=$(which graphify 2>/dev/null) - if [ -n "$GRAPHIFY_BIN" ]; then - PYTHON=$(head -1 "$GRAPHIFY_BIN" | tr -d '#!') - case "$PYTHON" in *[!a-zA-Z0-9/_.@-]*) PYTHON="python3" ;; esac - else - PYTHON="python3" - fi - mkdir -p graphify-out - "$PYTHON" -c "import sys; open('graphify-out/.graphify_python', 'w', encoding='utf-8').write(sys.executable)" -fi +```powershell +if (-not (Test-Path graphify-out\.graphify_python)) { + $GRAPHIFY_PYTHON = $null + $graphifyCmd = Get-Command graphify -ErrorAction SilentlyContinue + if ($graphifyCmd) { + # The interpreter that owns the graphify entry point sits next to it + # (\Scripts\python.exe for uv tool, pipx, and venv installs). + $py = Join-Path (Split-Path $graphifyCmd.Source) "python.exe" + if (Test-Path $py) { $GRAPHIFY_PYTHON = $py } + } + if (-not $GRAPHIFY_PYTHON) { $GRAPHIFY_PYTHON = "python" } + New-Item -ItemType Directory -Force -Path graphify-out | Out-Null + & $GRAPHIFY_PYTHON -c "import sys; open('graphify-out/.graphify_python', 'w', encoding='utf-8').write(sys.executable)" +} ``` ## For --update and --cluster-only @@ -703,7 +708,7 @@ Both are non-default subcommands. `--update` re-extracts only new or changed fil When `graphify-out/graph.json` already exists and the user asks a question about the corpus, answer from the graph rather than rebuilding it: -```bash +```powershell graphify query "" ``` diff --git a/tools/skillgen/expected/graphify__skill.md b/tools/skillgen/expected/graphify__skill.md index cef4d5c..abd2811 100644 --- a/tools/skillgen/expected/graphify__skill.md +++ b/tools/skillgen/expected/graphify__skill.md @@ -112,8 +112,11 @@ import json from graphify.detect import detect from pathlib import Path result = detect(Path('INPUT_PATH')) -print(json.dumps(result, ensure_ascii=False)) -" > graphify-out/.graphify_detect.json +# Write the sidecar from Python, not a shell redirect, so the same block renders +# on PowerShell hosts without console-encoding drift (#2528). +Path('graphify-out/.graphify_detect.json').write_text(json.dumps(result, ensure_ascii=False), encoding=\"utf-8\") +print(f'Detected {result[\"total_files\"]} files') +" ``` Replace INPUT_PATH with the actual path the user provided. Do NOT cat or print the JSON - read it silently and present a clean summary instead: diff --git a/tools/skillgen/fragments/core/core.md b/tools/skillgen/fragments/core/core.md index e9a4c1b..c527a12 100644 --- a/tools/skillgen/fragments/core/core.md +++ b/tools/skillgen/fragments/core/core.md @@ -71,8 +71,11 @@ import json from graphify.detect import detect from pathlib import Path result = detect(Path('INPUT_PATH')) -print(json.dumps(result, ensure_ascii=False)) -" > graphify-out/.graphify_detect.json +# Write the sidecar from Python, not a shell redirect, so the same block renders +# on PowerShell hosts without console-encoding drift (#2528). +Path('graphify-out/.graphify_detect.json').write_text(json.dumps(result, ensure_ascii=False), encoding=\"utf-8\") +print(f'Detected {result[\"total_files\"]} files') +" ``` Replace INPUT_PATH with the actual path the user provided. Do NOT cat or print the JSON - read it silently and present a clean summary instead: @@ -592,19 +595,7 @@ The graph is the map. Your job after the pipeline is to be the guide. Before running any subcommand below (`--update`, `--cluster-only`, `query`, `path`, `explain`, `add`), check that `.graphify_python` exists. If it's missing (e.g. user deleted `graphify-out/`), re-resolve the interpreter first: -```bash -if [ ! -f graphify-out/.graphify_python ]; then - GRAPHIFY_BIN=$(which graphify 2>/dev/null) - if [ -n "$GRAPHIFY_BIN" ]; then - PYTHON=$(head -1 "$GRAPHIFY_BIN" | tr -d '#!') - case "$PYTHON" in *[!a-zA-Z0-9/_.@-]*) PYTHON="python3" ;; esac - else - PYTHON="python3" - fi - mkdir -p graphify-out - "$PYTHON" -c "import sys; open('graphify-out/.graphify_python', 'w', encoding='utf-8').write(sys.executable)" -fi -``` +@@INTERP_GUARD@@ ## For --update and --cluster-only diff --git a/tools/skillgen/fragments/shell/interpreter-guard-posix.md b/tools/skillgen/fragments/shell/interpreter-guard-posix.md new file mode 100644 index 0000000..f7e3448 --- /dev/null +++ b/tools/skillgen/fragments/shell/interpreter-guard-posix.md @@ -0,0 +1,13 @@ +```bash +if [ ! -f graphify-out/.graphify_python ]; then + GRAPHIFY_BIN=$(which graphify 2>/dev/null) + if [ -n "$GRAPHIFY_BIN" ]; then + PYTHON=$(head -1 "$GRAPHIFY_BIN" | tr -d '#!') + case "$PYTHON" in *[!a-zA-Z0-9/_.@-]*) PYTHON="python3" ;; esac + else + PYTHON="python3" + fi + mkdir -p graphify-out + "$PYTHON" -c "import sys; open('graphify-out/.graphify_python', 'w', encoding='utf-8').write(sys.executable)" +fi +``` diff --git a/tools/skillgen/fragments/shell/interpreter-guard-powershell.md b/tools/skillgen/fragments/shell/interpreter-guard-powershell.md new file mode 100644 index 0000000..dbd80f6 --- /dev/null +++ b/tools/skillgen/fragments/shell/interpreter-guard-powershell.md @@ -0,0 +1,15 @@ +```powershell +if (-not (Test-Path graphify-out\.graphify_python)) { + $GRAPHIFY_PYTHON = $null + $graphifyCmd = Get-Command graphify -ErrorAction SilentlyContinue + if ($graphifyCmd) { + # The interpreter that owns the graphify entry point sits next to it + # (\Scripts\python.exe for uv tool, pipx, and venv installs). + $py = Join-Path (Split-Path $graphifyCmd.Source) "python.exe" + if (Test-Path $py) { $GRAPHIFY_PYTHON = $py } + } + if (-not $GRAPHIFY_PYTHON) { $GRAPHIFY_PYTHON = "python" } + New-Item -ItemType Directory -Force -Path graphify-out | Out-Null + & $GRAPHIFY_PYTHON -c "import sys; open('graphify-out/.graphify_python', 'w', encoding='utf-8').write(sys.executable)" +} +``` diff --git a/tools/skillgen/gen.py b/tools/skillgen/gen.py index 4141a02..09e19ed 100644 --- a/tools/skillgen/gen.py +++ b/tools/skillgen/gen.py @@ -357,14 +357,174 @@ def _render_frontmatter(platform: Platform) -> str: return "\n".join(lines) +# --- POSIX -> PowerShell core translation (#2528) -------------------------------- +# +# The lean core template is written once, in POSIX shell. A host that declares +# ``shell = "powershell"`` (windows) used to get a PowerShell Step 1 but bash for +# every later step (``$(cat ...) -c "..."``, ``rm -f``, ``find ... -delete``), so +# on a strict-PowerShell host Steps 2+ failed. Rather than fork the core (which +# would drift), the composed body is translated deterministically: +# +# * ```` ```bash ```` fences become ```` ```powershell ```` fences; +# * the inline ``$(cat graphify-out/.graphify_python) -c "..."`` python blocks +# become single-quoted here-strings piped to the interpreter's stdin +# (``@'...'@ | & (Get-Content graphify-out\.graphify_python) -``). A +# single-quoted here-string is verbatim, so the bash ``\"`` escapes are +# dropped and no PowerShell escaping is introduced; piping the program to +# stdin (``python -``) sidesteps Windows PowerShell 5.1's native-argument +# quote mangling that would corrupt ``-c @'...'@``; +# * the ``rm -f`` / ``find ... -delete`` cleanup lines become ``Remove-Item`` +# / ``Get-ChildItem | Remove-Item``, and ``mkdir -p`` becomes ``New-Item``. +# +# The translator is deliberately STRICT: any bash line it does not recognize +# raises at render time, so a future core edit cannot silently ship +# untranslated bash to the Windows variant. ``_POWERSHELL_BANNED_TOKENS`` is the +# belt-and-braces post-check on the final body. + +_PY_INVOKE_POSIX = '$(cat graphify-out/.graphify_python) -c "' +_PY_INVOKE_PS_OPEN = "@'" +_PY_INVOKE_PS_CLOSE = "'@ | & (Get-Content graphify-out\\.graphify_python) -" +_MKDIR_POSIX = "mkdir -p graphify-out" +_MKDIR_PS = "New-Item -ItemType Directory -Force -Path graphify-out | Out-Null" +_FIND_CHUNKS_POSIX = "find graphify-out -maxdepth 1 -name '.graphify_chunk_*.json' -delete 2>/dev/null" +_FIND_CHUNKS_PS = ( + "Get-ChildItem graphify-out -Filter '.graphify_chunk_*.json' -File " + "-ErrorAction SilentlyContinue | Remove-Item -Force" +) + +# Bash-only tokens that must never survive in a powershell-shell render. +_POWERSHELL_BANNED_TOKENS = ("$(cat ", "rm -f ", "2>/dev/null", "```bash") + + +def _unescape_bash_dq(line: str) -> str: + """Turn one line of a bash ``-c "..."`` body into its verbatim (here-string) form. + + Inside bash double quotes every literal ``"`` is written ``\\"``; a + single-quoted here-string is verbatim, so those escapes are dropped. Anything + else backslashed (except a Python ``'\\n'`` string literal, the only other + backslash the core bodies carry) is unexpected bash escaping and fails loudly, + as does a line that would terminate the here-string early. + """ + body = line.replace('\\"', '"') + if body.lstrip().startswith("'@"): + raise ValueError(f"python body line would close the here-string: {line!r}") + for m in re.finditer(r"\\(.)", body): + if m.group(1) != "n": + raise ValueError(f"unexpected backslash escape in python body line: {line!r}") + return body + + +def _rm_to_remove_item(cmd: str) -> str: + """Translate a ``rm -f FILE...`` line (with optional ``2>/dev/null [|| true]``).""" + rest = cmd.strip() + if not rest.startswith("rm -f "): + raise ValueError(f"not an rm -f command: {cmd!r}") + rest = rest[len("rm -f "):].replace("2>/dev/null", " ").replace("|| true", " ") + files = rest.split() + for f in files: + if not re.fullmatch(r"[\w./-]+", f): + raise ValueError(f"cannot translate rm -f operand to PowerShell: {f!r}") + joined = ", ".join(f.replace("/", "\\") for f in files) + return f"Remove-Item -Force -ErrorAction SilentlyContinue {joined}" + + +def _translate_bash_block(lines: list[str]) -> list[str]: + """Translate the body of one ```` ```bash ```` fence to PowerShell.""" + out: list[str] = [] + in_py = False + for line in lines: + if in_py: + if line == '"': + out.append(_PY_INVOKE_PS_CLOSE) + in_py = False + else: + out.append(_unescape_bash_dq(line)) + elif line == _PY_INVOKE_POSIX: + out.append(_PY_INVOKE_PS_OPEN) + in_py = True + elif line == _MKDIR_POSIX: + out.append(_MKDIR_PS) + elif line == _FIND_CHUNKS_POSIX: + out.append(_FIND_CHUNKS_PS) + elif line.strip().startswith("rm -f "): + out.append(_rm_to_remove_item(line)) + elif not line.strip() or line.lstrip().startswith("#") or line.startswith("graphify "): + out.append(line) # blank lines, comments, and graphify CLI calls are shell-neutral + else: + raise ValueError(f"cannot translate bash line to PowerShell: {line!r}") + if in_py: + raise ValueError("unterminated python -c body in bash fence") + return out + + +def _translate_prose_line(line: str) -> str: + """Translate inline `` `rm -f ...` `` code spans in prose to Remove-Item.""" + return re.sub( + r"`rm -f ([^`]+)`", + lambda m: "`" + _rm_to_remove_item("rm -f " + m.group(1)) + "`", + line, + ) + + +def _core_to_powershell(body: str) -> str: + """Render the composed POSIX core body as strict PowerShell (#2528).""" + lines = body.split("\n") + out: list[str] = [] + i, n = 0, len(lines) + while i < n: + line = lines[i] + stripped = line.strip() + if stripped.startswith("```") and stripped != "```": + # Opening fence with an info string. Find its closing fence. + j = i + 1 + while j < n and lines[j].strip() != "```": + j += 1 + if j >= n: + raise ValueError(f"unterminated code fence: {line!r}") + if stripped == "```bash": + out.append(line.replace("```bash", "```powershell")) + out.extend(_translate_bash_block(lines[i + 1:j])) + else: + # Non-bash fences (```powershell, plain ```` ``` ````) pass through. + out.append(line) + out.extend(lines[i + 1:j]) + out.append(lines[j]) + i = j + 1 + elif stripped == "```": + # Bare opening fence (Usage / summary blocks): opaque passthrough. + j = i + 1 + while j < n and lines[j].strip() != "```": + j += 1 + if j >= n: + raise ValueError("unterminated bare code fence") + out.extend(lines[i:j + 1]) + i = j + 1 + else: + out.append(_translate_prose_line(line)) + i += 1 + translated = "\n".join(out) + leftovers = [t for t in _POWERSHELL_BANNED_TOKENS if t in translated] + if re.search(r"\bfind\b[^\n]*-delete", translated): + leftovers.append("find ... -delete") + if leftovers: + raise ValueError(f"bash-only tokens survived the PowerShell render: {leftovers}") + return translated + + def _render_core(platform: Platform) -> str: - """Fill the shared core template's per-platform slots for this platform.""" + """Fill the shared core template's per-platform slots for this platform. + + The template is authored once in POSIX shell; a ``shell = "powershell"`` + platform gets the same composed body translated to strict PowerShell (see + ``_core_to_powershell``), so the two variants cannot drift apart. + """ template = _read_fragment(f"core/{platform.core}.md") if platform.dispatch is None: raise ValueError(f"split platform '{platform.key}' is missing a dispatch variant") install = _read_fragment(f"shell/{platform.shell}.md").rstrip("\n") + interp_guard = _read_fragment(f"shell/interpreter-guard-{platform.shell}.md").rstrip("\n") dispatch = _read_fragment(f"dispatch/{platform.dispatch}.md").rstrip("\n") query_stub = _read_fragment(_QUERY_STUB).rstrip("\n") @@ -379,6 +539,7 @@ def _render_core(platform: Platform) -> str: body = ( template.replace("@@FRONTMATTER@@", _render_frontmatter(platform)) .replace("@@INSTALL@@", install) + .replace("@@INTERP_GUARD@@", interp_guard) .replace("@@DISPATCH@@", dispatch) .replace("@@QUERY_STUB@@", query_stub) .replace("@@HOOKS_TARGET@@", platform.hooks_target) @@ -387,6 +548,8 @@ def _render_core(platform: Platform) -> str: if "@@" in body: leftover = sorted(set(re.findall(r"@@\w+@@", body))) raise ValueError(f"unfilled core slots for '{platform.key}': {leftover}") + if platform.shell == "powershell": + body = _core_to_powershell(body) return _normalise(body)