fix(windows): unblock pipeline on Windows consoles + missing __main__ guards (#788)

Three independent Windows compatibility fixes shipped together because they
all surface during the same first /graphify run on Windows.

graphify/benchmark.py
  print_benchmark() unconditionally printed U+2500 (box-drawing) and U+2192
  (rightwards arrow), which UnicodeEncodeError'd on stdouts that can't encode
  them — most notably the legacy Windows console at cp1252. New _safe()
  helper falls back to ASCII when the active stdout encoding can't carry the
  glyph; _hr() uses it. Two regression tests cover both paths and prove
  print_benchmark survives a cp1252-strict stream.

graphify/extract.py
  ProcessPoolExecutor on Windows uses spawn, so worker subprocesses
  re-import the calling __main__. When the caller is `python -c "..."` or a
  script without an `if __name__ == "__main__":` guard, the workers
  recursively spawn themselves and the pool dies. The user-visible failure
  was a 290-line traceback ending in BrokenProcessPool, hiding the actual
  cause. _extract_parallel now catches BrokenProcessPool, prints a one-line
  warning that names the __main__-guard idiom, and returns False so the
  public extract() routes to the existing _extract_sequential fallback. Two
  tests cover the parallel-returns-False contract and the sequential
  fallback wiring.

graphify/skill-windows.md
  Every `python -c "..."` block (30 in total) is replaced with a
  Write+run+delete pattern using PowerShell's literal here-string @'...'@.
  The old form was a quote-escaping minefield: any double-quote inside the
  Python source had to be backslash-escaped for the shell, and PowerShell's
  parser ate them inconsistently — failing on f-strings like
  `f'AST: {len(result["nodes"])} nodes'`. The new form passes Python source
  to disk literally, so what the model writes is what Python sees. The AST
  step's script template now includes an explicit `if __name__ == "__main__":`
  guard so multi-core extraction works even before the runtime fallback above
  kicks in. All 31 resulting heredoc blocks parse cleanly under
  `ast.parse`.

Co-authored-by: Nauman Hameed <Nauman.Hameed@enghouse.com>
This commit is contained in:
nauman73
2026-05-09 12:59:38 +01:00
committed by GitHub
co-authored by Nauman Hameed
parent f88567b114
commit e5f263ba98
5 changed files with 333 additions and 107 deletions
+23 -2
View File
@@ -1,6 +1,7 @@
"""Token-reduction benchmark - measures how much context graphify saves vs naive full-corpus approach."""
from __future__ import annotations
import json
import sys
from pathlib import Path
import networkx as nx
from networkx.readwrite import json_graph
@@ -9,6 +10,25 @@ from networkx.readwrite import json_graph
_CHARS_PER_TOKEN = 4 # standard approximation
def _safe(unicode_char: str, ascii_fallback: str) -> str:
"""Return unicode_char if stdout can encode it, else ascii_fallback.
Windows consoles often default to cp1252 which cannot encode box-drawing
or arrow glyphs; printing them raises UnicodeEncodeError mid-output.
"""
encoding = getattr(sys.stdout, "encoding", None) or ""
try:
unicode_char.encode(encoding)
return unicode_char
except (UnicodeEncodeError, LookupError):
return ascii_fallback
def _hr(width: int = 50) -> str:
"""Horizontal rule that survives non-UTF-8 stdout (e.g. Windows cp1252 console)."""
return _safe("─", "-") * width
def _estimate_tokens(text: str) -> int:
return max(1, len(text) // _CHARS_PER_TOKEN)
@@ -118,8 +138,9 @@ def print_benchmark(result: dict) -> None:
return
print(f"\ngraphify token reduction benchmark")
print(f"{'─' * 50}")
print(f" Corpus: {result['corpus_words']:,} words → ~{result['corpus_tokens']:,} tokens (naive)")
print(_hr(50))
arrow = _safe("→", "->")
print(f" Corpus: {result['corpus_words']:,} words {arrow} ~{result['corpus_tokens']:,} tokens (naive)")
print(f" Graph: {result['nodes']:,} nodes, {result['edges']:,} edges")
print(f" Avg query cost: ~{result['avg_query_tokens']:,} tokens")
print(f" Reduction: {result['reduction_ratio']}x fewer tokens per query")
+44 -21
View File
@@ -4660,8 +4660,14 @@ def _extract_parallel(
effective_root: Path,
max_workers: int | None,
total_files: int,
) -> None:
"""Extract uncached files in parallel using ProcessPoolExecutor."""
) -> bool:
"""Extract uncached files in parallel using ProcessPoolExecutor.
Returns True if the pool ran to completion. Returns False if the pool
failed in a recoverable way (typically Windows-spawn without an
``if __name__ == "__main__"`` guard in the calling script, which causes
BrokenProcessPool); the caller should fall back to sequential extraction.
"""
import concurrent.futures
if max_workers is None:
@@ -4672,28 +4678,44 @@ def _extract_parallel(
done_count = 0
_PROGRESS_INTERVAL = 100
with concurrent.futures.ProcessPoolExecutor(max_workers=max_workers) as pool:
futures = {
pool.submit(_extract_single_file, item): item[0] for item in work_items
}
for future in concurrent.futures.as_completed(futures):
idx, result = future.result()
per_file[idx] = result
done_count += 1
if (
total_files >= _PROGRESS_INTERVAL
and done_count % _PROGRESS_INTERVAL == 0
):
print(
f" AST extraction: {done_count}/{len(uncached_work)} uncached files "
f"({done_count * 100 // len(uncached_work)}%) [{max_workers} workers]",
flush=True,
)
try:
with concurrent.futures.ProcessPoolExecutor(max_workers=max_workers) as pool:
futures = {
pool.submit(_extract_single_file, item): item[0] for item in work_items
}
for future in concurrent.futures.as_completed(futures):
idx, result = future.result()
per_file[idx] = result
done_count += 1
if (
total_files >= _PROGRESS_INTERVAL
and done_count % _PROGRESS_INTERVAL == 0
):
print(
f" AST extraction: {done_count}/{len(uncached_work)} uncached files "
f"({done_count * 100 // len(uncached_work)}%) [{max_workers} workers]",
flush=True,
)
except concurrent.futures.process.BrokenProcessPool:
# On Windows (spawn start method) the worker subprocesses re-import the
# caller's __main__. Inline invocations like `python -c "..."` have no
# __main__ guard, so worker bootstrap raises and the pool dies before
# any work completes. Fall back to in-process sequential extraction —
# slower but correct.
print(
" warning: parallel extraction failed (BrokenProcessPool); "
"falling back to sequential. On Windows this usually means the "
'caller is missing an `if __name__ == "__main__":` guard. Pass '
"parallel=False to extract() to skip the pool entirely.",
flush=True,
)
return False
if total_files >= _PROGRESS_INTERVAL:
print(
f" AST extraction: {total_files}/{total_files} files (100%) [{max_workers} workers]",
flush=True,
)
return True
def _extract_sequential(
@@ -4793,11 +4815,12 @@ def extract(
# Phase 2: extract uncached files (parallel or sequential)
if uncached_work:
ran_parallel = False
if parallel and len(uncached_work) >= _PARALLEL_THRESHOLD:
_extract_parallel(
ran_parallel = _extract_parallel(
uncached_work, per_file, effective_root, max_workers, total
)
else:
if not ran_parallel:
_extract_sequential(uncached_work, per_file, effective_root, total)
# Fill any remaining None slots (shouldn't happen, but defensive)
+159 -83
View File
@@ -62,10 +62,18 @@ Follow these steps in order. Do not skip steps.
```powershell
# Detect Python and install graphify if needed
python -c "import graphify" 2>$null
@'
import graphify
'@ | Out-File -FilePath .graphify_step_1_ensure_graphify_is_installed_1.py -Encoding utf8
python .graphify_step_1_ensure_graphify_is_installed_1.py 2>$null
Remove-Item -ErrorAction SilentlyContinue .graphify_step_1_ensure_graphify_is_installed_1.py
if ($LASTEXITCODE -ne 0) { pip install graphifyy -q 2>&1 | Select-Object -Last 3 }
# Write interpreter path for all subsequent steps
python -c "import sys; open('.graphify_python', 'w').write(sys.executable)"
@'
import sys; open('.graphify_python', 'w').write(sys.executable)
'@ | Out-File -FilePath .graphify_step_1_ensure_graphify_is_installed_2.py -Encoding utf8
python .graphify_step_1_ensure_graphify_is_installed_2.py
Remove-Item -ErrorAction SilentlyContinue .graphify_step_1_ensure_graphify_is_installed_2.py
```
If the import succeeds, print nothing and move straight to Step 2.
@@ -73,13 +81,15 @@ If the import succeeds, print nothing and move straight to Step 2.
### Step 2 - Detect files
```powershell
python -c "
@'
import json
from graphify.detect import detect
from pathlib import Path
result = detect(Path('INPUT_PATH'))
print(json.dumps(result))
" > .graphify_detect.json
'@ | Out-File -FilePath .graphify_step_2_detect_files_3.py -Encoding utf8
python .graphify_step_2_detect_files_3.py > .graphify_detect.json
Remove-Item -ErrorAction SilentlyContinue .graphify_step_2_detect_files_3.py
```
Replace INPUT_PATH with the actual path the user provided. Do NOT cat or print the JSON - read it silently and present a clean summary instead:
@@ -123,7 +133,7 @@ Set it as `$env:GRAPHIFY_WHISPER_PROMPT` before running the transcription comman
**Step 2 - Transcribe (PowerShell):**
```powershell
& (Get-Content graphify-out\.graphify_python) -c "
@'
import json, os
from pathlib import Path
from graphify.transcribe import transcribe_all
@@ -134,7 +144,9 @@ prompt = os.environ.get('GRAPHIFY_WHISPER_PROMPT', 'Use proper punctuation and p
transcript_paths = transcribe_all(video_files, initial_prompt=prompt)
print(json.dumps(transcript_paths))
" | Out-File -FilePath graphify-out\.graphify_transcripts.json -Encoding utf8
'@ | Out-File -FilePath .graphify_step_transcribe.py -Encoding utf8
& (Get-Content graphify-out\.graphify_python) .graphify_step_transcribe.py | Out-File -FilePath graphify-out\.graphify_transcripts.json -Encoding utf8
Remove-Item -ErrorAction SilentlyContinue .graphify_step_transcribe.py
```
After transcription:
@@ -160,25 +172,37 @@ Note: Parallelizing AST + semantic saves 5-15s on large corpora. AST is determin
For any code files detected, run AST extraction in parallel with Part B subagents:
```powershell
python -c "
import sys, json
@'
import json
from graphify.extract import collect_files, extract
from pathlib import Path
import json
code_files = []
detect = json.loads(Path('.graphify_detect.json').read_text())
for f in detect.get('files', {}).get('code', []):
code_files.extend(collect_files(Path(f)) if Path(f).is_dir() else [Path(f)])
if code_files:
result = extract(code_files)
Path('.graphify_ast.json').write_text(json.dumps(result, indent=2))
print(f'AST: {len(result[\"nodes\"])} nodes, {len(result[\"edges\"])} edges')
else:
Path('.graphify_ast.json').write_text(json.dumps({'nodes':[],'edges':[],'input_tokens':0,'output_tokens':0}))
print('No code files - skipping AST extraction')
"
def main():
code_files = []
detect = json.loads(Path('.graphify_detect.json').read_text())
for f in detect.get('files', {}).get('code', []):
code_files.extend(collect_files(Path(f)) if Path(f).is_dir() else [Path(f)])
if code_files:
result = extract(code_files)
Path('.graphify_ast.json').write_text(json.dumps(result, indent=2))
print(f'AST: {len(result["nodes"])} nodes, {len(result["edges"])} edges')
else:
Path('.graphify_ast.json').write_text(json.dumps({'nodes':[],'edges':[],'input_tokens':0,'output_tokens':0}))
print('No code files - skipping AST extraction')
# Windows-spawn ProcessPoolExecutor (used inside extract()) re-imports this
# script in each worker; without an `if __name__ == "__main__":` guard the
# pool would recursively spawn itself. graphify v0.7.11+ falls back to
# sequential extraction if the pool dies, but the guard keeps multi-core
# extraction working on Windows.
if __name__ == '__main__':
main()
'@ | Out-File -FilePath .graphify_step_ast.py -Encoding utf8
python .graphify_step_ast.py
Remove-Item -ErrorAction SilentlyContinue .graphify_step_ast.py
```
#### Part B - Semantic extraction (parallel subagents)
@@ -198,7 +222,7 @@ Before dispatching subagents, print a timing estimate:
Before dispatching any subagents, check which files already have cached extraction results:
```powershell
python -c "
@'
import json
from graphify.cache import check_semantic_cache
from pathlib import Path
@@ -212,7 +236,9 @@ if cached_nodes or cached_edges or cached_hyperedges:
Path('.graphify_cached.json').write_text(json.dumps({'nodes': cached_nodes, 'edges': cached_edges, 'hyperedges': cached_hyperedges}))
Path('.graphify_uncached.txt').write_text('\n'.join(uncached))
print(f'Cache: {len(all_files)-len(uncached)} files hit, {len(uncached)} files need extraction')
"
'@ | Out-File -FilePath .graphify_step_3_extract_entities_and_relations_5.py -Encoding utf8
python .graphify_step_3_extract_entities_and_relations_5.py
Remove-Item -ErrorAction SilentlyContinue .graphify_step_3_extract_entities_and_relations_5.py
```
Only dispatch subagents for files listed in `.graphify_uncached.txt`. If all files are cached, skip to Part C directly.
@@ -331,7 +357,7 @@ print(f'Merged {len(chunks)} chunks: {total_in:,} in / {total_out:,} out tokens'
Save new results to cache:
```powershell
python -c "
@'
import json
from graphify.cache import save_semantic_cache
from pathlib import Path
@@ -339,12 +365,14 @@ from pathlib import Path
new = json.loads(Path('.graphify_semantic_new.json').read_text()) if Path('.graphify_semantic_new.json').exists() else {'nodes':[],'edges':[],'hyperedges':[]}
saved = save_semantic_cache(new.get('nodes', []), new.get('edges', []), new.get('hyperedges', []))
print(f'Cached {saved} files')
"
'@ | Out-File -FilePath .graphify_step_3_extract_entities_and_relations_6.py -Encoding utf8
python .graphify_step_3_extract_entities_and_relations_6.py
Remove-Item -ErrorAction SilentlyContinue .graphify_step_3_extract_entities_and_relations_6.py
```
Merge cached + new results into `.graphify_semantic.json`:
```powershell
python -c "
@'
import json
from pathlib import Path
@@ -369,15 +397,17 @@ merged = {
'output_tokens': new.get('output_tokens', 0),
}
Path('.graphify_semantic.json').write_text(json.dumps(merged, indent=2))
print(f'Extraction complete - {len(deduped)} nodes, {len(all_edges)} edges ({len(cached[\"nodes\"])} from cache, {len(new.get(\"nodes\",[]))} new)')
"
print(f'Extraction complete - {len(deduped)} nodes, {len(all_edges)} edges ({len(cached["nodes"])} from cache, {len(new.get("nodes",[]))} new)')
'@ | Out-File -FilePath .graphify_step_3_extract_entities_and_relations_7.py -Encoding utf8
python .graphify_step_3_extract_entities_and_relations_7.py
Remove-Item -ErrorAction SilentlyContinue .graphify_step_3_extract_entities_and_relations_7.py
```
Clean up temp files: `Remove-Item -ErrorAction SilentlyContinue .graphify_cached.json, .graphify_uncached.txt, .graphify_semantic_new.json`
#### Part C - Merge AST + semantic into final extraction
```powershell
python -c "
@'
import sys, json
from pathlib import Path
@@ -404,15 +434,17 @@ merged = {
Path('.graphify_extract.json').write_text(json.dumps(merged, indent=2))
total = len(merged_nodes)
edges = len(merged_edges)
print(f'Merged: {total} nodes, {edges} edges ({len(ast[\"nodes\"])} AST + {len(sem[\"nodes\"])} semantic)')
"
print(f'Merged: {total} nodes, {edges} edges ({len(ast["nodes"])} AST + {len(sem["nodes"])} semantic)')
'@ | Out-File -FilePath .graphify_step_3_extract_entities_and_relations_8.py -Encoding utf8
python .graphify_step_3_extract_entities_and_relations_8.py
Remove-Item -ErrorAction SilentlyContinue .graphify_step_3_extract_entities_and_relations_8.py
```
### Step 4 - Build graph, cluster, analyze, generate outputs
```powershell
New-Item -ItemType Directory -Force -Path graphify-out | Out-Null
python -c "
@'
import sys, json
from graphify.build import build_from_json
from graphify.cluster import cluster, score_all
@@ -451,7 +483,9 @@ if G.number_of_nodes() == 0:
print('Possible causes: all files were skipped, binary-only corpus, or extraction failed.')
raise SystemExit(1)
print(f'Graph: {G.number_of_nodes()} nodes, {G.number_of_edges()} edges, {len(communities)} communities')
"
'@ | Out-File -FilePath .graphify_step_4_build_graph_cluster_analyze_ge_9.py -Encoding utf8
python .graphify_step_4_build_graph_cluster_analyze_ge_9.py
Remove-Item -ErrorAction SilentlyContinue .graphify_step_4_build_graph_cluster_analyze_ge_9.py
```
If this step prints `ERROR: Graph is empty`, stop and tell the user what happened - do not proceed to labeling or visualization.
@@ -465,7 +499,7 @@ Read `.graphify_analysis.json`. For each community key, look at its node labels
Then regenerate the report and save the labels for the visualizer:
```powershell
python -c "
@'
import sys, json
from graphify.build import build_from_json
from graphify.cluster import score_all
@@ -492,7 +526,9 @@ report = generate(G, communities, cohesion, labels, analysis['gods'], analysis['
Path('graphify-out/GRAPH_REPORT.md').write_text(report)
Path('.graphify_labels.json').write_text(json.dumps({str(k): v for k, v in labels.items()}))
print('Report updated with community labels')
"
'@ | Out-File -FilePath .graphify_step_5_label_communities_10.py -Encoding utf8
python .graphify_step_5_label_communities_10.py
Remove-Item -ErrorAction SilentlyContinue .graphify_step_5_label_communities_10.py
```
Replace `LABELS_DICT` with the actual dict you constructed (e.g. `{0: "Attention Mechanism", 1: "Training Pipeline"}`).
@@ -507,7 +543,7 @@ If `--obsidian` was given:
- If `--obsidian-dir <path>` was also given, use that path as the vault directory. Otherwise default to `graphify-out/obsidian`.
```powershell
python -c "
@'
import sys, json
from graphify.build import build_from_json
from graphify.export import to_obsidian, to_canvas
@@ -534,13 +570,15 @@ print(f'Open {obsidian_dir}/ as a vault in Obsidian.')
print(' Graph view - nodes colored by community (set automatically)')
print(' graph.canvas - structured layout with communities as groups')
print(' _COMMUNITY_* - overview notes with cohesion scores and dataview queries')
"
'@ | Out-File -FilePath .graphify_step_6_generate_obsidian_vault_opt_in_11.py -Encoding utf8
python .graphify_step_6_generate_obsidian_vault_opt_in_11.py
Remove-Item -ErrorAction SilentlyContinue .graphify_step_6_generate_obsidian_vault_opt_in_11.py
```
Generate the HTML graph (always, unless `--no-viz`):
```powershell
python -c "
@'
import sys, json
from graphify.build import build_from_json
from graphify.export import to_html
@@ -559,7 +597,9 @@ if G.number_of_nodes() > 5000:
else:
to_html(G, communities, 'graphify-out/graph.html', community_labels=labels or None)
print('graph.html written - open in any browser, no server needed')
"
'@ | Out-File -FilePath .graphify_step_6_generate_obsidian_vault_opt_in_12.py -Encoding utf8
python .graphify_step_6_generate_obsidian_vault_opt_in_12.py
Remove-Item -ErrorAction SilentlyContinue .graphify_step_6_generate_obsidian_vault_opt_in_12.py
```
### Step 7 - Neo4j export (only if --neo4j or --neo4j-push flag)
@@ -567,7 +607,7 @@ else:
**If `--neo4j`** - generate a Cypher file for manual import:
```powershell
python -c "
@'
import sys, json
from graphify.build import build_from_json
from graphify.export import to_cypher
@@ -576,13 +616,15 @@ from pathlib import Path
G = build_from_json(json.loads(Path('.graphify_extract.json').read_text()))
to_cypher(G, 'graphify-out/cypher.txt')
print('cypher.txt written - import with: cypher-shell < graphify-out/cypher.txt')
"
'@ | Out-File -FilePath .graphify_step_7_neo4j_export_only_if_neo4j_or__13.py -Encoding utf8
python .graphify_step_7_neo4j_export_only_if_neo4j_or__13.py
Remove-Item -ErrorAction SilentlyContinue .graphify_step_7_neo4j_export_only_if_neo4j_or__13.py
```
**If `--neo4j-push <uri>`** - push directly to a running Neo4j instance. Ask the user for credentials if not provided:
```powershell
python -c "
@'
import sys, json
from graphify.build import build_from_json
from graphify.cluster import cluster
@@ -595,8 +637,10 @@ G = build_from_json(extraction)
communities = {int(k): v for k, v in analysis['communities'].items()}
result = push_to_neo4j(G, uri='NEO4J_URI', user='NEO4J_USER', password='NEO4J_PASSWORD', communities=communities)
print(f'Pushed to Neo4j: {result[\"nodes\"]} nodes, {result[\"edges\"]} edges')
"
print(f'Pushed to Neo4j: {result["nodes"]} nodes, {result["edges"]} edges')
'@ | Out-File -FilePath .graphify_step_7_neo4j_export_only_if_neo4j_or__14.py -Encoding utf8
python .graphify_step_7_neo4j_export_only_if_neo4j_or__14.py
Remove-Item -ErrorAction SilentlyContinue .graphify_step_7_neo4j_export_only_if_neo4j_or__14.py
```
Replace `NEO4J_URI`, `NEO4J_USER`, `NEO4J_PASSWORD` with actual values. Default URI is `bolt://localhost:7687`, default user is `neo4j`. Uses MERGE - safe to re-run without creating duplicates.
@@ -604,7 +648,7 @@ Replace `NEO4J_URI`, `NEO4J_USER`, `NEO4J_PASSWORD` with actual values. Default
### Step 7b - SVG export (only if --svg flag)
```powershell
python -c "
@'
import sys, json
from graphify.build import build_from_json
from graphify.export import to_svg
@@ -620,13 +664,15 @@ labels = {int(k): v for k, v in labels_raw.items()}
to_svg(G, communities, 'graphify-out/graph.svg', community_labels=labels or None)
print('graph.svg written - embeds in Obsidian, Notion, GitHub READMEs')
"
'@ | Out-File -FilePath .graphify_step_7b_svg_export_only_if_svg_flag_15.py -Encoding utf8
python .graphify_step_7b_svg_export_only_if_svg_flag_15.py
Remove-Item -ErrorAction SilentlyContinue .graphify_step_7b_svg_export_only_if_svg_flag_15.py
```
### Step 7c - GraphML export (only if --graphml flag)
```powershell
python -c "
@'
import json
from graphify.build import build_from_json
from graphify.export import to_graphml
@@ -640,7 +686,9 @@ communities = {int(k): v for k, v in analysis['communities'].items()}
to_graphml(G, communities, 'graphify-out/graph.graphml')
print('graph.graphml written - open in Gephi, yEd, or any GraphML tool')
"
'@ | Out-File -FilePath .graphify_step_7c_graphml_export_only_if_graphml_16.py -Encoding utf8
python .graphify_step_7c_graphml_export_only_if_graphml_16.py
Remove-Item -ErrorAction SilentlyContinue .graphify_step_7c_graphml_export_only_if_graphml_16.py
```
### Step 7d - MCP server (only if --mcp flag)
@@ -668,7 +716,7 @@ To configure in Claude Desktop, add to `claude_desktop_config.json`:
If `total_words` from `.graphify_detect.json` is greater than 5,000, run:
```powershell
python -c "
@'
import json
from graphify.benchmark import run_benchmark, print_benchmark
from pathlib import Path
@@ -676,7 +724,9 @@ from pathlib import Path
detection = json.loads(Path('.graphify_detect.json').read_text())
result = run_benchmark('graphify-out/graph.json', corpus_words=detection['total_words'])
print_benchmark(result)
"
'@ | Out-File -FilePath .graphify_step_8_token_reduction_benchmark_only_17.py -Encoding utf8
python .graphify_step_8_token_reduction_benchmark_only_17.py
Remove-Item -ErrorAction SilentlyContinue .graphify_step_8_token_reduction_benchmark_only_17.py
```
Print the output directly in chat. If `total_words <= 5000`, skip silently - the graph value is structural clarity, not token compression, for small corpora.
@@ -686,7 +736,7 @@ Print the output directly in chat. If `total_words <= 5000`, skip silently - the
### Step 9 - Save manifest, update cost tracker, clean up, and report
```powershell
python -c "
@'
import json
from pathlib import Path
from datetime import datetime, timezone
@@ -718,8 +768,10 @@ cost['total_output_tokens'] += output_tok
cost_path.write_text(json.dumps(cost, indent=2))
print(f'This run: {input_tok:,} input tokens, {output_tok:,} output tokens')
print(f'All time: {cost[\"total_input_tokens\"]:,} input, {cost[\"total_output_tokens\"]:,} output ({len(cost[\"runs\"])} runs)')
"
print(f'All time: {cost["total_input_tokens"]:,} input, {cost["total_output_tokens"]:,} output ({len(cost["runs"])} runs)')
'@ | Out-File -FilePath .graphify_step_9_save_manifest_update_cost_trac_18.py -Encoding utf8
python .graphify_step_9_save_manifest_update_cost_trac_18.py
Remove-Item -ErrorAction SilentlyContinue .graphify_step_9_save_manifest_update_cost_trac_18.py
Remove-Item -ErrorAction SilentlyContinue .graphify_detect.json, .graphify_extract.json, .graphify_ast.json, .graphify_semantic.json, .graphify_analysis.json, .graphify_labels.json
Remove-Item -ErrorAction SilentlyContinue graphify-out/.needs_update
```
@@ -760,7 +812,7 @@ The graph is the map. Your job after the pipeline is to be the guide.
Use when you've added or modified files since the last run. Only re-extracts changed files - saves tokens and time.
```powershell
python -c "
@'
import sys, json
from graphify.detect import detect_incremental, save_manifest
from pathlib import Path
@@ -773,13 +825,15 @@ if new_total == 0:
print('No files changed since last run. Nothing to update.')
raise SystemExit(0)
print(f'{new_total} new/changed file(s) to re-extract.')
"
'@ | Out-File -FilePath .graphify_step_for_update_incremental_re_extracti_19.py -Encoding utf8
python .graphify_step_for_update_incremental_re_extracti_19.py
Remove-Item -ErrorAction SilentlyContinue .graphify_step_for_update_incremental_re_extracti_19.py
```
If new files exist, first check whether all changed files are code files:
```powershell
python -c "
@'
import json
from pathlib import Path
@@ -789,7 +843,9 @@ new_files = result.get('new_files', {})
all_changed = [f for files in new_files.values() for f in files]
code_only = all(Path(f).suffix.lower() in code_exts for f in all_changed)
print('code_only:', code_only)
"
'@ | Out-File -FilePath .graphify_step_for_update_incremental_re_extracti_20.py -Encoding utf8
python .graphify_step_for_update_incremental_re_extracti_20.py
Remove-Item -ErrorAction SilentlyContinue .graphify_step_for_update_incremental_re_extracti_20.py
```
If `code_only` is True: print `[graphify update] Code-only changes detected - skipping semantic extraction (no LLM needed)`, run only Step 3A (AST) on the changed files, skip Step 3B entirely (no subagents), then go straight to merge and Steps 4–8.
@@ -799,7 +855,7 @@ If `code_only` is False (any changed file is a doc/paper/image): run the full St
Then:
```powershell
python -c "
@'
import sys, json
from graphify.build import build_from_json
from graphify.export import to_json
@@ -837,7 +893,9 @@ print(f'Merged: {G_existing.number_of_nodes()} nodes, {G_existing.number_of_edge
from graphify.detect import save_manifest
save_manifest(incremental['files'])
print('[graphify update] Manifest saved.')
"
'@ | Out-File -FilePath .graphify_step_for_update_incremental_re_extracti_21.py -Encoding utf8
python .graphify_step_for_update_incremental_re_extracti_21.py
Remove-Item -ErrorAction SilentlyContinue .graphify_step_for_update_incremental_re_extracti_21.py
```
Then run Steps 4–8 on the merged graph as normal.
@@ -845,7 +903,7 @@ Then run Steps 4–8 on the merged graph as normal.
After Step 4, show the graph diff:
```powershell
python -c "
@'
import json
from graphify.analyze import graph_diff
from graphify.build import build_from_json
@@ -866,7 +924,9 @@ if old_data:
print('New nodes:', ', '.join(n['label'] for n in diff['new_nodes'][:5]))
if diff['new_edges']:
print('New edges:', len(diff['new_edges']))
"
'@ | Out-File -FilePath .graphify_step_for_update_incremental_re_extracti_22.py -Encoding utf8
python .graphify_step_for_update_incremental_re_extracti_22.py
Remove-Item -ErrorAction SilentlyContinue .graphify_step_for_update_incremental_re_extracti_22.py
```
Before the merge step, save the old graph: `Copy-Item graphify-out/graph.json .graphify_old.json`
@@ -879,7 +939,7 @@ Clean up after: `Remove-Item -ErrorAction SilentlyContinue .graphify_old.json`
Skip Steps 1–3. Load the existing graph from `graphify-out/graph.json` and re-run clustering:
```powershell
python -c "
@'
import sys, json
from graphify.cluster import cluster, score_all
from graphify.analyze import god_nodes, surprising_connections
@@ -914,7 +974,9 @@ analysis = {
}
Path('.graphify_analysis.json').write_text(json.dumps(analysis, indent=2))
print(f'Re-clustered: {len(communities)} communities')
"
'@ | Out-File -FilePath .graphify_step_for_cluster_only_23.py -Encoding utf8
python .graphify_step_for_cluster_only_23.py
Remove-Item -ErrorAction SilentlyContinue .graphify_step_for_cluster_only_23.py
```
Then run Steps 5–9 as normal (label communities, generate viz, benchmark, clean up, report).
@@ -932,12 +994,14 @@ Two traversal modes - choose based on the question:
First check the graph exists:
```powershell
python -c "
@'
from pathlib import Path
if not Path('graphify-out/graph.json').exists():
print('ERROR: No graph found. Run /graphify <path> first to build the graph.')
raise SystemExit(1)
"
'@ | Out-File -FilePath .graphify_step_for_graphify_query_24.py -Encoding utf8
python .graphify_step_for_graphify_query_24.py
Remove-Item -ErrorAction SilentlyContinue .graphify_step_for_graphify_query_24.py
```
If it fails, stop and tell the user to run `/graphify <path>` first.
@@ -950,7 +1014,7 @@ Load `graphify-out/graph.json`, then:
5. If the graph lacks enough information, say so - do not hallucinate edges.
```powershell
python -c "
@'
import sys, json
from networkx.readwrite import json_graph
import networkx as nx
@@ -1020,20 +1084,22 @@ def relevance(nid):
ranked_nodes = sorted(subgraph_nodes, key=relevance, reverse=True)
lines = [f'Traversal: {mode.upper()} | Start: {[G.nodes[n].get(\"label\",n) for n in start_nodes]} | {len(subgraph_nodes)} nodes']
lines = [f'Traversal: {mode.upper()} | Start: {[G.nodes[n].get("label",n) for n in start_nodes]} | {len(subgraph_nodes)} nodes']
for nid in ranked_nodes:
d = G.nodes[nid]
lines.append(f' NODE {d.get(\"label\", nid)} [src={d.get(\"source_file\",\"\")} loc={d.get(\"source_location\",\"\")}]')
lines.append(f' NODE {d.get("label", nid)} [src={d.get("source_file","")} loc={d.get("source_location","")}]')
for u, v in subgraph_edges:
if u in subgraph_nodes and v in subgraph_nodes:
d = G.edges[u, v]
lines.append(f' EDGE {G.nodes[u].get(\"label\",u)} --{d.get(\"relation\",\"\")} [{d.get(\"confidence\",\"\")}]--> {G.nodes[v].get(\"label\",v)}')
lines.append(f' EDGE {G.nodes[u].get("label",u)} --{d.get("relation","")} [{d.get("confidence","")}]--> {G.nodes[v].get("label",v)}')
output = '\n'.join(lines)
if len(output) > char_budget:
output = output[:char_budget] + f'\n... (truncated at ~{token_budget} token budget - use --budget N for more)'
print(output)
"
'@ | Out-File -FilePath .graphify_step_for_graphify_query_25.py -Encoding utf8
python .graphify_step_for_graphify_query_25.py
Remove-Item -ErrorAction SilentlyContinue .graphify_step_for_graphify_query_25.py
```
Replace `QUESTION` with the user's actual question, `MODE` with `bfs` or `dfs`, and `BUDGET` with the token budget (default `2000`, or whatever `--budget N` specifies). Then answer based on the subgraph output above.
@@ -1054,17 +1120,19 @@ Find the shortest path between two named concepts in the graph.
First check the graph exists:
```powershell
python -c "
@'
from pathlib import Path
if not Path('graphify-out/graph.json').exists():
print('ERROR: No graph found. Run /graphify <path> first to build the graph.')
raise SystemExit(1)
"
'@ | Out-File -FilePath .graphify_step_for_graphify_path_26.py -Encoding utf8
python .graphify_step_for_graphify_path_26.py
Remove-Item -ErrorAction SilentlyContinue .graphify_step_for_graphify_path_26.py
```
If it fails, stop and tell the user to run `/graphify <path>` first.
```powershell
python -c "
@'
import json, sys
import networkx as nx
from networkx.readwrite import json_graph
@@ -1108,7 +1176,9 @@ except nx.NetworkXNoPath:
print(f'No path found between {a_term!r} and {b_term!r}')
except nx.NodeNotFound as e:
print(f'Node not found: {e}')
"
'@ | Out-File -FilePath .graphify_step_for_graphify_path_27.py -Encoding utf8
python .graphify_step_for_graphify_path_27.py
Remove-Item -ErrorAction SilentlyContinue .graphify_step_for_graphify_path_27.py
```
Replace `NODE_A` and `NODE_B` with the actual concept names from the user. Then explain the path in plain language - what each hop means, why it's significant.
@@ -1127,17 +1197,19 @@ Give a plain-language explanation of a single node - everything connected to it.
First check the graph exists:
```powershell
python -c "
@'
from pathlib import Path
if not Path('graphify-out/graph.json').exists():
print('ERROR: No graph found. Run /graphify <path> first to build the graph.')
raise SystemExit(1)
"
'@ | Out-File -FilePath .graphify_step_for_graphify_explain_28.py -Encoding utf8
python .graphify_step_for_graphify_explain_28.py
Remove-Item -ErrorAction SilentlyContinue .graphify_step_for_graphify_explain_28.py
```
If it fails, stop and tell the user to run `/graphify <path>` first.
```powershell
python -c "
@'
import json, sys
import networkx as nx
from networkx.readwrite import json_graph
@@ -1161,9 +1233,9 @@ if not scored or scored[0][0] == 0:
nid = scored[0][1]
data_n = G.nodes[nid]
print(f'NODE: {data_n.get(\"label\", nid)}')
print(f' source: {data_n.get(\"source_file\",\"unknown\")}')
print(f' type: {data_n.get(\"file_type\",\"unknown\")}')
print(f'NODE: {data_n.get("label", nid)}')
print(f' source: {data_n.get("source_file","unknown")}')
print(f' type: {data_n.get("file_type","unknown")}')
print(f' degree: {G.degree(nid)}')
print()
print('CONNECTIONS:')
@@ -1174,7 +1246,9 @@ for neighbor in G.neighbors(nid):
conf = edge.get('confidence', '')
src_file = G.nodes[neighbor].get('source_file', '')
print(f' --{rel}--> {nlabel} [{conf}] ({src_file})')
"
'@ | Out-File -FilePath .graphify_step_for_graphify_explain_29.py -Encoding utf8
python .graphify_step_for_graphify_explain_29.py
Remove-Item -ErrorAction SilentlyContinue .graphify_step_for_graphify_explain_29.py
```
Replace `NODE_NAME` with the concept the user asked about. Then write a 3-5 sentence explanation of what this node is, what it connects to, and why those connections are significant. Use the source locations as citations.
@@ -1192,7 +1266,7 @@ python -m graphify save-result --question "Explain NODE_NAME" --answer "ANSWER"
Fetch a URL and add it to the corpus, then update the graph.
```powershell
python -c "
@'
import sys
from graphify.ingest import ingest
from pathlib import Path
@@ -1206,7 +1280,9 @@ except ValueError as e:
except RuntimeError as e:
print(f'error: {e}', file=sys.stderr)
sys.exit(1)
"
'@ | Out-File -FilePath .graphify_step_for_graphify_add_30.py -Encoding utf8
python .graphify_step_for_graphify_add_30.py
Remove-Item -ErrorAction SilentlyContinue .graphify_step_for_graphify_add_30.py
```
Replace `URL` with the actual URL, `AUTHOR` with the user's name if provided, `CONTRIBUTOR` likewise. If the command exits with an error, tell the user what went wrong - do not silently continue. After a successful save, automatically run the `--update` pipeline on `./raw` to merge the new file into the existing graph.
+47 -1
View File
@@ -5,7 +5,7 @@ import pytest
import networkx as nx
from networkx.readwrite import json_graph
from graphify.benchmark import run_benchmark, print_benchmark, _query_subgraph_tokens, _SAMPLE_QUESTIONS
from graphify.benchmark import run_benchmark, print_benchmark, _query_subgraph_tokens, _SAMPLE_QUESTIONS, _safe, _hr
def _make_graph() -> nx.Graph:
@@ -117,3 +117,49 @@ def test_print_benchmark_error_message(capsys):
print_benchmark({"error": "test error message"})
out = capsys.readouterr().out
assert "test error message" in out
# --- cp1252 / Windows-console encoding compatibility (regression for #?) ---
# print_benchmark previously crashed on Windows consoles (cp1252) because it
# unconditionally printed U+2500 and U+2192. _safe() falls back to ASCII when
# stdout cannot encode the glyph.
def test_safe_returns_unicode_when_encodable():
import io, sys
real_stdout = sys.stdout
try:
sys.stdout = io.TextIOWrapper(io.BytesIO(), encoding="utf-8")
assert _safe("→", "->") == "→"
assert _hr(5) == "─" * 5
finally:
sys.stdout = real_stdout
def test_safe_falls_back_when_unencodable():
import io, sys
real_stdout = sys.stdout
try:
sys.stdout = io.TextIOWrapper(io.BytesIO(), encoding="cp1252")
assert _safe("→", "->") == "->"
assert _hr(5) == "-" * 5
finally:
sys.stdout = real_stdout
def test_print_benchmark_survives_cp1252_stdout(tmp_path, monkeypatch, capsys):
"""Regression: U+2500 / U+2192 used to crash with UnicodeEncodeError on cp1252."""
import io, sys
G = _make_graph()
graph_file = tmp_path / "graph.json"
_write_graph(G, graph_file)
result = run_benchmark(str(graph_file), corpus_words=5_000)
# Replace stdout with a strict cp1252 stream — same behaviour as the
# legacy Windows console that surfaced this bug.
cp1252_stdout = io.TextIOWrapper(io.BytesIO(), encoding="cp1252", errors="strict")
monkeypatch.setattr(sys, "stdout", cp1252_stdout)
print_benchmark(result) # must not raise UnicodeEncodeError
cp1252_stdout.flush()
written = cp1252_stdout.buffer.getvalue().decode("cp1252")
assert "reduction" in written.lower()
# ASCII fallbacks must be present, fancy glyphs must not.
assert "─" not in written
assert "→" not in written
+60
View File
@@ -365,3 +365,63 @@ def test_extract_tsx_uses_tsx_grammar():
from graphify.extract import _TSX_CONFIG, _TS_CONFIG
assert _TSX_CONFIG.ts_language_fn == "language_tsx"
assert _TS_CONFIG.ts_language_fn == "language_typescript"
# --- Windows-spawn ProcessPool fallback (regression for #?) ---
# When the caller has no `if __name__ == "__main__":` guard, ProcessPoolExecutor
# on Windows raises BrokenProcessPool before any work completes. extract() must
# detect this, warn, and fall back to sequential extraction rather than
# propagating a 290-line traceback.
def test_extract_falls_back_to_sequential_when_parallel_returns_false(tmp_path, monkeypatch):
"""extract() must run sequential when _extract_parallel signals failure (returns False)."""
from graphify import extract as extract_mod
files = [FIXTURES / "sample.py"] * 25 # >= _PARALLEL_THRESHOLD triggers parallel branch
cache_root = tmp_path / "cache"
cache_root.mkdir()
calls = {"parallel": 0, "sequential": 0}
real_sequential = extract_mod._extract_sequential
def fake_parallel(uncached_work, per_file, effective_root, max_workers, total_files):
calls["parallel"] += 1
return False # simulate the post-fix BrokenProcessPool branch
def wrapped_sequential(*args, **kwargs):
calls["sequential"] += 1
return real_sequential(*args, **kwargs)
monkeypatch.setattr(extract_mod, "_extract_parallel", fake_parallel)
monkeypatch.setattr(extract_mod, "_extract_sequential", wrapped_sequential)
result = extract_mod.extract(files, cache_root=cache_root)
assert calls["parallel"] == 1, "parallel path should have been attempted once"
assert calls["sequential"] == 1, "sequential fallback should have run exactly once"
assert result["nodes"], "extract should still produce nodes after fallback"
def test_extract_parallel_returns_false_on_broken_pool(tmp_path, monkeypatch, capsys):
"""_extract_parallel must catch BrokenProcessPool internally and return False."""
from concurrent.futures.process import BrokenProcessPool
import concurrent.futures
from graphify import extract as extract_mod
class FakePool:
def __init__(self, *a, **kw): pass
def __enter__(self): return self
def __exit__(self, *a): return False
def submit(self, *a, **kw):
raise BrokenProcessPool("simulated spawn failure")
monkeypatch.setattr(
concurrent.futures, "ProcessPoolExecutor", lambda *a, **kw: FakePool()
)
uncached = [(0, FIXTURES / "sample.py")]
per_file: list = [None]
ok = extract_mod._extract_parallel(uncached, per_file, tmp_path, 2, 1)
assert ok is False, "function should report failure via return value, not raise"
out = capsys.readouterr().out
assert "BrokenProcessPool" in out, "user-facing warning must mention the failure"
assert "__main__" in out, "warning must hint at the Windows __main__ guard idiom"