From e5f263ba98e81ed68e0d02558bf5cf5062291f46 Mon Sep 17 00:00:00 2001 From: nauman73 Date: Sat, 9 May 2026 16:59:38 +0500 Subject: [PATCH] fix(windows): unblock pipeline on Windows consoles + missing __main__ guards (#788) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Three independent Windows compatibility fixes shipped together because they all surface during the same first /graphify run on Windows. graphify/benchmark.py print_benchmark() unconditionally printed U+2500 (box-drawing) and U+2192 (rightwards arrow), which UnicodeEncodeError'd on stdouts that can't encode them — most notably the legacy Windows console at cp1252. New _safe() helper falls back to ASCII when the active stdout encoding can't carry the glyph; _hr() uses it. Two regression tests cover both paths and prove print_benchmark survives a cp1252-strict stream. graphify/extract.py ProcessPoolExecutor on Windows uses spawn, so worker subprocesses re-import the calling __main__. When the caller is `python -c "..."` or a script without an `if __name__ == "__main__":` guard, the workers recursively spawn themselves and the pool dies. The user-visible failure was a 290-line traceback ending in BrokenProcessPool, hiding the actual cause. _extract_parallel now catches BrokenProcessPool, prints a one-line warning that names the __main__-guard idiom, and returns False so the public extract() routes to the existing _extract_sequential fallback. Two tests cover the parallel-returns-False contract and the sequential fallback wiring. graphify/skill-windows.md Every `python -c "..."` block (30 in total) is replaced with a Write+run+delete pattern using PowerShell's literal here-string @'...'@. The old form was a quote-escaping minefield: any double-quote inside the Python source had to be backslash-escaped for the shell, and PowerShell's parser ate them inconsistently — failing on f-strings like `f'AST: {len(result["nodes"])} nodes'`. The new form passes Python source to disk literally, so what the model writes is what Python sees. The AST step's script template now includes an explicit `if __name__ == "__main__":` guard so multi-core extraction works even before the runtime fallback above kicks in. All 31 resulting heredoc blocks parse cleanly under `ast.parse`. Co-authored-by: Nauman Hameed --- graphify/benchmark.py | 25 +++- graphify/extract.py | 65 ++++++---- graphify/skill-windows.md | 242 +++++++++++++++++++++++++------------- tests/test_benchmark.py | 48 +++++++- tests/test_extract.py | 60 ++++++++++ 5 files changed, 333 insertions(+), 107 deletions(-) diff --git a/graphify/benchmark.py b/graphify/benchmark.py index dc42056..2fb161a 100644 --- a/graphify/benchmark.py +++ b/graphify/benchmark.py @@ -1,6 +1,7 @@ """Token-reduction benchmark - measures how much context graphify saves vs naive full-corpus approach.""" from __future__ import annotations import json +import sys from pathlib import Path import networkx as nx from networkx.readwrite import json_graph @@ -9,6 +10,25 @@ from networkx.readwrite import json_graph _CHARS_PER_TOKEN = 4 # standard approximation +def _safe(unicode_char: str, ascii_fallback: str) -> str: + """Return unicode_char if stdout can encode it, else ascii_fallback. + + Windows consoles often default to cp1252 which cannot encode box-drawing + or arrow glyphs; printing them raises UnicodeEncodeError mid-output. + """ + encoding = getattr(sys.stdout, "encoding", None) or "" + try: + unicode_char.encode(encoding) + return unicode_char + except (UnicodeEncodeError, LookupError): + return ascii_fallback + + +def _hr(width: int = 50) -> str: + """Horizontal rule that survives non-UTF-8 stdout (e.g. Windows cp1252 console).""" + return _safe("─", "-") * width + + def _estimate_tokens(text: str) -> int: return max(1, len(text) // _CHARS_PER_TOKEN) @@ -118,8 +138,9 @@ def print_benchmark(result: dict) -> None: return print(f"\ngraphify token reduction benchmark") - print(f"{'─' * 50}") - print(f" Corpus: {result['corpus_words']:,} words → ~{result['corpus_tokens']:,} tokens (naive)") + print(_hr(50)) + arrow = _safe("→", "->") + print(f" Corpus: {result['corpus_words']:,} words {arrow} ~{result['corpus_tokens']:,} tokens (naive)") print(f" Graph: {result['nodes']:,} nodes, {result['edges']:,} edges") print(f" Avg query cost: ~{result['avg_query_tokens']:,} tokens") print(f" Reduction: {result['reduction_ratio']}x fewer tokens per query") diff --git a/graphify/extract.py b/graphify/extract.py index 6000e02..00c26f1 100644 --- a/graphify/extract.py +++ b/graphify/extract.py @@ -4660,8 +4660,14 @@ def _extract_parallel( effective_root: Path, max_workers: int | None, total_files: int, -) -> None: - """Extract uncached files in parallel using ProcessPoolExecutor.""" +) -> bool: + """Extract uncached files in parallel using ProcessPoolExecutor. + + Returns True if the pool ran to completion. Returns False if the pool + failed in a recoverable way (typically Windows-spawn without an + ``if __name__ == "__main__"`` guard in the calling script, which causes + BrokenProcessPool); the caller should fall back to sequential extraction. + """ import concurrent.futures if max_workers is None: @@ -4672,28 +4678,44 @@ def _extract_parallel( done_count = 0 _PROGRESS_INTERVAL = 100 - with concurrent.futures.ProcessPoolExecutor(max_workers=max_workers) as pool: - futures = { - pool.submit(_extract_single_file, item): item[0] for item in work_items - } - for future in concurrent.futures.as_completed(futures): - idx, result = future.result() - per_file[idx] = result - done_count += 1 - if ( - total_files >= _PROGRESS_INTERVAL - and done_count % _PROGRESS_INTERVAL == 0 - ): - print( - f" AST extraction: {done_count}/{len(uncached_work)} uncached files " - f"({done_count * 100 // len(uncached_work)}%) [{max_workers} workers]", - flush=True, - ) + try: + with concurrent.futures.ProcessPoolExecutor(max_workers=max_workers) as pool: + futures = { + pool.submit(_extract_single_file, item): item[0] for item in work_items + } + for future in concurrent.futures.as_completed(futures): + idx, result = future.result() + per_file[idx] = result + done_count += 1 + if ( + total_files >= _PROGRESS_INTERVAL + and done_count % _PROGRESS_INTERVAL == 0 + ): + print( + f" AST extraction: {done_count}/{len(uncached_work)} uncached files " + f"({done_count * 100 // len(uncached_work)}%) [{max_workers} workers]", + flush=True, + ) + except concurrent.futures.process.BrokenProcessPool: + # On Windows (spawn start method) the worker subprocesses re-import the + # caller's __main__. Inline invocations like `python -c "..."` have no + # __main__ guard, so worker bootstrap raises and the pool dies before + # any work completes. Fall back to in-process sequential extraction — + # slower but correct. + print( + " warning: parallel extraction failed (BrokenProcessPool); " + "falling back to sequential. On Windows this usually means the " + 'caller is missing an `if __name__ == "__main__":` guard. Pass ' + "parallel=False to extract() to skip the pool entirely.", + flush=True, + ) + return False if total_files >= _PROGRESS_INTERVAL: print( f" AST extraction: {total_files}/{total_files} files (100%) [{max_workers} workers]", flush=True, ) + return True def _extract_sequential( @@ -4793,11 +4815,12 @@ def extract( # Phase 2: extract uncached files (parallel or sequential) if uncached_work: + ran_parallel = False if parallel and len(uncached_work) >= _PARALLEL_THRESHOLD: - _extract_parallel( + ran_parallel = _extract_parallel( uncached_work, per_file, effective_root, max_workers, total ) - else: + if not ran_parallel: _extract_sequential(uncached_work, per_file, effective_root, total) # Fill any remaining None slots (shouldn't happen, but defensive) diff --git a/graphify/skill-windows.md b/graphify/skill-windows.md index fb727df..e01a99c 100644 --- a/graphify/skill-windows.md +++ b/graphify/skill-windows.md @@ -62,10 +62,18 @@ Follow these steps in order. Do not skip steps. ```powershell # Detect Python and install graphify if needed -python -c "import graphify" 2>$null +@' +import graphify +'@ | Out-File -FilePath .graphify_step_1_ensure_graphify_is_installed_1.py -Encoding utf8 +python .graphify_step_1_ensure_graphify_is_installed_1.py 2>$null +Remove-Item -ErrorAction SilentlyContinue .graphify_step_1_ensure_graphify_is_installed_1.py if ($LASTEXITCODE -ne 0) { pip install graphifyy -q 2>&1 | Select-Object -Last 3 } # Write interpreter path for all subsequent steps -python -c "import sys; open('.graphify_python', 'w').write(sys.executable)" +@' +import sys; open('.graphify_python', 'w').write(sys.executable) +'@ | Out-File -FilePath .graphify_step_1_ensure_graphify_is_installed_2.py -Encoding utf8 +python .graphify_step_1_ensure_graphify_is_installed_2.py +Remove-Item -ErrorAction SilentlyContinue .graphify_step_1_ensure_graphify_is_installed_2.py ``` If the import succeeds, print nothing and move straight to Step 2. @@ -73,13 +81,15 @@ If the import succeeds, print nothing and move straight to Step 2. ### Step 2 - Detect files ```powershell -python -c " +@' import json from graphify.detect import detect from pathlib import Path result = detect(Path('INPUT_PATH')) print(json.dumps(result)) -" > .graphify_detect.json +'@ | Out-File -FilePath .graphify_step_2_detect_files_3.py -Encoding utf8 +python .graphify_step_2_detect_files_3.py > .graphify_detect.json +Remove-Item -ErrorAction SilentlyContinue .graphify_step_2_detect_files_3.py ``` Replace INPUT_PATH with the actual path the user provided. Do NOT cat or print the JSON - read it silently and present a clean summary instead: @@ -123,7 +133,7 @@ Set it as `$env:GRAPHIFY_WHISPER_PROMPT` before running the transcription comman **Step 2 - Transcribe (PowerShell):** ```powershell -& (Get-Content graphify-out\.graphify_python) -c " +@' import json, os from pathlib import Path from graphify.transcribe import transcribe_all @@ -134,7 +144,9 @@ prompt = os.environ.get('GRAPHIFY_WHISPER_PROMPT', 'Use proper punctuation and p transcript_paths = transcribe_all(video_files, initial_prompt=prompt) print(json.dumps(transcript_paths)) -" | Out-File -FilePath graphify-out\.graphify_transcripts.json -Encoding utf8 +'@ | Out-File -FilePath .graphify_step_transcribe.py -Encoding utf8 +& (Get-Content graphify-out\.graphify_python) .graphify_step_transcribe.py | Out-File -FilePath graphify-out\.graphify_transcripts.json -Encoding utf8 +Remove-Item -ErrorAction SilentlyContinue .graphify_step_transcribe.py ``` After transcription: @@ -160,25 +172,37 @@ Note: Parallelizing AST + semantic saves 5-15s on large corpora. AST is determin For any code files detected, run AST extraction in parallel with Part B subagents: ```powershell -python -c " -import sys, json +@' +import json from graphify.extract import collect_files, extract from pathlib import Path -import json -code_files = [] -detect = json.loads(Path('.graphify_detect.json').read_text()) -for f in detect.get('files', {}).get('code', []): - code_files.extend(collect_files(Path(f)) if Path(f).is_dir() else [Path(f)]) -if code_files: - result = extract(code_files) - Path('.graphify_ast.json').write_text(json.dumps(result, indent=2)) - print(f'AST: {len(result[\"nodes\"])} nodes, {len(result[\"edges\"])} edges') -else: - Path('.graphify_ast.json').write_text(json.dumps({'nodes':[],'edges':[],'input_tokens':0,'output_tokens':0})) - print('No code files - skipping AST extraction') -" +def main(): + code_files = [] + detect = json.loads(Path('.graphify_detect.json').read_text()) + for f in detect.get('files', {}).get('code', []): + code_files.extend(collect_files(Path(f)) if Path(f).is_dir() else [Path(f)]) + + if code_files: + result = extract(code_files) + Path('.graphify_ast.json').write_text(json.dumps(result, indent=2)) + print(f'AST: {len(result["nodes"])} nodes, {len(result["edges"])} edges') + else: + Path('.graphify_ast.json').write_text(json.dumps({'nodes':[],'edges':[],'input_tokens':0,'output_tokens':0})) + print('No code files - skipping AST extraction') + + +# Windows-spawn ProcessPoolExecutor (used inside extract()) re-imports this +# script in each worker; without an `if __name__ == "__main__":` guard the +# pool would recursively spawn itself. graphify v0.7.11+ falls back to +# sequential extraction if the pool dies, but the guard keeps multi-core +# extraction working on Windows. +if __name__ == '__main__': + main() +'@ | Out-File -FilePath .graphify_step_ast.py -Encoding utf8 +python .graphify_step_ast.py +Remove-Item -ErrorAction SilentlyContinue .graphify_step_ast.py ``` #### Part B - Semantic extraction (parallel subagents) @@ -198,7 +222,7 @@ Before dispatching subagents, print a timing estimate: Before dispatching any subagents, check which files already have cached extraction results: ```powershell -python -c " +@' import json from graphify.cache import check_semantic_cache from pathlib import Path @@ -212,7 +236,9 @@ if cached_nodes or cached_edges or cached_hyperedges: Path('.graphify_cached.json').write_text(json.dumps({'nodes': cached_nodes, 'edges': cached_edges, 'hyperedges': cached_hyperedges})) Path('.graphify_uncached.txt').write_text('\n'.join(uncached)) print(f'Cache: {len(all_files)-len(uncached)} files hit, {len(uncached)} files need extraction') -" +'@ | Out-File -FilePath .graphify_step_3_extract_entities_and_relations_5.py -Encoding utf8 +python .graphify_step_3_extract_entities_and_relations_5.py +Remove-Item -ErrorAction SilentlyContinue .graphify_step_3_extract_entities_and_relations_5.py ``` Only dispatch subagents for files listed in `.graphify_uncached.txt`. If all files are cached, skip to Part C directly. @@ -331,7 +357,7 @@ print(f'Merged {len(chunks)} chunks: {total_in:,} in / {total_out:,} out tokens' Save new results to cache: ```powershell -python -c " +@' import json from graphify.cache import save_semantic_cache from pathlib import Path @@ -339,12 +365,14 @@ from pathlib import Path new = json.loads(Path('.graphify_semantic_new.json').read_text()) if Path('.graphify_semantic_new.json').exists() else {'nodes':[],'edges':[],'hyperedges':[]} saved = save_semantic_cache(new.get('nodes', []), new.get('edges', []), new.get('hyperedges', [])) print(f'Cached {saved} files') -" +'@ | Out-File -FilePath .graphify_step_3_extract_entities_and_relations_6.py -Encoding utf8 +python .graphify_step_3_extract_entities_and_relations_6.py +Remove-Item -ErrorAction SilentlyContinue .graphify_step_3_extract_entities_and_relations_6.py ``` Merge cached + new results into `.graphify_semantic.json`: ```powershell -python -c " +@' import json from pathlib import Path @@ -369,15 +397,17 @@ merged = { 'output_tokens': new.get('output_tokens', 0), } Path('.graphify_semantic.json').write_text(json.dumps(merged, indent=2)) -print(f'Extraction complete - {len(deduped)} nodes, {len(all_edges)} edges ({len(cached[\"nodes\"])} from cache, {len(new.get(\"nodes\",[]))} new)') -" +print(f'Extraction complete - {len(deduped)} nodes, {len(all_edges)} edges ({len(cached["nodes"])} from cache, {len(new.get("nodes",[]))} new)') +'@ | Out-File -FilePath .graphify_step_3_extract_entities_and_relations_7.py -Encoding utf8 +python .graphify_step_3_extract_entities_and_relations_7.py +Remove-Item -ErrorAction SilentlyContinue .graphify_step_3_extract_entities_and_relations_7.py ``` Clean up temp files: `Remove-Item -ErrorAction SilentlyContinue .graphify_cached.json, .graphify_uncached.txt, .graphify_semantic_new.json` #### Part C - Merge AST + semantic into final extraction ```powershell -python -c " +@' import sys, json from pathlib import Path @@ -404,15 +434,17 @@ merged = { Path('.graphify_extract.json').write_text(json.dumps(merged, indent=2)) total = len(merged_nodes) edges = len(merged_edges) -print(f'Merged: {total} nodes, {edges} edges ({len(ast[\"nodes\"])} AST + {len(sem[\"nodes\"])} semantic)') -" +print(f'Merged: {total} nodes, {edges} edges ({len(ast["nodes"])} AST + {len(sem["nodes"])} semantic)') +'@ | Out-File -FilePath .graphify_step_3_extract_entities_and_relations_8.py -Encoding utf8 +python .graphify_step_3_extract_entities_and_relations_8.py +Remove-Item -ErrorAction SilentlyContinue .graphify_step_3_extract_entities_and_relations_8.py ``` ### Step 4 - Build graph, cluster, analyze, generate outputs ```powershell New-Item -ItemType Directory -Force -Path graphify-out | Out-Null -python -c " +@' import sys, json from graphify.build import build_from_json from graphify.cluster import cluster, score_all @@ -451,7 +483,9 @@ if G.number_of_nodes() == 0: print('Possible causes: all files were skipped, binary-only corpus, or extraction failed.') raise SystemExit(1) print(f'Graph: {G.number_of_nodes()} nodes, {G.number_of_edges()} edges, {len(communities)} communities') -" +'@ | Out-File -FilePath .graphify_step_4_build_graph_cluster_analyze_ge_9.py -Encoding utf8 +python .graphify_step_4_build_graph_cluster_analyze_ge_9.py +Remove-Item -ErrorAction SilentlyContinue .graphify_step_4_build_graph_cluster_analyze_ge_9.py ``` If this step prints `ERROR: Graph is empty`, stop and tell the user what happened - do not proceed to labeling or visualization. @@ -465,7 +499,7 @@ Read `.graphify_analysis.json`. For each community key, look at its node labels Then regenerate the report and save the labels for the visualizer: ```powershell -python -c " +@' import sys, json from graphify.build import build_from_json from graphify.cluster import score_all @@ -492,7 +526,9 @@ report = generate(G, communities, cohesion, labels, analysis['gods'], analysis[' Path('graphify-out/GRAPH_REPORT.md').write_text(report) Path('.graphify_labels.json').write_text(json.dumps({str(k): v for k, v in labels.items()})) print('Report updated with community labels') -" +'@ | Out-File -FilePath .graphify_step_5_label_communities_10.py -Encoding utf8 +python .graphify_step_5_label_communities_10.py +Remove-Item -ErrorAction SilentlyContinue .graphify_step_5_label_communities_10.py ``` Replace `LABELS_DICT` with the actual dict you constructed (e.g. `{0: "Attention Mechanism", 1: "Training Pipeline"}`). @@ -507,7 +543,7 @@ If `--obsidian` was given: - If `--obsidian-dir ` was also given, use that path as the vault directory. Otherwise default to `graphify-out/obsidian`. ```powershell -python -c " +@' import sys, json from graphify.build import build_from_json from graphify.export import to_obsidian, to_canvas @@ -534,13 +570,15 @@ print(f'Open {obsidian_dir}/ as a vault in Obsidian.') print(' Graph view - nodes colored by community (set automatically)') print(' graph.canvas - structured layout with communities as groups') print(' _COMMUNITY_* - overview notes with cohesion scores and dataview queries') -" +'@ | Out-File -FilePath .graphify_step_6_generate_obsidian_vault_opt_in_11.py -Encoding utf8 +python .graphify_step_6_generate_obsidian_vault_opt_in_11.py +Remove-Item -ErrorAction SilentlyContinue .graphify_step_6_generate_obsidian_vault_opt_in_11.py ``` Generate the HTML graph (always, unless `--no-viz`): ```powershell -python -c " +@' import sys, json from graphify.build import build_from_json from graphify.export import to_html @@ -559,7 +597,9 @@ if G.number_of_nodes() > 5000: else: to_html(G, communities, 'graphify-out/graph.html', community_labels=labels or None) print('graph.html written - open in any browser, no server needed') -" +'@ | Out-File -FilePath .graphify_step_6_generate_obsidian_vault_opt_in_12.py -Encoding utf8 +python .graphify_step_6_generate_obsidian_vault_opt_in_12.py +Remove-Item -ErrorAction SilentlyContinue .graphify_step_6_generate_obsidian_vault_opt_in_12.py ``` ### Step 7 - Neo4j export (only if --neo4j or --neo4j-push flag) @@ -567,7 +607,7 @@ else: **If `--neo4j`** - generate a Cypher file for manual import: ```powershell -python -c " +@' import sys, json from graphify.build import build_from_json from graphify.export import to_cypher @@ -576,13 +616,15 @@ from pathlib import Path G = build_from_json(json.loads(Path('.graphify_extract.json').read_text())) to_cypher(G, 'graphify-out/cypher.txt') print('cypher.txt written - import with: cypher-shell < graphify-out/cypher.txt') -" +'@ | Out-File -FilePath .graphify_step_7_neo4j_export_only_if_neo4j_or__13.py -Encoding utf8 +python .graphify_step_7_neo4j_export_only_if_neo4j_or__13.py +Remove-Item -ErrorAction SilentlyContinue .graphify_step_7_neo4j_export_only_if_neo4j_or__13.py ``` **If `--neo4j-push `** - push directly to a running Neo4j instance. Ask the user for credentials if not provided: ```powershell -python -c " +@' import sys, json from graphify.build import build_from_json from graphify.cluster import cluster @@ -595,8 +637,10 @@ G = build_from_json(extraction) communities = {int(k): v for k, v in analysis['communities'].items()} result = push_to_neo4j(G, uri='NEO4J_URI', user='NEO4J_USER', password='NEO4J_PASSWORD', communities=communities) -print(f'Pushed to Neo4j: {result[\"nodes\"]} nodes, {result[\"edges\"]} edges') -" +print(f'Pushed to Neo4j: {result["nodes"]} nodes, {result["edges"]} edges') +'@ | Out-File -FilePath .graphify_step_7_neo4j_export_only_if_neo4j_or__14.py -Encoding utf8 +python .graphify_step_7_neo4j_export_only_if_neo4j_or__14.py +Remove-Item -ErrorAction SilentlyContinue .graphify_step_7_neo4j_export_only_if_neo4j_or__14.py ``` Replace `NEO4J_URI`, `NEO4J_USER`, `NEO4J_PASSWORD` with actual values. Default URI is `bolt://localhost:7687`, default user is `neo4j`. Uses MERGE - safe to re-run without creating duplicates. @@ -604,7 +648,7 @@ Replace `NEO4J_URI`, `NEO4J_USER`, `NEO4J_PASSWORD` with actual values. Default ### Step 7b - SVG export (only if --svg flag) ```powershell -python -c " +@' import sys, json from graphify.build import build_from_json from graphify.export import to_svg @@ -620,13 +664,15 @@ labels = {int(k): v for k, v in labels_raw.items()} to_svg(G, communities, 'graphify-out/graph.svg', community_labels=labels or None) print('graph.svg written - embeds in Obsidian, Notion, GitHub READMEs') -" +'@ | Out-File -FilePath .graphify_step_7b_svg_export_only_if_svg_flag_15.py -Encoding utf8 +python .graphify_step_7b_svg_export_only_if_svg_flag_15.py +Remove-Item -ErrorAction SilentlyContinue .graphify_step_7b_svg_export_only_if_svg_flag_15.py ``` ### Step 7c - GraphML export (only if --graphml flag) ```powershell -python -c " +@' import json from graphify.build import build_from_json from graphify.export import to_graphml @@ -640,7 +686,9 @@ communities = {int(k): v for k, v in analysis['communities'].items()} to_graphml(G, communities, 'graphify-out/graph.graphml') print('graph.graphml written - open in Gephi, yEd, or any GraphML tool') -" +'@ | Out-File -FilePath .graphify_step_7c_graphml_export_only_if_graphml_16.py -Encoding utf8 +python .graphify_step_7c_graphml_export_only_if_graphml_16.py +Remove-Item -ErrorAction SilentlyContinue .graphify_step_7c_graphml_export_only_if_graphml_16.py ``` ### Step 7d - MCP server (only if --mcp flag) @@ -668,7 +716,7 @@ To configure in Claude Desktop, add to `claude_desktop_config.json`: If `total_words` from `.graphify_detect.json` is greater than 5,000, run: ```powershell -python -c " +@' import json from graphify.benchmark import run_benchmark, print_benchmark from pathlib import Path @@ -676,7 +724,9 @@ from pathlib import Path detection = json.loads(Path('.graphify_detect.json').read_text()) result = run_benchmark('graphify-out/graph.json', corpus_words=detection['total_words']) print_benchmark(result) -" +'@ | Out-File -FilePath .graphify_step_8_token_reduction_benchmark_only_17.py -Encoding utf8 +python .graphify_step_8_token_reduction_benchmark_only_17.py +Remove-Item -ErrorAction SilentlyContinue .graphify_step_8_token_reduction_benchmark_only_17.py ``` Print the output directly in chat. If `total_words <= 5000`, skip silently - the graph value is structural clarity, not token compression, for small corpora. @@ -686,7 +736,7 @@ Print the output directly in chat. If `total_words <= 5000`, skip silently - the ### Step 9 - Save manifest, update cost tracker, clean up, and report ```powershell -python -c " +@' import json from pathlib import Path from datetime import datetime, timezone @@ -718,8 +768,10 @@ cost['total_output_tokens'] += output_tok cost_path.write_text(json.dumps(cost, indent=2)) print(f'This run: {input_tok:,} input tokens, {output_tok:,} output tokens') -print(f'All time: {cost[\"total_input_tokens\"]:,} input, {cost[\"total_output_tokens\"]:,} output ({len(cost[\"runs\"])} runs)') -" +print(f'All time: {cost["total_input_tokens"]:,} input, {cost["total_output_tokens"]:,} output ({len(cost["runs"])} runs)') +'@ | Out-File -FilePath .graphify_step_9_save_manifest_update_cost_trac_18.py -Encoding utf8 +python .graphify_step_9_save_manifest_update_cost_trac_18.py +Remove-Item -ErrorAction SilentlyContinue .graphify_step_9_save_manifest_update_cost_trac_18.py Remove-Item -ErrorAction SilentlyContinue .graphify_detect.json, .graphify_extract.json, .graphify_ast.json, .graphify_semantic.json, .graphify_analysis.json, .graphify_labels.json Remove-Item -ErrorAction SilentlyContinue graphify-out/.needs_update ``` @@ -760,7 +812,7 @@ The graph is the map. Your job after the pipeline is to be the guide. Use when you've added or modified files since the last run. Only re-extracts changed files - saves tokens and time. ```powershell -python -c " +@' import sys, json from graphify.detect import detect_incremental, save_manifest from pathlib import Path @@ -773,13 +825,15 @@ if new_total == 0: print('No files changed since last run. Nothing to update.') raise SystemExit(0) print(f'{new_total} new/changed file(s) to re-extract.') -" +'@ | Out-File -FilePath .graphify_step_for_update_incremental_re_extracti_19.py -Encoding utf8 +python .graphify_step_for_update_incremental_re_extracti_19.py +Remove-Item -ErrorAction SilentlyContinue .graphify_step_for_update_incremental_re_extracti_19.py ``` If new files exist, first check whether all changed files are code files: ```powershell -python -c " +@' import json from pathlib import Path @@ -789,7 +843,9 @@ new_files = result.get('new_files', {}) all_changed = [f for files in new_files.values() for f in files] code_only = all(Path(f).suffix.lower() in code_exts for f in all_changed) print('code_only:', code_only) -" +'@ | Out-File -FilePath .graphify_step_for_update_incremental_re_extracti_20.py -Encoding utf8 +python .graphify_step_for_update_incremental_re_extracti_20.py +Remove-Item -ErrorAction SilentlyContinue .graphify_step_for_update_incremental_re_extracti_20.py ``` If `code_only` is True: print `[graphify update] Code-only changes detected - skipping semantic extraction (no LLM needed)`, run only Step 3A (AST) on the changed files, skip Step 3B entirely (no subagents), then go straight to merge and Steps 4–8. @@ -799,7 +855,7 @@ If `code_only` is False (any changed file is a doc/paper/image): run the full St Then: ```powershell -python -c " +@' import sys, json from graphify.build import build_from_json from graphify.export import to_json @@ -837,7 +893,9 @@ print(f'Merged: {G_existing.number_of_nodes()} nodes, {G_existing.number_of_edge from graphify.detect import save_manifest save_manifest(incremental['files']) print('[graphify update] Manifest saved.') -" +'@ | Out-File -FilePath .graphify_step_for_update_incremental_re_extracti_21.py -Encoding utf8 +python .graphify_step_for_update_incremental_re_extracti_21.py +Remove-Item -ErrorAction SilentlyContinue .graphify_step_for_update_incremental_re_extracti_21.py ``` Then run Steps 4–8 on the merged graph as normal. @@ -845,7 +903,7 @@ Then run Steps 4–8 on the merged graph as normal. After Step 4, show the graph diff: ```powershell -python -c " +@' import json from graphify.analyze import graph_diff from graphify.build import build_from_json @@ -866,7 +924,9 @@ if old_data: print('New nodes:', ', '.join(n['label'] for n in diff['new_nodes'][:5])) if diff['new_edges']: print('New edges:', len(diff['new_edges'])) -" +'@ | Out-File -FilePath .graphify_step_for_update_incremental_re_extracti_22.py -Encoding utf8 +python .graphify_step_for_update_incremental_re_extracti_22.py +Remove-Item -ErrorAction SilentlyContinue .graphify_step_for_update_incremental_re_extracti_22.py ``` Before the merge step, save the old graph: `Copy-Item graphify-out/graph.json .graphify_old.json` @@ -879,7 +939,7 @@ Clean up after: `Remove-Item -ErrorAction SilentlyContinue .graphify_old.json` Skip Steps 1–3. Load the existing graph from `graphify-out/graph.json` and re-run clustering: ```powershell -python -c " +@' import sys, json from graphify.cluster import cluster, score_all from graphify.analyze import god_nodes, surprising_connections @@ -914,7 +974,9 @@ analysis = { } Path('.graphify_analysis.json').write_text(json.dumps(analysis, indent=2)) print(f'Re-clustered: {len(communities)} communities') -" +'@ | Out-File -FilePath .graphify_step_for_cluster_only_23.py -Encoding utf8 +python .graphify_step_for_cluster_only_23.py +Remove-Item -ErrorAction SilentlyContinue .graphify_step_for_cluster_only_23.py ``` Then run Steps 5–9 as normal (label communities, generate viz, benchmark, clean up, report). @@ -932,12 +994,14 @@ Two traversal modes - choose based on the question: First check the graph exists: ```powershell -python -c " +@' from pathlib import Path if not Path('graphify-out/graph.json').exists(): print('ERROR: No graph found. Run /graphify first to build the graph.') raise SystemExit(1) -" +'@ | Out-File -FilePath .graphify_step_for_graphify_query_24.py -Encoding utf8 +python .graphify_step_for_graphify_query_24.py +Remove-Item -ErrorAction SilentlyContinue .graphify_step_for_graphify_query_24.py ``` If it fails, stop and tell the user to run `/graphify ` first. @@ -950,7 +1014,7 @@ Load `graphify-out/graph.json`, then: 5. If the graph lacks enough information, say so - do not hallucinate edges. ```powershell -python -c " +@' import sys, json from networkx.readwrite import json_graph import networkx as nx @@ -1020,20 +1084,22 @@ def relevance(nid): ranked_nodes = sorted(subgraph_nodes, key=relevance, reverse=True) -lines = [f'Traversal: {mode.upper()} | Start: {[G.nodes[n].get(\"label\",n) for n in start_nodes]} | {len(subgraph_nodes)} nodes'] +lines = [f'Traversal: {mode.upper()} | Start: {[G.nodes[n].get("label",n) for n in start_nodes]} | {len(subgraph_nodes)} nodes'] for nid in ranked_nodes: d = G.nodes[nid] - lines.append(f' NODE {d.get(\"label\", nid)} [src={d.get(\"source_file\",\"\")} loc={d.get(\"source_location\",\"\")}]') + lines.append(f' NODE {d.get("label", nid)} [src={d.get("source_file","")} loc={d.get("source_location","")}]') for u, v in subgraph_edges: if u in subgraph_nodes and v in subgraph_nodes: d = G.edges[u, v] - lines.append(f' EDGE {G.nodes[u].get(\"label\",u)} --{d.get(\"relation\",\"\")} [{d.get(\"confidence\",\"\")}]--> {G.nodes[v].get(\"label\",v)}') + lines.append(f' EDGE {G.nodes[u].get("label",u)} --{d.get("relation","")} [{d.get("confidence","")}]--> {G.nodes[v].get("label",v)}') output = '\n'.join(lines) if len(output) > char_budget: output = output[:char_budget] + f'\n... (truncated at ~{token_budget} token budget - use --budget N for more)' print(output) -" +'@ | Out-File -FilePath .graphify_step_for_graphify_query_25.py -Encoding utf8 +python .graphify_step_for_graphify_query_25.py +Remove-Item -ErrorAction SilentlyContinue .graphify_step_for_graphify_query_25.py ``` Replace `QUESTION` with the user's actual question, `MODE` with `bfs` or `dfs`, and `BUDGET` with the token budget (default `2000`, or whatever `--budget N` specifies). Then answer based on the subgraph output above. @@ -1054,17 +1120,19 @@ Find the shortest path between two named concepts in the graph. First check the graph exists: ```powershell -python -c " +@' from pathlib import Path if not Path('graphify-out/graph.json').exists(): print('ERROR: No graph found. Run /graphify first to build the graph.') raise SystemExit(1) -" +'@ | Out-File -FilePath .graphify_step_for_graphify_path_26.py -Encoding utf8 +python .graphify_step_for_graphify_path_26.py +Remove-Item -ErrorAction SilentlyContinue .graphify_step_for_graphify_path_26.py ``` If it fails, stop and tell the user to run `/graphify ` first. ```powershell -python -c " +@' import json, sys import networkx as nx from networkx.readwrite import json_graph @@ -1108,7 +1176,9 @@ except nx.NetworkXNoPath: print(f'No path found between {a_term!r} and {b_term!r}') except nx.NodeNotFound as e: print(f'Node not found: {e}') -" +'@ | Out-File -FilePath .graphify_step_for_graphify_path_27.py -Encoding utf8 +python .graphify_step_for_graphify_path_27.py +Remove-Item -ErrorAction SilentlyContinue .graphify_step_for_graphify_path_27.py ``` Replace `NODE_A` and `NODE_B` with the actual concept names from the user. Then explain the path in plain language - what each hop means, why it's significant. @@ -1127,17 +1197,19 @@ Give a plain-language explanation of a single node - everything connected to it. First check the graph exists: ```powershell -python -c " +@' from pathlib import Path if not Path('graphify-out/graph.json').exists(): print('ERROR: No graph found. Run /graphify first to build the graph.') raise SystemExit(1) -" +'@ | Out-File -FilePath .graphify_step_for_graphify_explain_28.py -Encoding utf8 +python .graphify_step_for_graphify_explain_28.py +Remove-Item -ErrorAction SilentlyContinue .graphify_step_for_graphify_explain_28.py ``` If it fails, stop and tell the user to run `/graphify ` first. ```powershell -python -c " +@' import json, sys import networkx as nx from networkx.readwrite import json_graph @@ -1161,9 +1233,9 @@ if not scored or scored[0][0] == 0: nid = scored[0][1] data_n = G.nodes[nid] -print(f'NODE: {data_n.get(\"label\", nid)}') -print(f' source: {data_n.get(\"source_file\",\"unknown\")}') -print(f' type: {data_n.get(\"file_type\",\"unknown\")}') +print(f'NODE: {data_n.get("label", nid)}') +print(f' source: {data_n.get("source_file","unknown")}') +print(f' type: {data_n.get("file_type","unknown")}') print(f' degree: {G.degree(nid)}') print() print('CONNECTIONS:') @@ -1174,7 +1246,9 @@ for neighbor in G.neighbors(nid): conf = edge.get('confidence', '') src_file = G.nodes[neighbor].get('source_file', '') print(f' --{rel}--> {nlabel} [{conf}] ({src_file})') -" +'@ | Out-File -FilePath .graphify_step_for_graphify_explain_29.py -Encoding utf8 +python .graphify_step_for_graphify_explain_29.py +Remove-Item -ErrorAction SilentlyContinue .graphify_step_for_graphify_explain_29.py ``` Replace `NODE_NAME` with the concept the user asked about. Then write a 3-5 sentence explanation of what this node is, what it connects to, and why those connections are significant. Use the source locations as citations. @@ -1192,7 +1266,7 @@ python -m graphify save-result --question "Explain NODE_NAME" --answer "ANSWER" Fetch a URL and add it to the corpus, then update the graph. ```powershell -python -c " +@' import sys from graphify.ingest import ingest from pathlib import Path @@ -1206,7 +1280,9 @@ except ValueError as e: except RuntimeError as e: print(f'error: {e}', file=sys.stderr) sys.exit(1) -" +'@ | Out-File -FilePath .graphify_step_for_graphify_add_30.py -Encoding utf8 +python .graphify_step_for_graphify_add_30.py +Remove-Item -ErrorAction SilentlyContinue .graphify_step_for_graphify_add_30.py ``` Replace `URL` with the actual URL, `AUTHOR` with the user's name if provided, `CONTRIBUTOR` likewise. If the command exits with an error, tell the user what went wrong - do not silently continue. After a successful save, automatically run the `--update` pipeline on `./raw` to merge the new file into the existing graph. diff --git a/tests/test_benchmark.py b/tests/test_benchmark.py index d5e1808..a1e3a4c 100644 --- a/tests/test_benchmark.py +++ b/tests/test_benchmark.py @@ -5,7 +5,7 @@ import pytest import networkx as nx from networkx.readwrite import json_graph -from graphify.benchmark import run_benchmark, print_benchmark, _query_subgraph_tokens, _SAMPLE_QUESTIONS +from graphify.benchmark import run_benchmark, print_benchmark, _query_subgraph_tokens, _SAMPLE_QUESTIONS, _safe, _hr def _make_graph() -> nx.Graph: @@ -117,3 +117,49 @@ def test_print_benchmark_error_message(capsys): print_benchmark({"error": "test error message"}) out = capsys.readouterr().out assert "test error message" in out + + +# --- cp1252 / Windows-console encoding compatibility (regression for #?) --- +# print_benchmark previously crashed on Windows consoles (cp1252) because it +# unconditionally printed U+2500 and U+2192. _safe() falls back to ASCII when +# stdout cannot encode the glyph. + +def test_safe_returns_unicode_when_encodable(): + import io, sys + real_stdout = sys.stdout + try: + sys.stdout = io.TextIOWrapper(io.BytesIO(), encoding="utf-8") + assert _safe("→", "->") == "→" + assert _hr(5) == "─" * 5 + finally: + sys.stdout = real_stdout + +def test_safe_falls_back_when_unencodable(): + import io, sys + real_stdout = sys.stdout + try: + sys.stdout = io.TextIOWrapper(io.BytesIO(), encoding="cp1252") + assert _safe("→", "->") == "->" + assert _hr(5) == "-" * 5 + finally: + sys.stdout = real_stdout + +def test_print_benchmark_survives_cp1252_stdout(tmp_path, monkeypatch, capsys): + """Regression: U+2500 / U+2192 used to crash with UnicodeEncodeError on cp1252.""" + import io, sys + G = _make_graph() + graph_file = tmp_path / "graph.json" + _write_graph(G, graph_file) + result = run_benchmark(str(graph_file), corpus_words=5_000) + + # Replace stdout with a strict cp1252 stream — same behaviour as the + # legacy Windows console that surfaced this bug. + cp1252_stdout = io.TextIOWrapper(io.BytesIO(), encoding="cp1252", errors="strict") + monkeypatch.setattr(sys, "stdout", cp1252_stdout) + print_benchmark(result) # must not raise UnicodeEncodeError + cp1252_stdout.flush() + written = cp1252_stdout.buffer.getvalue().decode("cp1252") + assert "reduction" in written.lower() + # ASCII fallbacks must be present, fancy glyphs must not. + assert "─" not in written + assert "→" not in written diff --git a/tests/test_extract.py b/tests/test_extract.py index dd062b8..a0b897c 100644 --- a/tests/test_extract.py +++ b/tests/test_extract.py @@ -365,3 +365,63 @@ def test_extract_tsx_uses_tsx_grammar(): from graphify.extract import _TSX_CONFIG, _TS_CONFIG assert _TSX_CONFIG.ts_language_fn == "language_tsx" assert _TS_CONFIG.ts_language_fn == "language_typescript" + + +# --- Windows-spawn ProcessPool fallback (regression for #?) --- +# When the caller has no `if __name__ == "__main__":` guard, ProcessPoolExecutor +# on Windows raises BrokenProcessPool before any work completes. extract() must +# detect this, warn, and fall back to sequential extraction rather than +# propagating a 290-line traceback. + +def test_extract_falls_back_to_sequential_when_parallel_returns_false(tmp_path, monkeypatch): + """extract() must run sequential when _extract_parallel signals failure (returns False).""" + from graphify import extract as extract_mod + + files = [FIXTURES / "sample.py"] * 25 # >= _PARALLEL_THRESHOLD triggers parallel branch + cache_root = tmp_path / "cache" + cache_root.mkdir() + + calls = {"parallel": 0, "sequential": 0} + real_sequential = extract_mod._extract_sequential + + def fake_parallel(uncached_work, per_file, effective_root, max_workers, total_files): + calls["parallel"] += 1 + return False # simulate the post-fix BrokenProcessPool branch + + def wrapped_sequential(*args, **kwargs): + calls["sequential"] += 1 + return real_sequential(*args, **kwargs) + + monkeypatch.setattr(extract_mod, "_extract_parallel", fake_parallel) + monkeypatch.setattr(extract_mod, "_extract_sequential", wrapped_sequential) + + result = extract_mod.extract(files, cache_root=cache_root) + assert calls["parallel"] == 1, "parallel path should have been attempted once" + assert calls["sequential"] == 1, "sequential fallback should have run exactly once" + assert result["nodes"], "extract should still produce nodes after fallback" + + +def test_extract_parallel_returns_false_on_broken_pool(tmp_path, monkeypatch, capsys): + """_extract_parallel must catch BrokenProcessPool internally and return False.""" + from concurrent.futures.process import BrokenProcessPool + import concurrent.futures + from graphify import extract as extract_mod + + class FakePool: + def __init__(self, *a, **kw): pass + def __enter__(self): return self + def __exit__(self, *a): return False + def submit(self, *a, **kw): + raise BrokenProcessPool("simulated spawn failure") + + monkeypatch.setattr( + concurrent.futures, "ProcessPoolExecutor", lambda *a, **kw: FakePool() + ) + + uncached = [(0, FIXTURES / "sample.py")] + per_file: list = [None] + ok = extract_mod._extract_parallel(uncached, per_file, tmp_path, 2, 1) + assert ok is False, "function should report failure via return value, not raise" + out = capsys.readouterr().out + assert "BrokenProcessPool" in out, "user-facing warning must mention the failure" + assert "__main__" in out, "warning must hint at the Windows __main__ guard idiom"