fix(windows): unblock pipeline on Windows consoles + missing __main__ guards (#788)
Three independent Windows compatibility fixes shipped together because they
all surface during the same first /graphify run on Windows.
graphify/benchmark.py
print_benchmark() unconditionally printed U+2500 (box-drawing) and U+2192
(rightwards arrow), which UnicodeEncodeError'd on stdouts that can't encode
them — most notably the legacy Windows console at cp1252. New _safe()
helper falls back to ASCII when the active stdout encoding can't carry the
glyph; _hr() uses it. Two regression tests cover both paths and prove
print_benchmark survives a cp1252-strict stream.
graphify/extract.py
ProcessPoolExecutor on Windows uses spawn, so worker subprocesses
re-import the calling __main__. When the caller is `python -c "..."` or a
script without an `if __name__ == "__main__":` guard, the workers
recursively spawn themselves and the pool dies. The user-visible failure
was a 290-line traceback ending in BrokenProcessPool, hiding the actual
cause. _extract_parallel now catches BrokenProcessPool, prints a one-line
warning that names the __main__-guard idiom, and returns False so the
public extract() routes to the existing _extract_sequential fallback. Two
tests cover the parallel-returns-False contract and the sequential
fallback wiring.
graphify/skill-windows.md
Every `python -c "..."` block (30 in total) is replaced with a
Write+run+delete pattern using PowerShell's literal here-string @'...'@.
The old form was a quote-escaping minefield: any double-quote inside the
Python source had to be backslash-escaped for the shell, and PowerShell's
parser ate them inconsistently — failing on f-strings like
`f'AST: {len(result["nodes"])} nodes'`. The new form passes Python source
to disk literally, so what the model writes is what Python sees. The AST
step's script template now includes an explicit `if __name__ == "__main__":`
guard so multi-core extraction works even before the runtime fallback above
kicks in. All 31 resulting heredoc blocks parse cleanly under
`ast.parse`.
Co-authored-by: Nauman Hameed <Nauman.Hameed@enghouse.com>
This commit is contained in:
co-authored by
Nauman Hameed
parent
f88567b114
commit
e5f263ba98
+23
-2
@@ -1,6 +1,7 @@
|
||||
"""Token-reduction benchmark - measures how much context graphify saves vs naive full-corpus approach."""
|
||||
from __future__ import annotations
|
||||
import json
|
||||
import sys
|
||||
from pathlib import Path
|
||||
import networkx as nx
|
||||
from networkx.readwrite import json_graph
|
||||
@@ -9,6 +10,25 @@ from networkx.readwrite import json_graph
|
||||
_CHARS_PER_TOKEN = 4 # standard approximation
|
||||
|
||||
|
||||
def _safe(unicode_char: str, ascii_fallback: str) -> str:
|
||||
"""Return unicode_char if stdout can encode it, else ascii_fallback.
|
||||
|
||||
Windows consoles often default to cp1252 which cannot encode box-drawing
|
||||
or arrow glyphs; printing them raises UnicodeEncodeError mid-output.
|
||||
"""
|
||||
encoding = getattr(sys.stdout, "encoding", None) or ""
|
||||
try:
|
||||
unicode_char.encode(encoding)
|
||||
return unicode_char
|
||||
except (UnicodeEncodeError, LookupError):
|
||||
return ascii_fallback
|
||||
|
||||
|
||||
def _hr(width: int = 50) -> str:
|
||||
"""Horizontal rule that survives non-UTF-8 stdout (e.g. Windows cp1252 console)."""
|
||||
return _safe("─", "-") * width
|
||||
|
||||
|
||||
def _estimate_tokens(text: str) -> int:
|
||||
return max(1, len(text) // _CHARS_PER_TOKEN)
|
||||
|
||||
@@ -118,8 +138,9 @@ def print_benchmark(result: dict) -> None:
|
||||
return
|
||||
|
||||
print(f"\ngraphify token reduction benchmark")
|
||||
print(f"{'─' * 50}")
|
||||
print(f" Corpus: {result['corpus_words']:,} words → ~{result['corpus_tokens']:,} tokens (naive)")
|
||||
print(_hr(50))
|
||||
arrow = _safe("→", "->")
|
||||
print(f" Corpus: {result['corpus_words']:,} words {arrow} ~{result['corpus_tokens']:,} tokens (naive)")
|
||||
print(f" Graph: {result['nodes']:,} nodes, {result['edges']:,} edges")
|
||||
print(f" Avg query cost: ~{result['avg_query_tokens']:,} tokens")
|
||||
print(f" Reduction: {result['reduction_ratio']}x fewer tokens per query")
|
||||
|
||||
+44
-21
@@ -4660,8 +4660,14 @@ def _extract_parallel(
|
||||
effective_root: Path,
|
||||
max_workers: int | None,
|
||||
total_files: int,
|
||||
) -> None:
|
||||
"""Extract uncached files in parallel using ProcessPoolExecutor."""
|
||||
) -> bool:
|
||||
"""Extract uncached files in parallel using ProcessPoolExecutor.
|
||||
|
||||
Returns True if the pool ran to completion. Returns False if the pool
|
||||
failed in a recoverable way (typically Windows-spawn without an
|
||||
``if __name__ == "__main__"`` guard in the calling script, which causes
|
||||
BrokenProcessPool); the caller should fall back to sequential extraction.
|
||||
"""
|
||||
import concurrent.futures
|
||||
|
||||
if max_workers is None:
|
||||
@@ -4672,28 +4678,44 @@ def _extract_parallel(
|
||||
|
||||
done_count = 0
|
||||
_PROGRESS_INTERVAL = 100
|
||||
with concurrent.futures.ProcessPoolExecutor(max_workers=max_workers) as pool:
|
||||
futures = {
|
||||
pool.submit(_extract_single_file, item): item[0] for item in work_items
|
||||
}
|
||||
for future in concurrent.futures.as_completed(futures):
|
||||
idx, result = future.result()
|
||||
per_file[idx] = result
|
||||
done_count += 1
|
||||
if (
|
||||
total_files >= _PROGRESS_INTERVAL
|
||||
and done_count % _PROGRESS_INTERVAL == 0
|
||||
):
|
||||
print(
|
||||
f" AST extraction: {done_count}/{len(uncached_work)} uncached files "
|
||||
f"({done_count * 100 // len(uncached_work)}%) [{max_workers} workers]",
|
||||
flush=True,
|
||||
)
|
||||
try:
|
||||
with concurrent.futures.ProcessPoolExecutor(max_workers=max_workers) as pool:
|
||||
futures = {
|
||||
pool.submit(_extract_single_file, item): item[0] for item in work_items
|
||||
}
|
||||
for future in concurrent.futures.as_completed(futures):
|
||||
idx, result = future.result()
|
||||
per_file[idx] = result
|
||||
done_count += 1
|
||||
if (
|
||||
total_files >= _PROGRESS_INTERVAL
|
||||
and done_count % _PROGRESS_INTERVAL == 0
|
||||
):
|
||||
print(
|
||||
f" AST extraction: {done_count}/{len(uncached_work)} uncached files "
|
||||
f"({done_count * 100 // len(uncached_work)}%) [{max_workers} workers]",
|
||||
flush=True,
|
||||
)
|
||||
except concurrent.futures.process.BrokenProcessPool:
|
||||
# On Windows (spawn start method) the worker subprocesses re-import the
|
||||
# caller's __main__. Inline invocations like `python -c "..."` have no
|
||||
# __main__ guard, so worker bootstrap raises and the pool dies before
|
||||
# any work completes. Fall back to in-process sequential extraction —
|
||||
# slower but correct.
|
||||
print(
|
||||
" warning: parallel extraction failed (BrokenProcessPool); "
|
||||
"falling back to sequential. On Windows this usually means the "
|
||||
'caller is missing an `if __name__ == "__main__":` guard. Pass '
|
||||
"parallel=False to extract() to skip the pool entirely.",
|
||||
flush=True,
|
||||
)
|
||||
return False
|
||||
if total_files >= _PROGRESS_INTERVAL:
|
||||
print(
|
||||
f" AST extraction: {total_files}/{total_files} files (100%) [{max_workers} workers]",
|
||||
flush=True,
|
||||
)
|
||||
return True
|
||||
|
||||
|
||||
def _extract_sequential(
|
||||
@@ -4793,11 +4815,12 @@ def extract(
|
||||
|
||||
# Phase 2: extract uncached files (parallel or sequential)
|
||||
if uncached_work:
|
||||
ran_parallel = False
|
||||
if parallel and len(uncached_work) >= _PARALLEL_THRESHOLD:
|
||||
_extract_parallel(
|
||||
ran_parallel = _extract_parallel(
|
||||
uncached_work, per_file, effective_root, max_workers, total
|
||||
)
|
||||
else:
|
||||
if not ran_parallel:
|
||||
_extract_sequential(uncached_work, per_file, effective_root, total)
|
||||
|
||||
# Fill any remaining None slots (shouldn't happen, but defensive)
|
||||
|
||||
+159
-83
@@ -62,10 +62,18 @@ Follow these steps in order. Do not skip steps.
|
||||
|
||||
```powershell
|
||||
# Detect Python and install graphify if needed
|
||||
python -c "import graphify" 2>$null
|
||||
@'
|
||||
import graphify
|
||||
'@ | Out-File -FilePath .graphify_step_1_ensure_graphify_is_installed_1.py -Encoding utf8
|
||||
python .graphify_step_1_ensure_graphify_is_installed_1.py 2>$null
|
||||
Remove-Item -ErrorAction SilentlyContinue .graphify_step_1_ensure_graphify_is_installed_1.py
|
||||
if ($LASTEXITCODE -ne 0) { pip install graphifyy -q 2>&1 | Select-Object -Last 3 }
|
||||
# Write interpreter path for all subsequent steps
|
||||
python -c "import sys; open('.graphify_python', 'w').write(sys.executable)"
|
||||
@'
|
||||
import sys; open('.graphify_python', 'w').write(sys.executable)
|
||||
'@ | Out-File -FilePath .graphify_step_1_ensure_graphify_is_installed_2.py -Encoding utf8
|
||||
python .graphify_step_1_ensure_graphify_is_installed_2.py
|
||||
Remove-Item -ErrorAction SilentlyContinue .graphify_step_1_ensure_graphify_is_installed_2.py
|
||||
```
|
||||
|
||||
If the import succeeds, print nothing and move straight to Step 2.
|
||||
@@ -73,13 +81,15 @@ If the import succeeds, print nothing and move straight to Step 2.
|
||||
### Step 2 - Detect files
|
||||
|
||||
```powershell
|
||||
python -c "
|
||||
@'
|
||||
import json
|
||||
from graphify.detect import detect
|
||||
from pathlib import Path
|
||||
result = detect(Path('INPUT_PATH'))
|
||||
print(json.dumps(result))
|
||||
" > .graphify_detect.json
|
||||
'@ | Out-File -FilePath .graphify_step_2_detect_files_3.py -Encoding utf8
|
||||
python .graphify_step_2_detect_files_3.py > .graphify_detect.json
|
||||
Remove-Item -ErrorAction SilentlyContinue .graphify_step_2_detect_files_3.py
|
||||
```
|
||||
|
||||
Replace INPUT_PATH with the actual path the user provided. Do NOT cat or print the JSON - read it silently and present a clean summary instead:
|
||||
@@ -123,7 +133,7 @@ Set it as `$env:GRAPHIFY_WHISPER_PROMPT` before running the transcription comman
|
||||
**Step 2 - Transcribe (PowerShell):**
|
||||
|
||||
```powershell
|
||||
& (Get-Content graphify-out\.graphify_python) -c "
|
||||
@'
|
||||
import json, os
|
||||
from pathlib import Path
|
||||
from graphify.transcribe import transcribe_all
|
||||
@@ -134,7 +144,9 @@ prompt = os.environ.get('GRAPHIFY_WHISPER_PROMPT', 'Use proper punctuation and p
|
||||
|
||||
transcript_paths = transcribe_all(video_files, initial_prompt=prompt)
|
||||
print(json.dumps(transcript_paths))
|
||||
" | Out-File -FilePath graphify-out\.graphify_transcripts.json -Encoding utf8
|
||||
'@ | Out-File -FilePath .graphify_step_transcribe.py -Encoding utf8
|
||||
& (Get-Content graphify-out\.graphify_python) .graphify_step_transcribe.py | Out-File -FilePath graphify-out\.graphify_transcripts.json -Encoding utf8
|
||||
Remove-Item -ErrorAction SilentlyContinue .graphify_step_transcribe.py
|
||||
```
|
||||
|
||||
After transcription:
|
||||
@@ -160,25 +172,37 @@ Note: Parallelizing AST + semantic saves 5-15s on large corpora. AST is determin
|
||||
For any code files detected, run AST extraction in parallel with Part B subagents:
|
||||
|
||||
```powershell
|
||||
python -c "
|
||||
import sys, json
|
||||
@'
|
||||
import json
|
||||
from graphify.extract import collect_files, extract
|
||||
from pathlib import Path
|
||||
import json
|
||||
|
||||
code_files = []
|
||||
detect = json.loads(Path('.graphify_detect.json').read_text())
|
||||
for f in detect.get('files', {}).get('code', []):
|
||||
code_files.extend(collect_files(Path(f)) if Path(f).is_dir() else [Path(f)])
|
||||
|
||||
if code_files:
|
||||
result = extract(code_files)
|
||||
Path('.graphify_ast.json').write_text(json.dumps(result, indent=2))
|
||||
print(f'AST: {len(result[\"nodes\"])} nodes, {len(result[\"edges\"])} edges')
|
||||
else:
|
||||
Path('.graphify_ast.json').write_text(json.dumps({'nodes':[],'edges':[],'input_tokens':0,'output_tokens':0}))
|
||||
print('No code files - skipping AST extraction')
|
||||
"
|
||||
def main():
|
||||
code_files = []
|
||||
detect = json.loads(Path('.graphify_detect.json').read_text())
|
||||
for f in detect.get('files', {}).get('code', []):
|
||||
code_files.extend(collect_files(Path(f)) if Path(f).is_dir() else [Path(f)])
|
||||
|
||||
if code_files:
|
||||
result = extract(code_files)
|
||||
Path('.graphify_ast.json').write_text(json.dumps(result, indent=2))
|
||||
print(f'AST: {len(result["nodes"])} nodes, {len(result["edges"])} edges')
|
||||
else:
|
||||
Path('.graphify_ast.json').write_text(json.dumps({'nodes':[],'edges':[],'input_tokens':0,'output_tokens':0}))
|
||||
print('No code files - skipping AST extraction')
|
||||
|
||||
|
||||
# Windows-spawn ProcessPoolExecutor (used inside extract()) re-imports this
|
||||
# script in each worker; without an `if __name__ == "__main__":` guard the
|
||||
# pool would recursively spawn itself. graphify v0.7.11+ falls back to
|
||||
# sequential extraction if the pool dies, but the guard keeps multi-core
|
||||
# extraction working on Windows.
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
'@ | Out-File -FilePath .graphify_step_ast.py -Encoding utf8
|
||||
python .graphify_step_ast.py
|
||||
Remove-Item -ErrorAction SilentlyContinue .graphify_step_ast.py
|
||||
```
|
||||
|
||||
#### Part B - Semantic extraction (parallel subagents)
|
||||
@@ -198,7 +222,7 @@ Before dispatching subagents, print a timing estimate:
|
||||
Before dispatching any subagents, check which files already have cached extraction results:
|
||||
|
||||
```powershell
|
||||
python -c "
|
||||
@'
|
||||
import json
|
||||
from graphify.cache import check_semantic_cache
|
||||
from pathlib import Path
|
||||
@@ -212,7 +236,9 @@ if cached_nodes or cached_edges or cached_hyperedges:
|
||||
Path('.graphify_cached.json').write_text(json.dumps({'nodes': cached_nodes, 'edges': cached_edges, 'hyperedges': cached_hyperedges}))
|
||||
Path('.graphify_uncached.txt').write_text('\n'.join(uncached))
|
||||
print(f'Cache: {len(all_files)-len(uncached)} files hit, {len(uncached)} files need extraction')
|
||||
"
|
||||
'@ | Out-File -FilePath .graphify_step_3_extract_entities_and_relations_5.py -Encoding utf8
|
||||
python .graphify_step_3_extract_entities_and_relations_5.py
|
||||
Remove-Item -ErrorAction SilentlyContinue .graphify_step_3_extract_entities_and_relations_5.py
|
||||
```
|
||||
|
||||
Only dispatch subagents for files listed in `.graphify_uncached.txt`. If all files are cached, skip to Part C directly.
|
||||
@@ -331,7 +357,7 @@ print(f'Merged {len(chunks)} chunks: {total_in:,} in / {total_out:,} out tokens'
|
||||
|
||||
Save new results to cache:
|
||||
```powershell
|
||||
python -c "
|
||||
@'
|
||||
import json
|
||||
from graphify.cache import save_semantic_cache
|
||||
from pathlib import Path
|
||||
@@ -339,12 +365,14 @@ from pathlib import Path
|
||||
new = json.loads(Path('.graphify_semantic_new.json').read_text()) if Path('.graphify_semantic_new.json').exists() else {'nodes':[],'edges':[],'hyperedges':[]}
|
||||
saved = save_semantic_cache(new.get('nodes', []), new.get('edges', []), new.get('hyperedges', []))
|
||||
print(f'Cached {saved} files')
|
||||
"
|
||||
'@ | Out-File -FilePath .graphify_step_3_extract_entities_and_relations_6.py -Encoding utf8
|
||||
python .graphify_step_3_extract_entities_and_relations_6.py
|
||||
Remove-Item -ErrorAction SilentlyContinue .graphify_step_3_extract_entities_and_relations_6.py
|
||||
```
|
||||
|
||||
Merge cached + new results into `.graphify_semantic.json`:
|
||||
```powershell
|
||||
python -c "
|
||||
@'
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
@@ -369,15 +397,17 @@ merged = {
|
||||
'output_tokens': new.get('output_tokens', 0),
|
||||
}
|
||||
Path('.graphify_semantic.json').write_text(json.dumps(merged, indent=2))
|
||||
print(f'Extraction complete - {len(deduped)} nodes, {len(all_edges)} edges ({len(cached[\"nodes\"])} from cache, {len(new.get(\"nodes\",[]))} new)')
|
||||
"
|
||||
print(f'Extraction complete - {len(deduped)} nodes, {len(all_edges)} edges ({len(cached["nodes"])} from cache, {len(new.get("nodes",[]))} new)')
|
||||
'@ | Out-File -FilePath .graphify_step_3_extract_entities_and_relations_7.py -Encoding utf8
|
||||
python .graphify_step_3_extract_entities_and_relations_7.py
|
||||
Remove-Item -ErrorAction SilentlyContinue .graphify_step_3_extract_entities_and_relations_7.py
|
||||
```
|
||||
Clean up temp files: `Remove-Item -ErrorAction SilentlyContinue .graphify_cached.json, .graphify_uncached.txt, .graphify_semantic_new.json`
|
||||
|
||||
#### Part C - Merge AST + semantic into final extraction
|
||||
|
||||
```powershell
|
||||
python -c "
|
||||
@'
|
||||
import sys, json
|
||||
from pathlib import Path
|
||||
|
||||
@@ -404,15 +434,17 @@ merged = {
|
||||
Path('.graphify_extract.json').write_text(json.dumps(merged, indent=2))
|
||||
total = len(merged_nodes)
|
||||
edges = len(merged_edges)
|
||||
print(f'Merged: {total} nodes, {edges} edges ({len(ast[\"nodes\"])} AST + {len(sem[\"nodes\"])} semantic)')
|
||||
"
|
||||
print(f'Merged: {total} nodes, {edges} edges ({len(ast["nodes"])} AST + {len(sem["nodes"])} semantic)')
|
||||
'@ | Out-File -FilePath .graphify_step_3_extract_entities_and_relations_8.py -Encoding utf8
|
||||
python .graphify_step_3_extract_entities_and_relations_8.py
|
||||
Remove-Item -ErrorAction SilentlyContinue .graphify_step_3_extract_entities_and_relations_8.py
|
||||
```
|
||||
|
||||
### Step 4 - Build graph, cluster, analyze, generate outputs
|
||||
|
||||
```powershell
|
||||
New-Item -ItemType Directory -Force -Path graphify-out | Out-Null
|
||||
python -c "
|
||||
@'
|
||||
import sys, json
|
||||
from graphify.build import build_from_json
|
||||
from graphify.cluster import cluster, score_all
|
||||
@@ -451,7 +483,9 @@ if G.number_of_nodes() == 0:
|
||||
print('Possible causes: all files were skipped, binary-only corpus, or extraction failed.')
|
||||
raise SystemExit(1)
|
||||
print(f'Graph: {G.number_of_nodes()} nodes, {G.number_of_edges()} edges, {len(communities)} communities')
|
||||
"
|
||||
'@ | Out-File -FilePath .graphify_step_4_build_graph_cluster_analyze_ge_9.py -Encoding utf8
|
||||
python .graphify_step_4_build_graph_cluster_analyze_ge_9.py
|
||||
Remove-Item -ErrorAction SilentlyContinue .graphify_step_4_build_graph_cluster_analyze_ge_9.py
|
||||
```
|
||||
|
||||
If this step prints `ERROR: Graph is empty`, stop and tell the user what happened - do not proceed to labeling or visualization.
|
||||
@@ -465,7 +499,7 @@ Read `.graphify_analysis.json`. For each community key, look at its node labels
|
||||
Then regenerate the report and save the labels for the visualizer:
|
||||
|
||||
```powershell
|
||||
python -c "
|
||||
@'
|
||||
import sys, json
|
||||
from graphify.build import build_from_json
|
||||
from graphify.cluster import score_all
|
||||
@@ -492,7 +526,9 @@ report = generate(G, communities, cohesion, labels, analysis['gods'], analysis['
|
||||
Path('graphify-out/GRAPH_REPORT.md').write_text(report)
|
||||
Path('.graphify_labels.json').write_text(json.dumps({str(k): v for k, v in labels.items()}))
|
||||
print('Report updated with community labels')
|
||||
"
|
||||
'@ | Out-File -FilePath .graphify_step_5_label_communities_10.py -Encoding utf8
|
||||
python .graphify_step_5_label_communities_10.py
|
||||
Remove-Item -ErrorAction SilentlyContinue .graphify_step_5_label_communities_10.py
|
||||
```
|
||||
|
||||
Replace `LABELS_DICT` with the actual dict you constructed (e.g. `{0: "Attention Mechanism", 1: "Training Pipeline"}`).
|
||||
@@ -507,7 +543,7 @@ If `--obsidian` was given:
|
||||
- If `--obsidian-dir <path>` was also given, use that path as the vault directory. Otherwise default to `graphify-out/obsidian`.
|
||||
|
||||
```powershell
|
||||
python -c "
|
||||
@'
|
||||
import sys, json
|
||||
from graphify.build import build_from_json
|
||||
from graphify.export import to_obsidian, to_canvas
|
||||
@@ -534,13 +570,15 @@ print(f'Open {obsidian_dir}/ as a vault in Obsidian.')
|
||||
print(' Graph view - nodes colored by community (set automatically)')
|
||||
print(' graph.canvas - structured layout with communities as groups')
|
||||
print(' _COMMUNITY_* - overview notes with cohesion scores and dataview queries')
|
||||
"
|
||||
'@ | Out-File -FilePath .graphify_step_6_generate_obsidian_vault_opt_in_11.py -Encoding utf8
|
||||
python .graphify_step_6_generate_obsidian_vault_opt_in_11.py
|
||||
Remove-Item -ErrorAction SilentlyContinue .graphify_step_6_generate_obsidian_vault_opt_in_11.py
|
||||
```
|
||||
|
||||
Generate the HTML graph (always, unless `--no-viz`):
|
||||
|
||||
```powershell
|
||||
python -c "
|
||||
@'
|
||||
import sys, json
|
||||
from graphify.build import build_from_json
|
||||
from graphify.export import to_html
|
||||
@@ -559,7 +597,9 @@ if G.number_of_nodes() > 5000:
|
||||
else:
|
||||
to_html(G, communities, 'graphify-out/graph.html', community_labels=labels or None)
|
||||
print('graph.html written - open in any browser, no server needed')
|
||||
"
|
||||
'@ | Out-File -FilePath .graphify_step_6_generate_obsidian_vault_opt_in_12.py -Encoding utf8
|
||||
python .graphify_step_6_generate_obsidian_vault_opt_in_12.py
|
||||
Remove-Item -ErrorAction SilentlyContinue .graphify_step_6_generate_obsidian_vault_opt_in_12.py
|
||||
```
|
||||
|
||||
### Step 7 - Neo4j export (only if --neo4j or --neo4j-push flag)
|
||||
@@ -567,7 +607,7 @@ else:
|
||||
**If `--neo4j`** - generate a Cypher file for manual import:
|
||||
|
||||
```powershell
|
||||
python -c "
|
||||
@'
|
||||
import sys, json
|
||||
from graphify.build import build_from_json
|
||||
from graphify.export import to_cypher
|
||||
@@ -576,13 +616,15 @@ from pathlib import Path
|
||||
G = build_from_json(json.loads(Path('.graphify_extract.json').read_text()))
|
||||
to_cypher(G, 'graphify-out/cypher.txt')
|
||||
print('cypher.txt written - import with: cypher-shell < graphify-out/cypher.txt')
|
||||
"
|
||||
'@ | Out-File -FilePath .graphify_step_7_neo4j_export_only_if_neo4j_or__13.py -Encoding utf8
|
||||
python .graphify_step_7_neo4j_export_only_if_neo4j_or__13.py
|
||||
Remove-Item -ErrorAction SilentlyContinue .graphify_step_7_neo4j_export_only_if_neo4j_or__13.py
|
||||
```
|
||||
|
||||
**If `--neo4j-push <uri>`** - push directly to a running Neo4j instance. Ask the user for credentials if not provided:
|
||||
|
||||
```powershell
|
||||
python -c "
|
||||
@'
|
||||
import sys, json
|
||||
from graphify.build import build_from_json
|
||||
from graphify.cluster import cluster
|
||||
@@ -595,8 +637,10 @@ G = build_from_json(extraction)
|
||||
communities = {int(k): v for k, v in analysis['communities'].items()}
|
||||
|
||||
result = push_to_neo4j(G, uri='NEO4J_URI', user='NEO4J_USER', password='NEO4J_PASSWORD', communities=communities)
|
||||
print(f'Pushed to Neo4j: {result[\"nodes\"]} nodes, {result[\"edges\"]} edges')
|
||||
"
|
||||
print(f'Pushed to Neo4j: {result["nodes"]} nodes, {result["edges"]} edges')
|
||||
'@ | Out-File -FilePath .graphify_step_7_neo4j_export_only_if_neo4j_or__14.py -Encoding utf8
|
||||
python .graphify_step_7_neo4j_export_only_if_neo4j_or__14.py
|
||||
Remove-Item -ErrorAction SilentlyContinue .graphify_step_7_neo4j_export_only_if_neo4j_or__14.py
|
||||
```
|
||||
|
||||
Replace `NEO4J_URI`, `NEO4J_USER`, `NEO4J_PASSWORD` with actual values. Default URI is `bolt://localhost:7687`, default user is `neo4j`. Uses MERGE - safe to re-run without creating duplicates.
|
||||
@@ -604,7 +648,7 @@ Replace `NEO4J_URI`, `NEO4J_USER`, `NEO4J_PASSWORD` with actual values. Default
|
||||
### Step 7b - SVG export (only if --svg flag)
|
||||
|
||||
```powershell
|
||||
python -c "
|
||||
@'
|
||||
import sys, json
|
||||
from graphify.build import build_from_json
|
||||
from graphify.export import to_svg
|
||||
@@ -620,13 +664,15 @@ labels = {int(k): v for k, v in labels_raw.items()}
|
||||
|
||||
to_svg(G, communities, 'graphify-out/graph.svg', community_labels=labels or None)
|
||||
print('graph.svg written - embeds in Obsidian, Notion, GitHub READMEs')
|
||||
"
|
||||
'@ | Out-File -FilePath .graphify_step_7b_svg_export_only_if_svg_flag_15.py -Encoding utf8
|
||||
python .graphify_step_7b_svg_export_only_if_svg_flag_15.py
|
||||
Remove-Item -ErrorAction SilentlyContinue .graphify_step_7b_svg_export_only_if_svg_flag_15.py
|
||||
```
|
||||
|
||||
### Step 7c - GraphML export (only if --graphml flag)
|
||||
|
||||
```powershell
|
||||
python -c "
|
||||
@'
|
||||
import json
|
||||
from graphify.build import build_from_json
|
||||
from graphify.export import to_graphml
|
||||
@@ -640,7 +686,9 @@ communities = {int(k): v for k, v in analysis['communities'].items()}
|
||||
|
||||
to_graphml(G, communities, 'graphify-out/graph.graphml')
|
||||
print('graph.graphml written - open in Gephi, yEd, or any GraphML tool')
|
||||
"
|
||||
'@ | Out-File -FilePath .graphify_step_7c_graphml_export_only_if_graphml_16.py -Encoding utf8
|
||||
python .graphify_step_7c_graphml_export_only_if_graphml_16.py
|
||||
Remove-Item -ErrorAction SilentlyContinue .graphify_step_7c_graphml_export_only_if_graphml_16.py
|
||||
```
|
||||
|
||||
### Step 7d - MCP server (only if --mcp flag)
|
||||
@@ -668,7 +716,7 @@ To configure in Claude Desktop, add to `claude_desktop_config.json`:
|
||||
If `total_words` from `.graphify_detect.json` is greater than 5,000, run:
|
||||
|
||||
```powershell
|
||||
python -c "
|
||||
@'
|
||||
import json
|
||||
from graphify.benchmark import run_benchmark, print_benchmark
|
||||
from pathlib import Path
|
||||
@@ -676,7 +724,9 @@ from pathlib import Path
|
||||
detection = json.loads(Path('.graphify_detect.json').read_text())
|
||||
result = run_benchmark('graphify-out/graph.json', corpus_words=detection['total_words'])
|
||||
print_benchmark(result)
|
||||
"
|
||||
'@ | Out-File -FilePath .graphify_step_8_token_reduction_benchmark_only_17.py -Encoding utf8
|
||||
python .graphify_step_8_token_reduction_benchmark_only_17.py
|
||||
Remove-Item -ErrorAction SilentlyContinue .graphify_step_8_token_reduction_benchmark_only_17.py
|
||||
```
|
||||
|
||||
Print the output directly in chat. If `total_words <= 5000`, skip silently - the graph value is structural clarity, not token compression, for small corpora.
|
||||
@@ -686,7 +736,7 @@ Print the output directly in chat. If `total_words <= 5000`, skip silently - the
|
||||
### Step 9 - Save manifest, update cost tracker, clean up, and report
|
||||
|
||||
```powershell
|
||||
python -c "
|
||||
@'
|
||||
import json
|
||||
from pathlib import Path
|
||||
from datetime import datetime, timezone
|
||||
@@ -718,8 +768,10 @@ cost['total_output_tokens'] += output_tok
|
||||
cost_path.write_text(json.dumps(cost, indent=2))
|
||||
|
||||
print(f'This run: {input_tok:,} input tokens, {output_tok:,} output tokens')
|
||||
print(f'All time: {cost[\"total_input_tokens\"]:,} input, {cost[\"total_output_tokens\"]:,} output ({len(cost[\"runs\"])} runs)')
|
||||
"
|
||||
print(f'All time: {cost["total_input_tokens"]:,} input, {cost["total_output_tokens"]:,} output ({len(cost["runs"])} runs)')
|
||||
'@ | Out-File -FilePath .graphify_step_9_save_manifest_update_cost_trac_18.py -Encoding utf8
|
||||
python .graphify_step_9_save_manifest_update_cost_trac_18.py
|
||||
Remove-Item -ErrorAction SilentlyContinue .graphify_step_9_save_manifest_update_cost_trac_18.py
|
||||
Remove-Item -ErrorAction SilentlyContinue .graphify_detect.json, .graphify_extract.json, .graphify_ast.json, .graphify_semantic.json, .graphify_analysis.json, .graphify_labels.json
|
||||
Remove-Item -ErrorAction SilentlyContinue graphify-out/.needs_update
|
||||
```
|
||||
@@ -760,7 +812,7 @@ The graph is the map. Your job after the pipeline is to be the guide.
|
||||
Use when you've added or modified files since the last run. Only re-extracts changed files - saves tokens and time.
|
||||
|
||||
```powershell
|
||||
python -c "
|
||||
@'
|
||||
import sys, json
|
||||
from graphify.detect import detect_incremental, save_manifest
|
||||
from pathlib import Path
|
||||
@@ -773,13 +825,15 @@ if new_total == 0:
|
||||
print('No files changed since last run. Nothing to update.')
|
||||
raise SystemExit(0)
|
||||
print(f'{new_total} new/changed file(s) to re-extract.')
|
||||
"
|
||||
'@ | Out-File -FilePath .graphify_step_for_update_incremental_re_extracti_19.py -Encoding utf8
|
||||
python .graphify_step_for_update_incremental_re_extracti_19.py
|
||||
Remove-Item -ErrorAction SilentlyContinue .graphify_step_for_update_incremental_re_extracti_19.py
|
||||
```
|
||||
|
||||
If new files exist, first check whether all changed files are code files:
|
||||
|
||||
```powershell
|
||||
python -c "
|
||||
@'
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
@@ -789,7 +843,9 @@ new_files = result.get('new_files', {})
|
||||
all_changed = [f for files in new_files.values() for f in files]
|
||||
code_only = all(Path(f).suffix.lower() in code_exts for f in all_changed)
|
||||
print('code_only:', code_only)
|
||||
"
|
||||
'@ | Out-File -FilePath .graphify_step_for_update_incremental_re_extracti_20.py -Encoding utf8
|
||||
python .graphify_step_for_update_incremental_re_extracti_20.py
|
||||
Remove-Item -ErrorAction SilentlyContinue .graphify_step_for_update_incremental_re_extracti_20.py
|
||||
```
|
||||
|
||||
If `code_only` is True: print `[graphify update] Code-only changes detected - skipping semantic extraction (no LLM needed)`, run only Step 3A (AST) on the changed files, skip Step 3B entirely (no subagents), then go straight to merge and Steps 4–8.
|
||||
@@ -799,7 +855,7 @@ If `code_only` is False (any changed file is a doc/paper/image): run the full St
|
||||
Then:
|
||||
|
||||
```powershell
|
||||
python -c "
|
||||
@'
|
||||
import sys, json
|
||||
from graphify.build import build_from_json
|
||||
from graphify.export import to_json
|
||||
@@ -837,7 +893,9 @@ print(f'Merged: {G_existing.number_of_nodes()} nodes, {G_existing.number_of_edge
|
||||
from graphify.detect import save_manifest
|
||||
save_manifest(incremental['files'])
|
||||
print('[graphify update] Manifest saved.')
|
||||
"
|
||||
'@ | Out-File -FilePath .graphify_step_for_update_incremental_re_extracti_21.py -Encoding utf8
|
||||
python .graphify_step_for_update_incremental_re_extracti_21.py
|
||||
Remove-Item -ErrorAction SilentlyContinue .graphify_step_for_update_incremental_re_extracti_21.py
|
||||
```
|
||||
|
||||
Then run Steps 4–8 on the merged graph as normal.
|
||||
@@ -845,7 +903,7 @@ Then run Steps 4–8 on the merged graph as normal.
|
||||
After Step 4, show the graph diff:
|
||||
|
||||
```powershell
|
||||
python -c "
|
||||
@'
|
||||
import json
|
||||
from graphify.analyze import graph_diff
|
||||
from graphify.build import build_from_json
|
||||
@@ -866,7 +924,9 @@ if old_data:
|
||||
print('New nodes:', ', '.join(n['label'] for n in diff['new_nodes'][:5]))
|
||||
if diff['new_edges']:
|
||||
print('New edges:', len(diff['new_edges']))
|
||||
"
|
||||
'@ | Out-File -FilePath .graphify_step_for_update_incremental_re_extracti_22.py -Encoding utf8
|
||||
python .graphify_step_for_update_incremental_re_extracti_22.py
|
||||
Remove-Item -ErrorAction SilentlyContinue .graphify_step_for_update_incremental_re_extracti_22.py
|
||||
```
|
||||
|
||||
Before the merge step, save the old graph: `Copy-Item graphify-out/graph.json .graphify_old.json`
|
||||
@@ -879,7 +939,7 @@ Clean up after: `Remove-Item -ErrorAction SilentlyContinue .graphify_old.json`
|
||||
Skip Steps 1–3. Load the existing graph from `graphify-out/graph.json` and re-run clustering:
|
||||
|
||||
```powershell
|
||||
python -c "
|
||||
@'
|
||||
import sys, json
|
||||
from graphify.cluster import cluster, score_all
|
||||
from graphify.analyze import god_nodes, surprising_connections
|
||||
@@ -914,7 +974,9 @@ analysis = {
|
||||
}
|
||||
Path('.graphify_analysis.json').write_text(json.dumps(analysis, indent=2))
|
||||
print(f'Re-clustered: {len(communities)} communities')
|
||||
"
|
||||
'@ | Out-File -FilePath .graphify_step_for_cluster_only_23.py -Encoding utf8
|
||||
python .graphify_step_for_cluster_only_23.py
|
||||
Remove-Item -ErrorAction SilentlyContinue .graphify_step_for_cluster_only_23.py
|
||||
```
|
||||
|
||||
Then run Steps 5–9 as normal (label communities, generate viz, benchmark, clean up, report).
|
||||
@@ -932,12 +994,14 @@ Two traversal modes - choose based on the question:
|
||||
|
||||
First check the graph exists:
|
||||
```powershell
|
||||
python -c "
|
||||
@'
|
||||
from pathlib import Path
|
||||
if not Path('graphify-out/graph.json').exists():
|
||||
print('ERROR: No graph found. Run /graphify <path> first to build the graph.')
|
||||
raise SystemExit(1)
|
||||
"
|
||||
'@ | Out-File -FilePath .graphify_step_for_graphify_query_24.py -Encoding utf8
|
||||
python .graphify_step_for_graphify_query_24.py
|
||||
Remove-Item -ErrorAction SilentlyContinue .graphify_step_for_graphify_query_24.py
|
||||
```
|
||||
If it fails, stop and tell the user to run `/graphify <path>` first.
|
||||
|
||||
@@ -950,7 +1014,7 @@ Load `graphify-out/graph.json`, then:
|
||||
5. If the graph lacks enough information, say so - do not hallucinate edges.
|
||||
|
||||
```powershell
|
||||
python -c "
|
||||
@'
|
||||
import sys, json
|
||||
from networkx.readwrite import json_graph
|
||||
import networkx as nx
|
||||
@@ -1020,20 +1084,22 @@ def relevance(nid):
|
||||
|
||||
ranked_nodes = sorted(subgraph_nodes, key=relevance, reverse=True)
|
||||
|
||||
lines = [f'Traversal: {mode.upper()} | Start: {[G.nodes[n].get(\"label\",n) for n in start_nodes]} | {len(subgraph_nodes)} nodes']
|
||||
lines = [f'Traversal: {mode.upper()} | Start: {[G.nodes[n].get("label",n) for n in start_nodes]} | {len(subgraph_nodes)} nodes']
|
||||
for nid in ranked_nodes:
|
||||
d = G.nodes[nid]
|
||||
lines.append(f' NODE {d.get(\"label\", nid)} [src={d.get(\"source_file\",\"\")} loc={d.get(\"source_location\",\"\")}]')
|
||||
lines.append(f' NODE {d.get("label", nid)} [src={d.get("source_file","")} loc={d.get("source_location","")}]')
|
||||
for u, v in subgraph_edges:
|
||||
if u in subgraph_nodes and v in subgraph_nodes:
|
||||
d = G.edges[u, v]
|
||||
lines.append(f' EDGE {G.nodes[u].get(\"label\",u)} --{d.get(\"relation\",\"\")} [{d.get(\"confidence\",\"\")}]--> {G.nodes[v].get(\"label\",v)}')
|
||||
lines.append(f' EDGE {G.nodes[u].get("label",u)} --{d.get("relation","")} [{d.get("confidence","")}]--> {G.nodes[v].get("label",v)}')
|
||||
|
||||
output = '\n'.join(lines)
|
||||
if len(output) > char_budget:
|
||||
output = output[:char_budget] + f'\n... (truncated at ~{token_budget} token budget - use --budget N for more)'
|
||||
print(output)
|
||||
"
|
||||
'@ | Out-File -FilePath .graphify_step_for_graphify_query_25.py -Encoding utf8
|
||||
python .graphify_step_for_graphify_query_25.py
|
||||
Remove-Item -ErrorAction SilentlyContinue .graphify_step_for_graphify_query_25.py
|
||||
```
|
||||
|
||||
Replace `QUESTION` with the user's actual question, `MODE` with `bfs` or `dfs`, and `BUDGET` with the token budget (default `2000`, or whatever `--budget N` specifies). Then answer based on the subgraph output above.
|
||||
@@ -1054,17 +1120,19 @@ Find the shortest path between two named concepts in the graph.
|
||||
|
||||
First check the graph exists:
|
||||
```powershell
|
||||
python -c "
|
||||
@'
|
||||
from pathlib import Path
|
||||
if not Path('graphify-out/graph.json').exists():
|
||||
print('ERROR: No graph found. Run /graphify <path> first to build the graph.')
|
||||
raise SystemExit(1)
|
||||
"
|
||||
'@ | Out-File -FilePath .graphify_step_for_graphify_path_26.py -Encoding utf8
|
||||
python .graphify_step_for_graphify_path_26.py
|
||||
Remove-Item -ErrorAction SilentlyContinue .graphify_step_for_graphify_path_26.py
|
||||
```
|
||||
If it fails, stop and tell the user to run `/graphify <path>` first.
|
||||
|
||||
```powershell
|
||||
python -c "
|
||||
@'
|
||||
import json, sys
|
||||
import networkx as nx
|
||||
from networkx.readwrite import json_graph
|
||||
@@ -1108,7 +1176,9 @@ except nx.NetworkXNoPath:
|
||||
print(f'No path found between {a_term!r} and {b_term!r}')
|
||||
except nx.NodeNotFound as e:
|
||||
print(f'Node not found: {e}')
|
||||
"
|
||||
'@ | Out-File -FilePath .graphify_step_for_graphify_path_27.py -Encoding utf8
|
||||
python .graphify_step_for_graphify_path_27.py
|
||||
Remove-Item -ErrorAction SilentlyContinue .graphify_step_for_graphify_path_27.py
|
||||
```
|
||||
|
||||
Replace `NODE_A` and `NODE_B` with the actual concept names from the user. Then explain the path in plain language - what each hop means, why it's significant.
|
||||
@@ -1127,17 +1197,19 @@ Give a plain-language explanation of a single node - everything connected to it.
|
||||
|
||||
First check the graph exists:
|
||||
```powershell
|
||||
python -c "
|
||||
@'
|
||||
from pathlib import Path
|
||||
if not Path('graphify-out/graph.json').exists():
|
||||
print('ERROR: No graph found. Run /graphify <path> first to build the graph.')
|
||||
raise SystemExit(1)
|
||||
"
|
||||
'@ | Out-File -FilePath .graphify_step_for_graphify_explain_28.py -Encoding utf8
|
||||
python .graphify_step_for_graphify_explain_28.py
|
||||
Remove-Item -ErrorAction SilentlyContinue .graphify_step_for_graphify_explain_28.py
|
||||
```
|
||||
If it fails, stop and tell the user to run `/graphify <path>` first.
|
||||
|
||||
```powershell
|
||||
python -c "
|
||||
@'
|
||||
import json, sys
|
||||
import networkx as nx
|
||||
from networkx.readwrite import json_graph
|
||||
@@ -1161,9 +1233,9 @@ if not scored or scored[0][0] == 0:
|
||||
|
||||
nid = scored[0][1]
|
||||
data_n = G.nodes[nid]
|
||||
print(f'NODE: {data_n.get(\"label\", nid)}')
|
||||
print(f' source: {data_n.get(\"source_file\",\"unknown\")}')
|
||||
print(f' type: {data_n.get(\"file_type\",\"unknown\")}')
|
||||
print(f'NODE: {data_n.get("label", nid)}')
|
||||
print(f' source: {data_n.get("source_file","unknown")}')
|
||||
print(f' type: {data_n.get("file_type","unknown")}')
|
||||
print(f' degree: {G.degree(nid)}')
|
||||
print()
|
||||
print('CONNECTIONS:')
|
||||
@@ -1174,7 +1246,9 @@ for neighbor in G.neighbors(nid):
|
||||
conf = edge.get('confidence', '')
|
||||
src_file = G.nodes[neighbor].get('source_file', '')
|
||||
print(f' --{rel}--> {nlabel} [{conf}] ({src_file})')
|
||||
"
|
||||
'@ | Out-File -FilePath .graphify_step_for_graphify_explain_29.py -Encoding utf8
|
||||
python .graphify_step_for_graphify_explain_29.py
|
||||
Remove-Item -ErrorAction SilentlyContinue .graphify_step_for_graphify_explain_29.py
|
||||
```
|
||||
|
||||
Replace `NODE_NAME` with the concept the user asked about. Then write a 3-5 sentence explanation of what this node is, what it connects to, and why those connections are significant. Use the source locations as citations.
|
||||
@@ -1192,7 +1266,7 @@ python -m graphify save-result --question "Explain NODE_NAME" --answer "ANSWER"
|
||||
Fetch a URL and add it to the corpus, then update the graph.
|
||||
|
||||
```powershell
|
||||
python -c "
|
||||
@'
|
||||
import sys
|
||||
from graphify.ingest import ingest
|
||||
from pathlib import Path
|
||||
@@ -1206,7 +1280,9 @@ except ValueError as e:
|
||||
except RuntimeError as e:
|
||||
print(f'error: {e}', file=sys.stderr)
|
||||
sys.exit(1)
|
||||
"
|
||||
'@ | Out-File -FilePath .graphify_step_for_graphify_add_30.py -Encoding utf8
|
||||
python .graphify_step_for_graphify_add_30.py
|
||||
Remove-Item -ErrorAction SilentlyContinue .graphify_step_for_graphify_add_30.py
|
||||
```
|
||||
|
||||
Replace `URL` with the actual URL, `AUTHOR` with the user's name if provided, `CONTRIBUTOR` likewise. If the command exits with an error, tell the user what went wrong - do not silently continue. After a successful save, automatically run the `--update` pipeline on `./raw` to merge the new file into the existing graph.
|
||||
|
||||
+47
-1
@@ -5,7 +5,7 @@ import pytest
|
||||
import networkx as nx
|
||||
from networkx.readwrite import json_graph
|
||||
|
||||
from graphify.benchmark import run_benchmark, print_benchmark, _query_subgraph_tokens, _SAMPLE_QUESTIONS
|
||||
from graphify.benchmark import run_benchmark, print_benchmark, _query_subgraph_tokens, _SAMPLE_QUESTIONS, _safe, _hr
|
||||
|
||||
|
||||
def _make_graph() -> nx.Graph:
|
||||
@@ -117,3 +117,49 @@ def test_print_benchmark_error_message(capsys):
|
||||
print_benchmark({"error": "test error message"})
|
||||
out = capsys.readouterr().out
|
||||
assert "test error message" in out
|
||||
|
||||
|
||||
# --- cp1252 / Windows-console encoding compatibility (regression for #?) ---
|
||||
# print_benchmark previously crashed on Windows consoles (cp1252) because it
|
||||
# unconditionally printed U+2500 and U+2192. _safe() falls back to ASCII when
|
||||
# stdout cannot encode the glyph.
|
||||
|
||||
def test_safe_returns_unicode_when_encodable():
|
||||
import io, sys
|
||||
real_stdout = sys.stdout
|
||||
try:
|
||||
sys.stdout = io.TextIOWrapper(io.BytesIO(), encoding="utf-8")
|
||||
assert _safe("→", "->") == "→"
|
||||
assert _hr(5) == "─" * 5
|
||||
finally:
|
||||
sys.stdout = real_stdout
|
||||
|
||||
def test_safe_falls_back_when_unencodable():
|
||||
import io, sys
|
||||
real_stdout = sys.stdout
|
||||
try:
|
||||
sys.stdout = io.TextIOWrapper(io.BytesIO(), encoding="cp1252")
|
||||
assert _safe("→", "->") == "->"
|
||||
assert _hr(5) == "-" * 5
|
||||
finally:
|
||||
sys.stdout = real_stdout
|
||||
|
||||
def test_print_benchmark_survives_cp1252_stdout(tmp_path, monkeypatch, capsys):
|
||||
"""Regression: U+2500 / U+2192 used to crash with UnicodeEncodeError on cp1252."""
|
||||
import io, sys
|
||||
G = _make_graph()
|
||||
graph_file = tmp_path / "graph.json"
|
||||
_write_graph(G, graph_file)
|
||||
result = run_benchmark(str(graph_file), corpus_words=5_000)
|
||||
|
||||
# Replace stdout with a strict cp1252 stream — same behaviour as the
|
||||
# legacy Windows console that surfaced this bug.
|
||||
cp1252_stdout = io.TextIOWrapper(io.BytesIO(), encoding="cp1252", errors="strict")
|
||||
monkeypatch.setattr(sys, "stdout", cp1252_stdout)
|
||||
print_benchmark(result) # must not raise UnicodeEncodeError
|
||||
cp1252_stdout.flush()
|
||||
written = cp1252_stdout.buffer.getvalue().decode("cp1252")
|
||||
assert "reduction" in written.lower()
|
||||
# ASCII fallbacks must be present, fancy glyphs must not.
|
||||
assert "─" not in written
|
||||
assert "→" not in written
|
||||
|
||||
@@ -365,3 +365,63 @@ def test_extract_tsx_uses_tsx_grammar():
|
||||
from graphify.extract import _TSX_CONFIG, _TS_CONFIG
|
||||
assert _TSX_CONFIG.ts_language_fn == "language_tsx"
|
||||
assert _TS_CONFIG.ts_language_fn == "language_typescript"
|
||||
|
||||
|
||||
# --- Windows-spawn ProcessPool fallback (regression for #?) ---
|
||||
# When the caller has no `if __name__ == "__main__":` guard, ProcessPoolExecutor
|
||||
# on Windows raises BrokenProcessPool before any work completes. extract() must
|
||||
# detect this, warn, and fall back to sequential extraction rather than
|
||||
# propagating a 290-line traceback.
|
||||
|
||||
def test_extract_falls_back_to_sequential_when_parallel_returns_false(tmp_path, monkeypatch):
|
||||
"""extract() must run sequential when _extract_parallel signals failure (returns False)."""
|
||||
from graphify import extract as extract_mod
|
||||
|
||||
files = [FIXTURES / "sample.py"] * 25 # >= _PARALLEL_THRESHOLD triggers parallel branch
|
||||
cache_root = tmp_path / "cache"
|
||||
cache_root.mkdir()
|
||||
|
||||
calls = {"parallel": 0, "sequential": 0}
|
||||
real_sequential = extract_mod._extract_sequential
|
||||
|
||||
def fake_parallel(uncached_work, per_file, effective_root, max_workers, total_files):
|
||||
calls["parallel"] += 1
|
||||
return False # simulate the post-fix BrokenProcessPool branch
|
||||
|
||||
def wrapped_sequential(*args, **kwargs):
|
||||
calls["sequential"] += 1
|
||||
return real_sequential(*args, **kwargs)
|
||||
|
||||
monkeypatch.setattr(extract_mod, "_extract_parallel", fake_parallel)
|
||||
monkeypatch.setattr(extract_mod, "_extract_sequential", wrapped_sequential)
|
||||
|
||||
result = extract_mod.extract(files, cache_root=cache_root)
|
||||
assert calls["parallel"] == 1, "parallel path should have been attempted once"
|
||||
assert calls["sequential"] == 1, "sequential fallback should have run exactly once"
|
||||
assert result["nodes"], "extract should still produce nodes after fallback"
|
||||
|
||||
|
||||
def test_extract_parallel_returns_false_on_broken_pool(tmp_path, monkeypatch, capsys):
|
||||
"""_extract_parallel must catch BrokenProcessPool internally and return False."""
|
||||
from concurrent.futures.process import BrokenProcessPool
|
||||
import concurrent.futures
|
||||
from graphify import extract as extract_mod
|
||||
|
||||
class FakePool:
|
||||
def __init__(self, *a, **kw): pass
|
||||
def __enter__(self): return self
|
||||
def __exit__(self, *a): return False
|
||||
def submit(self, *a, **kw):
|
||||
raise BrokenProcessPool("simulated spawn failure")
|
||||
|
||||
monkeypatch.setattr(
|
||||
concurrent.futures, "ProcessPoolExecutor", lambda *a, **kw: FakePool()
|
||||
)
|
||||
|
||||
uncached = [(0, FIXTURES / "sample.py")]
|
||||
per_file: list = [None]
|
||||
ok = extract_mod._extract_parallel(uncached, per_file, tmp_path, 2, 1)
|
||||
assert ok is False, "function should report failure via return value, not raise"
|
||||
out = capsys.readouterr().out
|
||||
assert "BrokenProcessPool" in out, "user-facing warning must mention the failure"
|
||||
assert "__main__" in out, "warning must hint at the Windows __main__ guard idiom"
|
||||
|
||||
Reference in New Issue
Block a user