fix(extract): make --mode deep effective over a warm cache; add --force (#1894)

`graphify extract --mode deep` over a warm tree was a silent no-op, for
three stacked reasons:

1. The semantic cache ignored mode: deep runs were served standard-mode
   entries (and vice versa). check_semantic_cache/save_semantic_cache now
   take `mode` (default None, byte-identical when omitted so older
   installed callers keep working) and map it to a namespaced kind —
   cache/semantic/ for None, cache/semantic-{mode}/ otherwise. The
   per-chunk checkpoint in llm.extract_corpus_parallel and the extract /
   cache-check consumers thread the run's mode through (cache-check grows
   --mode/--deep). cached_files, clear_cache, and prune_semantic_cache
   sweep BOTH namespaces; prune uses the same live-hash set for both
   (liveness is content-based, mode-independent) so semantic-deep/ can't
   regrow the #1527 unbounded-orphan problem and inherits the
   files_by_type-derived exclusion gating for free.

2. extract had no --force — the flag was silently swallowed by the
   parser's unknown-arg fallthrough. It is now real (plus GRAPHIFY_FORCE
   env parity with `update`): force disables the incremental gate so
   detection is a full scan and skips the semantic cache READ so every
   semantic file re-dispatches, while the post-run save and manifest
   stamping still happen.

3. The incremental gate dispatched zero files on a warm unchanged tree
   before the cache was ever consulted, so namespacing alone couldn't fix
   the repro. In deep+incremental runs the semantic pass now widens to the
   full live doc/paper/image set from detect_incremental's files_by_type
   (already exclusion-filtered, #1908/#1909) and lets the mode-namespaced
   cache decide hits/misses, with a loud count line so the first deep
   run's full re-dispatch is visible.

Skill-side threading of mode is deliberately deferred to PR-2; mode
defaults keep generated skills byte-compatible.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
safishamsi
2026-07-15 14:14:13 +01:00
co-authored by Claude Opus 4.8
parent d916c61559
commit b7ddee3c28
7 changed files with 462 additions and 32 deletions
+2
View File
@@ -597,6 +597,8 @@ def _run_cli() -> None:
print(" proxy, gateways): set ANTHROPIC_BASE_URL and ANTHROPIC_MODEL")
print(" --model M override backend default model")
print(" --mode deep aggressive INFERRED-edge semantic extraction")
print(" --force full re-scan and re-dispatch: skip the incremental")
print(" manifest gate and semantic cache reads (env: GRAPHIFY_FORCE=1)")
print(" --max-workers N AST extraction subprocess count (default: cpu_count)")
print(" --token-budget N per-chunk token cap for semantic extraction (default: 60000)")
print(" --max-concurrency N parallel semantic chunks in flight (default: 4; set 1 for local LLMs)")
+58 -26
View File
@@ -90,7 +90,8 @@ def _body_content(content: bytes) -> bytes:
# size+mtime_ns are unchanged — same trade-off as make(1).
# Correctness risks: `touch` causes a harmless extra re-hash; same-size edits
# within NFS second-resolution mtime have a 1-second window (same as make).
# Use `graphify extract --force` to bypass when needed.
# `graphify extract --force` / `graphify update --force` (or GRAPHIFY_FORCE=1)
# skip the cache reads and re-dispatch everything when needed (#1894).
_stat_index: dict[str, dict] = {}
_stat_index_root: Path | None = None
_stat_index_dirty: bool = False
@@ -340,13 +341,15 @@ def _absolutize_source_files_in(payload: dict, root: Path) -> None:
def cache_dir(root: Path = Path("."), kind: str = "ast") -> Path:
"""Returns the cache directory for ``kind`` - creates it if needed.
kind is "ast" or "semantic". Separate subdirectories prevent semantic cache
kind is "ast", "semantic", or a mode-namespaced semantic kind such as
"semantic-deep" (#1894). Separate subdirectories prevent semantic cache
entries from overwriting AST cache entries for the same source_file (#582).
AST entries live in graphify-out/cache/ast/v{version}/ — namespaced by
graphify version because they depend on extractor code, not just file
contents. Semantic entries live unversioned in graphify-out/cache/semantic/
(re-extraction costs LLM calls).
(re-extraction costs LLM calls); deep-mode entries live beside them in
graphify-out/cache/semantic-deep/.
"""
_out = Path(_GRAPHIFY_OUT)
base = _out if _out.is_absolute() else Path(root).resolve() / _out
@@ -467,8 +470,10 @@ def cached_files(root: Path = Path(".")) -> set[str]:
# Legacy flat entries
if base.is_dir():
hashes.update(p.stem for p in base.glob("*.json"))
# Namespaced entries (ast/ recursively, covering per-version subdirs)
for kind, pattern in (("ast", "**/*.json"), ("semantic", "*.json")):
# Namespaced entries (ast/ recursively, covering per-version subdirs;
# semantic-deep/ holds --mode deep entries, #1894)
for kind, pattern in (("ast", "**/*.json"), ("semantic", "*.json"),
("semantic-deep", "*.json")):
d = base / kind
if d.is_dir():
hashes.update(p.stem for p in d.glob(pattern))
@@ -476,14 +481,17 @@ def cached_files(root: Path = Path(".")) -> set[str]:
def clear_cache(root: Path = Path(".")) -> None:
"""Delete all cache entries (ast/, semantic/, and legacy flat entries)."""
"""Delete all cache entries (ast/, semantic/, semantic-deep/, and legacy
flat entries)."""
base = Path(root).resolve() / _GRAPHIFY_OUT / "cache"
# Legacy flat entries
if base.is_dir():
for f in base.glob("*.json"):
f.unlink()
# Namespaced entries (ast/ recursively, covering per-version subdirs)
for kind, pattern in (("ast", "**/*.json"), ("semantic", "*.json")):
# Namespaced entries (ast/ recursively, covering per-version subdirs;
# semantic-deep/ holds --mode deep entries, #1894)
for kind, pattern in (("ast", "**/*.json"), ("semantic", "*.json"),
("semantic-deep", "*.json")):
d = base / kind
if d.is_dir():
for f in d.glob(pattern):
@@ -500,11 +508,16 @@ def prune_semantic_cache(root: Path, live_hashes: set[str]) -> int:
AST version-cleanup, so every content change or file deletion leaves a
permanent orphan entry that accumulates unbounded.
This sweeps ``cache/semantic/*.json`` and deletes any entry whose stem (the
content hash) is not in ``live_hashes`` — the hashes of the current live
document set. ``*.tmp`` atomic-write temporaries are skipped, and only this
directory is touched (never ``cache/ast/**`` or anything else). The
unversioned design is preserved: we prune by liveness, not by version.
This sweeps ``cache/semantic/*.json`` AND ``cache/semantic-deep/*.json``
(the ``--mode deep`` namespace, #1894) and deletes any entry whose stem
(the content hash) is not in ``live_hashes`` — the hashes of the current
live document set. Both namespaces are pruned against the SAME live set:
liveness is content-based and mode-independent, so a hash that is live for
one namespace is live for both. Skipping the deep namespace would re-grow
the unbounded-orphan problem this function fixed (#1527). ``*.tmp``
atomic-write temporaries are skipped, and only these directories are
touched (never ``cache/ast/**`` or anything else). The unversioned design
is preserved: we prune by liveness, not by version.
Best-effort, mirroring :func:`_cleanup_stale_ast_entries`: each unlink is
wrapped in ``try/except OSError`` and a failure is ignored. The worst-case
@@ -513,30 +526,40 @@ def prune_semantic_cache(root: Path, live_hashes: set[str]) -> int:
"""
_out = Path(_GRAPHIFY_OUT)
base = _out if _out.is_absolute() else Path(root).resolve() / _out
semantic_dir = base / "cache" / "semantic"
if not semantic_dir.is_dir():
return 0
pruned = 0
for entry in semantic_dir.glob("*.json"):
if entry.stem in live_hashes:
for kind in ("semantic", "semantic-deep"):
semantic_dir = base / "cache" / kind
if not semantic_dir.is_dir():
continue
try:
entry.unlink()
pruned += 1
except OSError:
pass
for entry in semantic_dir.glob("*.json"):
if entry.stem in live_hashes:
continue
try:
entry.unlink()
pruned += 1
except OSError:
pass
return pruned
def check_semantic_cache(
files: list[str],
root: Path = Path("."),
mode: str | None = None,
) -> tuple[list[dict], list[dict], list[dict], list[str]]:
"""Check semantic extraction cache for a list of absolute file paths.
Returns (cached_nodes, cached_edges, cached_hyperedges, uncached_files).
Uncached files need Claude extraction; cached files are merged directly.
``mode`` selects the cache namespace: ``None`` (the default) reads
``cache/semantic/`` — byte-identical to the historical behavior, so
existing callers that omit it (including older installed skill flows)
are unaffected. A non-None mode (e.g. ``"deep"``) reads
``cache/semantic-{mode}/`` instead, so deep-mode results never shadow
(or get shadowed by) standard-mode entries for the same content (#1894).
"""
kind = "semantic" if mode is None else f"semantic-{mode}"
cached_nodes: list[dict] = []
cached_edges: list[dict] = []
cached_hyperedges: list[dict] = []
@@ -546,7 +569,7 @@ def check_semantic_cache(
p = Path(fpath)
if not p.is_absolute():
p = Path(root) / p
result = load_cached(p, root, kind="semantic")
result = load_cached(p, root, kind=kind)
if result is not None:
cached_nodes.extend(result.get("nodes", []))
cached_edges.extend(result.get("edges", []))
@@ -564,6 +587,7 @@ def save_semantic_cache(
root: Path = Path("."),
merge_existing: bool = False,
allowed_source_files: Iterable[str | Path] | None = None,
mode: str | None = None,
) -> int:
"""Save semantic extraction results to cache, keyed by source_file.
@@ -571,6 +595,13 @@ def save_semantic_cache(
under cache/semantic/ (separate from AST entries in cache/ast/) to prevent
hash-key collisions (#582).
``mode`` selects the cache namespace, mirroring
:func:`check_semantic_cache`: ``None`` (the default) writes
``cache/semantic/`` — byte-identical to the historical behavior for
existing callers that omit it — while a non-None mode (e.g. ``"deep"``)
writes ``cache/semantic-{mode}/`` so richer deep-mode results never
overwrite standard-mode entries and vice versa (#1894).
When ``merge_existing`` is True, any already-cached entry for a file is
unioned with the new results before saving instead of being overwritten.
This lets callers checkpoint incrementally (e.g. once per chunk) without
@@ -584,6 +615,7 @@ def save_semantic_cache(
"""
from collections import defaultdict
kind = "semantic" if mode is None else f"semantic-{mode}"
by_file: dict[str, dict] = defaultdict(lambda: {"nodes": [], "edges": [], "hyperedges": []})
for n in nodes:
src = n.get("source_file", "")
@@ -684,13 +716,13 @@ def save_semantic_cache(
)
continue
if merge_existing:
prev = load_cached(p, root, kind="semantic")
prev = load_cached(p, root, kind=kind)
if prev:
result = {
"nodes": (prev.get("nodes", []) or []) + result["nodes"],
"edges": (prev.get("edges", []) or []) + result["edges"],
"hyperedges": (prev.get("hyperedges", []) or []) + result["hyperedges"],
}
save_cached(p, result, root, kind="semantic")
save_cached(p, result, root, kind=kind)
saved += 1
return saved
+65 -6
View File
@@ -2138,6 +2138,9 @@ def dispatch_command(cmd: str) -> None:
cli_exclude_hubs: float | None = None
cli_excludes: list[str] = []
cli_timing: bool = False
# --force parity with `graphify update`: the flag or GRAPHIFY_FORCE=1
# disables the incremental gate and skips semantic-cache reads (#1894).
force = os.environ.get("GRAPHIFY_FORCE", "").lower() in ("1", "true", "yes")
def _parse_int(name: str, raw: str) -> int:
try:
@@ -2228,6 +2231,8 @@ def dispatch_command(cmd: str) -> None:
elif a == "--cargo":
cli_cargo = True
i += 1
elif a == "--force":
force = True; i += 1
elif a == "--timing":
cli_timing = True; i += 1
else:
@@ -2277,6 +2282,11 @@ def dispatch_command(cmd: str) -> None:
manifest_path = graphify_out / "manifest.json"
existing_graph_path = graphify_out / "graph.json"
incremental_mode = manifest_path.exists() and existing_graph_path.exists() if has_path else False
# --force: full scan, not the manifest-gated incremental diff — a warm
# unchanged tree would otherwise dispatch zero files (#1894).
incremental_mode = incremental_mode and not force
if force:
print("[graphify extract] --force: full re-scan, semantic cache reads skipped")
if not has_path:
code_files = []
@@ -2343,6 +2353,30 @@ def dispatch_command(cmd: str) -> None:
doc_files = []
paper_files = []
image_files = []
if deep_mode and incremental_mode and not code_only:
# Deep mode reads/writes its own cache namespace
# (cache/semantic-deep/), so the manifest's changed-file gate is
# not a valid proxy for deep coverage: over a warm unchanged tree
# it dispatches zero files and `--mode deep` silently no-ops
# (#1894). Widen the semantic pass to the FULL live
# doc/paper/image set (``files_by_type`` from detect_incremental,
# which already excludes excluded files) and let the
# mode-namespaced cache decide hits/misses — the first deep run
# re-dispatches everything (deep namespace cold), later deep runs
# hit the deep cache.
_deep_all = [
Path(p)
for _ftype in ("document", "paper", "image")
for p in files_by_type.get(_ftype, [])
]
if len(_deep_all) != len(semantic_files):
print(
f"[graphify extract] deep mode: widening semantic pass from "
f"{len(semantic_files)} changed to {len(_deep_all)} live "
f"doc/paper/image file(s); the deep semantic cache decides "
f"what is re-extracted"
)
semantic_files = _deep_all
if incremental_mode:
# Excluded-but-alive files are reported separately from deletions
# (#1908): they still exist on disk, the scan just stopped
@@ -2495,11 +2529,21 @@ def dispatch_command(cmd: str) -> None:
}
sem_cache_hits = 0
sem_cache_misses = 0
# Deep mode uses its own namespace (cache/semantic-deep/) so deep and
# standard results for the same content never shadow each other (#1894).
sem_cache_mode = "deep" if deep_mode else None
if semantic_files:
sem_paths_str = [str(p) for p in semantic_files]
cached_nodes, cached_edges, cached_hyperedges, uncached_paths = (
_check_semantic_cache(sem_paths_str, root=out_root)
)
if force:
# --force: skip the cache READ so every semantic file is
# re-dispatched; the save below still runs so the fresh
# results replace the stale entries.
cached_nodes, cached_edges, cached_hyperedges = [], [], []
uncached_paths = list(sem_paths_str)
else:
cached_nodes, cached_edges, cached_hyperedges, uncached_paths = (
_check_semantic_cache(sem_paths_str, root=out_root, mode=sem_cache_mode)
)
sem_cache_hits = len(semantic_files) - len(uncached_paths)
sem_cache_misses = len(uncached_paths)
sem_result["nodes"].extend(cached_nodes)
@@ -2570,6 +2614,7 @@ def dispatch_command(cmd: str) -> None:
fresh.get("hyperedges", []),
root=out_root,
allowed_source_files=uncached_paths,
mode=sem_cache_mode,
)
except Exception as exc:
print(f"[graphify extract] warning: could not write semantic cache: {exc}", file=sys.stderr)
@@ -2880,27 +2925,41 @@ def dispatch_command(cmd: str) -> None:
stages.total()
elif cmd == "cache-check":
# graphify cache-check <files_from> [--root <dir>]
# graphify cache-check <files_from> [--root <dir>] [--mode <m> | --deep]
# Reads file paths (one per line) from <files_from>, checks semantic cache.
# --mode deep (or --deep) checks the cache/semantic-deep/ namespace
# written by `extract --mode deep` instead of cache/semantic/ (#1894).
# Writes:
# graphify-out/.graphify_cached.json — already-cached nodes/edges/hyperedges
# graphify-out/.graphify_uncached.txt — paths that need extraction
# Stdout: "Cache: N hit, M miss"
from graphify.cache import check_semantic_cache
if len(sys.argv) < 3:
print("Usage: graphify cache-check <files_from> [--root <dir>]", file=sys.stderr)
print("Usage: graphify cache-check <files_from> [--root <dir>] [--mode <m> | --deep]", file=sys.stderr)
sys.exit(1)
files_from = Path(sys.argv[2])
root = Path(".")
cache_mode: str | None = None
i = 3
while i < len(sys.argv):
if sys.argv[i] == "--root" and i + 1 < len(sys.argv):
root = Path(sys.argv[i + 1])
i += 2
elif sys.argv[i] == "--mode" and i + 1 < len(sys.argv):
cache_mode = sys.argv[i + 1]
i += 2
elif sys.argv[i].startswith("--mode="):
cache_mode = sys.argv[i].split("=", 1)[1]
i += 1
elif sys.argv[i] == "--deep":
cache_mode = "deep"
i += 1
else:
i += 1
files = [f for f in files_from.read_text(encoding="utf-8").splitlines() if f.strip()]
cached_nodes, cached_edges, cached_hyperedges, uncached = check_semantic_cache(files, root)
cached_nodes, cached_edges, cached_hyperedges, uncached = check_semantic_cache(
files, root, mode=cache_mode
)
out = root / _GRAPHIFY_OUT
out.mkdir(parents=True, exist_ok=True)
if cached_nodes or cached_edges or cached_hyperedges:
+4
View File
@@ -1924,6 +1924,9 @@ def extract_corpus_parallel(
# chunk leaked the FileSlice object into the allowlist and the write
# raised TypeError, silently defeating the checkpoint.)
allowed = [unit_path(item) for item in chunk]
# Deep-mode results checkpoint into their own namespace
# (cache/semantic-deep/) so a deep run never overwrites standard
# entries — and a later standard run never serves deep ones (#1894).
_scs(
result.get("nodes", []),
result.get("edges", []),
@@ -1931,6 +1934,7 @@ def extract_corpus_parallel(
root=root,
merge_existing=True,
allowed_source_files=allowed,
mode="deep" if deep_mode else None,
)
except Exception as _exc: # noqa: BLE001 — checkpoint is best-effort
print(f"[graphify] incremental cache checkpoint failed: {_exc}", file=sys.stderr)
+143
View File
@@ -570,6 +570,149 @@ def test_save_semantic_cache_rejects_out_of_scope_source_file(tmp_path):
assert protected_cache["hyperedges"] == []
# --- #1894: mode-namespaced semantic cache -----------------------------------
# `extract --mode deep` produces richer results than standard extraction, so
# deep entries live in their own namespace (cache/semantic-deep/). mode=None
# must stay byte-identical to the historical behavior: older installed skill
# flows call check/save without the parameter and must be unaffected.
def test_semantic_cache_deep_mode_roundtrip_under_deep_namespace(tmp_path):
"""mode='deep' saves under cache/semantic-deep/ and reads back from it."""
from graphify.cache import check_semantic_cache, save_semantic_cache
f = tmp_path / "doc.md"
f.write_text("# Doc\n\nBody.\n")
saved = save_semantic_cache(
[{"id": "deep_n", "source_file": "doc.md"}], [], root=tmp_path, mode="deep"
)
assert saved == 1
deep_dir = tmp_path / "graphify-out" / "cache" / "semantic-deep"
h = file_hash(f, tmp_path)
assert (deep_dir / f"{h}.json").exists(), (
"deep entry must land under cache/semantic-deep/"
)
# And NOT in the plain namespace.
plain_dir = tmp_path / "graphify-out" / "cache" / "semantic"
assert not (plain_dir / f"{h}.json").exists()
nodes, edges, hyper, uncached = check_semantic_cache(
[str(f)], root=tmp_path, mode="deep"
)
assert [n["id"] for n in nodes] == ["deep_n"]
assert uncached == []
def test_semantic_cache_deep_invisible_to_plain_reads_and_vice_versa(tmp_path):
"""Deep entries must not satisfy mode=None reads (and plain entries must
not satisfy deep reads) — the namespaces are fully isolated."""
from graphify.cache import check_semantic_cache, save_semantic_cache
deep_doc = tmp_path / "deep.md"
deep_doc.write_text("# Deep\n")
plain_doc = tmp_path / "plain.md"
plain_doc.write_text("# Plain\n")
save_semantic_cache([{"id": "d", "source_file": "deep.md"}], [],
root=tmp_path, mode="deep")
save_semantic_cache([{"id": "p", "source_file": "plain.md"}], [],
root=tmp_path) # mode omitted: historical call shape
# Plain read: deep entry is a miss, plain entry is a hit.
nodes, _, _, uncached = check_semantic_cache(
[str(deep_doc), str(plain_doc)], root=tmp_path
)
assert [n["id"] for n in nodes] == ["p"]
assert uncached == [str(deep_doc)]
# Deep read: mirror image.
nodes, _, _, uncached = check_semantic_cache(
[str(deep_doc), str(plain_doc)], root=tmp_path, mode="deep"
)
assert [n["id"] for n in nodes] == ["d"]
assert uncached == [str(plain_doc)]
def test_semantic_cache_mode_none_layout_unchanged(tmp_path):
"""Omitting mode writes exactly the historical cache/semantic/ layout —
forward-compat for older installed callers that never pass mode."""
from graphify.cache import check_semantic_cache, save_semantic_cache
f = tmp_path / "doc.md"
f.write_text("# Doc\n")
save_semantic_cache([{"id": "n", "source_file": "doc.md"}], [], root=tmp_path)
h = file_hash(f, tmp_path)
assert (tmp_path / "graphify-out" / "cache" / "semantic" / f"{h}.json").exists()
assert not (tmp_path / "graphify-out" / "cache" / "semantic-deep").exists(), (
"mode=None must never create the deep namespace"
)
nodes, _, _, uncached = check_semantic_cache([str(f)], root=tmp_path)
assert [n["id"] for n in nodes] == ["n"] and uncached == []
def test_clear_cache_removes_deep_namespace(tmp_path):
"""clear_cache sweeps cache/semantic-deep/ alongside semantic/ and ast/."""
from graphify.cache import save_semantic_cache
f = tmp_path / "doc.md"
f.write_text("# Doc\n")
save_semantic_cache([{"id": "p", "source_file": "doc.md"}], [], root=tmp_path)
save_semantic_cache([{"id": "d", "source_file": "doc.md"}], [],
root=tmp_path, mode="deep")
base = tmp_path / "graphify-out" / "cache"
assert list((base / "semantic").glob("*.json"))
assert list((base / "semantic-deep").glob("*.json"))
clear_cache(tmp_path)
assert not list(base.rglob("*.json")), (
"clear_cache must remove entries in BOTH semantic namespaces"
)
def test_cached_files_includes_deep_namespace(tmp_path):
"""cached_files reports deep-namespace entries too."""
from graphify.cache import save_semantic_cache
f = tmp_path / "doc.md"
f.write_text("# Doc\n")
save_semantic_cache([{"id": "d", "source_file": "doc.md"}], [],
root=tmp_path, mode="deep")
assert file_hash(f, tmp_path) in cached_files(tmp_path)
def test_semantic_prune_sweeps_both_namespaces_against_same_live_set(tmp_path):
"""#1894 follow-up to #1527: prune must sweep cache/semantic/ AND
cache/semantic-deep/ against the SAME live-hash set (liveness is
content-based, mode-independent). Orphans go in both namespaces; live
entries survive in both."""
from graphify.cache import prune_semantic_cache, save_semantic_cache
f = tmp_path / "doc.md"
f.write_text("# A\n\nContent A.\n")
h_old = file_hash(f, tmp_path)
save_semantic_cache([{"id": "pa", "source_file": "doc.md"}], [], root=tmp_path)
save_semantic_cache([{"id": "da", "source_file": "doc.md"}], [],
root=tmp_path, mode="deep")
f.write_text("# B\n\nContent B.\n")
h_live = file_hash(f, tmp_path)
save_semantic_cache([{"id": "pb", "source_file": "doc.md"}], [], root=tmp_path)
save_semantic_cache([{"id": "db", "source_file": "doc.md"}], [],
root=tmp_path, mode="deep")
plain_dir = tmp_path / "graphify-out" / "cache" / "semantic"
deep_dir = tmp_path / "graphify-out" / "cache" / "semantic-deep"
for d in (plain_dir, deep_dir):
assert (d / f"{h_old}.json").exists()
assert (d / f"{h_live}.json").exists()
pruned = prune_semantic_cache(tmp_path, {h_live})
assert pruned == 2, "one orphan in EACH namespace must be pruned"
for d in (plain_dir, deep_dir):
assert not (d / f"{h_old}.json").exists(), f"orphan survived in {d.name}"
assert (d / f"{h_live}.json").exists(), f"live entry pruned from {d.name}"
def test_save_semantic_cache_merge_existing_unions(tmp_path):
"""#1715: merge_existing=True unions with the prior entry so a file split
across chunks (checkpointed per chunk) keeps every slice."""
+32
View File
@@ -326,6 +326,38 @@ def test_checkpoint_scopes_cache_writes_to_chunk_files(tmp_path):
assert a_cache and any(n["id"] == "a_ok" for n in a_cache["nodes"])
def test_checkpoint_writes_deep_namespace_in_deep_mode(tmp_path):
"""#1894: the per-chunk checkpoint must follow the run's mode — a
deep_mode=True run checkpoints into cache/semantic-deep/, leaving the
standard cache/semantic/ namespace untouched (and vice versa)."""
from graphify.llm import extract_corpus_parallel
from graphify.cache import load_cached
doc = tmp_path / "doc.md"
doc.write_text("# Doc\n\nsome content\n")
def ok(chunk, **kwargs):
return {
"nodes": [{"id": "d1", "source_file": "doc.md", "file_type": "document"}],
"edges": [], "hyperedges": [], "input_tokens": 1, "output_tokens": 1,
}
with patch("graphify.llm.extract_files_direct", side_effect=ok):
extract_corpus_parallel(
[doc], backend="kimi", root=tmp_path,
token_budget=None, chunk_size=1, max_concurrency=1,
deep_mode=True,
)
deep = load_cached(doc, tmp_path, kind="semantic-deep")
assert deep and [n["id"] for n in deep["nodes"]] == ["d1"], (
"deep-mode checkpoint must land in cache/semantic-deep/"
)
assert load_cached(doc, tmp_path, kind="semantic") is None, (
"deep-mode checkpoint must not write the standard semantic namespace"
)
def test_omitted_documents_are_reconciled_and_warned(tmp_path, capsys):
"""#1890: a chunk can return a clean, non-empty response that omits some of the
documents it was given. Those docs must not vanish silently — the run reports
+158
View File
@@ -222,6 +222,164 @@ def test_stamped_manifest_files_normalizes_both_sides(tmp_path):
assert out["document"] == [str(fresh_doc), str(cached_doc)]
# --- #1894: --force and deep-mode dispatch over a warm cache -----------------
def _recording_extractor(calls):
"""extract_corpus_parallel stand-in that records each dispatch."""
def _extract(paths, **kwargs):
calls.append({"paths": [str(p) for p in paths], "kwargs": kwargs})
on_chunk = kwargs.get("on_chunk_done")
if on_chunk:
on_chunk(0, 1, {"nodes": [], "edges": [], "hyperedges": []})
return {
"nodes": [{"id": "readme", "source_file": "README.md",
"file_type": "document"}],
"edges": [],
"hyperedges": [],
"input_tokens": 10,
"output_tokens": 5,
}
return _extract
def _run_extract(monkeypatch, argv):
monkeypatch.setattr(mainmod.sys, "argv", argv)
try:
mainmod.main()
except SystemExit as exc:
assert exc.code in (None, 0), f"unexpected exit code {exc.code}"
def test_extract_mode_deep_dispatches_over_warm_cache(monkeypatch, tmp_path):
"""#1894 repro: over a warm manifest + warm standard semantic cache,
`extract --mode deep` was a silent no-op — the incremental gate dispatched
zero files before the cache was ever consulted, and the cache key ignored
mode anyway. Deep must re-dispatch on the first deep run (deep namespace
cold) and be served from cache/semantic-deep/ on the second."""
corpus = _make_corpus(tmp_path)
monkeypatch.setenv("ANTHROPIC_API_KEY", "sk-test-fake-key")
monkeypatch.delenv("GRAPHIFY_FORCE", raising=False)
calls: list[dict] = []
monkeypatch.setattr("graphify.llm.extract_corpus_parallel",
_recording_extractor(calls))
monkeypatch.setattr(mainmod, "_check_skill_version", lambda _: None)
# No --out: the default layout (graphify-out/ beside the sources) keeps the
# CLI-level cache write's root anchored at the corpus, so the stub's
# root-relative source_file resolves (real runs also checkpoint per chunk
# inside llm.extract_corpus_parallel, which this stub replaces).
base = ["graphify", "extract", str(corpus), "--backend", "claude",
"--no-cluster"]
# Run 1: cold standard extraction — warms manifest + plain semantic cache.
_run_extract(monkeypatch, base)
assert len(calls) == 1
# Sanity: a warm standard re-run dispatches nothing (expected behavior).
_run_extract(monkeypatch, base)
assert len(calls) == 1
# The repro: warm tree + --mode deep MUST dispatch.
_run_extract(monkeypatch, base + ["--mode", "deep"])
assert len(calls) == 2, (
"--mode deep over a warm cache must re-dispatch (#1894)"
)
assert calls[1]["paths"] == [str(corpus / "README.md")]
assert calls[1]["kwargs"].get("deep_mode") is True
# Second deep run: served from the (now warm) deep namespace, no dispatch.
_run_extract(monkeypatch, base + ["--mode", "deep"])
assert len(calls) == 2, (
"second deep run must be served from cache/semantic-deep/"
)
# The deep entry landed in its own namespace, not cache/semantic/.
assert any((corpus / "graphify-out" / "cache" / "semantic-deep").glob("*.json"))
def test_extract_force_flag_redispatches_and_stamps_manifest(monkeypatch, tmp_path):
"""extract accepts --force: a warm tree re-dispatches every semantic file
(cache read skipped, incremental gate off) and the manifest is still
stamped afterward (#1897-compatible full coverage)."""
import json
corpus = _make_corpus(tmp_path)
monkeypatch.setenv("ANTHROPIC_API_KEY", "sk-test-fake-key")
monkeypatch.delenv("GRAPHIFY_FORCE", raising=False)
calls: list[dict] = []
monkeypatch.setattr("graphify.llm.extract_corpus_parallel",
_recording_extractor(calls))
monkeypatch.setattr(mainmod, "_check_skill_version", lambda _: None)
base = ["graphify", "extract", str(corpus), "--backend", "claude",
"--no-cluster"]
_run_extract(monkeypatch, base)
assert len(calls) == 1
_run_extract(monkeypatch, base) # warm: no dispatch
assert len(calls) == 1
_run_extract(monkeypatch, base + ["--force"])
assert len(calls) == 2, (
"--force over a warm tree must re-dispatch every semantic file"
)
assert calls[1]["paths"] == [str(corpus / "README.md")]
# The forced run still wrote the semantic cache and stamped the manifest.
assert any((corpus / "graphify-out" / "cache" / "semantic").glob("*.json"))
manifest = json.loads(
(corpus / "graphify-out" / "manifest.json").read_text()
)
assert manifest.get("README.md", {}).get("semantic_hash"), (
"forced re-dispatch must still stamp the manifest"
)
assert manifest.get("main.go", {}).get("semantic_hash")
def test_extract_graphify_force_env_redispatches(monkeypatch, tmp_path):
"""GRAPHIFY_FORCE=1 behaves like --force (env parity with `update`)."""
corpus = _make_corpus(tmp_path)
monkeypatch.setenv("ANTHROPIC_API_KEY", "sk-test-fake-key")
monkeypatch.delenv("GRAPHIFY_FORCE", raising=False)
calls: list[dict] = []
monkeypatch.setattr("graphify.llm.extract_corpus_parallel",
_recording_extractor(calls))
monkeypatch.setattr(mainmod, "_check_skill_version", lambda _: None)
base = ["graphify", "extract", str(corpus), "--backend", "claude",
"--no-cluster"]
_run_extract(monkeypatch, base)
assert len(calls) == 1
_run_extract(monkeypatch, base) # warm: no dispatch
assert len(calls) == 1
monkeypatch.setenv("GRAPHIFY_FORCE", "1")
_run_extract(monkeypatch, base)
assert len(calls) == 2, "GRAPHIFY_FORCE=1 must force a re-dispatch"
def test_cache_check_mode_deep_reads_deep_namespace(monkeypatch, tmp_path, capsys):
"""cache-check --mode deep consults cache/semantic-deep/; without the flag
it keeps reading cache/semantic/ (deep entries are invisible to it)."""
from graphify.cache import save_semantic_cache
doc = tmp_path / "doc.md"
doc.write_text("# Doc\n")
save_semantic_cache([{"id": "d", "source_file": "doc.md"}], [],
root=tmp_path, mode="deep")
files_from = tmp_path / "files.txt"
files_from.write_text(str(doc) + "\n")
monkeypatch.setattr(mainmod, "_check_skill_version", lambda _: None)
_run_extract(monkeypatch, ["graphify", "cache-check", str(files_from),
"--root", str(tmp_path)])
assert "Cache: 0 hit, 1 miss" in capsys.readouterr().out
_run_extract(monkeypatch, ["graphify", "cache-check", str(files_from),
"--root", str(tmp_path), "--mode", "deep"])
assert "Cache: 1 hit, 0 miss" in capsys.readouterr().out
def _code_only_corpus(tmp_path):
"""A corpus with only code — no docs/papers/images."""
(tmp_path / "auth.py").write_text(