diff --git a/graphify/__main__.py b/graphify/__main__.py index c5151b8..b424327 100644 --- a/graphify/__main__.py +++ b/graphify/__main__.py @@ -597,6 +597,8 @@ def _run_cli() -> None: print(" proxy, gateways): set ANTHROPIC_BASE_URL and ANTHROPIC_MODEL") print(" --model M override backend default model") print(" --mode deep aggressive INFERRED-edge semantic extraction") + print(" --force full re-scan and re-dispatch: skip the incremental") + print(" manifest gate and semantic cache reads (env: GRAPHIFY_FORCE=1)") print(" --max-workers N AST extraction subprocess count (default: cpu_count)") print(" --token-budget N per-chunk token cap for semantic extraction (default: 60000)") print(" --max-concurrency N parallel semantic chunks in flight (default: 4; set 1 for local LLMs)") diff --git a/graphify/cache.py b/graphify/cache.py index 3055db7..15795d4 100644 --- a/graphify/cache.py +++ b/graphify/cache.py @@ -90,7 +90,8 @@ def _body_content(content: bytes) -> bytes: # size+mtime_ns are unchanged — same trade-off as make(1). # Correctness risks: `touch` causes a harmless extra re-hash; same-size edits # within NFS second-resolution mtime have a 1-second window (same as make). -# Use `graphify extract --force` to bypass when needed. +# `graphify extract --force` / `graphify update --force` (or GRAPHIFY_FORCE=1) +# skip the cache reads and re-dispatch everything when needed (#1894). _stat_index: dict[str, dict] = {} _stat_index_root: Path | None = None _stat_index_dirty: bool = False @@ -340,13 +341,15 @@ def _absolutize_source_files_in(payload: dict, root: Path) -> None: def cache_dir(root: Path = Path("."), kind: str = "ast") -> Path: """Returns the cache directory for ``kind`` - creates it if needed. - kind is "ast" or "semantic". Separate subdirectories prevent semantic cache + kind is "ast", "semantic", or a mode-namespaced semantic kind such as + "semantic-deep" (#1894). Separate subdirectories prevent semantic cache entries from overwriting AST cache entries for the same source_file (#582). AST entries live in graphify-out/cache/ast/v{version}/ — namespaced by graphify version because they depend on extractor code, not just file contents. Semantic entries live unversioned in graphify-out/cache/semantic/ - (re-extraction costs LLM calls). + (re-extraction costs LLM calls); deep-mode entries live beside them in + graphify-out/cache/semantic-deep/. """ _out = Path(_GRAPHIFY_OUT) base = _out if _out.is_absolute() else Path(root).resolve() / _out @@ -467,8 +470,10 @@ def cached_files(root: Path = Path(".")) -> set[str]: # Legacy flat entries if base.is_dir(): hashes.update(p.stem for p in base.glob("*.json")) - # Namespaced entries (ast/ recursively, covering per-version subdirs) - for kind, pattern in (("ast", "**/*.json"), ("semantic", "*.json")): + # Namespaced entries (ast/ recursively, covering per-version subdirs; + # semantic-deep/ holds --mode deep entries, #1894) + for kind, pattern in (("ast", "**/*.json"), ("semantic", "*.json"), + ("semantic-deep", "*.json")): d = base / kind if d.is_dir(): hashes.update(p.stem for p in d.glob(pattern)) @@ -476,14 +481,17 @@ def cached_files(root: Path = Path(".")) -> set[str]: def clear_cache(root: Path = Path(".")) -> None: - """Delete all cache entries (ast/, semantic/, and legacy flat entries).""" + """Delete all cache entries (ast/, semantic/, semantic-deep/, and legacy + flat entries).""" base = Path(root).resolve() / _GRAPHIFY_OUT / "cache" # Legacy flat entries if base.is_dir(): for f in base.glob("*.json"): f.unlink() - # Namespaced entries (ast/ recursively, covering per-version subdirs) - for kind, pattern in (("ast", "**/*.json"), ("semantic", "*.json")): + # Namespaced entries (ast/ recursively, covering per-version subdirs; + # semantic-deep/ holds --mode deep entries, #1894) + for kind, pattern in (("ast", "**/*.json"), ("semantic", "*.json"), + ("semantic-deep", "*.json")): d = base / kind if d.is_dir(): for f in d.glob(pattern): @@ -500,11 +508,16 @@ def prune_semantic_cache(root: Path, live_hashes: set[str]) -> int: AST version-cleanup, so every content change or file deletion leaves a permanent orphan entry that accumulates unbounded. - This sweeps ``cache/semantic/*.json`` and deletes any entry whose stem (the - content hash) is not in ``live_hashes`` — the hashes of the current live - document set. ``*.tmp`` atomic-write temporaries are skipped, and only this - directory is touched (never ``cache/ast/**`` or anything else). The - unversioned design is preserved: we prune by liveness, not by version. + This sweeps ``cache/semantic/*.json`` AND ``cache/semantic-deep/*.json`` + (the ``--mode deep`` namespace, #1894) and deletes any entry whose stem + (the content hash) is not in ``live_hashes`` — the hashes of the current + live document set. Both namespaces are pruned against the SAME live set: + liveness is content-based and mode-independent, so a hash that is live for + one namespace is live for both. Skipping the deep namespace would re-grow + the unbounded-orphan problem this function fixed (#1527). ``*.tmp`` + atomic-write temporaries are skipped, and only these directories are + touched (never ``cache/ast/**`` or anything else). The unversioned design + is preserved: we prune by liveness, not by version. Best-effort, mirroring :func:`_cleanup_stale_ast_entries`: each unlink is wrapped in ``try/except OSError`` and a failure is ignored. The worst-case @@ -513,30 +526,40 @@ def prune_semantic_cache(root: Path, live_hashes: set[str]) -> int: """ _out = Path(_GRAPHIFY_OUT) base = _out if _out.is_absolute() else Path(root).resolve() / _out - semantic_dir = base / "cache" / "semantic" - if not semantic_dir.is_dir(): - return 0 pruned = 0 - for entry in semantic_dir.glob("*.json"): - if entry.stem in live_hashes: + for kind in ("semantic", "semantic-deep"): + semantic_dir = base / "cache" / kind + if not semantic_dir.is_dir(): continue - try: - entry.unlink() - pruned += 1 - except OSError: - pass + for entry in semantic_dir.glob("*.json"): + if entry.stem in live_hashes: + continue + try: + entry.unlink() + pruned += 1 + except OSError: + pass return pruned def check_semantic_cache( files: list[str], root: Path = Path("."), + mode: str | None = None, ) -> tuple[list[dict], list[dict], list[dict], list[str]]: """Check semantic extraction cache for a list of absolute file paths. Returns (cached_nodes, cached_edges, cached_hyperedges, uncached_files). Uncached files need Claude extraction; cached files are merged directly. + + ``mode`` selects the cache namespace: ``None`` (the default) reads + ``cache/semantic/`` — byte-identical to the historical behavior, so + existing callers that omit it (including older installed skill flows) + are unaffected. A non-None mode (e.g. ``"deep"``) reads + ``cache/semantic-{mode}/`` instead, so deep-mode results never shadow + (or get shadowed by) standard-mode entries for the same content (#1894). """ + kind = "semantic" if mode is None else f"semantic-{mode}" cached_nodes: list[dict] = [] cached_edges: list[dict] = [] cached_hyperedges: list[dict] = [] @@ -546,7 +569,7 @@ def check_semantic_cache( p = Path(fpath) if not p.is_absolute(): p = Path(root) / p - result = load_cached(p, root, kind="semantic") + result = load_cached(p, root, kind=kind) if result is not None: cached_nodes.extend(result.get("nodes", [])) cached_edges.extend(result.get("edges", [])) @@ -564,6 +587,7 @@ def save_semantic_cache( root: Path = Path("."), merge_existing: bool = False, allowed_source_files: Iterable[str | Path] | None = None, + mode: str | None = None, ) -> int: """Save semantic extraction results to cache, keyed by source_file. @@ -571,6 +595,13 @@ def save_semantic_cache( under cache/semantic/ (separate from AST entries in cache/ast/) to prevent hash-key collisions (#582). + ``mode`` selects the cache namespace, mirroring + :func:`check_semantic_cache`: ``None`` (the default) writes + ``cache/semantic/`` — byte-identical to the historical behavior for + existing callers that omit it — while a non-None mode (e.g. ``"deep"``) + writes ``cache/semantic-{mode}/`` so richer deep-mode results never + overwrite standard-mode entries and vice versa (#1894). + When ``merge_existing`` is True, any already-cached entry for a file is unioned with the new results before saving instead of being overwritten. This lets callers checkpoint incrementally (e.g. once per chunk) without @@ -584,6 +615,7 @@ def save_semantic_cache( """ from collections import defaultdict + kind = "semantic" if mode is None else f"semantic-{mode}" by_file: dict[str, dict] = defaultdict(lambda: {"nodes": [], "edges": [], "hyperedges": []}) for n in nodes: src = n.get("source_file", "") @@ -684,13 +716,13 @@ def save_semantic_cache( ) continue if merge_existing: - prev = load_cached(p, root, kind="semantic") + prev = load_cached(p, root, kind=kind) if prev: result = { "nodes": (prev.get("nodes", []) or []) + result["nodes"], "edges": (prev.get("edges", []) or []) + result["edges"], "hyperedges": (prev.get("hyperedges", []) or []) + result["hyperedges"], } - save_cached(p, result, root, kind="semantic") + save_cached(p, result, root, kind=kind) saved += 1 return saved diff --git a/graphify/cli.py b/graphify/cli.py index 5061d85..f2ae1da 100644 --- a/graphify/cli.py +++ b/graphify/cli.py @@ -2138,6 +2138,9 @@ def dispatch_command(cmd: str) -> None: cli_exclude_hubs: float | None = None cli_excludes: list[str] = [] cli_timing: bool = False + # --force parity with `graphify update`: the flag or GRAPHIFY_FORCE=1 + # disables the incremental gate and skips semantic-cache reads (#1894). + force = os.environ.get("GRAPHIFY_FORCE", "").lower() in ("1", "true", "yes") def _parse_int(name: str, raw: str) -> int: try: @@ -2228,6 +2231,8 @@ def dispatch_command(cmd: str) -> None: elif a == "--cargo": cli_cargo = True i += 1 + elif a == "--force": + force = True; i += 1 elif a == "--timing": cli_timing = True; i += 1 else: @@ -2277,6 +2282,11 @@ def dispatch_command(cmd: str) -> None: manifest_path = graphify_out / "manifest.json" existing_graph_path = graphify_out / "graph.json" incremental_mode = manifest_path.exists() and existing_graph_path.exists() if has_path else False + # --force: full scan, not the manifest-gated incremental diff — a warm + # unchanged tree would otherwise dispatch zero files (#1894). + incremental_mode = incremental_mode and not force + if force: + print("[graphify extract] --force: full re-scan, semantic cache reads skipped") if not has_path: code_files = [] @@ -2343,6 +2353,30 @@ def dispatch_command(cmd: str) -> None: doc_files = [] paper_files = [] image_files = [] + if deep_mode and incremental_mode and not code_only: + # Deep mode reads/writes its own cache namespace + # (cache/semantic-deep/), so the manifest's changed-file gate is + # not a valid proxy for deep coverage: over a warm unchanged tree + # it dispatches zero files and `--mode deep` silently no-ops + # (#1894). Widen the semantic pass to the FULL live + # doc/paper/image set (``files_by_type`` from detect_incremental, + # which already excludes excluded files) and let the + # mode-namespaced cache decide hits/misses — the first deep run + # re-dispatches everything (deep namespace cold), later deep runs + # hit the deep cache. + _deep_all = [ + Path(p) + for _ftype in ("document", "paper", "image") + for p in files_by_type.get(_ftype, []) + ] + if len(_deep_all) != len(semantic_files): + print( + f"[graphify extract] deep mode: widening semantic pass from " + f"{len(semantic_files)} changed to {len(_deep_all)} live " + f"doc/paper/image file(s); the deep semantic cache decides " + f"what is re-extracted" + ) + semantic_files = _deep_all if incremental_mode: # Excluded-but-alive files are reported separately from deletions # (#1908): they still exist on disk, the scan just stopped @@ -2495,11 +2529,21 @@ def dispatch_command(cmd: str) -> None: } sem_cache_hits = 0 sem_cache_misses = 0 + # Deep mode uses its own namespace (cache/semantic-deep/) so deep and + # standard results for the same content never shadow each other (#1894). + sem_cache_mode = "deep" if deep_mode else None if semantic_files: sem_paths_str = [str(p) for p in semantic_files] - cached_nodes, cached_edges, cached_hyperedges, uncached_paths = ( - _check_semantic_cache(sem_paths_str, root=out_root) - ) + if force: + # --force: skip the cache READ so every semantic file is + # re-dispatched; the save below still runs so the fresh + # results replace the stale entries. + cached_nodes, cached_edges, cached_hyperedges = [], [], [] + uncached_paths = list(sem_paths_str) + else: + cached_nodes, cached_edges, cached_hyperedges, uncached_paths = ( + _check_semantic_cache(sem_paths_str, root=out_root, mode=sem_cache_mode) + ) sem_cache_hits = len(semantic_files) - len(uncached_paths) sem_cache_misses = len(uncached_paths) sem_result["nodes"].extend(cached_nodes) @@ -2570,6 +2614,7 @@ def dispatch_command(cmd: str) -> None: fresh.get("hyperedges", []), root=out_root, allowed_source_files=uncached_paths, + mode=sem_cache_mode, ) except Exception as exc: print(f"[graphify extract] warning: could not write semantic cache: {exc}", file=sys.stderr) @@ -2880,27 +2925,41 @@ def dispatch_command(cmd: str) -> None: stages.total() elif cmd == "cache-check": - # graphify cache-check [--root ] + # graphify cache-check [--root ] [--mode | --deep] # Reads file paths (one per line) from , checks semantic cache. + # --mode deep (or --deep) checks the cache/semantic-deep/ namespace + # written by `extract --mode deep` instead of cache/semantic/ (#1894). # Writes: # graphify-out/.graphify_cached.json — already-cached nodes/edges/hyperedges # graphify-out/.graphify_uncached.txt — paths that need extraction # Stdout: "Cache: N hit, M miss" from graphify.cache import check_semantic_cache if len(sys.argv) < 3: - print("Usage: graphify cache-check [--root ]", file=sys.stderr) + print("Usage: graphify cache-check [--root ] [--mode | --deep]", file=sys.stderr) sys.exit(1) files_from = Path(sys.argv[2]) root = Path(".") + cache_mode: str | None = None i = 3 while i < len(sys.argv): if sys.argv[i] == "--root" and i + 1 < len(sys.argv): root = Path(sys.argv[i + 1]) i += 2 + elif sys.argv[i] == "--mode" and i + 1 < len(sys.argv): + cache_mode = sys.argv[i + 1] + i += 2 + elif sys.argv[i].startswith("--mode="): + cache_mode = sys.argv[i].split("=", 1)[1] + i += 1 + elif sys.argv[i] == "--deep": + cache_mode = "deep" + i += 1 else: i += 1 files = [f for f in files_from.read_text(encoding="utf-8").splitlines() if f.strip()] - cached_nodes, cached_edges, cached_hyperedges, uncached = check_semantic_cache(files, root) + cached_nodes, cached_edges, cached_hyperedges, uncached = check_semantic_cache( + files, root, mode=cache_mode + ) out = root / _GRAPHIFY_OUT out.mkdir(parents=True, exist_ok=True) if cached_nodes or cached_edges or cached_hyperedges: diff --git a/graphify/llm.py b/graphify/llm.py index 52e2b48..e3dbf68 100644 --- a/graphify/llm.py +++ b/graphify/llm.py @@ -1924,6 +1924,9 @@ def extract_corpus_parallel( # chunk leaked the FileSlice object into the allowlist and the write # raised TypeError, silently defeating the checkpoint.) allowed = [unit_path(item) for item in chunk] + # Deep-mode results checkpoint into their own namespace + # (cache/semantic-deep/) so a deep run never overwrites standard + # entries — and a later standard run never serves deep ones (#1894). _scs( result.get("nodes", []), result.get("edges", []), @@ -1931,6 +1934,7 @@ def extract_corpus_parallel( root=root, merge_existing=True, allowed_source_files=allowed, + mode="deep" if deep_mode else None, ) except Exception as _exc: # noqa: BLE001 — checkpoint is best-effort print(f"[graphify] incremental cache checkpoint failed: {_exc}", file=sys.stderr) diff --git a/tests/test_cache.py b/tests/test_cache.py index 8588df9..ff89a32 100644 --- a/tests/test_cache.py +++ b/tests/test_cache.py @@ -570,6 +570,149 @@ def test_save_semantic_cache_rejects_out_of_scope_source_file(tmp_path): assert protected_cache["hyperedges"] == [] +# --- #1894: mode-namespaced semantic cache ----------------------------------- +# `extract --mode deep` produces richer results than standard extraction, so +# deep entries live in their own namespace (cache/semantic-deep/). mode=None +# must stay byte-identical to the historical behavior: older installed skill +# flows call check/save without the parameter and must be unaffected. + +def test_semantic_cache_deep_mode_roundtrip_under_deep_namespace(tmp_path): + """mode='deep' saves under cache/semantic-deep/ and reads back from it.""" + from graphify.cache import check_semantic_cache, save_semantic_cache + + f = tmp_path / "doc.md" + f.write_text("# Doc\n\nBody.\n") + saved = save_semantic_cache( + [{"id": "deep_n", "source_file": "doc.md"}], [], root=tmp_path, mode="deep" + ) + assert saved == 1 + + deep_dir = tmp_path / "graphify-out" / "cache" / "semantic-deep" + h = file_hash(f, tmp_path) + assert (deep_dir / f"{h}.json").exists(), ( + "deep entry must land under cache/semantic-deep/" + ) + # And NOT in the plain namespace. + plain_dir = tmp_path / "graphify-out" / "cache" / "semantic" + assert not (plain_dir / f"{h}.json").exists() + + nodes, edges, hyper, uncached = check_semantic_cache( + [str(f)], root=tmp_path, mode="deep" + ) + assert [n["id"] for n in nodes] == ["deep_n"] + assert uncached == [] + + +def test_semantic_cache_deep_invisible_to_plain_reads_and_vice_versa(tmp_path): + """Deep entries must not satisfy mode=None reads (and plain entries must + not satisfy deep reads) — the namespaces are fully isolated.""" + from graphify.cache import check_semantic_cache, save_semantic_cache + + deep_doc = tmp_path / "deep.md" + deep_doc.write_text("# Deep\n") + plain_doc = tmp_path / "plain.md" + plain_doc.write_text("# Plain\n") + + save_semantic_cache([{"id": "d", "source_file": "deep.md"}], [], + root=tmp_path, mode="deep") + save_semantic_cache([{"id": "p", "source_file": "plain.md"}], [], + root=tmp_path) # mode omitted: historical call shape + + # Plain read: deep entry is a miss, plain entry is a hit. + nodes, _, _, uncached = check_semantic_cache( + [str(deep_doc), str(plain_doc)], root=tmp_path + ) + assert [n["id"] for n in nodes] == ["p"] + assert uncached == [str(deep_doc)] + + # Deep read: mirror image. + nodes, _, _, uncached = check_semantic_cache( + [str(deep_doc), str(plain_doc)], root=tmp_path, mode="deep" + ) + assert [n["id"] for n in nodes] == ["d"] + assert uncached == [str(plain_doc)] + + +def test_semantic_cache_mode_none_layout_unchanged(tmp_path): + """Omitting mode writes exactly the historical cache/semantic/ layout — + forward-compat for older installed callers that never pass mode.""" + from graphify.cache import check_semantic_cache, save_semantic_cache + + f = tmp_path / "doc.md" + f.write_text("# Doc\n") + save_semantic_cache([{"id": "n", "source_file": "doc.md"}], [], root=tmp_path) + h = file_hash(f, tmp_path) + assert (tmp_path / "graphify-out" / "cache" / "semantic" / f"{h}.json").exists() + assert not (tmp_path / "graphify-out" / "cache" / "semantic-deep").exists(), ( + "mode=None must never create the deep namespace" + ) + nodes, _, _, uncached = check_semantic_cache([str(f)], root=tmp_path) + assert [n["id"] for n in nodes] == ["n"] and uncached == [] + + +def test_clear_cache_removes_deep_namespace(tmp_path): + """clear_cache sweeps cache/semantic-deep/ alongside semantic/ and ast/.""" + from graphify.cache import save_semantic_cache + + f = tmp_path / "doc.md" + f.write_text("# Doc\n") + save_semantic_cache([{"id": "p", "source_file": "doc.md"}], [], root=tmp_path) + save_semantic_cache([{"id": "d", "source_file": "doc.md"}], [], + root=tmp_path, mode="deep") + base = tmp_path / "graphify-out" / "cache" + assert list((base / "semantic").glob("*.json")) + assert list((base / "semantic-deep").glob("*.json")) + + clear_cache(tmp_path) + assert not list(base.rglob("*.json")), ( + "clear_cache must remove entries in BOTH semantic namespaces" + ) + + +def test_cached_files_includes_deep_namespace(tmp_path): + """cached_files reports deep-namespace entries too.""" + from graphify.cache import save_semantic_cache + + f = tmp_path / "doc.md" + f.write_text("# Doc\n") + save_semantic_cache([{"id": "d", "source_file": "doc.md"}], [], + root=tmp_path, mode="deep") + assert file_hash(f, tmp_path) in cached_files(tmp_path) + + +def test_semantic_prune_sweeps_both_namespaces_against_same_live_set(tmp_path): + """#1894 follow-up to #1527: prune must sweep cache/semantic/ AND + cache/semantic-deep/ against the SAME live-hash set (liveness is + content-based, mode-independent). Orphans go in both namespaces; live + entries survive in both.""" + from graphify.cache import prune_semantic_cache, save_semantic_cache + + f = tmp_path / "doc.md" + f.write_text("# A\n\nContent A.\n") + h_old = file_hash(f, tmp_path) + save_semantic_cache([{"id": "pa", "source_file": "doc.md"}], [], root=tmp_path) + save_semantic_cache([{"id": "da", "source_file": "doc.md"}], [], + root=tmp_path, mode="deep") + + f.write_text("# B\n\nContent B.\n") + h_live = file_hash(f, tmp_path) + save_semantic_cache([{"id": "pb", "source_file": "doc.md"}], [], root=tmp_path) + save_semantic_cache([{"id": "db", "source_file": "doc.md"}], [], + root=tmp_path, mode="deep") + + plain_dir = tmp_path / "graphify-out" / "cache" / "semantic" + deep_dir = tmp_path / "graphify-out" / "cache" / "semantic-deep" + for d in (plain_dir, deep_dir): + assert (d / f"{h_old}.json").exists() + assert (d / f"{h_live}.json").exists() + + pruned = prune_semantic_cache(tmp_path, {h_live}) + assert pruned == 2, "one orphan in EACH namespace must be pruned" + for d in (plain_dir, deep_dir): + assert not (d / f"{h_old}.json").exists(), f"orphan survived in {d.name}" + assert (d / f"{h_live}.json").exists(), f"live entry pruned from {d.name}" + + def test_save_semantic_cache_merge_existing_unions(tmp_path): """#1715: merge_existing=True unions with the prior entry so a file split across chunks (checkpointed per chunk) keeps every slice.""" diff --git a/tests/test_chunking.py b/tests/test_chunking.py index 546503a..a978a5d 100644 --- a/tests/test_chunking.py +++ b/tests/test_chunking.py @@ -326,6 +326,38 @@ def test_checkpoint_scopes_cache_writes_to_chunk_files(tmp_path): assert a_cache and any(n["id"] == "a_ok" for n in a_cache["nodes"]) +def test_checkpoint_writes_deep_namespace_in_deep_mode(tmp_path): + """#1894: the per-chunk checkpoint must follow the run's mode — a + deep_mode=True run checkpoints into cache/semantic-deep/, leaving the + standard cache/semantic/ namespace untouched (and vice versa).""" + from graphify.llm import extract_corpus_parallel + from graphify.cache import load_cached + + doc = tmp_path / "doc.md" + doc.write_text("# Doc\n\nsome content\n") + + def ok(chunk, **kwargs): + return { + "nodes": [{"id": "d1", "source_file": "doc.md", "file_type": "document"}], + "edges": [], "hyperedges": [], "input_tokens": 1, "output_tokens": 1, + } + + with patch("graphify.llm.extract_files_direct", side_effect=ok): + extract_corpus_parallel( + [doc], backend="kimi", root=tmp_path, + token_budget=None, chunk_size=1, max_concurrency=1, + deep_mode=True, + ) + + deep = load_cached(doc, tmp_path, kind="semantic-deep") + assert deep and [n["id"] for n in deep["nodes"]] == ["d1"], ( + "deep-mode checkpoint must land in cache/semantic-deep/" + ) + assert load_cached(doc, tmp_path, kind="semantic") is None, ( + "deep-mode checkpoint must not write the standard semantic namespace" + ) + + def test_omitted_documents_are_reconciled_and_warned(tmp_path, capsys): """#1890: a chunk can return a clean, non-empty response that omits some of the documents it was given. Those docs must not vanish silently — the run reports diff --git a/tests/test_extract_cli.py b/tests/test_extract_cli.py index 3442a48..6ef1b2c 100644 --- a/tests/test_extract_cli.py +++ b/tests/test_extract_cli.py @@ -222,6 +222,164 @@ def test_stamped_manifest_files_normalizes_both_sides(tmp_path): assert out["document"] == [str(fresh_doc), str(cached_doc)] +# --- #1894: --force and deep-mode dispatch over a warm cache ----------------- + +def _recording_extractor(calls): + """extract_corpus_parallel stand-in that records each dispatch.""" + def _extract(paths, **kwargs): + calls.append({"paths": [str(p) for p in paths], "kwargs": kwargs}) + on_chunk = kwargs.get("on_chunk_done") + if on_chunk: + on_chunk(0, 1, {"nodes": [], "edges": [], "hyperedges": []}) + return { + "nodes": [{"id": "readme", "source_file": "README.md", + "file_type": "document"}], + "edges": [], + "hyperedges": [], + "input_tokens": 10, + "output_tokens": 5, + } + return _extract + + +def _run_extract(monkeypatch, argv): + monkeypatch.setattr(mainmod.sys, "argv", argv) + try: + mainmod.main() + except SystemExit as exc: + assert exc.code in (None, 0), f"unexpected exit code {exc.code}" + + +def test_extract_mode_deep_dispatches_over_warm_cache(monkeypatch, tmp_path): + """#1894 repro: over a warm manifest + warm standard semantic cache, + `extract --mode deep` was a silent no-op — the incremental gate dispatched + zero files before the cache was ever consulted, and the cache key ignored + mode anyway. Deep must re-dispatch on the first deep run (deep namespace + cold) and be served from cache/semantic-deep/ on the second.""" + corpus = _make_corpus(tmp_path) + monkeypatch.setenv("ANTHROPIC_API_KEY", "sk-test-fake-key") + monkeypatch.delenv("GRAPHIFY_FORCE", raising=False) + calls: list[dict] = [] + monkeypatch.setattr("graphify.llm.extract_corpus_parallel", + _recording_extractor(calls)) + monkeypatch.setattr(mainmod, "_check_skill_version", lambda _: None) + + # No --out: the default layout (graphify-out/ beside the sources) keeps the + # CLI-level cache write's root anchored at the corpus, so the stub's + # root-relative source_file resolves (real runs also checkpoint per chunk + # inside llm.extract_corpus_parallel, which this stub replaces). + base = ["graphify", "extract", str(corpus), "--backend", "claude", + "--no-cluster"] + + # Run 1: cold standard extraction — warms manifest + plain semantic cache. + _run_extract(monkeypatch, base) + assert len(calls) == 1 + + # Sanity: a warm standard re-run dispatches nothing (expected behavior). + _run_extract(monkeypatch, base) + assert len(calls) == 1 + + # The repro: warm tree + --mode deep MUST dispatch. + _run_extract(monkeypatch, base + ["--mode", "deep"]) + assert len(calls) == 2, ( + "--mode deep over a warm cache must re-dispatch (#1894)" + ) + assert calls[1]["paths"] == [str(corpus / "README.md")] + assert calls[1]["kwargs"].get("deep_mode") is True + + # Second deep run: served from the (now warm) deep namespace, no dispatch. + _run_extract(monkeypatch, base + ["--mode", "deep"]) + assert len(calls) == 2, ( + "second deep run must be served from cache/semantic-deep/" + ) + # The deep entry landed in its own namespace, not cache/semantic/. + assert any((corpus / "graphify-out" / "cache" / "semantic-deep").glob("*.json")) + + +def test_extract_force_flag_redispatches_and_stamps_manifest(monkeypatch, tmp_path): + """extract accepts --force: a warm tree re-dispatches every semantic file + (cache read skipped, incremental gate off) and the manifest is still + stamped afterward (#1897-compatible full coverage).""" + import json + + corpus = _make_corpus(tmp_path) + monkeypatch.setenv("ANTHROPIC_API_KEY", "sk-test-fake-key") + monkeypatch.delenv("GRAPHIFY_FORCE", raising=False) + calls: list[dict] = [] + monkeypatch.setattr("graphify.llm.extract_corpus_parallel", + _recording_extractor(calls)) + monkeypatch.setattr(mainmod, "_check_skill_version", lambda _: None) + + base = ["graphify", "extract", str(corpus), "--backend", "claude", + "--no-cluster"] + + _run_extract(monkeypatch, base) + assert len(calls) == 1 + _run_extract(monkeypatch, base) # warm: no dispatch + assert len(calls) == 1 + + _run_extract(monkeypatch, base + ["--force"]) + assert len(calls) == 2, ( + "--force over a warm tree must re-dispatch every semantic file" + ) + assert calls[1]["paths"] == [str(corpus / "README.md")] + + # The forced run still wrote the semantic cache and stamped the manifest. + assert any((corpus / "graphify-out" / "cache" / "semantic").glob("*.json")) + manifest = json.loads( + (corpus / "graphify-out" / "manifest.json").read_text() + ) + assert manifest.get("README.md", {}).get("semantic_hash"), ( + "forced re-dispatch must still stamp the manifest" + ) + assert manifest.get("main.go", {}).get("semantic_hash") + + +def test_extract_graphify_force_env_redispatches(monkeypatch, tmp_path): + """GRAPHIFY_FORCE=1 behaves like --force (env parity with `update`).""" + corpus = _make_corpus(tmp_path) + monkeypatch.setenv("ANTHROPIC_API_KEY", "sk-test-fake-key") + monkeypatch.delenv("GRAPHIFY_FORCE", raising=False) + calls: list[dict] = [] + monkeypatch.setattr("graphify.llm.extract_corpus_parallel", + _recording_extractor(calls)) + monkeypatch.setattr(mainmod, "_check_skill_version", lambda _: None) + + base = ["graphify", "extract", str(corpus), "--backend", "claude", + "--no-cluster"] + + _run_extract(monkeypatch, base) + assert len(calls) == 1 + _run_extract(monkeypatch, base) # warm: no dispatch + assert len(calls) == 1 + + monkeypatch.setenv("GRAPHIFY_FORCE", "1") + _run_extract(monkeypatch, base) + assert len(calls) == 2, "GRAPHIFY_FORCE=1 must force a re-dispatch" + + +def test_cache_check_mode_deep_reads_deep_namespace(monkeypatch, tmp_path, capsys): + """cache-check --mode deep consults cache/semantic-deep/; without the flag + it keeps reading cache/semantic/ (deep entries are invisible to it).""" + from graphify.cache import save_semantic_cache + + doc = tmp_path / "doc.md" + doc.write_text("# Doc\n") + save_semantic_cache([{"id": "d", "source_file": "doc.md"}], [], + root=tmp_path, mode="deep") + files_from = tmp_path / "files.txt" + files_from.write_text(str(doc) + "\n") + monkeypatch.setattr(mainmod, "_check_skill_version", lambda _: None) + + _run_extract(monkeypatch, ["graphify", "cache-check", str(files_from), + "--root", str(tmp_path)]) + assert "Cache: 0 hit, 1 miss" in capsys.readouterr().out + + _run_extract(monkeypatch, ["graphify", "cache-check", str(files_from), + "--root", str(tmp_path), "--mode", "deep"]) + assert "Cache: 1 hit, 0 miss" in capsys.readouterr().out + + def _code_only_corpus(tmp_path): """A corpus with only code — no docs/papers/images.""" (tmp_path / "auth.py").write_text(