From a4ab6ed3f6986451ee2e5653ee5aa0388865ae5a Mon Sep 17 00:00:00 2001 From: safishamsi Date: Tue, 14 Jul 2026 23:45:46 +0100 Subject: [PATCH] fix(llm): reconcile dispatched vs returned files in semantic extract (#1890) A semantic chunk can return a clean, non-empty response that omits some of the documents it was given. Those docs vanished from the graph with no node, no warning, and no cache/manifest stamp, so they were silently re-dispatched (and typically re-omitted) on every run. extract_corpus_ parallel now diffs the dispatched file set against the source_files that returned, records the gap in merged["uncovered_files"], and prints a loud warning naming the omitted files. Smallest visibility guard; routing docs through the deterministic extractor for a guaranteed file node is a separate, larger change. Adds a regression test: a chunk that omits odd-numbered docs is caught and warned, not silently dropped. Co-Authored-By: Claude Opus 4.8 (1M context) --- CHANGELOG.md | 2 ++ graphify/llm.py | 27 +++++++++++++++++++++++++++ tests/test_chunking.py | 34 ++++++++++++++++++++++++++++++++++ 3 files changed, 63 insertions(+) diff --git a/CHANGELOG.md b/CHANGELOG.md index b276d1e..00d0b3f 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,8 @@ Full release notes with details on each version: [GitHub Releases](https://githu ## 0.9.16 (unreleased) +- Fix: semantic extraction now reconciles dispatched files against returned results, so a document the model silently omits is no longer lost without a trace (#1890). A chunk can return a clean, non-empty response that simply leaves out some of the documents it was given; those docs previously produced no node, no warning, and no cache/manifest stamp, so they were re-dispatched and re-omitted on every run. `extract_corpus_parallel` now diffs the dispatched file set against the `source_file`s that came back, records the gap in `uncovered_files`, and prints a loud warning listing the omitted files. (This is the visibility guard; routing documents through the deterministic extractor so they always get at least a file node is tracked separately.) + - Fix: close two residual paths where an absolute scan path (including the OS username) still leaked into a committed `graph.json`, completing #1789 (#1899). (a) A reference target outside the scan root (an out-of-root `.csproj` ProjectReference, `.sln` project, or bash `source`) kept its absolute `source_file` and an absolute-derived id, because the relativization post-passes silently skipped anything `relative_to(root)` could not handle; such targets now get a portable walk-up relative path and an `ext_`-namespaced id (bare basename when the target is far outside the corpus or on another drive). (b) A symbol whose name normalizes to nothing (a minified `$` function, a JSONC `"//"` comment key) collapsed `_make_id(stem, name)` down to the bare absolute file stem; those no-signal symbols are now skipped at mint time. - Fix: uppercase TypeScript extensions (`.TS`/`.TSX`/`.MTS`/`.CTS`) are now parsed with the TypeScript grammar instead of falling through to the JavaScript grammar, which silently dropped interfaces and type aliases (#1881, thanks @xkam7ar). Detection and dispatch already lowercased, but the grammar selection inside `extract_js` compared the suffix case-sensitively. diff --git a/graphify/llm.py b/graphify/llm.py index 05e2186..d019b70 100644 --- a/graphify/llm.py +++ b/graphify/llm.py @@ -1987,6 +1987,33 @@ def extract_corpus_parallel( " — see errors above. Partial results returned.", file=sys.stderr, ) + + # Dispatch/return reconciliation (#1890). A chunk can return a clean, non-empty + # response that simply omits some of the documents it was given; those docs then + # vanish from the graph with no node, no warning, and no cache/manifest stamp, so + # they are silently re-dispatched (and re-omitted) forever. Diff the files we + # dispatched against the source_files that actually came back and surface the gap. + dispatched = {unit_path(f) for chunk in chunks for f in chunk} + covered: set[Path] = set() + for n in merged.get("nodes", []): + sf = n.get("source_file") + if sf: + p = Path(sf) + covered.add(p if p.is_absolute() else (root / p)) + uncovered = sorted( + p for p in dispatched + if p.resolve() not in {c.resolve() for c in covered} + ) + merged["uncovered_files"] = [str(p) for p in uncovered] + if uncovered: + shown = ", ".join(p.name for p in uncovered[:5]) + more = f" (+{len(uncovered) - 5} more)" if len(uncovered) > 5 else "" + print( + f"[graphify] WARNING: {len(uncovered)}/{len(dispatched)} dispatched file(s) " + f"produced no nodes and are absent from the graph: {shown}{more}. The model " + "returned a response but omitted them; a re-run will retry them.", + file=sys.stderr, + ) return merged diff --git a/tests/test_chunking.py b/tests/test_chunking.py index 42b4651..f0ec4f5 100644 --- a/tests/test_chunking.py +++ b/tests/test_chunking.py @@ -326,6 +326,40 @@ def test_checkpoint_scopes_cache_writes_to_chunk_files(tmp_path): assert a_cache and any(n["id"] == "a_ok" for n in a_cache["nodes"]) +def test_omitted_documents_are_reconciled_and_warned(tmp_path, capsys): + """#1890: a chunk can return a clean, non-empty response that omits some of the + documents it was given. Those docs must not vanish silently — the run reports + them in `uncovered_files` and warns, instead of dropping them with no signal.""" + from graphify.llm import extract_corpus_parallel + + docs = [] + for i in range(4): + f = tmp_path / f"doc{i}.md" + f.write_text(f"# Doc {i}\n\nsome content\n", encoding="utf-8") + docs.append(f) + + def omit_odd(chunk, **kwargs): + # Return nodes only for even-numbered docs; a clean response, not a failure. + nodes = [] + for u in chunk: + name = getattr(u, "path", u).name + idx = int(name[len("doc")]) + if idx % 2 == 0: + nodes.append({"id": f"n{idx}", "source_file": name, "file_type": "document"}) + return {"nodes": nodes, "edges": [], "hyperedges": [], "input_tokens": 1, "output_tokens": 1} + + with patch("graphify.llm.extract_files_direct", side_effect=omit_odd): + result = extract_corpus_parallel( + docs, backend="kimi", root=tmp_path, + token_budget=None, chunk_size=1, max_concurrency=1, + ) + + uncovered = {Path(p).name for p in result.get("uncovered_files", [])} + assert uncovered == {"doc1.md", "doc3.md"}, f"reconciliation missed omissions: {uncovered}" + err = capsys.readouterr().err + assert "produced no nodes" in err and "doc1.md" in err + + def test_checkpoint_caches_sliced_document_chunks(tmp_path, capsys): """#1870: the checkpoint's allowlist must resolve a FileSlice to its parent path (via unit_path), not read a non-existent `.rel`. An oversized doc is