fix(llm): reconcile dispatched vs returned files in semantic extract (#1890)
A semantic chunk can return a clean, non-empty response that omits some of the documents it was given. Those docs vanished from the graph with no node, no warning, and no cache/manifest stamp, so they were silently re-dispatched (and typically re-omitted) on every run. extract_corpus_ parallel now diffs the dispatched file set against the source_files that returned, records the gap in merged["uncovered_files"], and prints a loud warning naming the omitted files. Smallest visibility guard; routing docs through the deterministic extractor for a guaranteed file node is a separate, larger change. Adds a regression test: a chunk that omits odd-numbered docs is caught and warned, not silently dropped. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 4.8
parent
b3dc15b839
commit
a4ab6ed3f6
@@ -4,6 +4,8 @@ Full release notes with details on each version: [GitHub Releases](https://githu
|
||||
|
||||
## 0.9.16 (unreleased)
|
||||
|
||||
- Fix: semantic extraction now reconciles dispatched files against returned results, so a document the model silently omits is no longer lost without a trace (#1890). A chunk can return a clean, non-empty response that simply leaves out some of the documents it was given; those docs previously produced no node, no warning, and no cache/manifest stamp, so they were re-dispatched and re-omitted on every run. `extract_corpus_parallel` now diffs the dispatched file set against the `source_file`s that came back, records the gap in `uncovered_files`, and prints a loud warning listing the omitted files. (This is the visibility guard; routing documents through the deterministic extractor so they always get at least a file node is tracked separately.)
|
||||
|
||||
- Fix: close two residual paths where an absolute scan path (including the OS username) still leaked into a committed `graph.json`, completing #1789 (#1899). (a) A reference target outside the scan root (an out-of-root `.csproj` ProjectReference, `.sln` project, or bash `source`) kept its absolute `source_file` and an absolute-derived id, because the relativization post-passes silently skipped anything `relative_to(root)` could not handle; such targets now get a portable walk-up relative path and an `ext_`-namespaced id (bare basename when the target is far outside the corpus or on another drive). (b) A symbol whose name normalizes to nothing (a minified `$` function, a JSONC `"//"` comment key) collapsed `_make_id(stem, name)` down to the bare absolute file stem; those no-signal symbols are now skipped at mint time.
|
||||
|
||||
- Fix: uppercase TypeScript extensions (`.TS`/`.TSX`/`.MTS`/`.CTS`) are now parsed with the TypeScript grammar instead of falling through to the JavaScript grammar, which silently dropped interfaces and type aliases (#1881, thanks @xkam7ar). Detection and dispatch already lowercased, but the grammar selection inside `extract_js` compared the suffix case-sensitively.
|
||||
|
||||
@@ -1987,6 +1987,33 @@ def extract_corpus_parallel(
|
||||
" — see errors above. Partial results returned.",
|
||||
file=sys.stderr,
|
||||
)
|
||||
|
||||
# Dispatch/return reconciliation (#1890). A chunk can return a clean, non-empty
|
||||
# response that simply omits some of the documents it was given; those docs then
|
||||
# vanish from the graph with no node, no warning, and no cache/manifest stamp, so
|
||||
# they are silently re-dispatched (and re-omitted) forever. Diff the files we
|
||||
# dispatched against the source_files that actually came back and surface the gap.
|
||||
dispatched = {unit_path(f) for chunk in chunks for f in chunk}
|
||||
covered: set[Path] = set()
|
||||
for n in merged.get("nodes", []):
|
||||
sf = n.get("source_file")
|
||||
if sf:
|
||||
p = Path(sf)
|
||||
covered.add(p if p.is_absolute() else (root / p))
|
||||
uncovered = sorted(
|
||||
p for p in dispatched
|
||||
if p.resolve() not in {c.resolve() for c in covered}
|
||||
)
|
||||
merged["uncovered_files"] = [str(p) for p in uncovered]
|
||||
if uncovered:
|
||||
shown = ", ".join(p.name for p in uncovered[:5])
|
||||
more = f" (+{len(uncovered) - 5} more)" if len(uncovered) > 5 else ""
|
||||
print(
|
||||
f"[graphify] WARNING: {len(uncovered)}/{len(dispatched)} dispatched file(s) "
|
||||
f"produced no nodes and are absent from the graph: {shown}{more}. The model "
|
||||
"returned a response but omitted them; a re-run will retry them.",
|
||||
file=sys.stderr,
|
||||
)
|
||||
return merged
|
||||
|
||||
|
||||
|
||||
@@ -326,6 +326,40 @@ def test_checkpoint_scopes_cache_writes_to_chunk_files(tmp_path):
|
||||
assert a_cache and any(n["id"] == "a_ok" for n in a_cache["nodes"])
|
||||
|
||||
|
||||
def test_omitted_documents_are_reconciled_and_warned(tmp_path, capsys):
|
||||
"""#1890: a chunk can return a clean, non-empty response that omits some of the
|
||||
documents it was given. Those docs must not vanish silently — the run reports
|
||||
them in `uncovered_files` and warns, instead of dropping them with no signal."""
|
||||
from graphify.llm import extract_corpus_parallel
|
||||
|
||||
docs = []
|
||||
for i in range(4):
|
||||
f = tmp_path / f"doc{i}.md"
|
||||
f.write_text(f"# Doc {i}\n\nsome content\n", encoding="utf-8")
|
||||
docs.append(f)
|
||||
|
||||
def omit_odd(chunk, **kwargs):
|
||||
# Return nodes only for even-numbered docs; a clean response, not a failure.
|
||||
nodes = []
|
||||
for u in chunk:
|
||||
name = getattr(u, "path", u).name
|
||||
idx = int(name[len("doc")])
|
||||
if idx % 2 == 0:
|
||||
nodes.append({"id": f"n{idx}", "source_file": name, "file_type": "document"})
|
||||
return {"nodes": nodes, "edges": [], "hyperedges": [], "input_tokens": 1, "output_tokens": 1}
|
||||
|
||||
with patch("graphify.llm.extract_files_direct", side_effect=omit_odd):
|
||||
result = extract_corpus_parallel(
|
||||
docs, backend="kimi", root=tmp_path,
|
||||
token_budget=None, chunk_size=1, max_concurrency=1,
|
||||
)
|
||||
|
||||
uncovered = {Path(p).name for p in result.get("uncovered_files", [])}
|
||||
assert uncovered == {"doc1.md", "doc3.md"}, f"reconciliation missed omissions: {uncovered}"
|
||||
err = capsys.readouterr().err
|
||||
assert "produced no nodes" in err and "doc1.md" in err
|
||||
|
||||
|
||||
def test_checkpoint_caches_sliced_document_chunks(tmp_path, capsys):
|
||||
"""#1870: the checkpoint's allowlist must resolve a FileSlice to its parent
|
||||
path (via unit_path), not read a non-existent `.rel`. An oversized doc is
|
||||
|
||||
Reference in New Issue
Block a user