fix(extract): don't cache zero-node results; warn on empty source files (#1666)
krishnateja7 reported that on a full-repo run a stable subset of Ruby files yields zero nodes (not even a file node), each fine in isolation, drop set byte-stable across runs. Root cause is a transient batch/parallel extraction that produces an empty result, which then gets cached and persists. Every extractable file yields at least a file node, so a zero-node result is anomalous. Both extraction paths (parallel worker and sequential fallback) now skip the cache write when a non-error result has no nodes, so a rerun re-extracts and self-heals instead of loading the stale empty. extract() also warns, listing the files that landed in the graph with zero nodes, so the previously-silent blindness in affected/explain is visible. This addresses the persistence and the silent blindness. The underlying trigger (why a valid file occasionally extracts empty when co-processed with certain others) was not reproducible with synthetic corpora; the warning now surfaces it for a concrete report if it recurs. Full suite: 2912 passed, 3 skipped. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 4.8
parent
3140b2ecce
commit
1288a557e3
@@ -4,6 +4,7 @@ Full release notes with details on each version: [GitHub Releases](https://githu
|
||||
|
||||
## Unreleased
|
||||
|
||||
- Fix: an extractable source file that produces zero nodes is no longer cached, and is surfaced with a warning (#1666, thanks @krishnateja7). Every supported file yields at least a file node, so a zero-node result is anomalous (a transient batch/parallel hiccup). Caching it made the empty byte-stable across runs and silently blinded `affected`/`explain` to and through the file. The cache write is now skipped for a zero-node result so a rerun self-heals, and `extract` warns when an accepted source file lands in the graph with no nodes. This addresses the persistence and the silent blindness; if the underlying zero-node extraction still reproduces on a specific corpus, the warning now makes it visible to report.
|
||||
- Fix: the Windows skill variant now declares `name: graphify` instead of `name: graphify-windows` (#1635, thanks @ray8875). `graphify install --platform windows` writes the variant to `~/.claude/skills/graphify/SKILL.md`, but Claude Code requires the skill folder name to equal the frontmatter `name`, so the `-windows` suffix broke discovery/validation. The variant suffix is a packaging detail, not part of the skill's identity.
|
||||
- Fix: the OpenCode plugin joins its reminder to the user's command with `;` instead of `&&` (#1646, thanks @gonaik). Windows PowerShell 5.1 rejects `&&` as a statement separator (`not a valid statement separator`), so the first bash command of every OpenCode session on Windows failed. `;` works in PowerShell 5.1, Bash, and POSIX shells. (Both the OpenCode and Kilo plugin templates are fixed.)
|
||||
- Fix: the `GRAPH_REPORT.md` "Import Cycles" section is now emitted only when the graph contains code (#1657, thanks @Ns2384-star). On a documents-only corpus there are no imports, so the section was pure noise ("None detected") on every run; it is now conditioned on code nodes or import edges being present. (The same report also confirms the mojibake and stdout-encoding items in that issue are already addressed on the current branch: manifest.json and `GRAPH_REPORT.md` are written UTF-8, and the CLI reconfigures stdout/stderr to UTF-8 with `errors="replace"`.)
|
||||
|
||||
+29
-2
@@ -15904,7 +15904,12 @@ def _extract_single_file(args: tuple) -> tuple[int, dict]:
|
||||
return idx, {"nodes": [], "edges": []}
|
||||
|
||||
result = _safe_extract_with_xaml_root(extractor, path, cache_root)
|
||||
if not bypass_cache and "error" not in result:
|
||||
# Never cache a zero-node result for an extractable file. Every supported
|
||||
# source produces at least a file node, so an empty node list is anomalous
|
||||
# (e.g. a transient batch/parallel hiccup). Caching it makes the empty
|
||||
# byte-stable across runs and silently blinds affected/explain to and
|
||||
# through the file (#1666); skipping the write lets a rerun self-heal.
|
||||
if not bypass_cache and "error" not in result and result.get("nodes"):
|
||||
save_cached(path, result, cache_root)
|
||||
return idx, result
|
||||
|
||||
@@ -16028,7 +16033,8 @@ def _extract_sequential(
|
||||
continue
|
||||
bypass_cache = path.suffix in _JS_CACHE_BYPASS_SUFFIXES
|
||||
result = _safe_extract_with_xaml_root(extractor, path, effective_root)
|
||||
if not bypass_cache and "error" not in result:
|
||||
# See _extract_single_file: don't cache an anomalous zero-node result (#1666).
|
||||
if not bypass_cache and "error" not in result and result.get("nodes"):
|
||||
save_cached(path, result, effective_root)
|
||||
per_file[idx] = result
|
||||
if total_files >= _PROGRESS_INTERVAL:
|
||||
@@ -16124,6 +16130,27 @@ def extract(
|
||||
if per_file[i] is None:
|
||||
per_file[i] = {"nodes": [], "edges": []}
|
||||
|
||||
# #1666: surface any source file an extractor accepted but that produced zero
|
||||
# nodes (not even a file node). Such a file is silently absent from the graph,
|
||||
# so affected/explain are blind to and through it with no other signal.
|
||||
_empty_sources: list[str] = []
|
||||
for i, _p in enumerate(paths):
|
||||
_res = per_file[i] or {}
|
||||
if _res.get("nodes") or _res.get("error"):
|
||||
continue
|
||||
if _get_extractor(_p) is not None:
|
||||
_empty_sources.append(str(_p))
|
||||
if _empty_sources:
|
||||
_shown = ", ".join(Path(x).name for x in _empty_sources[:5])
|
||||
_more = f" (+{len(_empty_sources) - 5} more)" if len(_empty_sources) > 5 else ""
|
||||
print(
|
||||
f" warning: {len(_empty_sources)} source file(s) produced zero nodes and "
|
||||
f"are absent from the graph: {_shown}{_more}. A re-run will retry them "
|
||||
f"(empties are no longer cached); if it persists, please report the "
|
||||
f"file(s) (#1666).",
|
||||
file=sys.stderr, flush=True,
|
||||
)
|
||||
|
||||
all_nodes: list[dict] = []
|
||||
all_edges: list[dict] = []
|
||||
all_raw_calls: list[dict] = []
|
||||
|
||||
@@ -0,0 +1,54 @@
|
||||
"""#1666 — an extractable source file that yields zero nodes must not be cached,
|
||||
and must be surfaced.
|
||||
|
||||
Every supported file produces at least a file node, so a zero-node result is
|
||||
anomalous (a transient batch/parallel hiccup). Caching it made the empty
|
||||
byte-stable across runs and silently blinded affected/explain to the file. We
|
||||
now skip the cache write for a zero-node result (so a rerun self-heals) and warn.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
import graphify.extract as ex
|
||||
|
||||
|
||||
def test_zero_node_result_not_cached_then_self_heals(tmp_path, capsys, monkeypatch):
|
||||
f = tmp_path / "thing.rb"
|
||||
f.write_text("class Foo\n def bar; end\nend\n")
|
||||
|
||||
real = ex._safe_extract_with_xaml_root
|
||||
|
||||
def _empty(extractor, path, root):
|
||||
return {"nodes": [], "edges": []}
|
||||
|
||||
# First run: force a zero-node extraction for this file.
|
||||
monkeypatch.setattr(ex, "_safe_extract_with_xaml_root", _empty)
|
||||
ex.extract([f], cache_root=tmp_path / "out", parallel=False)
|
||||
|
||||
err = capsys.readouterr().err
|
||||
assert "zero nodes" in err and "thing.rb" in err, err
|
||||
|
||||
# Second run with the real extractor: because the empty was NOT cached, the
|
||||
# file re-extracts and lands in the graph (self-heal).
|
||||
monkeypatch.setattr(ex, "_safe_extract_with_xaml_root", real)
|
||||
r2 = ex.extract([f], cache_root=tmp_path / "out", parallel=False)
|
||||
assert any(str(n.get("source_file", "")).endswith("thing.rb") for n in r2["nodes"])
|
||||
|
||||
|
||||
def test_normal_file_still_cached(tmp_path):
|
||||
# Guard against over-correction: a normal (non-empty) result must still cache.
|
||||
f = tmp_path / "ok.rb"
|
||||
f.write_text("class Bar\n def baz; end\nend\n")
|
||||
r1 = ex.extract([f], cache_root=tmp_path / "out", parallel=False)
|
||||
assert r1["nodes"]
|
||||
from graphify.cache import load_cached
|
||||
assert load_cached(f, tmp_path / "out") is not None, "non-empty result should be cached"
|
||||
|
||||
|
||||
def test_no_warning_when_all_files_produce_nodes(tmp_path, capsys):
|
||||
f = tmp_path / "fine.rb"
|
||||
f.write_text("module M\n def self.go; end\nend\n")
|
||||
ex.extract([f], cache_root=tmp_path / "out", parallel=False)
|
||||
err = capsys.readouterr().err
|
||||
assert "zero nodes" not in err
|
||||
Reference in New Issue
Block a user