diff --git a/CHANGELOG.md b/CHANGELOG.md index 0006c4f..ce253d7 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,7 @@ Full release notes with details on each version: [GitHub Releases](https://githu ## Unreleased +- Fix: an extractable source file that produces zero nodes is no longer cached, and is surfaced with a warning (#1666, thanks @krishnateja7). Every supported file yields at least a file node, so a zero-node result is anomalous (a transient batch/parallel hiccup). Caching it made the empty byte-stable across runs and silently blinded `affected`/`explain` to and through the file. The cache write is now skipped for a zero-node result so a rerun self-heals, and `extract` warns when an accepted source file lands in the graph with no nodes. This addresses the persistence and the silent blindness; if the underlying zero-node extraction still reproduces on a specific corpus, the warning now makes it visible to report. - Fix: the Windows skill variant now declares `name: graphify` instead of `name: graphify-windows` (#1635, thanks @ray8875). `graphify install --platform windows` writes the variant to `~/.claude/skills/graphify/SKILL.md`, but Claude Code requires the skill folder name to equal the frontmatter `name`, so the `-windows` suffix broke discovery/validation. The variant suffix is a packaging detail, not part of the skill's identity. - Fix: the OpenCode plugin joins its reminder to the user's command with `;` instead of `&&` (#1646, thanks @gonaik). Windows PowerShell 5.1 rejects `&&` as a statement separator (`not a valid statement separator`), so the first bash command of every OpenCode session on Windows failed. `;` works in PowerShell 5.1, Bash, and POSIX shells. (Both the OpenCode and Kilo plugin templates are fixed.) - Fix: the `GRAPH_REPORT.md` "Import Cycles" section is now emitted only when the graph contains code (#1657, thanks @Ns2384-star). On a documents-only corpus there are no imports, so the section was pure noise ("None detected") on every run; it is now conditioned on code nodes or import edges being present. (The same report also confirms the mojibake and stdout-encoding items in that issue are already addressed on the current branch: manifest.json and `GRAPH_REPORT.md` are written UTF-8, and the CLI reconfigures stdout/stderr to UTF-8 with `errors="replace"`.) diff --git a/graphify/extract.py b/graphify/extract.py index ab67154..0f2c17c 100644 --- a/graphify/extract.py +++ b/graphify/extract.py @@ -15904,7 +15904,12 @@ def _extract_single_file(args: tuple) -> tuple[int, dict]: return idx, {"nodes": [], "edges": []} result = _safe_extract_with_xaml_root(extractor, path, cache_root) - if not bypass_cache and "error" not in result: + # Never cache a zero-node result for an extractable file. Every supported + # source produces at least a file node, so an empty node list is anomalous + # (e.g. a transient batch/parallel hiccup). Caching it makes the empty + # byte-stable across runs and silently blinds affected/explain to and + # through the file (#1666); skipping the write lets a rerun self-heal. + if not bypass_cache and "error" not in result and result.get("nodes"): save_cached(path, result, cache_root) return idx, result @@ -16028,7 +16033,8 @@ def _extract_sequential( continue bypass_cache = path.suffix in _JS_CACHE_BYPASS_SUFFIXES result = _safe_extract_with_xaml_root(extractor, path, effective_root) - if not bypass_cache and "error" not in result: + # See _extract_single_file: don't cache an anomalous zero-node result (#1666). + if not bypass_cache and "error" not in result and result.get("nodes"): save_cached(path, result, effective_root) per_file[idx] = result if total_files >= _PROGRESS_INTERVAL: @@ -16124,6 +16130,27 @@ def extract( if per_file[i] is None: per_file[i] = {"nodes": [], "edges": []} + # #1666: surface any source file an extractor accepted but that produced zero + # nodes (not even a file node). Such a file is silently absent from the graph, + # so affected/explain are blind to and through it with no other signal. + _empty_sources: list[str] = [] + for i, _p in enumerate(paths): + _res = per_file[i] or {} + if _res.get("nodes") or _res.get("error"): + continue + if _get_extractor(_p) is not None: + _empty_sources.append(str(_p)) + if _empty_sources: + _shown = ", ".join(Path(x).name for x in _empty_sources[:5]) + _more = f" (+{len(_empty_sources) - 5} more)" if len(_empty_sources) > 5 else "" + print( + f" warning: {len(_empty_sources)} source file(s) produced zero nodes and " + f"are absent from the graph: {_shown}{_more}. A re-run will retry them " + f"(empties are no longer cached); if it persists, please report the " + f"file(s) (#1666).", + file=sys.stderr, flush=True, + ) + all_nodes: list[dict] = [] all_edges: list[dict] = [] all_raw_calls: list[dict] = [] diff --git a/tests/test_zero_node_no_cache.py b/tests/test_zero_node_no_cache.py new file mode 100644 index 0000000..5e62547 --- /dev/null +++ b/tests/test_zero_node_no_cache.py @@ -0,0 +1,54 @@ +"""#1666 — an extractable source file that yields zero nodes must not be cached, +and must be surfaced. + +Every supported file produces at least a file node, so a zero-node result is +anomalous (a transient batch/parallel hiccup). Caching it made the empty +byte-stable across runs and silently blinded affected/explain to the file. We +now skip the cache write for a zero-node result (so a rerun self-heals) and warn. +""" +from __future__ import annotations + +from pathlib import Path + +import graphify.extract as ex + + +def test_zero_node_result_not_cached_then_self_heals(tmp_path, capsys, monkeypatch): + f = tmp_path / "thing.rb" + f.write_text("class Foo\n def bar; end\nend\n") + + real = ex._safe_extract_with_xaml_root + + def _empty(extractor, path, root): + return {"nodes": [], "edges": []} + + # First run: force a zero-node extraction for this file. + monkeypatch.setattr(ex, "_safe_extract_with_xaml_root", _empty) + ex.extract([f], cache_root=tmp_path / "out", parallel=False) + + err = capsys.readouterr().err + assert "zero nodes" in err and "thing.rb" in err, err + + # Second run with the real extractor: because the empty was NOT cached, the + # file re-extracts and lands in the graph (self-heal). + monkeypatch.setattr(ex, "_safe_extract_with_xaml_root", real) + r2 = ex.extract([f], cache_root=tmp_path / "out", parallel=False) + assert any(str(n.get("source_file", "")).endswith("thing.rb") for n in r2["nodes"]) + + +def test_normal_file_still_cached(tmp_path): + # Guard against over-correction: a normal (non-empty) result must still cache. + f = tmp_path / "ok.rb" + f.write_text("class Bar\n def baz; end\nend\n") + r1 = ex.extract([f], cache_root=tmp_path / "out", parallel=False) + assert r1["nodes"] + from graphify.cache import load_cached + assert load_cached(f, tmp_path / "out") is not None, "non-empty result should be cached" + + +def test_no_warning_when_all_files_produce_nodes(tmp_path, capsys): + f = tmp_path / "fine.rb" + f.write_text("module M\n def self.go; end\nend\n") + ex.extract([f], cache_root=tmp_path / "out", parallel=False) + err = capsys.readouterr().err + assert "zero nodes" not in err