fix(extract): warn when code files have no AST extractor instead of silently dropping them (#1689)

Extensions like .r/.R (also .ejs, .ets) are in CODE_EXTENSIONS, so those files
are classified as code and counted in the scan, but there is no entry for them
in the extractor dispatch — so they produce zero nodes and are silently absent
from the graph while the CLI still reports success. The #1666 zero-node warning
deliberately skips them (it only fires when an extractor exists), so nothing
surfaced the gap.

extract() now emits a warning listing the offending extensions and counts
("N file(s) are classified as code but graphify has no AST extractor ...: .r (2)")
so a primarily-R (or .ejs/.ets) corpus no longer looks fully mapped when it is
not. Grouped by extension, fires only for files actually present with no
extractor (today: .ejs, .ets, .r). Adding real grammars for these remains the
follow-up; this removes the silent-data-loss now.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
safishamsi
2026-07-07 00:06:49 +01:00
co-authored by Claude Opus 4.8
parent d1d1f412b2
commit 377dc7f384
2 changed files with 49 additions and 0 deletions
+23
View File
@@ -16403,6 +16403,29 @@ def extract(
file=sys.stderr, flush=True,
)
# #1689: a file counted as code (extension in CODE_EXTENSIONS) but with no AST
# extractor wired up (e.g. .r/.R — there is no tree-sitter-r dispatch) silently
# contributes zero nodes. The #1666 warning above deliberately skips these (it
# only fires when an extractor exists), so surface them explicitly, grouped by
# extension, rather than reporting success as if the language were mapped.
from graphify.detect import CODE_EXTENSIONS as _CODE_EXTS
_no_extractor: dict[str, int] = {}
for _p in paths:
_ext = _p.suffix.lower()
if _ext in _CODE_EXTS and _get_extractor(_p) is None:
_no_extractor[_ext] = _no_extractor.get(_ext, 0) + 1
if _no_extractor:
_by_count = ", ".join(
f"{ext} ({n})" for ext, n in sorted(_no_extractor.items(), key=lambda kv: (-kv[1], kv[0]))
)
_tot = sum(_no_extractor.values())
print(
f" warning: {_tot} file(s) are classified as code but graphify has no AST "
f"extractor for their language, so they contributed nothing to the graph: "
f"{_by_count}. Please open an issue to request support for these (#1689).",
file=sys.stderr, flush=True,
)
all_nodes: list[dict] = []
all_edges: list[dict] = []
all_raw_calls: list[dict] = []
+26
View File
@@ -1769,3 +1769,29 @@ def test_case_insensitive_suffix_filtering(tmp_path):
assert "myJSFunction()" in labels
assert "MyTSClass" in labels
def test_extract_warns_on_code_files_with_no_ast_extractor(tmp_path, capsys):
# #1689: .r/.R is in CODE_EXTENSIONS (counted as code) but has no AST extractor,
# so R files silently contribute nothing. extract() must surface that instead of
# reporting success as if the language were mapped.
r1 = tmp_path / "analysis.R"; r1.write_text("f <- function(x) x + 1\n")
r2 = tmp_path / "helper.r"; r2.write_text("g <- function(y) y * 2\n")
py = tmp_path / "main.py"; py.write_text("def main():\n return 1\n")
result = extract([r1, r2, py], cache_root=tmp_path)
err = capsys.readouterr().err
assert "no AST extractor" in err
assert ".r (2)" in err # both R files grouped under the lowercased ext
assert "#1689" in err
# the Python file still extracts normally
labels = [n.get("label") for n in result["nodes"]]
assert any(str(l).startswith("main") for l in labels)
def test_extract_no_warning_when_all_code_has_extractors(tmp_path, capsys):
py = tmp_path / "a.py"; py.write_text("def a():\n return 1\n")
extract([py], cache_root=tmp_path)
err = capsys.readouterr().err
assert "no AST extractor" not in err