fix(extract): warn when code files have no AST extractor instead of silently dropping them (#1689)
Extensions like .r/.R (also .ejs, .ets) are in CODE_EXTENSIONS, so those files are classified as code and counted in the scan, but there is no entry for them in the extractor dispatch — so they produce zero nodes and are silently absent from the graph while the CLI still reports success. The #1666 zero-node warning deliberately skips them (it only fires when an extractor exists), so nothing surfaced the gap. extract() now emits a warning listing the offending extensions and counts ("N file(s) are classified as code but graphify has no AST extractor ...: .r (2)") so a primarily-R (or .ejs/.ets) corpus no longer looks fully mapped when it is not. Grouped by extension, fires only for files actually present with no extractor (today: .ejs, .ets, .r). Adding real grammars for these remains the follow-up; this removes the silent-data-loss now. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 4.8
parent
d1d1f412b2
commit
377dc7f384
@@ -16403,6 +16403,29 @@ def extract(
|
||||
file=sys.stderr, flush=True,
|
||||
)
|
||||
|
||||
# #1689: a file counted as code (extension in CODE_EXTENSIONS) but with no AST
|
||||
# extractor wired up (e.g. .r/.R — there is no tree-sitter-r dispatch) silently
|
||||
# contributes zero nodes. The #1666 warning above deliberately skips these (it
|
||||
# only fires when an extractor exists), so surface them explicitly, grouped by
|
||||
# extension, rather than reporting success as if the language were mapped.
|
||||
from graphify.detect import CODE_EXTENSIONS as _CODE_EXTS
|
||||
_no_extractor: dict[str, int] = {}
|
||||
for _p in paths:
|
||||
_ext = _p.suffix.lower()
|
||||
if _ext in _CODE_EXTS and _get_extractor(_p) is None:
|
||||
_no_extractor[_ext] = _no_extractor.get(_ext, 0) + 1
|
||||
if _no_extractor:
|
||||
_by_count = ", ".join(
|
||||
f"{ext} ({n})" for ext, n in sorted(_no_extractor.items(), key=lambda kv: (-kv[1], kv[0]))
|
||||
)
|
||||
_tot = sum(_no_extractor.values())
|
||||
print(
|
||||
f" warning: {_tot} file(s) are classified as code but graphify has no AST "
|
||||
f"extractor for their language, so they contributed nothing to the graph: "
|
||||
f"{_by_count}. Please open an issue to request support for these (#1689).",
|
||||
file=sys.stderr, flush=True,
|
||||
)
|
||||
|
||||
all_nodes: list[dict] = []
|
||||
all_edges: list[dict] = []
|
||||
all_raw_calls: list[dict] = []
|
||||
|
||||
@@ -1769,3 +1769,29 @@ def test_case_insensitive_suffix_filtering(tmp_path):
|
||||
assert "myJSFunction()" in labels
|
||||
assert "MyTSClass" in labels
|
||||
|
||||
|
||||
|
||||
def test_extract_warns_on_code_files_with_no_ast_extractor(tmp_path, capsys):
|
||||
# #1689: .r/.R is in CODE_EXTENSIONS (counted as code) but has no AST extractor,
|
||||
# so R files silently contribute nothing. extract() must surface that instead of
|
||||
# reporting success as if the language were mapped.
|
||||
r1 = tmp_path / "analysis.R"; r1.write_text("f <- function(x) x + 1\n")
|
||||
r2 = tmp_path / "helper.r"; r2.write_text("g <- function(y) y * 2\n")
|
||||
py = tmp_path / "main.py"; py.write_text("def main():\n return 1\n")
|
||||
|
||||
result = extract([r1, r2, py], cache_root=tmp_path)
|
||||
err = capsys.readouterr().err
|
||||
|
||||
assert "no AST extractor" in err
|
||||
assert ".r (2)" in err # both R files grouped under the lowercased ext
|
||||
assert "#1689" in err
|
||||
# the Python file still extracts normally
|
||||
labels = [n.get("label") for n in result["nodes"]]
|
||||
assert any(str(l).startswith("main") for l in labels)
|
||||
|
||||
|
||||
def test_extract_no_warning_when_all_code_has_extractors(tmp_path, capsys):
|
||||
py = tmp_path / "a.py"; py.write_text("def a():\n return 1\n")
|
||||
extract([py], cache_root=tmp_path)
|
||||
err = capsys.readouterr().err
|
||||
assert "no AST extractor" not in err
|
||||
|
||||
Reference in New Issue
Block a user