feat(detect): surface unclassified files instead of dropping them silently (#1692)
When classify_file() returned None — an extensionless, non-shebang file (Dockerfile, Gemfile, Makefile, Rakefile, LICENSE, ...) or an unsupported extension — the file left no trace at all: not counted, not listed, nothing. A user had no way to tell from graphify's output that those files were even considered. detect() now collects these into an "unclassified" list in its result, and `graphify extract` prints a one-line summary after the scan counts: "N file(s) not classified (no supported extension or shebang), skipped: Dockerfile, Makefile, ...". Real code/docs are unaffected. This is the visibility half of the issue; wiring up extractors/manifest handling for Dockerfile/Makefile-style files remains a separate feature. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 4.8
parent
f84a0af1bf
commit
23457d11ff
@@ -4673,6 +4673,17 @@ def main() -> None:
|
||||
f"{len(doc_files)} docs, {len(paper_files)} papers, "
|
||||
f"{len(image_files)} images"
|
||||
)
|
||||
# Surface files that were seen but not classified (extensionless non-shebang
|
||||
# project files like Dockerfile/Makefile, or unsupported extensions), so they
|
||||
# are no longer invisible in graphify's own output (#1692).
|
||||
_unclassified = detection.get("unclassified", []) if isinstance(detection, dict) else []
|
||||
if _unclassified:
|
||||
_names = ", ".join(sorted({Path(p).name for p in _unclassified})[:6])
|
||||
_more = f" (+{len(_unclassified) - 6} more)" if len(_unclassified) > 6 else ""
|
||||
print(
|
||||
f"[graphify extract] {len(_unclassified)} file(s) not classified "
|
||||
f"(no supported extension or shebang), skipped: {_names}{_more}"
|
||||
)
|
||||
stages.mark("detect")
|
||||
|
||||
# Resolve the LLM backend only now that we know whether the corpus
|
||||
|
||||
@@ -1086,6 +1086,7 @@ def detect(root: Path, *, follow_symlinks: bool | None = None, google_workspace:
|
||||
return _cache.cached_word_count(path, root, count_words)
|
||||
|
||||
skipped_sensitive: list[str] = []
|
||||
unclassified: list[str] = []
|
||||
ignore_patterns = _load_graphifyignore(root)
|
||||
ignore_cache: dict[Path, bool] = {} # shared across all _is_ignored calls in this scan
|
||||
# CLI --exclude patterns are anchored at the scan root and appended last
|
||||
@@ -1171,6 +1172,13 @@ def detect(root: Path, *, follow_symlinks: bool | None = None, google_workspace:
|
||||
skipped_sensitive.append(str(p))
|
||||
continue
|
||||
ftype = classify_file(p)
|
||||
if not ftype:
|
||||
# Considered but unclassifiable: an extension not in any supported set,
|
||||
# or an extensionless, non-shebang file (Dockerfile, Gemfile, Makefile,
|
||||
# Rakefile, LICENSE, ...). Previously these left no trace at all — not
|
||||
# counted, not listed — so a user couldn't tell they were seen (#1692).
|
||||
unclassified.append(str(p))
|
||||
continue
|
||||
if ftype:
|
||||
if p.suffix.lower() in GOOGLE_WORKSPACE_EXTENSIONS:
|
||||
if not google_workspace:
|
||||
@@ -1236,6 +1244,7 @@ def detect(root: Path, *, follow_symlinks: bool | None = None, google_workspace:
|
||||
"needs_graph": needs_graph,
|
||||
"warning": warning,
|
||||
"skipped_sensitive": skipped_sensitive,
|
||||
"unclassified": sorted(unclassified),
|
||||
"graphifyignore_patterns": len(ignore_patterns),
|
||||
"scan_root": str(root.resolve()),
|
||||
}
|
||||
|
||||
@@ -1575,3 +1575,25 @@ def test_convert_office_file_does_not_rewrite_existing_sidecar(tmp_path, monkeyp
|
||||
second = detect_mod.convert_office_file(src, out_dir)
|
||||
assert second == first
|
||||
assert second.stat().st_mtime_ns == mtime_before
|
||||
|
||||
|
||||
def test_detect_records_unclassified_extensionless_files(tmp_path):
|
||||
# #1692: extensionless, non-shebang project files (Dockerfile, Makefile, ...)
|
||||
# were considered but left no trace. detect() now lists them under
|
||||
# "unclassified" so they can be surfaced instead of silently vanishing.
|
||||
(tmp_path / "app.py").write_text("def f():\n return 1\n")
|
||||
(tmp_path / "Dockerfile").write_text("FROM python:3.12\nRUN pip install x\n")
|
||||
(tmp_path / "Makefile").write_text("build:\n\techo hi\n")
|
||||
(tmp_path / "LICENSE").write_text("MIT License\n")
|
||||
res = detect(tmp_path)
|
||||
unclassified = sorted(Path(p).name for p in res.get("unclassified", []))
|
||||
assert unclassified == ["Dockerfile", "LICENSE", "Makefile"]
|
||||
# real code is still classified, not swept into unclassified
|
||||
assert any("app.py" in f for f in res["files"].get("code", []))
|
||||
|
||||
|
||||
def test_detect_unclassified_empty_when_all_supported(tmp_path):
|
||||
(tmp_path / "a.py").write_text("x = 1\n")
|
||||
(tmp_path / "README.md").write_text("# hi\n")
|
||||
res = detect(tmp_path)
|
||||
assert res.get("unclassified", []) == []
|
||||
|
||||
Reference in New Issue
Block a user