feat(detect): surface unclassified files instead of dropping them silently (#1692)

When classify_file() returned None — an extensionless, non-shebang file
(Dockerfile, Gemfile, Makefile, Rakefile, LICENSE, ...) or an unsupported
extension — the file left no trace at all: not counted, not listed, nothing.
A user had no way to tell from graphify's output that those files were even
considered.

detect() now collects these into an "unclassified" list in its result, and
`graphify extract` prints a one-line summary after the scan counts:
"N file(s) not classified (no supported extension or shebang), skipped:
Dockerfile, Makefile, ...". Real code/docs are unaffected. This is the
visibility half of the issue; wiring up extractors/manifest handling for
Dockerfile/Makefile-style files remains a separate feature.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
safishamsi
2026-07-07 11:19:18 +01:00
co-authored by Claude Opus 4.8
parent f84a0af1bf
commit 23457d11ff
3 changed files with 42 additions and 0 deletions
+11
View File
@@ -4673,6 +4673,17 @@ def main() -> None:
f"{len(doc_files)} docs, {len(paper_files)} papers, "
f"{len(image_files)} images"
)
# Surface files that were seen but not classified (extensionless non-shebang
# project files like Dockerfile/Makefile, or unsupported extensions), so they
# are no longer invisible in graphify's own output (#1692).
_unclassified = detection.get("unclassified", []) if isinstance(detection, dict) else []
if _unclassified:
_names = ", ".join(sorted({Path(p).name for p in _unclassified})[:6])
_more = f" (+{len(_unclassified) - 6} more)" if len(_unclassified) > 6 else ""
print(
f"[graphify extract] {len(_unclassified)} file(s) not classified "
f"(no supported extension or shebang), skipped: {_names}{_more}"
)
stages.mark("detect")
# Resolve the LLM backend only now that we know whether the corpus
+9
View File
@@ -1086,6 +1086,7 @@ def detect(root: Path, *, follow_symlinks: bool | None = None, google_workspace:
return _cache.cached_word_count(path, root, count_words)
skipped_sensitive: list[str] = []
unclassified: list[str] = []
ignore_patterns = _load_graphifyignore(root)
ignore_cache: dict[Path, bool] = {} # shared across all _is_ignored calls in this scan
# CLI --exclude patterns are anchored at the scan root and appended last
@@ -1171,6 +1172,13 @@ def detect(root: Path, *, follow_symlinks: bool | None = None, google_workspace:
skipped_sensitive.append(str(p))
continue
ftype = classify_file(p)
if not ftype:
# Considered but unclassifiable: an extension not in any supported set,
# or an extensionless, non-shebang file (Dockerfile, Gemfile, Makefile,
# Rakefile, LICENSE, ...). Previously these left no trace at all — not
# counted, not listed — so a user couldn't tell they were seen (#1692).
unclassified.append(str(p))
continue
if ftype:
if p.suffix.lower() in GOOGLE_WORKSPACE_EXTENSIONS:
if not google_workspace:
@@ -1236,6 +1244,7 @@ def detect(root: Path, *, follow_symlinks: bool | None = None, google_workspace:
"needs_graph": needs_graph,
"warning": warning,
"skipped_sensitive": skipped_sensitive,
"unclassified": sorted(unclassified),
"graphifyignore_patterns": len(ignore_patterns),
"scan_root": str(root.resolve()),
}
+22
View File
@@ -1575,3 +1575,25 @@ def test_convert_office_file_does_not_rewrite_existing_sidecar(tmp_path, monkeyp
second = detect_mod.convert_office_file(src, out_dir)
assert second == first
assert second.stat().st_mtime_ns == mtime_before
def test_detect_records_unclassified_extensionless_files(tmp_path):
# #1692: extensionless, non-shebang project files (Dockerfile, Makefile, ...)
# were considered but left no trace. detect() now lists them under
# "unclassified" so they can be surfaced instead of silently vanishing.
(tmp_path / "app.py").write_text("def f():\n return 1\n")
(tmp_path / "Dockerfile").write_text("FROM python:3.12\nRUN pip install x\n")
(tmp_path / "Makefile").write_text("build:\n\techo hi\n")
(tmp_path / "LICENSE").write_text("MIT License\n")
res = detect(tmp_path)
unclassified = sorted(Path(p).name for p in res.get("unclassified", []))
assert unclassified == ["Dockerfile", "LICENSE", "Makefile"]
# real code is still classified, not swept into unclassified
assert any("app.py" in f for f in res["files"].get("code", []))
def test_detect_unclassified_empty_when_all_supported(tmp_path):
(tmp_path / "a.py").write_text("x = 1\n")
(tmp_path / "README.md").write_text("# hi\n")
res = detect(tmp_path)
assert res.get("unclassified", []) == []