diff --git a/graphify/__main__.py b/graphify/__main__.py index 8d37c91..285d19c 100644 --- a/graphify/__main__.py +++ b/graphify/__main__.py @@ -4673,6 +4673,17 @@ def main() -> None: f"{len(doc_files)} docs, {len(paper_files)} papers, " f"{len(image_files)} images" ) + # Surface files that were seen but not classified (extensionless non-shebang + # project files like Dockerfile/Makefile, or unsupported extensions), so they + # are no longer invisible in graphify's own output (#1692). + _unclassified = detection.get("unclassified", []) if isinstance(detection, dict) else [] + if _unclassified: + _names = ", ".join(sorted({Path(p).name for p in _unclassified})[:6]) + _more = f" (+{len(_unclassified) - 6} more)" if len(_unclassified) > 6 else "" + print( + f"[graphify extract] {len(_unclassified)} file(s) not classified " + f"(no supported extension or shebang), skipped: {_names}{_more}" + ) stages.mark("detect") # Resolve the LLM backend only now that we know whether the corpus diff --git a/graphify/detect.py b/graphify/detect.py index c9d7792..85f4b8b 100644 --- a/graphify/detect.py +++ b/graphify/detect.py @@ -1086,6 +1086,7 @@ def detect(root: Path, *, follow_symlinks: bool | None = None, google_workspace: return _cache.cached_word_count(path, root, count_words) skipped_sensitive: list[str] = [] + unclassified: list[str] = [] ignore_patterns = _load_graphifyignore(root) ignore_cache: dict[Path, bool] = {} # shared across all _is_ignored calls in this scan # CLI --exclude patterns are anchored at the scan root and appended last @@ -1171,6 +1172,13 @@ def detect(root: Path, *, follow_symlinks: bool | None = None, google_workspace: skipped_sensitive.append(str(p)) continue ftype = classify_file(p) + if not ftype: + # Considered but unclassifiable: an extension not in any supported set, + # or an extensionless, non-shebang file (Dockerfile, Gemfile, Makefile, + # Rakefile, LICENSE, ...). Previously these left no trace at all — not + # counted, not listed — so a user couldn't tell they were seen (#1692). + unclassified.append(str(p)) + continue if ftype: if p.suffix.lower() in GOOGLE_WORKSPACE_EXTENSIONS: if not google_workspace: @@ -1236,6 +1244,7 @@ def detect(root: Path, *, follow_symlinks: bool | None = None, google_workspace: "needs_graph": needs_graph, "warning": warning, "skipped_sensitive": skipped_sensitive, + "unclassified": sorted(unclassified), "graphifyignore_patterns": len(ignore_patterns), "scan_root": str(root.resolve()), } diff --git a/tests/test_detect.py b/tests/test_detect.py index 0c07e24..b26d37e 100644 --- a/tests/test_detect.py +++ b/tests/test_detect.py @@ -1575,3 +1575,25 @@ def test_convert_office_file_does_not_rewrite_existing_sidecar(tmp_path, monkeyp second = detect_mod.convert_office_file(src, out_dir) assert second == first assert second.stat().st_mtime_ns == mtime_before + + +def test_detect_records_unclassified_extensionless_files(tmp_path): + # #1692: extensionless, non-shebang project files (Dockerfile, Makefile, ...) + # were considered but left no trace. detect() now lists them under + # "unclassified" so they can be surfaced instead of silently vanishing. + (tmp_path / "app.py").write_text("def f():\n return 1\n") + (tmp_path / "Dockerfile").write_text("FROM python:3.12\nRUN pip install x\n") + (tmp_path / "Makefile").write_text("build:\n\techo hi\n") + (tmp_path / "LICENSE").write_text("MIT License\n") + res = detect(tmp_path) + unclassified = sorted(Path(p).name for p in res.get("unclassified", [])) + assert unclassified == ["Dockerfile", "LICENSE", "Makefile"] + # real code is still classified, not swept into unclassified + assert any("app.py" in f for f in res["files"].get("code", [])) + + +def test_detect_unclassified_empty_when_all_supported(tmp_path): + (tmp_path / "a.py").write_text("x = 1\n") + (tmp_path / "README.md").write_text("# hi\n") + res = detect(tmp_path) + assert res.get("unclassified", []) == []