From 0d1f25c22139fcfe869cb1293c2f678aa06e75b7 Mon Sep 17 00:00:00 2001 From: safishamsi Date: Wed, 22 Jul 2026 20:35:54 +0100 Subject: [PATCH] refactor(detect): remove dead .graphifyinclude handling (#2112) The .graphifyinclude loader and its two matcher helpers had no consumers: commit df40e4d (#873, index dot dirs) removed the blanket dot-prefix exclusion and with it the only call sites, leaving detect() parsing the file on every run and then discarding the result. A .graphifyinclude was silently a no-op. Delete _load_graphifyinclude, _is_included, _could_contain_included_path and the orphaned assignment; add .graphifyinclude to _SKIP_FILES so a leftover file no longer lands in unclassified; and print a one-time stderr note when one is present at the scan root, pointing to ! negation patterns in .graphifyignore. Bump to 0.9.25. Co-Authored-By: Claude Opus 4.8 (1M context) --- CHANGELOG.md | 4 ++ graphify/detect.py | 128 ++++++------------------------------------- pyproject.toml | 2 +- tests/test_detect.py | 25 +++++++++ uv.lock | 2 +- 5 files changed, 47 insertions(+), 114 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 7948466..88e884d 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,10 @@ Full release notes with details on each version: [GitHub Releases](https://github.com/safishamsi/graphify/releases) +## 0.9.25 (unreleased) + +- Removed: `.graphifyinclude` handling is gone (#2112). The file has been non-functional since dot directories became indexed by default (#873): the loader and its matchers had no consumers, so `detect` parsed the file on every run and then ignored it, and a `.graphifyinclude` was silently a no-op. The dead loader and matchers are deleted, a leftover `.graphifyinclude` no longer shows up in the `unclassified` list, and `detect` prints a one-time stderr note when one is present at the scan root. To re-include ignored paths, use `!` negation patterns in `.graphifyignore`. + ## 0.9.24 (2026-07-22) - Fix: the XAML code-behind `.cs` scan is now bounded and prunes noise dirs, so it can't hang. `_xaml_csharp_class_nodes` used `rglob("*.cs")` over a project root resolved by walking up for a `.csproj`/`.sln`; a standalone `extract_xaml` on a `.xaml` under a large or shared parent (a temp dir, a big monorepo) could resolve the root to a broad ancestor and then recursively scan the whole tree. It now walks with `node_modules`/`.venv`/`.git`/dot-dir pruning and a directory cap, so a real project scans fully while a runaway root degrades to a fast partial scan. diff --git a/graphify/detect.py b/graphify/detect.py index fe7c653..62efad0 100644 --- a/graphify/detect.py +++ b/graphify/detect.py @@ -797,6 +797,9 @@ _SKIP_FILES = { "package-lock.json", "yarn.lock", "pnpm-lock.yaml", "Cargo.lock", "poetry.lock", "Gemfile.lock", "composer.lock", "go.sum", "go.work.sum", + # Removed allowlist config (#2112) — no longer consumed, so keep a leftover + # file out of the unclassified list instead of surfacing it as scan input. + ".graphifyinclude", } # A bare "snapshots" dir is a Jest/Vitest artifact only when it actually holds @@ -1137,117 +1140,6 @@ def _is_ignored( return _eval(path) -def _load_graphifyinclude(root: Path) -> list[tuple[Path, str]]: - """Read .graphifyinclude allowlist patterns from root and ancestors. - - Include patterns opt matching hidden files/dirs into traversal. Sensitive - files and hard-skipped noise directories are still excluded later. - Uses the same VCS-root ceiling logic as _load_graphifyignore. - """ - root = root.resolve() - ceiling = _find_vcs_root(root) or root - - dirs: list[Path] = [] - current = root - while True: - dirs.append(current) - if current == ceiling: - break - current = current.parent - dirs.reverse() - - patterns: list[tuple[Path, str]] = [] - for d in dirs: - include_file = d / ".graphifyinclude" - if include_file.exists(): - for raw in include_file.read_text(encoding="utf-8", errors="ignore").splitlines(): - line = _parse_gitignore_line(raw) - if line: - patterns.append((d, line)) - return patterns - - -def _is_included(path: Path, root: Path, patterns: list[tuple[Path, str]]) -> bool: - """Return True if path matches any .graphifyinclude allowlist pattern.""" - if not patterns: - return False - - def _matches(rel: str, p: str, anchored: bool) -> bool: - if anchored: - return fnmatch.fnmatch(rel, p) - parts = rel.split("/") - if fnmatch.fnmatch(rel, p): - return True - if fnmatch.fnmatch(path.name, p): - return True - for i, part in enumerate(parts): - if fnmatch.fnmatch(part, p): - return True - if fnmatch.fnmatch("/".join(parts[:i + 1]), p): - return True - return False - - for anchor, pattern in patterns: - anchored = pattern.startswith("/") - p = pattern.strip("/") - if not p: - continue - if anchored: - try: - rel_anchor = str(path.relative_to(anchor)).replace(os.sep, "/") - if _matches(rel_anchor, p, anchored=True): - return True - except ValueError: - pass - else: - try: - rel = str(path.relative_to(root)).replace(os.sep, "/") - if _matches(rel, p, anchored=False): - return True - except ValueError: - pass - if anchor != root: - try: - rel_anchor = str(path.relative_to(anchor)).replace(os.sep, "/") - if _matches(rel_anchor, p, anchored=False): - return True - except ValueError: - pass - return False - - -def _could_contain_included_path(path: Path, root: Path, patterns: list[tuple[Path, str]]) -> bool: - """Return True if a directory may contain files matched by .graphifyinclude.""" - if not patterns: - return False - - rels: list[str] = [] - try: - rels.append(str(path.relative_to(root)).replace(os.sep, "/")) - except ValueError: - pass - for anchor, _ in patterns: - if anchor != root: - try: - rels.append(str(path.relative_to(anchor)).replace(os.sep, "/")) - except ValueError: - pass - - for rel in rels: - rel = rel.strip("/") - if not rel: - return True - for _, pattern in patterns: - p = pattern.strip("/") - if not p: - continue - if p == rel or p.startswith(rel + "/"): - return True - if fnmatch.fnmatch(rel, p): - return True - return False - - def _auto_follow_symlinks(root: Path) -> bool: """Return whether ``root`` has any direct symlinked child. @@ -1275,6 +1167,19 @@ def _resolves_under_root(path: Path, root: Path) -> bool: def detect(root: Path, *, follow_symlinks: bool | None = None, google_workspace: bool | None = None, extra_excludes: list[str] | None = None, cache_root: Path | None = None, gitignore: bool = True) -> dict: root = root.resolve() + # .graphifyinclude support was removed (#2112): its loader and matchers had + # no consumers, so the file has been a silent no-op since dot directories + # became indexed by default (#873). Surface that once per scan so a + # leftover allowlist file is not a silent behavior change. + if (root / ".graphifyinclude").is_file(): + import sys as _sys + print( + "[graphify] WARNING: .graphifyinclude is no longer supported " + "(it has been non-functional since dot directories became indexed " + "by default); to re-include ignored paths, use ! negation patterns " + "in .graphifyignore.", + file=_sys.stderr, + ) if follow_symlinks is None: follow_symlinks = False google_workspace = google_workspace_enabled() if google_workspace is None else google_workspace @@ -1312,7 +1217,6 @@ def detect(root: Path, *, follow_symlinks: bool | None = None, google_workspace: line = _parse_gitignore_line(pat) if line: ignore_patterns.append((root, line)) - include_patterns = _load_graphifyinclude(root) # Always include graphify-out/memory/ - query results filed back into the graph memory_dir = root / GRAPHIFY_OUT / "memory" diff --git a/pyproject.toml b/pyproject.toml index 4ad21bb..7a65af1 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta" [project] name = "graphifyy" -version = "0.9.24" +version = "0.9.25" description = "AI coding assistant skill (Claude Code, CodeBuddy, Codex, OpenCode, Kilo Code, Cursor, Gemini CLI, Aider, OpenClaw, Factory Droid, Trae, Hermes, Kiro, Pi, Devin CLI, Google Antigravity) - turn any folder of code, docs, papers, images, or videos into a queryable knowledge graph" readme = "README.md" license = { file = "LICENSE" } diff --git a/tests/test_detect.py b/tests/test_detect.py index 0298bea..2c6f9a7 100644 --- a/tests/test_detect.py +++ b/tests/test_detect.py @@ -2043,6 +2043,31 @@ def test_detect_unclassified_empty_when_all_supported(tmp_path): assert res.get("unclassified", []) == [] +def test_graphifyinclude_is_inert_and_not_unclassified(tmp_path, capsys): + """#2112: .graphifyinclude support was removed (dead since #873). + + A leftover .graphifyinclude must not error, must not surface in the + unclassified list, and must not change which real files are indexed. + detect() prints a one-time stderr note so the removal is not silent. + """ + (tmp_path / "main.py").write_text("x = 1\n") + + baseline = detect(tmp_path) + capsys.readouterr() # discard any baseline output + + (tmp_path / ".graphifyinclude").write_text(".github/\ndocs/**\n") + result = detect(tmp_path) + + # not surfaced as an unclassified scan input + assert not any(".graphifyinclude" in p for p in result["unclassified"]) + # real files are indexed exactly as before; the file changes nothing + assert result["files"] == baseline["files"] + assert any("main.py" in f for f in result["files"]["code"]) + # one-time stderr note, matching the [graphify] warning convention + err = capsys.readouterr().err + assert err.count("[graphify] WARNING: .graphifyinclude is no longer supported") == 1 + + def test_detect_reports_walk_errors_key(): """detect() always surfaces a walk_errors list so callers can tell whether enumeration was complete.""" diff --git a/uv.lock b/uv.lock index 9011529..ae89445 100644 --- a/uv.lock +++ b/uv.lock @@ -1090,7 +1090,7 @@ wheels = [ [[package]] name = "graphifyy" -version = "0.9.24" +version = "0.9.25" source = { editable = "." } dependencies = [ { name = "networkx", version = "3.4.2", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version < '3.11'" },