Add .graphifyignore support for excluding files and directories

This commit is contained in:
Safi
2026-04-07 09:41:01 +01:00
parent b079c6713a
commit f2423bc114
3 changed files with 105 additions and 2 deletions
+12
View File
@@ -23,6 +23,18 @@ graphify-out/
└── cache/ SHA256 cache - re-runs only process changed files
```
Add a `.graphifyignore` file to exclude folders you don't want in the graph:
```
# .graphifyignore
vendor/
node_modules/
dist/
*.generated.py
```
Same syntax as `.gitignore`. Patterns match against file paths relative to the folder you run graphify on.
## How it works
graphify runs in two passes. First, a deterministic AST pass extracts structure from code files (classes, functions, imports, call graphs, docstrings, rationale comments) with no LLM needed. Second, Claude subagents run in parallel over docs, papers, and images to extract concepts, relationships, and design rationale. The results are merged into a NetworkX graph, clustered with Leiden community detection, and exported as interactive HTML, queryable JSON, and a plain-language audit report.
+58 -1
View File
@@ -1,5 +1,6 @@
# file discovery, type classification, and corpus health checks
from __future__ import annotations
import fnmatch
import json
import os
import re
@@ -134,6 +135,56 @@ def _is_noise_dir(part: str) -> bool:
return False
def _load_graphifyignore(root: Path) -> list[str]:
"""Read .graphifyignore from root and return a list of patterns.
Lines starting with # are comments. Blank lines are ignored.
Patterns follow gitignore semantics: glob matched against the path
relative to root. A leading slash anchors to root. A trailing slash
matches directories only (we match both dir and file for simplicity).
"""
ignore_file = root / ".graphifyignore"
if not ignore_file.exists():
return []
patterns = []
for line in ignore_file.read_text(errors="ignore").splitlines():
line = line.strip()
if line and not line.startswith("#"):
patterns.append(line)
return patterns
def _is_ignored(path: Path, root: Path, patterns: list[str]) -> bool:
"""Return True if path matches any .graphifyignore pattern."""
if not patterns:
return False
try:
rel = str(path.relative_to(root))
except ValueError:
return False
rel = rel.replace(os.sep, "/")
parts = rel.split("/")
for pattern in patterns:
# Normalize: strip leading/trailing slashes for matching purposes
p = pattern.strip("/")
if not p:
continue
# Match against full relative path
if fnmatch.fnmatch(rel, p):
return True
# Match against filename alone
if fnmatch.fnmatch(path.name, p):
return True
# Match against any path segment or prefix
# e.g. "vendor" or "vendor/" should match "vendor/lib.py"
for i, part in enumerate(parts):
if fnmatch.fnmatch(part, p):
return True
if fnmatch.fnmatch("/".join(parts[:i + 1]), p):
return True
return False
def detect(root: Path) -> dict:
files: dict[FileType, list[str]] = {
FileType.CODE: [],
@@ -144,6 +195,7 @@ def detect(root: Path) -> dict:
total_words = 0
skipped_sensitive: list[str] = []
ignore_patterns = _load_graphifyignore(root)
# Always include graphify-out/memory/ - query results filed back into the graph
memory_dir = root / "graphify-out" / "memory"
@@ -162,7 +214,9 @@ def detect(root: Path) -> dict:
# Prune noise dirs in-place so os.walk never descends into them
dirnames[:] = [
d for d in dirnames
if not d.startswith(".") and not _is_noise_dir(d)
if not d.startswith(".")
and not _is_noise_dir(d)
and not _is_ignored(dp / d, root, ignore_patterns)
]
for fname in filenames:
p = dp / fname
@@ -178,6 +232,8 @@ def detect(root: Path) -> dict:
# but catch hidden files at the root level
if p.name.startswith("."):
continue
if _is_ignored(p, root, ignore_patterns):
continue
if _is_sensitive(p):
skipped_sensitive.append(str(p))
continue
@@ -210,6 +266,7 @@ def detect(root: Path) -> dict:
"needs_graph": needs_graph,
"warning": warning,
"skipped_sensitive": skipped_sensitive,
"graphifyignore_patterns": len(ignore_patterns),
}
+35 -1
View File
@@ -1,5 +1,5 @@
from pathlib import Path
from graphify.detect import classify_file, count_words, detect, FileType, _looks_like_paper
from graphify.detect import classify_file, count_words, detect, FileType, _looks_like_paper, _is_ignored, _load_graphifyignore
FIXTURES = Path(__file__).parent / "fixtures"
@@ -69,3 +69,37 @@ def test_classify_attention_paper():
if paper_path.exists():
result = classify_file(paper_path)
assert result == FileType.PAPER
def test_graphifyignore_excludes_file(tmp_path):
"""Files matching .graphifyignore patterns are excluded from detect()."""
(tmp_path / ".graphifyignore").write_text("vendor/\n*.generated.py\n")
vendor = tmp_path / "vendor"
vendor.mkdir()
(vendor / "lib.py").write_text("x = 1")
(tmp_path / "main.py").write_text("print('hi')")
(tmp_path / "schema.generated.py").write_text("x = 1")
result = detect(tmp_path)
file_list = result["files"]["code"]
assert any("main.py" in f for f in file_list)
assert not any("vendor" in f for f in file_list)
assert not any("generated" in f for f in file_list)
assert result["graphifyignore_patterns"] == 2
def test_graphifyignore_missing_is_fine(tmp_path):
"""No .graphifyignore is not an error."""
(tmp_path / "main.py").write_text("x = 1")
result = detect(tmp_path)
assert result["graphifyignore_patterns"] == 0
def test_graphifyignore_comments_ignored(tmp_path):
"""Comment lines in .graphifyignore are not treated as patterns."""
(tmp_path / ".graphifyignore").write_text("# this is a comment\n\nmain.py\n")
(tmp_path / "main.py").write_text("x = 1")
(tmp_path / "other.py").write_text("x = 2")
result = detect(tmp_path)
assert not any("main.py" in f for f in result["files"]["code"])
assert any("other.py" in f for f in result["files"]["code"])