diff --git a/CHANGELOG.md b/CHANGELOG.md index 840530e..1f73568 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,7 @@ Full release notes with details on each version: [GitHub Releases](https://githu ## Unreleased +- Fix: large text documents are no longer silently truncated during semantic extraction. `_read_files` capped every file at 20,000 characters, so a Markdown/text/rST document longer than that had everything past the cap dropped — the model never saw it, and the packer/adaptive-retry path couldn't recover ("packing can't shrink one big file"). Oversized splittable-text files are now sliced at heading/paragraph boundaries into units that each fit the cap and together cover the whole file; every slice reports its parent file as `source_file`, so the graph is not fragmented per-slice. A single slice that still overflows the model's output is bisected and retried. Code files and PDFs are never sliced (they keep whole-symbol / page handling). (#1369) - Fix: `/graphify --update` no longer deletes a changed file's freshly re-extracted nodes. The `0.8.41` `root=` fix (#1361) made `build_merge`'s prune actually match relative `source_file` values — which then matched the just-re-extracted nodes of *changed* files (still listed in `prune_sources`) and removed them, so an `--update` on a changed file could wipe its nodes. The update runbook now prunes only genuinely **deleted** files; changed files are reconciled by `build_merge`'s replace-on-re-extract (#1344). The full build also now passes `root=` to `build_from_json`, and the extraction-spec `source_file` is pinned to the verbatim path, so the full build and incremental updates never drift on node-key base. (#1366; thanks @RelywOo) - Fix: Java `record` declarations are now modeled as first-class type nodes (they share `class_declaration`'s name/body/interfaces fields), and `new Foo(...)` constructor calls now produce a `calls` edge to the constructed type. Previously a record appeared only as its file node (degree 0) with no incoming edges, and body-level `new` usages were dropped because `object_creation_expression` wasn't a recognized call type and its callee lives in the `type` field rather than `name`. (#1373) - Security fix: `.graphifyignore` and `.gitignore` are now **merged** per directory instead of `.graphifyignore` silently replacing that directory's `.gitignore`. Previously, adding a `.graphifyignore` (e.g. to exclude media) disabled the dir's `.gitignore` entirely, so a file excluded only by `.gitignore` — including neutrally-named secrets like `prod-dump.sql` or `customer-data.json` that the sensitive-file heuristic doesn't catch — got indexed into the graph, whose artifacts embed file contents and are routinely committed. `.gitignore` is read first and `.graphifyignore` last, so `.graphifyignore` patterns (including `!` negations) still win on conflict; adding one can only ever exclude more, never re-include a `.gitignore`-excluded file. (#1363) diff --git a/graphify/file_slice.py b/graphify/file_slice.py new file mode 100644 index 0000000..30dc49c --- /dev/null +++ b/graphify/file_slice.py @@ -0,0 +1,162 @@ +"""Intra-file slicing for oversized text documents (#1369). + +The extraction packer (`_pack_chunks_by_tokens`) treats each file as atomic and +`_read_files` caps every file at ``_FILE_CHAR_CAP`` characters, so a document +larger than that cap had everything past the cap silently dropped — the model +never saw it, and nothing in the adaptive-retry path could recover it ("a single +file larger than the budget ... packing can't shrink one big file"). + +This module splits an oversized *splittable text* document (Markdown, plain +text, reStructuredText) into contiguous ``FileSlice`` units at heading / +paragraph / line boundaries so the whole file gets extracted across several +units. Every slice of a file reports the **parent file path** as its source, so +the resulting nodes are never fragmented per-slice — they merge by source_file +exactly as if the file had been extracted in one pass. + +Only plain-text documents are sliced: code files need whole-symbol context, and +PDFs/images are read through their own extractors and have no char-offset model. +""" + +from __future__ import annotations + +from dataclasses import dataclass +from pathlib import Path + +# Plain-text document types where boundary-based slicing is meaningful and where +# `_file_to_text` is a straight ``read_text`` (so a char range matches the bytes +# the model is shown). Deliberately excludes code (.py, .ts, ...) and binary +# docs (.pdf) — those are never sliced. +_SPLITTABLE_TEXT_SUFFIXES = frozenset({".md", ".mdx", ".markdown", ".txt", ".rst"}) + +# Boundary preferences, strongest first. A Markdown heading (``\n#``) keeps a +# section with its title; a blank line keeps a paragraph intact; a bare newline +# avoids cutting mid-line. If none is found in the window we hard-cut. +_BOUNDARY_SEPARATORS = ("\n#", "\n\n", "\n") + + +@dataclass(frozen=True) +class FileSlice: + """A contiguous ``[start, end)`` character range of a splittable text file. + + ``index``/``total`` are for logging only. ``path`` is the real file on disk; + the slice always reports ``path`` as its source so slices don't fragment the + graph. + """ + + path: Path + start: int + end: int + index: int + total: int + + +# A unit of extraction work: either a whole file (``Path``) or one slice of one. +Unit = "Path | FileSlice" + + +def unit_path(unit: "Path | FileSlice") -> Path: + """The on-disk path a unit belongs to (the parent file for a slice).""" + return unit.path if isinstance(unit, FileSlice) else unit + + +def is_splittable_text(path: Path) -> bool: + """True for plain-text document types that may be sliced.""" + return path.suffix.lower() in _SPLITTABLE_TEXT_SUFFIXES + + +def _best_cut(text: str, start: int, end: int) -> int: + """Return a cut index in ``(start, end]`` at the strongest nearby boundary. + + Searches the window ``text[start:end]`` for the latest heading, then blank + line, then newline, and returns the index just *after* it (a heading cuts + just *before* the ``#`` so the heading leads the next slice). Falls back to a + hard cut at ``end`` when the window has no usable boundary, which still makes + forward progress because ``end > start``. + """ + window = text[start:end] + for sep in _BOUNDARY_SEPARATORS: + idx = window.rfind(sep) + if idx > 0: # a boundary strictly inside the window (non-empty prev slice) + if sep == "\n#": + return start + idx + 1 # keep the newline with the previous slice + return start + idx + len(sep) + return end + + +def slice_boundaries(text: str, max_chars: int) -> list[tuple[int, int]]: + """Contiguous ``(start, end)`` ranges covering all of ``text``, each ≤ max_chars. + + Ranges are gap-free and non-overlapping, so concatenating the slices + reproduces ``text`` exactly — no content is dropped. + """ + n = len(text) + if n <= max_chars: + return [(0, n)] + bounds: list[tuple[int, int]] = [] + pos = 0 + while pos < n: + hard = min(pos + max_chars, n) + end = _best_cut(text, pos, hard) if hard < n else n + if end <= pos: # defensive: never stall + end = hard + bounds.append((pos, end)) + pos = end + return bounds + + +def expand_oversized_files( + files: list[Path], max_chars: int +) -> list["Path | FileSlice"]: + """Replace each oversized splittable-text file with a list of ``FileSlice``s. + + Files at or below ``max_chars`` (and all non-splittable files) pass through + unchanged as ``Path``, so behaviour is identical for everything that already + fit. Unreadable files pass through untouched (the reader handles the error). + """ + out: list["Path | FileSlice"] = [] + for f in files: + if not is_splittable_text(f): + out.append(f) + continue + try: + text = f.read_text(encoding="utf-8", errors="replace") + except OSError: + out.append(f) + continue + if len(text) <= max_chars: + out.append(f) + continue + ranges = slice_boundaries(text, max_chars) + total = len(ranges) + for i, (s, e) in enumerate(ranges): + out.append(FileSlice(path=f, start=s, end=e, index=i, total=total)) + return out + + +def read_slice_text(fs: FileSlice) -> str: + """Read just this slice's characters from its parent file.""" + text = fs.path.read_text(encoding="utf-8", errors="replace") + return text[fs.start:fs.end] + + +def bisect_slice(fs: FileSlice) -> tuple[FileSlice, FileSlice] | None: + """Split a slice into two halves at a newline near its midpoint, or None. + + Used by the adaptive-retry path when a single slice still overflows the + model's output: halving it produces a smaller response. Returns None when the + slice is already too small to split meaningfully. + """ + if fs.end - fs.start <= 1: + return None + try: + text = fs.path.read_text(encoding="utf-8", errors="replace") + except OSError: + return None + mid = (fs.start + fs.end) // 2 + nl = text.find("\n", mid, fs.end) + cut = nl + 1 if (nl != -1 and fs.start < nl + 1 < fs.end) else mid + if not (fs.start < cut < fs.end): + return None + left = FileSlice(fs.path, fs.start, cut, fs.index, fs.total) + right = FileSlice(fs.path, cut, fs.end, fs.index, fs.total) + return left, right diff --git a/graphify/llm.py b/graphify/llm.py index 391ed28..6909d98 100644 --- a/graphify/llm.py +++ b/graphify/llm.py @@ -16,6 +16,14 @@ from concurrent.futures import ThreadPoolExecutor, as_completed from dataclasses import dataclass, replace from pathlib import Path +from graphify.file_slice import ( + FileSlice, + bisect_slice, + expand_oversized_files, + read_slice_text, + unit_path, +) + # `_read_files` truncates each file at this many characters before joining into # the user message. Token estimates use the same cap so packing matches reality. _FILE_CHAR_CAP = 20_000 @@ -454,23 +462,33 @@ def _wrap_untrusted(rel: str, content: str) -> str: ) -def _read_files(paths: list[Path], root: Path) -> str: - """Return file contents formatted for the extraction prompt. +def _read_files(units: "list[Path | FileSlice]", root: Path) -> str: + """Return file/slice contents formatted for the extraction prompt. - Each file is wrapped in an delimiter block and known + Each unit is wrapped in an delimiter block and known injection sentinels are defanged, so attacker-controlled source text cannot be confused with the trusted system instructions (see issue #1210). + + A ``FileSlice`` (one chunk of an oversized document, #1369) reports its + **parent file path** as ``rel`` so every slice of a file shares one + source_file and the graph isn't fragmented per-slice. """ parts: list[str] = [] - for p in paths: + for u in units: + p = unit_path(u) try: rel = str(p.relative_to(root)) except ValueError: rel = str(p) try: - content = _file_to_text(p) + if isinstance(u, FileSlice): + content = read_slice_text(u) + else: + content = _file_to_text(p) except OSError: continue + # Whole files are still capped (covers non-splittable large files like + # code); slices are already bounded to the cap, so the cap is a no-op. parts.append(_wrap_untrusted(rel, content[:_FILE_CHAR_CAP])) return "\n\n".join(parts) @@ -539,11 +557,17 @@ def _is_vision_image(path: Path) -> bool: return path.suffix.lower() in _VISION_IMAGE_EXTENSIONS -def _partition_semantic_files(files: list[Path]) -> tuple[list[Path], list[Path]]: - """Split a chunk into (text-like files, raster-image files).""" - text_files = [f for f in files if not _is_vision_image(f)] - image_files = [f for f in files if _is_vision_image(f)] - return text_files, image_files +def _partition_semantic_files( + units: "list[Path | FileSlice]", +) -> tuple["list[Path | FileSlice]", list[Path]]: + """Split a chunk into (text-like units, raster-image files). + + A ``FileSlice`` is always text (only splittable text is sliced), so it never + lands in the image partition. + """ + text_units = [u for u in units if isinstance(u, FileSlice) or not _is_vision_image(u)] + image_files = [u for u in units if not isinstance(u, FileSlice) and _is_vision_image(u)] + return text_units, image_files def _build_image_refs(image_files: list[Path], root: Path, *, read_bytes: bool = True) -> list[_ImageRef]: @@ -1370,15 +1394,26 @@ def extract_files_direct( ) -def _estimate_file_tokens(path: Path) -> int: - """Estimate the prompt-token cost of a single file under `_read_files` rules. +def _estimate_file_tokens(unit: "Path | FileSlice") -> int: + """Estimate the prompt-token cost of a file or slice under `_read_files` rules. Uses tiktoken (`cl100k_base`) when available for accurate counts. Falls back to the chars/4 heuristic if tiktoken is not installed. Both paths cap at `_FILE_CHAR_CAP` to match `_read_files`'s truncation, plus a constant for - the `=== rel ===` separator. Returns 0 for unreadable paths so they don't - blow up packing. + the wrapper. Returns 0 for unreadable paths so they don't blow up packing. """ + if isinstance(unit, FileSlice): + # A slice's size is its char range (already ≤ _FILE_CHAR_CAP). Use the + # tokenizer on its text when available, else the chars/4 heuristic. + if _TOKENIZER is None: + return (min(unit.end - unit.start, _FILE_CHAR_CAP) + _PER_FILE_OVERHEAD_CHARS) // _CHARS_PER_TOKEN + try: + content = read_slice_text(unit)[:_FILE_CHAR_CAP] + except OSError: + return 0 + return len(_TOKENIZER.encode(content)) + (_PER_FILE_OVERHEAD_CHARS // _CHARS_PER_TOKEN) + + path = unit # Raster images are not read as text; a vision model bills them at a roughly # fixed token cost, so estimate by image count rather than (binary) byte size. if _is_vision_image(path): @@ -1399,35 +1434,35 @@ def _estimate_file_tokens(path: Path) -> int: def _pack_chunks_by_tokens( - files: list[Path], + files: "list[Path | FileSlice]", token_budget: int, -) -> list[list[Path]]: - """Greedily pack files into chunks that fit a token budget. +) -> "list[list[Path | FileSlice]]": + """Greedily pack files/slices into chunks that fit a token budget. - Files are first grouped by parent directory so related artifacts share a + Units are first grouped by parent directory so related artifacts share a chunk (cross-file edges are more likely to be extracted within a chunk - than across chunks). Within each directory, files are added one at a - time; a chunk is closed when adding the next file would exceed the - budget. A single file larger than the budget gets its own chunk and the - caller is expected to handle the API error if it actually overflows the - model's context window — packing can't shrink one big file. + than across chunks). Within each directory, units are added one at a + time; a chunk is closed when adding the next would exceed the budget. + Oversized splittable documents are pre-split into ``FileSlice`` units by + ``expand_oversized_files`` before packing (#1369), so the old "one file + larger than the budget" case no longer silently drops content. """ if token_budget <= 0: raise ValueError(f"token_budget must be positive, got {token_budget}") - by_dir: dict[Path, list[Path]] = {} + by_dir: dict[Path, "list[Path | FileSlice]"] = {} for f in files: - by_dir.setdefault(f.parent, []).append(f) + by_dir.setdefault(unit_path(f).parent, []).append(f) - chunks: list[list[Path]] = [] - current: list[Path] = [] + chunks: "list[list[Path | FileSlice]]" = [] + current: "list[Path | FileSlice]" = [] current_tokens = 0 current_images = 0 for directory in sorted(by_dir): - for path in by_dir[directory]: - cost = _estimate_file_tokens(path) - is_image = _is_vision_image(path) + for unit in by_dir[directory]: + cost = _estimate_file_tokens(unit) + is_image = not isinstance(unit, FileSlice) and _is_vision_image(unit) over_budget = current_tokens + cost > token_budget over_images = is_image and current_images >= _MAX_IMAGES_PER_CHUNK if current and (over_budget or over_images): @@ -1435,7 +1470,7 @@ def _pack_chunks_by_tokens( current = [] current_tokens = 0 current_images = 0 - current.append(path) + current.append(unit) current_tokens += cost current_images += is_image @@ -1512,9 +1547,35 @@ def _extract_with_adaptive_retry( still failing at the cap, we surface the (likely empty) result with a warning rather than infinite-loop. - A single-file chunk that overflows is unrecoverable here — we can't make - one file smaller than itself, so we return what we got and warn. + A single-file chunk that overflows is recoverable only when it's a slice of + a splittable document: the slice is bisected and retried (#1369). A whole + non-splittable file (e.g. one huge code file) can't be made smaller than + itself, so we return what we got and warn. """ + def _merge_two(left_units, right_units) -> dict: + left = _extract_with_adaptive_retry( + left_units, backend, api_key, model, root, max_depth, _depth + 1, deep_mode=deep_mode + ) + right = _extract_with_adaptive_retry( + right_units, backend, api_key, model, root, max_depth, _depth + 1, deep_mode=deep_mode + ) + return { + "nodes": left.get("nodes", []) + right.get("nodes", []), + "edges": left.get("edges", []) + right.get("edges", []), + "hyperedges": left.get("hyperedges", []) + right.get("hyperedges", []), + "input_tokens": left.get("input_tokens", 0) + right.get("input_tokens", 0), + "output_tokens": left.get("output_tokens", 0) + right.get("output_tokens", 0), + "model": model, + "finish_reason": "stop", + } + + def _split_lone_slice() -> "tuple[FileSlice, FileSlice] | None": + # When a single-unit chunk is a slice, bisect the slice so we can retry + # on a smaller range rather than give up (#1369). + if len(chunk) == 1 and isinstance(chunk[0], FileSlice) and _depth < max_depth: + return bisect_slice(chunk[0]) + return None + try: result = extract_files_direct( chunk, backend=backend, api_key=api_key, model=model, root=root, deep_mode=deep_mode @@ -1523,8 +1584,16 @@ def _extract_with_adaptive_retry( if not _looks_like_context_exceeded(exc): raise if len(chunk) <= 1: + halves = _split_lone_slice() + if halves is not None: + print( + f"[graphify] slice of {unit_path(chunk[0])} exceeded context at " + f"depth {_depth}; splitting the slice and retrying", + file=sys.stderr, + ) + return _merge_two([halves[0]], [halves[1]]) print( - f"[graphify] single-file chunk {chunk[0]} exceeds model context " + f"[graphify] single-file chunk {unit_path(chunk[0])} exceeds model context " f"and cannot be split further: {exc}", file=sys.stderr, ) @@ -1562,8 +1631,16 @@ def _extract_with_adaptive_retry( return result if len(chunk) <= 1: + halves = _split_lone_slice() + if halves is not None: + print( + f"[graphify] slice of {unit_path(chunk[0])} truncated at depth {_depth}; " + f"splitting the slice and retrying", + file=sys.stderr, + ) + return _merge_two([halves[0]], [halves[1]]) print( - f"[graphify] single-file chunk {chunk[0]} truncated at " + f"[graphify] single-file chunk {unit_path(chunk[0])} truncated at " f"max_completion_tokens — partial result kept", file=sys.stderr, ) @@ -1652,6 +1729,10 @@ def extract_corpus_parallel( output_tokens. Failed chunks are logged to stderr and skipped — one bad chunk does not abort the run. """ + # Split oversized splittable documents into slices that cover the whole file + # before packing, so content past _FILE_CHAR_CAP is extracted instead of + # silently dropped (#1369). Files at/under the cap pass through unchanged. + files = expand_oversized_files(files, _FILE_CHAR_CAP) if token_budget is not None: chunks = _pack_chunks_by_tokens(files, token_budget=token_budget) else: diff --git a/tests/test_file_slice.py b/tests/test_file_slice.py new file mode 100644 index 0000000..619ffc1 --- /dev/null +++ b/tests/test_file_slice.py @@ -0,0 +1,165 @@ +from __future__ import annotations + +from pathlib import Path + +import pytest + +from graphify import llm +from graphify.file_slice import ( + FileSlice, + bisect_slice, + expand_oversized_files, + is_splittable_text, + read_slice_text, + slice_boundaries, + unit_path, +) + + +# ── slice_boundaries: coverage + bounds invariants ────────────────────────── + +def test_slice_boundaries_small_text_is_one_range(): + text = "short doc" + assert slice_boundaries(text, 100) == [(0, len(text))] + + +@pytest.mark.parametrize("max_chars", [50, 100, 500, 1000]) +def test_slice_boundaries_full_coverage_and_bounds(max_chars): + text = ("# Heading\n\n" + "lorem ipsum " * 40 + "\n\n") * 20 + bounds = slice_boundaries(text, max_chars) + # contiguous, gap-free, non-overlapping, covering the whole text + assert bounds[0][0] == 0 + assert bounds[-1][1] == len(text) + for (s0, e0), (s1, e1) in zip(bounds, bounds[1:]): + assert e0 == s1 + # concatenation reproduces the text exactly (no dropped content) + assert "".join(text[s:e] for s, e in bounds) == text + # every slice respects the budget + assert all((e - s) <= max_chars for s, e in bounds) + + +def test_slice_boundaries_single_huge_line_still_progresses(): + # No newline at all → must hard-cut and still cover everything. + text = "x" * 5000 + bounds = slice_boundaries(text, 1000) + assert "".join(text[s:e] for s, e in bounds) == text + assert all((e - s) <= 1000 for s, e in bounds) + + +def test_slice_boundaries_prefers_heading_boundary(): + a = "# A\n" + "a" * 30 + "\n" + b = "# B\n" + "b" * 30 + "\n" + text = a + b + bounds = slice_boundaries(text, len(a) + 5) # forces a split near the A/B seam + # the second slice should start at the "# B" heading + second_start = bounds[1][0] + assert text[second_start:second_start + 3] == "# B" + + +# ── expand_oversized_files ────────────────────────────────────────────────── + +def _write(p: Path, text: str) -> Path: + p.write_text(text, encoding="utf-8") + return p + + +def test_expand_small_file_stays_whole(tmp_path): + f = _write(tmp_path / "small.md", "# Tiny\n\nhi\n") + units = expand_oversized_files([f], max_chars=1000) + assert units == [f] + + +def test_expand_oversized_markdown_is_sliced_with_full_coverage(tmp_path): + text = ("# Section\n\n" + "word " * 200 + "\n\n") * 30 + f = _write(tmp_path / "big.md", text) + units = expand_oversized_files([f], max_chars=2000) + slices = [u for u in units if isinstance(u, FileSlice)] + assert len(slices) >= 2 + assert all(isinstance(u, FileSlice) for u in units) + # slices reconstruct the whole file + assert "".join(read_slice_text(s) for s in slices) == text + assert all((s.end - s.start) <= 2000 for s in slices) + # every slice points back at the parent file (anti-fragmentation) + assert all(s.path == f for s in slices) + assert slices[0].total == len(slices) + + +def test_expand_does_not_slice_code_even_when_oversized(tmp_path): + f = _write(tmp_path / "mod.py", "x = 1\n" * 6000) # >> max_chars but code + assert not is_splittable_text(f) + units = expand_oversized_files([f], max_chars=2000) + assert units == [f] # stays whole — code needs whole-symbol context + + +def test_expand_unreadable_file_passes_through(tmp_path): + missing = tmp_path / "nope.md" + units = expand_oversized_files([missing], max_chars=10) + assert units == [missing] + + +# ── anti-fragmentation: slices share one source_file in the prompt ────────── + +def test_read_files_keys_every_slice_to_parent_path(tmp_path): + import re + text = ("# H\n\n" + "lorem " * 300 + "\n\n") * 20 + f = _write(tmp_path / "doc.md", text) + units = expand_oversized_files([f], max_chars=llm._FILE_CHAR_CAP) + slices = [u for u in units if isinstance(u, FileSlice)] + assert len(slices) >= 2 + prompt = llm._read_files(units, tmp_path) + rels = re.findall(r'