Fix: tolerate tiktoken special-token text in token estimation (#1685)

`_TOKENIZER.encode(content)` raises ValueError by default when the text
contains a special token such as `<|endoftext|>`, so a doc or corpus that
merely mentions these strings crashed the entire semantic pass. Both
`encode` sites in `_estimate_file_tokens` now pass `disallowed_special=()`
so such text is tokenized as ordinary bytes. Thanks @Kyzcreig for the report.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
safishamsi
2026-07-06 12:42:39 +01:00
co-authored by Claude Opus 4.8
parent 97a1371478
commit 0ff584f070
3 changed files with 33 additions and 3 deletions
+1
View File
@@ -5,6 +5,7 @@ Full release notes with details on each version: [GitHub Releases](https://githu
## Unreleased
- Fix: `--update`-style section writes to `CLAUDE.md`/`AGENTS.md` no longer corrupt or drop content (#1688, thanks @bdfinst). `_replace_or_append_section` located its managed block by substring (`marker in content`) and `next(... if marker in line)`, so a heading that appeared as a substring of another line (or duplicate headings) matched the wrong offset and the rewrite could truncate the file. It now matches the section heading exactly (`line.strip() == marker`), appends when absent, and prefers the last exact match when several exist, so unrelated content is preserved.
- Fix: token estimation no longer crashes on files containing tiktoken special-token text like `<|endoftext|>` (#1685, thanks @Kyzcreig). `_TOKENIZER.encode(content)` raises `ValueError` by default when the text contains a special token, which aborted packing on docs/corpora that merely mention these strings. Both `encode` sites now pass `disallowed_special=()` so such text is tokenized as ordinary bytes.
## 0.9.7 (2026-07-06)
+2 -2
View File
@@ -1507,7 +1507,7 @@ def _estimate_file_tokens(unit: "Path | FileSlice") -> int:
content = read_slice_text(unit)[:_FILE_CHAR_CAP]
except OSError:
return 0
return len(_TOKENIZER.encode(content)) + (_PER_FILE_OVERHEAD_CHARS // _CHARS_PER_TOKEN)
return len(_TOKENIZER.encode(content, disallowed_special=())) + (_PER_FILE_OVERHEAD_CHARS // _CHARS_PER_TOKEN)
path = unit
# Raster images are not read as text; a vision model bills them at a roughly
@@ -1526,7 +1526,7 @@ def _estimate_file_tokens(unit: "Path | FileSlice") -> int:
content = path.read_text(encoding="utf-8", errors="replace")[:_FILE_CHAR_CAP]
except OSError:
return 0
return len(_TOKENIZER.encode(content)) + (_PER_FILE_OVERHEAD_CHARS // _CHARS_PER_TOKEN)
return len(_TOKENIZER.encode(content, disallowed_special=())) + (_PER_FILE_OVERHEAD_CHARS // _CHARS_PER_TOKEN)
def _pack_chunks_by_tokens(
+30 -1
View File
@@ -118,7 +118,9 @@ def test_estimate_file_tokens_uses_tiktoken_when_available(tmp_path):
# Force the tokenizer to be a mock that records calls and returns a known
# token list, so we can assert the tiktoken path is taken.
fake_encoder = type("E", (), {"encode": staticmethod(lambda s: [0] * 999)})()
# Match tiktoken's real signature: encode(text, *, disallowed_special=...)
# so the #1685 hardening call (disallowed_special=()) reaches the mock.
fake_encoder = type("E", (), {"encode": staticmethod(lambda s, **kw: [0] * 999)})()
with patch.object(llm, "_TOKENIZER", fake_encoder):
n = llm._estimate_file_tokens(f)
assert n == 999 + (llm._PER_FILE_OVERHEAD_CHARS // llm._CHARS_PER_TOKEN)
@@ -497,3 +499,30 @@ def test_corpus_parallel_uses_adaptive_retry(tmp_path):
assert len(chunk_done_args) == 1
assert chunk_done_args[0] == (0, 1, 4)
assert len(result["nodes"]) == 4
# ---- #1685: special-token strings in docs must not crash token estimation ----
def test_estimate_file_tokens_handles_tiktoken_special_token(tmp_path):
"""A doc containing a literal tiktoken special token (e.g. <|endoftext|>)
must not crash token estimation. tiktoken's default encode() raises on such
strings appearing as ordinary text; we pass disallowed_special=() since this
is only an estimate (#1685)."""
import graphify.llm as llm
if llm._TOKENIZER is None:
import pytest
pytest.skip("tiktoken not installed; estimation uses the char heuristic")
f = tmp_path / "tokenizer-notes.md"
f.write_text("The GPT end-of-text token is <|endoftext|> in the vocab.\n")
n = llm._estimate_file_tokens(f) # must not raise
assert isinstance(n, int) and n > 0
def test_pack_chunks_with_special_token_doc_does_not_crash(tmp_path):
"""End to end: packing a corpus that includes a special-token doc must not
raise (the crash in #1685 happened during token-budget packing)."""
from graphify.llm import _pack_chunks_by_tokens
doc = tmp_path / "doc.md"; doc.write_text("see <|endoftext|> and <|im_start|> tokens\n")
code = tmp_path / "code.py"; code.write_text("def f():\n return 1\n")
chunks = _pack_chunks_by_tokens([doc, code], token_budget=60_000)
assert chunks # produced at least one chunk, no exception