From 7722e23900f8ede638cd43a2a7a9c3ecba2e71c8 Mon Sep 17 00:00:00 2001 From: safishamsi Date: Mon, 17 Aug 2026 15:40:19 +0100 Subject: [PATCH] fix(detect): decode a UTF-16-BOM ignore file instead of latin-1 garbage (#2798) The #2799 fallback (utf-8 -> host codepage -> latin-1) turned a BOM'd UTF-16 ignore file (what PowerShell Set-Content / Notepad 'Unicode' write) into NUL-laden mojibake via latin-1, so its rules matched nothing. Detect the UTF-16 BOM and decode as utf-16 before the latin-1 fallback. Adds an end-to-end UTF-16 exclusion test and a direct no-NUL-garbage test. Co-Authored-By: Claude Opus 4.8 (1M context) --- CHANGELOG.md | 1 + graphify/detect.py | 8 ++++++++ tests/test_ignore_file_encoding.py | 20 ++++++++++++++++++++ 3 files changed, 29 insertions(+) diff --git a/CHANGELOG.md b/CHANGELOG.md index 4e31c62..d175c40 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,7 @@ Full release notes with details on each version: [GitHub Releases](https://githu ## 0.9.46 (unreleased) +- Fix: a `.gitignore`/`.graphifyignore` saved in a non-UTF-8 encoding no longer silently drops its rules (which let an explicitly-excluded directory get scanned anyway); the file is decoded UTF-8-first, then by a UTF-16 BOM, then the host codepage/latin-1, so the rule survives intact with a warning instead of being truncated (#2798, thanks @abhay-codes07). - Fix: when node dedup merges two nodes, any hyperedge that listed the merged-away node as a member now rewires that member to the survivor instead of silently dropping it, so a grouping no longer loses participants on dedup (#2805, thanks @abhay-codes07). - Fix: pruning a source file now also sweeps the external-import placeholder nodes it strands at degree 0, instead of leaving them to accumulate in the node count, `GRAPH_REPORT.md`, and exports; genuinely-isolated real nodes (which carry a `source_file`) are never touched (#2807, thanks @abhay-codes07). - Fix: when two edges connect the same node pair with different relations, the graph builder now keeps the more specific relation (`calls`, `imports`, `inherits`, ...) instead of letting a generic `references`/`uses`/`mentions` overwrite it; previously a real `calls` could be downgraded to `references` and then dropped from the call graph (#2803, thanks @abhay-codes07). diff --git a/graphify/detect.py b/graphify/detect.py index 3a300a3..e17809f 100644 --- a/graphify/detect.py +++ b/graphify/detect.py @@ -1148,6 +1148,14 @@ def _read_ignore_text(path: Path) -> str: return raw.decode("utf-8-sig") except UnicodeDecodeError: pass + # A BOM'd UTF-16 file (common from PowerShell `Set-Content` / Notepad "Unicode") + # is not valid UTF-8, and latin-1 would map its interleaved NUL bytes to a + # wall of \x00, garbling every rule. Decode it as UTF-16 by its BOM first. + if raw[:2] in (b"\xff\xfe", b"\xfe\xff"): + try: + return raw.decode("utf-16") + except UnicodeDecodeError: + pass import locale import sys as _sys fallback = locale.getpreferredencoding(False) or "latin-1" diff --git a/tests/test_ignore_file_encoding.py b/tests/test_ignore_file_encoding.py index ba624d1..cb22eaa 100644 --- a/tests/test_ignore_file_encoding.py +++ b/tests/test_ignore_file_encoding.py @@ -126,3 +126,23 @@ def test_utf8_rules_still_match_across_normalisation_forms(tmp_path, form): other = unicodedata.normalize("NFD" if form == "NFC" else "NFC", NAME) _corpus(tmp_path, f"{pattern}/\n".encode("utf-8"), dirname=other) assert "contrato.py" not in _scanned(detect(tmp_path)) + + +def test_utf16_bom_encoded_rule_still_excludes(tmp_path): + """A UTF-16 (BOM) .graphifyignore — what PowerShell Set-Content and Notepad + 'Unicode' emit — must decode by its BOM and apply, not fall to latin-1 and + garble every rule into NUL-laden noise.""" + _corpus(tmp_path, f"{NAME}/\n".encode("utf-16")) + scanned = _scanned(detect(tmp_path)) + assert "contrato.py" not in scanned, ( + "a UTF-16 ignore rule silently did nothing; scanned=" + repr(scanned)) + assert "main.py" in scanned + + +def test_utf16_is_decoded_without_nul_garbage(tmp_path): + """Direct check: the decoded text is clean UTF-16, not latin-1 mojibake.""" + p = tmp_path / ".graphifyignore" + p.write_bytes("build/\nsecret.py\n".encode("utf-16")) + text = _read_ignore_text(p) + assert "\x00" not in text + assert text.splitlines() == ["build/", "secret.py"]