diff --git a/CHANGELOG.md b/CHANGELOG.md index 9453fa4..a053e4e 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,8 @@ Full release notes with details on each version: [GitHub Releases](https://githu ## 0.9.46 (unreleased) +- Fix: node-id normalization is now caseless-stable for combining-mark sequences — `casefold` and NFKC don't commute, so a single pass left `normalize_id(s) != normalize_id(s.casefold())` for inputs like Greek ypogegrammeni followed by a combining accent; normalization now iterates casefold+NFKC to a fixpoint. Letter/digit-bearing ids are unchanged, so existing graphs are not re-keyed. + - Fix: Java annotations now emit `references` edges to their type — including class-literal arguments (`@Repeatable(Foo.class)`, `@Uses({A.class, B.class})`) and annotation-member return types — so a container annotation is no longer a disconnected island; string/enum arguments are not mistaken for type references (#2426, thanks @oleksii-tumanov). - Fix: `graphify query` treats `_` as a token separator (like `-`), so an underscore-spelled query (`user_service`) matches a hyphenated label (`user-service`); coverage-scaling keeps the broader tokenization from surfacing unrelated single-token noise (#2473, thanks @nadiadatepe-eng). - Fix: the `post-checkout` hook skips its rebuild when HEAD is unchanged (e.g. `git checkout -b` with no start point), so creating a branch no longer triggers a full graph rebuild (#2421, thanks @nothariharan). diff --git a/graphify/ids.py b/graphify/ids.py index cbc33ea..0143ee2 100644 --- a/graphify/ids.py +++ b/graphify/ids.py @@ -15,13 +15,14 @@ is exactly how the recurring ID-drift bug class crept in (#811 Unicode collapse, module exists so the recipe lives in one place and the two callers can no longer diverge. -The recipe: NFKC-normalize (so composed/decomposed Unicode forms collapse), -casefold, NFKC-normalize again (casefold can *expand* a character into a base -letter plus a combining mark — ``İ`` -> ``i`` + U+0307 — and the second pass -recomposes what can be recomposed), replace runs of non-word characters with a -single underscore (``re.UNICODE`` so CJK/Cyrillic/Arabic/accented-Latin letters -survive instead of collapsing to a per-file node), collapse repeated -underscores, then strip leading/trailing underscores. +The recipe: iterate ``casefold`` then NFKC-normalize to a fixpoint (casefold can +*expand* a character into a base letter plus a combining mark — ``İ`` -> ``i`` + +U+0307 — and NFKC then recomposes what can be recomposed; because the two do not +commute and neither is a fixpoint of the other, a single pass is not +caseless-stable), then replace runs of non-word characters with a single +underscore (``re.UNICODE`` so CJK/Cyrillic/Arabic/accented-Latin letters survive +instead of collapsing to a per-file node), collapse repeated underscores, then +strip leading/trailing underscores. Casefolding runs BEFORE the non-word filter, not after. With it last, the combining marks casefold introduces were never filtered: ``İslemYap`` produced @@ -29,6 +30,14 @@ combining marks casefold introduces were never filtered: ``İslemYap`` produced second pass collapsed it to ``i_slemyap``, so the function was not idempotent and the builder's re-normalization disagreed with the extractor's ``make_id`` for any Turkish identifier (#2614). + +Casefolding runs in a FIXPOINT LOOP, not once. A single ``NFKC(casefold(...))`` +left ``normalize_id(s) != normalize_id(s.casefold())`` for some combining-mark +sequences (Greek ypogegrammeni U+0345 followed by a combining accent): +pre-casefolding turns U+0345 into ``ι``, which NFKC composes with the accent into +a precomposed char the single pass never reached. Iterating to a fixpoint — +casefold first, on the raw input — makes the result caseless-stable regardless of +how many times the caller has already casefolded. """ from __future__ import annotations @@ -41,21 +50,37 @@ __all__ = ["normalize_id", "make_id"] def normalize_id(s: str) -> str: r"""Normalize a single ID string to its canonical form. - Guarantees, both enforced by tests: + Guarantees, all enforced by tests: - Idempotent: ``normalize_id(normalize_id(s)) == normalize_id(s)``. - The result contains only ``\w`` characters and ``_``. + - Caseless-stable: ``normalize_id(s) == normalize_id(s.casefold())``. - Casefolding before the ``[^\w]+`` filter is what makes both hold — see the - module docstring for why the reverse order silently broke them (#2614). + casefold and NFKC do not commute, and neither is a fixpoint of the other: + casefolding a char can expand it into a base letter plus a combining mark + (``İ`` -> ``i`` + U+0307), and NFKC can then recompose that mark with an + adjacent one into a different precomposed char. A single ``NFKC(casefold(...))`` + pass therefore left ``normalize_id(s) != normalize_id(s.casefold())`` for some + combining-mark sequences (e.g. Greek ypogegrammeni U+0345 followed by a + combining accent): pre-casefolding turned U+0345 into ``ι`` which NFKC then + composed with the accent, reaching a form the single-pass recipe never saw. + + So iterate ``casefold`` then ``NFKC`` to a fixpoint (casefold FIRST, on the + raw input, so a caller that pre-casefolds lands on the same fixpoint). The + loop is bounded — Unicode caseless folding converges in one or two steps — + with a hard cap as a termination guard. Only then apply the ``[^\w]+`` filter, + so every combining mark casefold introduced has been fully normalized before + it is filtered (#2614 and its combining-mark follow-on). """ - s = unicodedata.normalize("NFKC", s) - # casefold can expand one character into a letter + combining mark, so it - # must run while the non-word filter can still see the result. - s = unicodedata.normalize("NFKC", s.casefold()) - s = re.sub(r"[^\w]+", "_", s, flags=re.UNICODE) - s = re.sub(r"_+", "_", s) - return s.strip("_") + cur = s + for _ in range(6): + nxt = unicodedata.normalize("NFKC", cur.casefold()) + if nxt == cur: + break + cur = nxt + cur = re.sub(r"[^\w]+", "_", cur, flags=re.UNICODE) + cur = re.sub(r"_+", "_", cur) + return cur.strip("_") def make_id(*parts: str) -> str: diff --git a/tests/test_id_normalization_contract.py b/tests/test_id_normalization_contract.py index d69bf9d..230b695 100644 --- a/tests/test_id_normalization_contract.py +++ b/tests/test_id_normalization_contract.py @@ -160,6 +160,30 @@ def test_turkish_identifier_ids_match_between_extractor_and_builder(): assert minted == "islem_i_slemyap" +def test_normalize_id_caseless_stable_for_combining_mark_sequences(): + """Regression: casefold and NFKC do not commute, and a single + ``NFKC(casefold(...))`` pass left ``normalize_id(s) != normalize_id(s.casefold())`` + for combining-mark sequences — e.g. Greek ypogegrammeni (U+0345) followed by a + combining accent, where pre-casefolding turns U+0345 into ``ι`` which NFKC then + composes with the accent into a precomposed char the single pass never saw. + The fixpoint loop makes normalize_id caseless-stable. (Deterministic pin so the + fix does not rely on the hypothesis property re-drawing these codepoints.)""" + import re as _re + cases = [ + "\u0345\u0300", # ypogegrammeni + grave (minimal falsifying case) + "\u0345\u0301", # ypogegrammeni + acute + "\u0345\u0300\u0301", + "a\u0345\u0300b", + "\u01f0\u0f35\u0345", # ǰ + Tibetan mark + ypogegrammeni (hypothesis example) + ] + for s in cases: + assert normalize_id(s) == normalize_id(s.casefold()), ( + s.encode("unicode_escape"), normalize_id(s), normalize_id(s.casefold()), + ) + assert normalize_id(normalize_id(s)) == normalize_id(s) # still idempotent + assert not _re.search(r"[^\w]", normalize_id(s).replace("_", "")) # still word-only + + def test_both_callers_share_one_implementation(): """Guard against re-forking: the two public callers must resolve to the same underlying function object as graphify.ids.normalize_id."""