diff --git a/graphify/ids.py b/graphify/ids.py index 424f3ee9..cbc33ea2 100644 --- a/graphify/ids.py +++ b/graphify/ids.py @@ -28,7 +28,7 @@ combining marks casefold introduces were never filtered: ``İslemYap`` produced ``i̇slemyap`` — an id containing U+0307, which is not a ``\\w`` character — and a second pass collapsed it to ``i_slemyap``, so the function was not idempotent and the builder's re-normalization disagreed with the extractor's ``make_id`` -for any Turkish identifier. +for any Turkish identifier (#2614). """ from __future__ import annotations @@ -47,7 +47,7 @@ def normalize_id(s: str) -> str: - The result contains only ``\w`` characters and ``_``. Casefolding before the ``[^\w]+`` filter is what makes both hold — see the - module docstring for why the reverse order silently broke them. + module docstring for why the reverse order silently broke them (#2614). """ s = unicodedata.normalize("NFKC", s) # casefold can expand one character into a letter + combining mark, so it diff --git a/tests/test_id_normalization_contract.py b/tests/test_id_normalization_contract.py index 221305e9..d69bf9de 100644 --- a/tests/test_id_normalization_contract.py +++ b/tests/test_id_normalization_contract.py @@ -39,7 +39,7 @@ CONTRACT_CASES = [ "x_c1", # must NOT be treated as a chunk suffix here "__dunder__", # leading/trailing underscores stripped "tab\tnewline\nspace ", # whitespace runs -> single underscore - # Casefolding these EXPANDS them into a base letter plus a combining + # #2614: casefolding these EXPANDS them into a base letter plus a combining # mark. With casefold last, the mark landed in the id after the [^\w] filter # had already run, so the id carried a non-word character and a second pass # changed it. Turkish identifiers are the common real-world case. @@ -52,7 +52,7 @@ CONTRACT_CASES = [ ] # Characters whose casefold expands or recomposes — the exact class that broke -# the contract. Kept separate from CONTRACT_CASES because a few of them +# the contract in #2614. Kept separate from CONTRACT_CASES because a few of them # (e.g. U+01F0) legitimately normalize to a precomposed character that is not # equal to its own casefold, which the lowercase assertion below would reject. CASE_EXPANDING_CHARS = [ @@ -110,7 +110,7 @@ def test_normalized_ids_are_safe_node_ids(): @pytest.mark.parametrize("ch", CASE_EXPANDING_CHARS) def test_case_expanding_chars_yield_word_only_ids(ch): - """The postcondition the old recipe silently broke. + """#2614: the postcondition the old recipe silently broke. ``normalize_id`` must emit only ``\\w`` characters and ``_``. Casefolding last let the combining mark that ``İ``.casefold() produces slip past the @@ -145,7 +145,7 @@ def test_case_expanding_chars_normalize_case_insensitively(ch): def test_turkish_identifier_ids_match_between_extractor_and_builder(): - """End to end: the drift that split a Turkish symbol into ghost nodes. + """#2614 end to end: the drift that split a Turkish symbol into ghost nodes. ``make_id`` minted ``islem_i̇slemyap`` (with U+0307) while the builder's re-normalization produced ``islem_i_slemyap``, so ``_semantic_id_remap``'s @@ -195,7 +195,7 @@ def test_property_normalize_id_idempotent(s): assert normalize_id(once) == once -# The plain st.text() property above already existed when this bug shipped, and did +# The plain st.text() property above already existed when #2614 shipped, and did # not catch it: the bug needs one specific codepoint (U+0130) out of ~1.1M, which # a uniform draw essentially never produces. These strategies bias the search # toward the characters that actually stress the recipe — case-expanding letters @@ -216,7 +216,7 @@ def test_property_normalize_id_idempotent_under_case_stress(s): @given(_stress_text) def test_property_normalize_id_emits_only_word_chars(s): - """The postcondition the old recipe violated: only \\w and _ may survive.""" + """The postcondition #2614 violated: only \\w and _ may survive.""" out = normalize_id(s) assert not re.search(r"[^\w]", out.replace("_", "")), ( f"normalize_id({s!r}) -> {out!r} leaked a non-word character"