From d3335f8e2cc5b72c97fa1d1eef068ca53503c0ff Mon Sep 17 00:00:00 2001 From: Derek Gulbranson Date: Fri, 2 Oct 2026 15:10:52 -0700 Subject: [PATCH 1/5] fix(#542): read a combining mark as part of the letter before it in case repair MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A name typed decomposed (NFD) was split at each combining mark, since _WORD's \w matches no category-M character: 'josé garcía' repaired to 'José GarcíA'. Every clause now reads a mark as part of its letter: - _sub_words runs a _WORD match on through the marks after it - the Mac/Mc clause is decided on the composed spelling and applied to the word as written ('macée' gave 'MacÉe' composed, 'Macée' NFD) - _apply_mask's split-off-initial test and _letter_run_ge2 read a letter's neighbours past marks (_beside), two shapes the issue expected to need no change The output keeps the form it was typed in. A mark with no letter before it heads no word. As a side effect 'ǰ' (which upper-cases to J + combining caron) is now a fixpoint under a second forced pass. Co-Authored-By: Claude Opus 5.5 --- nameparser/_render.py | 113 +++++++++++++++++++++++++++--------- tests/v2/test_properties.py | 87 +++++++++++++++++++++++++++ tests/v2/test_regex_sync.py | 4 ++ tests/v2/test_render.py | 53 +++++++++++++++++ 4 files changed, 229 insertions(+), 28 deletions(-) diff --git a/nameparser/_render.py b/nameparser/_render.py index 20c5cfd1..140c7098 100644 --- a/nameparser/_render.py +++ b/nameparser/_render.py @@ -13,6 +13,8 @@ from __future__ import annotations import re +import unicodedata +from collections.abc import Callable from nameparser._lexicon import FULL_STOPS, Lexicon, _normalize from nameparser._types import (FOLDED_TAG, SHAPE_ACRONYM_TAG, @@ -25,6 +27,7 @@ _COMMA_CHAR = re.compile(r"[,،,]") # ASCII, Arabic, fullwidth _MAC = re.compile(r"^(ma?c)(\w{2,})", re.IGNORECASE) _WORD = re.compile(r"(\w|\.)+") +_NAME_RUN = re.compile(r"\w+") #: str.format keys render() accepts: the seven role fields in canonical #: order (derived from Role -- never restated) plus the derived views. @@ -226,14 +229,27 @@ def initials(name: ParsedName, spec: str, delimiter: str, separator: str) -> str return _format_spec(spec, values, "initials", _INITIALS_KEYS) +def _beside(text: str, at: int, step: int) -> str | None: + """The character beside text[at], one `step` (-1 or 1) away past + any combining marks, or None at the edge of `text`. A mark belongs + to the letter before it, so a decomposed letter's neighbours are + its composed twin's (#542).""" + at += step + while 0 <= at < len(text) and unicodedata.category(text[at])[0] == "M": + at += step + return text[at] if 0 <= at < len(text) else None + + def _letter_run_ge2(text: str) -> list[bool]: """One flag per alphanumeric character of `text`, in order: True - where it is a LETTER with a letter immediately before or after - it. Any other character -- a full stop, a space, a digit -- ends - a run, and a digit's own flag is always False.""" - last = len(text) - 1 - return [c.isalpha() and ((i > 0 and text[i - 1].isalpha()) - or (i < last and text[i + 1].isalpha())) + where it is a LETTER with a letter immediately before or after it + (_beside). Any other character -- a full stop, a space, a digit -- + ends a run, and a digit's own flag is always False.""" + def letter(c: str | None) -> bool: + return c is not None and c.isalpha() + + return [c.isalpha() and (letter(_beside(text, i, -1)) + or letter(_beside(text, i, 1))) for i, c in enumerate(text) if c.isalnum()] @@ -250,9 +266,10 @@ def _apply_mask(word: str, mask: str) -> str | None: The override cited above: a letter standing alone beside a full stop (any of FULL_STOPS, though through capitalized() only the - ASCII period reaches here, _WORD splitting a token at any other) - is written upper wherever the mask writes it inside a run of two - or more letters (_letter_run_ge2). + ASCII period reaches here, _WORD splitting a token at any other; + beside as _beside reads it, past combining marks) is written upper + wherever the mask writes it inside a run of two or more letters + (_letter_run_ge2). Casing goes through the whole word (word.lower()/word.upper()) when both keep its length, since per-character casing is @@ -266,7 +283,6 @@ def _apply_mask(word: str, mask: str) -> str | None: word_run = _letter_run_ge2(word) lowered, uppered = word.lower(), word.upper() same_length = len(lowered) == len(uppered) == len(word) - last = len(word) - 1 out: list[str] = [] at = 0 for i, c in enumerate(word): @@ -275,8 +291,8 @@ def _apply_mask(word: str, mask: str) -> str | None: continue upper = not mask_chars[at].islower() or ( mask_run[at] and c.isalpha() and not word_run[at] - and ((i > 0 and word[i - 1] in FULL_STOPS) - or (i < last and word[i + 1] in FULL_STOPS))) + and any(c is not None and c in FULL_STOPS + for c in (_beside(word, i, -1), _beside(word, i, 1)))) at += 1 if same_length: out.append(uppered[i] if upper else lowered[i]) @@ -439,13 +455,53 @@ def _cap_word(word: str, role: Role, tags: frozenset[str], # through ('Carod i' forced -> 'Carod I'). if role is Role.SUFFIX and _ROMAN.match(normalized): return word.upper() - if _MAC.match(word): - return _MAC.sub( - lambda m: m.group(1).capitalize() + m.group(2).capitalize(), - word) + # Decided on the composed spelling, so a word typed decomposed + # (NFD) is read as its composed twin is: 'mac' + 'e' + U+0301 + + # 'e' fails the regex's \w{2,} at the mark (#542). The name run + # after the prefix is then capitalized on the word as written, + # running on through its marks the way _sub_words does. + mac = _MAC.match(unicodedata.normalize("NFC", word)) + if mac: + cut = end = len(mac.group(1)) + while run := _NAME_RUN.match(word, end): + end = _past_marks(word, run.end()) + return (word[:cut].capitalize() + word[cut:end].capitalize() + + word[end:]) return word.capitalize() +def _past_marks(text: str, end: int) -> int: + """`end` moved past the combining marks (category M) at it.""" + while end < len(text) and unicodedata.category(text[end])[0] == "M": + end += 1 + return end + + +def _sub_words(cap: Callable[[str], str], text: str) -> str: + """`text` with each word replaced by cap(word): what + _WORD.sub(cap, text) does, except that a word runs on through the + combining marks after it (#542). A combining mark is category M, + which \\w does not match, so a name typed decomposed (NFD) -- + 'garci' + U+0301 + 'a' -- would otherwise be two words, and the + 'a' capitalized alone is 'A'. A mark heads no word of its own: one + with no word before it stays outside every word, since a word + starting with it would keep its letters lowercase under + str.capitalize().""" + spans: list[list[int]] = [] + for match in _WORD.finditer(text): + start, end = match.start(), _past_marks(text, match.end()) + if spans and spans[-1][1] == start: + spans[-1][1] = end + else: + spans.append([start, end]) + out: list[str] = [] + at = 0 + for start, end in spans: + out += (text[at:start], cap(text[start:end])) + at = end + return "".join(out) + text[at:] + + def _cap_text(text: str, role: Role, tags: frozenset[str], lex: Lexicon) -> str: # word-by-word within the token text: hyphenated names capitalize @@ -454,11 +510,11 @@ def _cap_text(text: str, role: Role, tags: frozenset[str], # vocabulary asked per word: the parse would have made one token # per word of that text, so this is the granularity its answer # would have had. - def cap(match: re.Match[str]) -> str: - return _cap_word(match.group(0), role, tags, lex) + def cap(word: str) -> str: + return _cap_word(word, role, tags, lex) if "-" not in text: - return _WORD.sub(cap, text) + return _sub_words(cap, text) parts = text.split("-") # A "named" part needs an alphanumeric, not just a _WORD match: # _WORD also matches a run of bare periods (or underscores), so a @@ -468,7 +524,7 @@ def cap(match: re.Match[str]) -> str: named = [at for at, part in enumerate(parts) if any(c.isalnum() for c in part)] if len(named) < 3: - return _WORD.sub(cap, text) + return _sub_words(cap, text) # rules.md#R4: "Inside a hyphenated word, a part that is # connective vocabulary with a worded part on each side of it # keeps its lowercase" (#478). Decided here, not in _cap_word, @@ -486,7 +542,7 @@ def cap(match: re.Match[str]) -> str: part.lower() if (first < at < last and not _DOTTED_INITIAL.fullmatch(part) and _normalize(part) in lex.conjunctions) - else _WORD.sub(cap, part) + else _sub_words(cap, part) for at, part in enumerate(parts)) @@ -532,7 +588,7 @@ def capitalized(name: ParsedName, lexicon: Lexicon | None, *, comes back unchanged whether or not the gate admits it again (a name whose non-suffix words are caseless, 'Kim Minjun' in hangul with a 'phd', is admitted every time) -- except where a - LETTER'S OWN CASE MAPPING changes its length or splits the word + LETTER'S OWN CASE MAPPING changes its length (decisions.md#R4's Unicode boundary; 'ß' recasing to 'SS' through a mask is one example, not the only one). Lengthening: a mask's own per-character casing fallback can turn one letter into a @@ -545,12 +601,13 @@ def capitalized(name: ParsedName, lexicon: Lexicon | None, *, CASED characters 'ʼN', so str.capitalize() on 'ʼna' gives 'ʼNa' but on THAT output gives 'ʼna' -- the 'N' is no longer the word's first character, so the second pass lower-cases it. - Splitting: some letters upper-case to a base letter plus a - COMBINING MARK, which _WORD does not match -- 'ǰ' upper-cases to - 'J' + a combining caron, so 'ǰo' capitalizes to 'J̌o', but - _cap_text reads THAT text as two separate words ('J', then 'o', - the combining mark between them matching neither), and 'o' - capitalized alone is 'O'.""" + A letter that upper-cases to a base letter plus a COMBINING MARK + ('ǰ' to 'J' + a combining caron) is not such a case since #542: + a word runs on through its marks (_sub_words), so 'J̌o' is one + word on the second pass and comes back unchanged. + A name typed decomposed (NFD) repairs as its composed twin does, + every clause reading a mark as part of the letter before it, and + keeps the form it was typed in.""" if lexicon is not None and not isinstance(lexicon, Lexicon): # eager, before the gate: a garbage argument must not become a # silent no-op on mixed-case input or a deep AttributeError diff --git a/tests/v2/test_properties.py b/tests/v2/test_properties.py index 9f75c8d9..ca3ad8f2 100644 --- a/tests/v2/test_properties.py +++ b/tests/v2/test_properties.py @@ -11,6 +11,7 @@ import hashlib import itertools import re +import unicodedata import warnings import pytest from hypothesis import given, settings @@ -3105,6 +3106,92 @@ def test_the_case_only_walk_can_fail( assert any("'Ph.D.' -> 'PhD'" in line for line in failures), failures +# --- #542: a decomposed name repairs as its composed twin does ------- +# The same texts as the walk above, those that decompose at all, each +# parsed and repaired in both forms and compared once the decomposed +# output is composed again. A text whose two forms PARSE differently +# is set aside rather than compared: repair follows the parse, so the +# comparison would report the parse, and the one class that does so is +# unspaced hangul, which segmentation matches as written by decision +# (docs/usage.rst, "Decomposed text"). The test below pins that filter. + +def _hangul(text: str) -> bool: + return any("가" <= c <= "힣" + for c in unicodedata.normalize("NFC", text)) + + +def _nfd_repair_findings() -> tuple[list[str], list[str], int]: + """(disagreements, texts set aside, texts compared).""" + def nfc(text: str) -> str: + return unicodedata.normalize("NFC", text) + + parser = Parser() + out: list[str] = [] + set_aside: list[str] = [] + compared = 0 + for text in _CASE_ONLY_TEXTS: + decomposed = unicodedata.normalize("NFD", text) + if decomposed == nfc(text): + continue + name, composed = parser.parse(decomposed), parser.parse(nfc(text)) + if ([(t.role, nfc(t.text)) for t in name.tokens] + != [(t.role, t.text) for t in composed.tokens]): + set_aside.append(text) + continue + compared += 1 + for force in (False, True): + got = [nfc(t.text) for t in + parser.capitalized(name, force=force).tokens] + want = [t.text for t in + parser.capitalized(composed, force=force).tokens] + if got != want: + out.append(f"[core, force={force}] {text!r}: " + f"{got!r} != {want!r}") + human, twin = HumanName(decomposed), HumanName(nfc(text)) + for force in (False, True): + human.capitalize(force=force) + twin.capitalize(force=force) + for attr in _FACADE_LISTS: + got = [nfc(s) for s in getattr(human, attr)] + if got != getattr(twin, attr): + out.append(f"[facade, force={force}] {text!r} {attr}: " + f"{got!r} != {getattr(twin, attr)!r}") + return out, set_aside, compared + + +def test_a_decomposed_name_repairs_as_its_composed_twin() -> None: + """rules.md#R4 read as an invariant over encodings: a letter written + as a base letter and its combining accent repairs as the letter + written whole does (#542). Core and facade, plain and forced. The output is compared composed + because repair keeps the form it was given, which the case-only + walk above holds separately. + + Recorded negative control, measured 2026-10-02 with 97af1f02's + _render.py, before a word ran on through its marks, over this + change's corpus (its own two R4 rows included): 144 disagreements + over the 137 texts compared, 47 more set aside, every one hangul + ('JOSÉ GARCÍA' repairing to 'José GarcíA'). The live control is the test below.""" + failures, set_aside, compared = _nfd_repair_findings() + assert compared, "no decomposable text was compared" + assert all(_hangul(text) for text in set_aside), ( + "a non-hangul text parses differently decomposed: " + f"{[t for t in set_aside if not _hangul(t)]}") + assert not failures, ( + f"{len(failures)} decomposed repair(s) disagree:\n" + + "\n".join(failures[:10])) + + +def test_the_decomposed_walk_can_fail( + monkeypatch: pytest.MonkeyPatch) -> None: + """Stopping a word at its first combining mark again -- #542's + defect -- has to be seen by the walk above.""" + import nameparser._render as render_module + monkeypatch.setattr(render_module, "_past_marks", + lambda text, end: end) + failures, _, _ = _nfd_repair_findings() + assert any("'José', 'GarcíA'" in line for line in failures), failures + + def test_a_delimiter_core_reads_as_if_it_were_not_written() -> None: """#538: under a core-bearing policy, a maiden clause reads exactly as the same text written without the core. The core is structure diff --git a/tests/v2/test_regex_sync.py b/tests/v2/test_regex_sync.py index 4aa99e53..f92cc49b 100644 --- a/tests/v2/test_regex_sync.py +++ b/tests/v2/test_regex_sync.py @@ -156,6 +156,10 @@ def test_dotted_initial_is_the_period_alternative_of_initial() -> None: ("_render", "_INITIAL"): None, # config's pattern minus one "?" ("_vocab", "_INITIAL"): None, # same ("_render", "_DOTTED_INITIAL"): None, # _INITIAL's period alternative + # #542: the name run after a Mac/Mc prefix, read on the word as + # written once _MAC has matched its composed spelling; no config + # key to mirror + ("_render", "_NAME_RUN"): None, ("_tokenize", "_BIDI"): None, # re_bidi, not a REGEXES key # Mirrors _pipeline._state.COMMA_CHARS, not nameparser.config ("_render", "_COMMA_CHAR"): None, diff --git a/tests/v2/test_render.py b/tests/v2/test_render.py index b92bed28..8a51f2bb 100644 --- a/tests/v2/test_render.py +++ b/tests/v2/test_render.py @@ -915,6 +915,59 @@ def test_a_decomposed_mask_value_reads_the_same_split_as_composed() -> None: "john smith pé.x.").suffix == "Pé.X." +def _nfc_texts(name: ParsedName) -> list[str]: + return [unicodedata.normalize("NFC", t.text) for t in name.tokens] + + +@pytest.mark.parametrize(("pairs", "text"), [ + # the plain word clause: 'a' after the mark is not a word of its own + ((), "josé garcía"), + # Mac/Mc, decided on the composed spelling: 'mac' + 'ée' passes + # _MAC's \w{2,} composed and fails it at the mark decomposed + ((), "macée smith"), + ((), "MACÉINRÍ SMITH"), + # the split-off-initial override reads the full stop past the mark + ((("éx", "éx"),), "john smith é.x."), + # a letter after a mark still has the letter before it as neighbour + ((("éex", "éeX"),), "john smith ée.x"), +]) +def test_a_decomposed_word_repairs_as_its_composed_twin( + pairs: tuple[tuple[str, str], ...], text: str) -> None: + """#542: a combining mark belongs to the letter before it, so every + clause of case repair reads a decomposed (NFD) word as it reads the + composed one -- the word splitter, the Mac/Mc clause, and the mask's + neighbour tests -- and the output keeps the decomposed form. These + are the shapes the corpus walk in test_properties.py does not + reach; each disagreed before the fix.""" + composed = unicodedata.normalize("NFC", text) + decomposed = unicodedata.normalize("NFD", text) + assert decomposed != composed + want = _repaired_under(pairs, composed) + got = _repaired_under(pairs, decomposed) + assert _nfc_texts(got) == [t.text for t in want.tokens] + assert all(unicodedata.is_normalized("NFD", t.text) + for t in got.tokens) + + +def test_a_letter_that_uppercases_to_a_combining_mark_is_a_fixpoint( +) -> None: + """'ǰ' upper-cases to 'J' + a combining caron. Before #542 the + second forced pass read 'J̌o' as two words and gave 'J̌O'; the mark + now stays with its letter, so repair is idempotent here.""" + p = Parser() + once = p.capitalized(p.parse("ǰo smith"), force=True) + assert once.given == "J̌o" + assert p.capitalized(once, force=True).given == "J̌o" + + +def test_a_mark_with_no_letter_before_it_heads_no_word() -> None: + """A leading combining mark stays outside the word after it: + str.capitalize() on a word starting with the mark would upper-case + the mark and leave the letter lowercase.""" + assert parse("́abc smith").capitalized(force=True).given == ( + "́Abc") + + def test_a_masks_upper_fallback_can_lengthen_a_word_through_ss() -> None: """decisions.md#R4 (2026-09-24 review): where the whole-word upper-casing does not keep the word's length, _apply_mask falls From 4de28051b2a14a22b160b32998d658a46a87f1de Mon Sep 17 00:00:00 2001 From: Derek Gulbranson Date: Fri, 2 Oct 2026 15:10:53 -0700 Subject: [PATCH 2/5] docs(#542): state that repair reads a decomposed letter whole MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit rules.md#R4 gains the statement and two NFD example lines (the expected values spell the mark as ́, so the form is visible and an editor that NFC-normalizes the file fails the example); corpus_rules.jsonl is regenerated to carry them. decisions.md#R4 records the choice, the declined NFC-compose option, the mask shapes measured against the issue's claim, and retires the 2026-09-24 SPLITTING boundary. Release log and usage.rst's decomposed-text section follow. Co-Authored-By: Claude Opus 5.5 --- docs/design/decisions.md | 1 + docs/design/rules.md | 8 +++++++- docs/release_log.rst | 2 ++ docs/usage.rst | 4 +++- tools/differential/corpus_rules.jsonl | 2 ++ 5 files changed, 15 insertions(+), 2 deletions(-) diff --git a/docs/design/decisions.md b/docs/design/decisions.md index cce231b4..7fc941c6 100644 --- a/docs/design/decisions.md +++ b/docs/design/decisions.md @@ -1442,6 +1442,7 @@ Accepted costs, deferred to the rescoped #459 rather than relitigated here: the - 2026-09-24 #478 — DECIDED (Derek, 2026-09-24): the hyphen clause keeps its reading in a name written wholly in one case, and the case where that disagrees with rules.md#P3 is a recorded boundary, not a change. The two collide on a marked letter. Spaced, a letter the vocabulary marks as reading both ways reads as an initial in a one-case name (P3), so `maria silva e sousa` repairs to `Maria Silva E Sousa`; hyphenated, the interior part reads as the connective whatever the name's case, so `maria silva-e-sousa` repairs to `Maria Silva-e-Sousa`. Derek's reasoning: hyphenating `Silva-e-Sousa` is the writer joining the surname on purpose, so the interior word is a connective by that act, whatever the one-case rule would say of the spaced form. THE COST, measured 2026-09-24: a one-case name whose hyphenated bare initials happen to spell a connective — `J-E-P DUPONT` gives `J-e-P Dupont`, where every release 1.4.0 through 2.3.0 and the parent 4d0680e6 gave `J-E-P Dupont`, and `JOHN A-Y-B SMITH` gives `John A-y-B Smith` (every release: `John A-Y-B Smith`). The period-marked spelling is unaffected (`j.-e.-p. dupont` keeps `J.-E.-P. Dupont`, the 2026-09-23 #478 bullet). `conjunctions_ambiguous`, P3's knob, does NOT reach a hyphenated word: under `Lexicon.default().add(conjunctions_ambiguous={"y"})` the spaced `JOSE ORTEGA Y GASSET` repairs to `Jose Ortega Y Gasset` (default lexicon: `Jose Ortega y Gasset`) while `JOSE ORTEGA-Y-GASSET` stays `Jose Ortega-y-Gasset` under both. rules.md#R4's hyphen sentence said the hyphens join the name "as the spaced connective would", which claimed an equality the one-case case breaks; it now says the hyphens are the writer's join and states the split, with an Accepted paragraph and the `J-E-P DUPONT` boundary line. Pinned by `tests/v2/test_render.py::test_the_hyphen_is_the_writers_join_even_in_a_one_case_name`. - 2026-10-02 #541 — THE "LEFT FOR THE ORCHESTRATOR TO FILE" FINDING of the 2026-09-24 branch-review bullet (d) above is #541, and it was wider than that bullet saw: besides the two raises it names, an entry carrying whitespace v1 never matched SILENTLY ACTIVATED in every set field the shim copies through and in `capitalization_exceptions` keys — measured 2026-10-02 against 1.4.0 (run outside the worktree, `PYTHONSAFEPATH=1`): `titles ' dean '` gave title `dean` on `dean john smith` where 1.4.0 gave first `dean`, and likewise `prefixes`, `suffix_acronyms`, `suffix_not_acronyms`, `conjunctions`, `bound_first_names` and a `' zzc '` key; the raise (`suffix_not_acronyms 'ma '`) and the activation (`titles ' dean '`) both reproduce on the 2.0.0, 2.2.0 and 2.3.0 wheels. DECIDED (Derek): `Constants._snapshot()` drops every entry `_config_shim._v1_matchable` rejects — empty after `lc()`, or not equal to its own single-spaced re-join — from all nine set fields and the keys, BEFORE its set algebra, and names them in one `UserWarning` whose remedy runs (`tests/v2/test_config_shim.py::test_the_offered_remedy_runs_and_silences_the_warning`). The test is exact because a v1 piece comes from a whitespace split re-joined only with single spaces; it generalizes the filter `given_name_titles` already had, and `tests/v2/test_config_shim.py::_UNFILTERED_OUTCOME` records what each swept row does with it off. One exception, decided when the 1.4 pickle test caught it: every release from at least 0.5.8 through 1.4.0 SHIPPED two such entries in `TITLES` (`'actor '`, `'television '`), so a restored 1.4 pickle carries them and its user never wrote them; they are dropped WITHOUT the warning (`_V14_SHIPPED_UNMATCHABLE`, held to the pickle's own unmatchable set by `test_the_1_4_shipped_unmatchable_roster_is_exactly_the_pickles`), keeping that pickle warning-free. The exemption is keyed by entry, not by provenance, so a user who writes `'actor '` or `'television '` themselves is not warned either — accepted, since the two strings are 1.x's own typos. The drop is a reading change for that pickle: `Actor John Smith` gave title `Actor` on 2.0.0 through 2.3.0 and gives first `Actor` now, as 1.4.0 did. DECLINED: stripping whitespace in `SetManager` (activates every v1-inert entry above), and a silent drop for user entries (Derek: the entry is certainly a typo and certainly dead, so say so). ACCEPTED WIDENING, deliberately left: a key written with capitals or edge periods (`'McDonald'`, `'phd.'`) never matched on 1.4.0, whose lookup is `lc(word)` against keys stored as written, but has matched since 2.0 through `Lexicon`'s fold; dropping it would break a working 2.x config to restore an inertness nobody wanted. Also left: a set entry `Lexicon` folds to empty that `lc()` does not, a lone non-ASCII full stop (`'。'`), still raises at the first parse in every field `Lexicon` receives directly — `titles`, `prefixes`, `suffix_acronyms`, `suffix_not_acronyms`, `conjunctions`, `bound_first_names` and a `capitalization_exceptions` key; `first_name_titles` drops it quietly (`if t`), and `non_first_name_prefixes` and `suffix_acronyms_ambiguous` never pass it to `Lexicon` at all — v1-matchable, so outside this rule. - 2026-10-02 #582 — SUPERSEDES the #541 bullet's "Also left" sentence above. DECIDED (Derek): an entry `Lexicon._normalize` folds to empty is dropped by `_config_shim._v1_matchable` with the #541 warning, in every set field and key. The case is a lone CJK full stop (`'。'`, `'.'`, `'。'`, or a run of them), which 2.3.0's `FULL_STOPS` fold (#322/#323) empties while v1's `lc()` strips only `.`. Measured 2026-10-02 on released wheels run outside the checkout with `PYTHONSAFEPATH=1`: 1.4.0 through 2.2.0 accept `c.titles.add('。')` and read `。 john smith` with title `。`, while 2.3.0 raises `ValueError` at the first parse, as did this tree before the change, in `titles`, `prefixes`, `suffix_acronyms`, `suffix_not_acronyms`, `conjunctions`, `bound_first_names` and a key. ACCEPTED DEPARTURE from 1.4.0, the only one this filter makes: such an entry used to act on a token that is nothing but that full stop. Since 2.3.0 `Lexicon` can neither hold the entry nor match the token, its lookup fold emptying both, so reproducing v1 is not on offer and the choice was drop or raise. The warning's wording moved with it, from "nameparser 1.x never matched" to "match no name word". `given_name_titles`' `if t` filter was deleted as unreachable: `_title_key` is empty only when every word folds away, and then the whole entry folds to empty and was already dropped. With no shim-reachable plain `ValueError` left from `_normpairs`, `test_an_unrelated_capitalization_exceptions_valueerror_has_no_v1_hint` turns the filter off to keep the except-by-type clause pinned. THE SAME FOLD, ONE CHARACTER IN: an entry with a CJK full stop at its EDGE (not only full stops) passed the filter, and the set algebra compared it raw while `Lexicon` folded it, so `c.suffix_not_acronyms.add('ma。')` (beside the ambiguous `ma`) raised the gate-bypass check and a bound `'zed。'` beside a never-given particle `zed` raised the contradiction check, both from 2.3.0 on; measured accepted on 1.4.0 and 2.2.0 by the review of this change. `_build_snapshot` now compares `_normalize`d spellings in exactly the two computations `Lexicon` re-checks after its own fold — dropping a `suffix_not_acronyms` entry whose fold is an ambiguous acronym's, and taking into `particles_ambiguous` every particle whose fold is a bound given name's — so such an entry reads exactly as its folded spelling (`test_an_edge_full_stop_collision_reads_as_its_folded_spelling`; the fold-off control is `_UNFOLDED_OUTCOME`). That reading is the `dean。` widening below, not v1's: 1.4.0 matched `ma。` only on a `ma。` token. DECLINED, after being written and reviewed: folding EVERY kept entry before the set algebra. `non_first_name_prefixes` and `suffix_acronyms_ambiguous` reach `Lexicon` only through raw set algebra, so folding them is a NEW match, not the 2.3.0 one: `prefixes zz` with `non_first_name_prefixes 'zz。'` moved `zz smith` from first `zz` to last `zz smith`, an NFD/NFC pair across those fields moved the same way, and `suffix_acronyms_ambiguous 'zq。'` beside `suffix_acronyms zq` moved `john zq` from suffix to last — the particle readings the same on 1.4.0, 2.2.0, 2.3.0 and the pre-fold tree, the `zq` one on 2.2.0, 2.3.0 and the pre-fold tree (1.4.0 gives last `zq` there by its own two-word reading, not by matching the entry). They are pinned at those readings by `test_an_edge_full_stop_entry_outside_the_two_checks_reads_as_before`. ACCEPTED WIDENING, left as it is (Derek): since 2.3.0 a `titles` entry `'dean。'` also matches a bare `dean`, where 1.4.0 matched only `dean。`. The same fold makes it, and dropping the entry would lose the `dean。` match too, so it sits beside the capitalized-key widening above. +- 2026-10-02 #542 — A DECOMPOSED WORD IS REPAIRED AS ITS COMPOSED TWIN, by reading a combining mark (Unicode category M) as part of the letter before it, in every clause of case repair, and never by composing the text. A name typed NFD (macOS file names, some databases) had been split at each mark by `_WORD`'s `(\w|\.)+`, `\w` matching no mark, so `josé garcía` repaired to `José GarcíA` on every release from 1.4.0. Python's `re` has no `\p{M}`, and a hand-written class of the combining blocks would be a copy of Unicode data that drifts (it would already miss Adlam, a cased script whose marks sit outside them), so the test is `unicodedata.category(ch)[0] == "M"`, asked by `_render._sub_words` after each `_WORD` match and by `_render._beside` for a letter's neighbours. DECLINED: NFC-composing before repair, the issue's option 2. It fixes the split but hands back text in a form the writer did not use, which is a change beyond case under rules.md#R4 — `tests/v2/test_properties.py::test_case_repair_changes_case_and_nothing_else` compares with `casefold()`, which does not normalize, and would report it. THE ISSUE'S CLAIM THAT THE MASK NEEDED NO CHANGE WAS MEASURED FALSE, in two shapes it did not try: `_apply_mask`'s split-off-initial override and `_letter_run_ge2` both read a letter's neighbour by raw index, which in NFD is the mark rather than the full stop or the letter past it. Under a caller's `('éx', 'éx')` mask, `john smith é.x.` forced gave `É.X.` composed and `é.X.` decomposed; under `('éex', 'éeX')`, `john smith ée.x` gave `ée.X` against `éE.X`. Both read past marks now. The Mac/Mc clause had the same split one level down, `_MAC`'s `\w{2,}` failing at the mark (`macée` gave `MacÉe` composed and `Macée` decomposed), and is now DECIDED on the composed spelling and applied to the word as written. AMENDS the 2026-09-24 branch-review bullet (a): its SPLITTING half is retired. `parse("ǰo smith").capitalized(force=True)` gives `J̌o`, and a second forced pass now gives `J̌o` back rather than `J̌O`, the mark staying with its letter (measured 2026-10-02); the LENGTHENING half (`ß`, `ʼn`) stands. A mark with no letter before it heads no word: `str.capitalize()` on a word beginning with one would upper-case the mark and leave the letter after it lowercase. Out of reach and left there: unspaced hangul, whose NFD parse differs from its NFC parse because segmentation matches the census list as written (docs/usage.rst, "Decomposed text"), so repair follows a different parse; hangul jamo are letters, not marks, and nothing here touches them. VERIFICATION: the differential gate cannot see this — `capitalized()` is not a compared surface (this entry, 2026-08-29) and no corpus name is written NFD. Standing in its place: rules.md#R4's two NFD example lines, `tests/v2/test_properties.py::test_a_decomposed_name_repairs_as_its_composed_twin`, which decomposes every corpus and case-table text that decomposes, compares its repair with the composed text's on both surfaces plain and forced, and records as its control 144 disagreements over 137 texts with 97af1f02's `_render.py` over this change's corpus (measured 2026-10-02), and `tests/v2/test_render.py::test_a_decomposed_word_repairs_as_its_composed_twin` for the shapes no corpus name reaches, whose two mask rows were mutation-checked by restoring raw-index neighbours. Excluded (CAPITALIZATION_EXCEPTIONS — meng, edd, lac, ded, left out of the 2026-09-24 masks, #459): diff --git a/docs/design/rules.md b/docs/design/rules.md index 4934d808..e911226b 100644 --- a/docs/design/rules.md +++ b/docs/design/rules.md @@ -2574,7 +2574,11 @@ R4. Rationale: case repair is a display concern, applied only on lowercase there like any other connective — the third part of a comma form is the shape that puts one there. Repair changes case and nothing else: the repaired word is the - word as written, recased. A vocabulary entry records casing as a + word as written, recased. A word is read whole however its letters + are encoded: a letter written as a base letter followed by its + combining accent is repaired as the same letter written whole, + and the repaired word keeps the encoding the writer used. A + vocabulary entry records casing as a mask — its word's letters, each in the case it takes — and repair lays it over the word as the writer punctuated it, so the one entry for phd repairs phd to PhD and ph.d. to Ph.D.; the mask @@ -2661,6 +2665,8 @@ R4. Rationale: case repair is a display concern, applied only on "Smith, John, and" → capitalized_forced="John Smith and" "Doe, Jane, and Jr." → capitalized_forced="Jane Doe and Jr." "juan de la vega" → capitalized="Juan de la Vega" · boundary + "josé garcía" → capitalized="Jose\u0301 Garci\u0301a" + "JOSÉ GARCÍA" → capitalized="Jose\u0301 Garci\u0301a" "jose ortega-y-gasset" → capitalized="Jose Ortega-y-Gasset" "JOSE ORTEGA-Y-GASSET" → capitalized="Jose Ortega-y-Gasset" "maria silva-e-sousa" → capitalized="Maria Silva-e-Sousa" diff --git a/docs/release_log.rst b/docs/release_log.rst index 9bd62191..d172bdc4 100644 --- a/docs/release_log.rst +++ b/docs/release_log.rst @@ -50,6 +50,8 @@ Release Log - **Change case repair to write an unlisted dotted credential and a roman numeral past iv in capitals.** ``HumanName("john smith x.y.z.").capitalize()`` gives ``John Smith X.Y.Z.`` where every release gave ``John Smith X.y.z.``, the dotted word being a suffix now (the ``unlisted_dotted_suffixes`` change above) and repaired as a listed acronym is; and ``john smith vi`` gives ``John Smith VI`` where every release gave ``John Smith Vi``, with ``vii``, ``viii`` and ``ix`` alike. Both are keyed on the suffix role: ``Jack X.Y.Z.``, which keeps its surname, still repairs as a name word (``Jack X.y.z.`` under ``force=True``), and ``john smith xi`` still gives ``John Smith Xi``, the parser reading ``xi`` as the surname. An unlisted dotted credential written in mixed case is kept as written on the default path, by the suffix change above (``john smith B.Tech.`` gives ``John Smith B.Tech.``), and reads all capitals under ``force=True`` (``John Smith B.TECH.``, where every release gave ``John Smith B.tech.``), which a ``capitalization_exceptions`` mask such as ``{"btech": "BTech"}`` undoes. Over the 1340 names in the differential corpora at the commit before this change (2026-09-23), 1 moves on the default path and 14 under ``force=True``. The recipe is the ``R4`` entry's 2026-09-23 MEASURED bullet in ``docs/design/decisions.md`` (#459) + - **Fix case repair breaking a name typed with decomposed accents at each accent.** ``HumanName(unicodedata.normalize("NFD", "josé garcía")).capitalize()`` gives ``José García``, where every release from 1.4.0 through 2.3.0 gave ``José GarcíA``: decomposed text (NFD, which macOS file names and some databases hand back) writes ``í`` as ``i`` followed by a combining accent, and the letters after the accent were repaired as a separate word. A decomposed name now repairs the way its composed spelling does, Mac/Mc names and case-repair masks included, and the output keeps the form it was typed in. See the ``R4`` entry of ``docs/design/decisions.md`` (closes #542) + - **Change the parse pipeline to copy its state without dataclasses.replace.** Every stage returns a copy of its frozen state, and several also copy tokens one at a time; ``dataclasses.replace`` goes through ``fields()`` and ``__init__`` on every one of those copies. The stages now copy fields directly through a small helper that is limited to the pipeline's own three dataclasses and checks them at import. One parse of the benchmark's reference name makes 36 fewer calls on py3.11 and 3.12 and 54 fewer from 3.13 (on 3.11, ``parse`` 406 to 370 and ``HumanName`` 443 to 407), and the call-count baselines move with them. Recomputable with ``uv run python tools/perf/call_count.py --against e0f1a2f``; the counts for every interpreter are in the ``parse-cost`` entry of ``docs/design/decisions.md``. No user-visible behavior changes (#546) - **Fix a long given part after a family comma costing quadratic time.** Since 2.3.0, parsing ``"Doe, Jane " + "Smith " * n`` took time growing with the square of the part's length: going from 1,600 to 6,400 words cost 8.4x the time, where 2.2.0 and the comma-less form cost 4x. A run of trailing titles in the same place (``Doe, Jane Smith Prof. Prof. ...``) cost 10x. Both cost 4x again (Python 3.11, measured 2026-09-28). No field moves (closes #553) diff --git a/docs/usage.rst b/docs/usage.rst index 194b14d2..35113819 100644 --- a/docs/usage.rst +++ b/docs/usage.rst @@ -334,7 +334,9 @@ decomposed name gets the same order rule as its composed twin. Vocabulary lookup does the same before matching a word against titles, honorifics and the rest, so a decomposed ``Señor`` or ``née`` — macOS-origin data again — is recognized as readily as its composed -spelling. +spelling. Case repair reads a decomposed accent as part of its letter, +so ``capitalized()`` gives a decomposed ``josé garcía`` the same +``José García`` as the composed spelling, still decomposed. Splitting is the exception. An unspaced decomposed hangul name is ordered correctly but not split, because surname matching runs against diff --git a/tools/differential/corpus_rules.jsonl b/tools/differential/corpus_rules.jsonl index c8b5a884..20b6213b 100644 --- a/tools/differential/corpus_rules.jsonl +++ b/tools/differential/corpus_rules.jsonl @@ -119,6 +119,7 @@ "JOHN SMITH PH.D." "JOHN SMITH, VD MA" "JOSE ORTEGA-Y-GASSET" +"JOSÉ GARCÍA" "JUAN GARCIA Jr." "JUAN GARCIA Y LOPEZ" "Jack MA" @@ -450,6 +451,7 @@ "john van der berg ma" "jose e maria santos" "jose ortega-y-gasset" +"josé garcía" "juan de la vega" "juan e-f smith" "juan garcia III" From cdd4afd12614ba319a8e0ce3d3babfcbf3769c1e Mon Sep 17 00:00:00 2001 From: Derek Gulbranson Date: Fri, 2 Oct 2026 15:19:52 -0700 Subject: [PATCH 3/5] fix(#542): read the two initial-shape tests on the composed spelling MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Docs review findings. The hyphen clause's _DOTTED_INITIAL and the unclassified-text fallback's _INITIAL matched the raw word, so a decomposed 'й.' (й being the one default conjunction that decomposes) read as the conjunction where its composed twin reads as an initial: 'ivan petrov-й.-sidorov' repaired to 'Petrov-й.-Sidorov' NFD against 'Petrov-Й.-Sidorov'. Both now match the NFC spelling. decisions.md#R4's #542 bullet quoted outputs from a splitter-only intermediate as pre-fix behavior; it now gives both, says the 'é.x.' mask row agreed before the fix, drops the false "no corpus name is NFD", and records initials() dropping a decomposed accent as out of reach. _lexicon's reason for storing mask values composed is updated. Co-Authored-By: Claude Opus 5.5 --- docs/design/decisions.md | 2 +- nameparser/_lexicon.py | 8 +++++--- nameparser/_render.py | 8 ++++++-- tests/v2/test_render.py | 19 ++++++++++++++++++- 4 files changed, 30 insertions(+), 7 deletions(-) diff --git a/docs/design/decisions.md b/docs/design/decisions.md index 7fc941c6..f3685fe7 100644 --- a/docs/design/decisions.md +++ b/docs/design/decisions.md @@ -1442,7 +1442,7 @@ Accepted costs, deferred to the rescoped #459 rather than relitigated here: the - 2026-09-24 #478 — DECIDED (Derek, 2026-09-24): the hyphen clause keeps its reading in a name written wholly in one case, and the case where that disagrees with rules.md#P3 is a recorded boundary, not a change. The two collide on a marked letter. Spaced, a letter the vocabulary marks as reading both ways reads as an initial in a one-case name (P3), so `maria silva e sousa` repairs to `Maria Silva E Sousa`; hyphenated, the interior part reads as the connective whatever the name's case, so `maria silva-e-sousa` repairs to `Maria Silva-e-Sousa`. Derek's reasoning: hyphenating `Silva-e-Sousa` is the writer joining the surname on purpose, so the interior word is a connective by that act, whatever the one-case rule would say of the spaced form. THE COST, measured 2026-09-24: a one-case name whose hyphenated bare initials happen to spell a connective — `J-E-P DUPONT` gives `J-e-P Dupont`, where every release 1.4.0 through 2.3.0 and the parent 4d0680e6 gave `J-E-P Dupont`, and `JOHN A-Y-B SMITH` gives `John A-y-B Smith` (every release: `John A-Y-B Smith`). The period-marked spelling is unaffected (`j.-e.-p. dupont` keeps `J.-E.-P. Dupont`, the 2026-09-23 #478 bullet). `conjunctions_ambiguous`, P3's knob, does NOT reach a hyphenated word: under `Lexicon.default().add(conjunctions_ambiguous={"y"})` the spaced `JOSE ORTEGA Y GASSET` repairs to `Jose Ortega Y Gasset` (default lexicon: `Jose Ortega y Gasset`) while `JOSE ORTEGA-Y-GASSET` stays `Jose Ortega-y-Gasset` under both. rules.md#R4's hyphen sentence said the hyphens join the name "as the spaced connective would", which claimed an equality the one-case case breaks; it now says the hyphens are the writer's join and states the split, with an Accepted paragraph and the `J-E-P DUPONT` boundary line. Pinned by `tests/v2/test_render.py::test_the_hyphen_is_the_writers_join_even_in_a_one_case_name`. - 2026-10-02 #541 — THE "LEFT FOR THE ORCHESTRATOR TO FILE" FINDING of the 2026-09-24 branch-review bullet (d) above is #541, and it was wider than that bullet saw: besides the two raises it names, an entry carrying whitespace v1 never matched SILENTLY ACTIVATED in every set field the shim copies through and in `capitalization_exceptions` keys — measured 2026-10-02 against 1.4.0 (run outside the worktree, `PYTHONSAFEPATH=1`): `titles ' dean '` gave title `dean` on `dean john smith` where 1.4.0 gave first `dean`, and likewise `prefixes`, `suffix_acronyms`, `suffix_not_acronyms`, `conjunctions`, `bound_first_names` and a `' zzc '` key; the raise (`suffix_not_acronyms 'ma '`) and the activation (`titles ' dean '`) both reproduce on the 2.0.0, 2.2.0 and 2.3.0 wheels. DECIDED (Derek): `Constants._snapshot()` drops every entry `_config_shim._v1_matchable` rejects — empty after `lc()`, or not equal to its own single-spaced re-join — from all nine set fields and the keys, BEFORE its set algebra, and names them in one `UserWarning` whose remedy runs (`tests/v2/test_config_shim.py::test_the_offered_remedy_runs_and_silences_the_warning`). The test is exact because a v1 piece comes from a whitespace split re-joined only with single spaces; it generalizes the filter `given_name_titles` already had, and `tests/v2/test_config_shim.py::_UNFILTERED_OUTCOME` records what each swept row does with it off. One exception, decided when the 1.4 pickle test caught it: every release from at least 0.5.8 through 1.4.0 SHIPPED two such entries in `TITLES` (`'actor '`, `'television '`), so a restored 1.4 pickle carries them and its user never wrote them; they are dropped WITHOUT the warning (`_V14_SHIPPED_UNMATCHABLE`, held to the pickle's own unmatchable set by `test_the_1_4_shipped_unmatchable_roster_is_exactly_the_pickles`), keeping that pickle warning-free. The exemption is keyed by entry, not by provenance, so a user who writes `'actor '` or `'television '` themselves is not warned either — accepted, since the two strings are 1.x's own typos. The drop is a reading change for that pickle: `Actor John Smith` gave title `Actor` on 2.0.0 through 2.3.0 and gives first `Actor` now, as 1.4.0 did. DECLINED: stripping whitespace in `SetManager` (activates every v1-inert entry above), and a silent drop for user entries (Derek: the entry is certainly a typo and certainly dead, so say so). ACCEPTED WIDENING, deliberately left: a key written with capitals or edge periods (`'McDonald'`, `'phd.'`) never matched on 1.4.0, whose lookup is `lc(word)` against keys stored as written, but has matched since 2.0 through `Lexicon`'s fold; dropping it would break a working 2.x config to restore an inertness nobody wanted. Also left: a set entry `Lexicon` folds to empty that `lc()` does not, a lone non-ASCII full stop (`'。'`), still raises at the first parse in every field `Lexicon` receives directly — `titles`, `prefixes`, `suffix_acronyms`, `suffix_not_acronyms`, `conjunctions`, `bound_first_names` and a `capitalization_exceptions` key; `first_name_titles` drops it quietly (`if t`), and `non_first_name_prefixes` and `suffix_acronyms_ambiguous` never pass it to `Lexicon` at all — v1-matchable, so outside this rule. - 2026-10-02 #582 — SUPERSEDES the #541 bullet's "Also left" sentence above. DECIDED (Derek): an entry `Lexicon._normalize` folds to empty is dropped by `_config_shim._v1_matchable` with the #541 warning, in every set field and key. The case is a lone CJK full stop (`'。'`, `'.'`, `'。'`, or a run of them), which 2.3.0's `FULL_STOPS` fold (#322/#323) empties while v1's `lc()` strips only `.`. Measured 2026-10-02 on released wheels run outside the checkout with `PYTHONSAFEPATH=1`: 1.4.0 through 2.2.0 accept `c.titles.add('。')` and read `。 john smith` with title `。`, while 2.3.0 raises `ValueError` at the first parse, as did this tree before the change, in `titles`, `prefixes`, `suffix_acronyms`, `suffix_not_acronyms`, `conjunctions`, `bound_first_names` and a key. ACCEPTED DEPARTURE from 1.4.0, the only one this filter makes: such an entry used to act on a token that is nothing but that full stop. Since 2.3.0 `Lexicon` can neither hold the entry nor match the token, its lookup fold emptying both, so reproducing v1 is not on offer and the choice was drop or raise. The warning's wording moved with it, from "nameparser 1.x never matched" to "match no name word". `given_name_titles`' `if t` filter was deleted as unreachable: `_title_key` is empty only when every word folds away, and then the whole entry folds to empty and was already dropped. With no shim-reachable plain `ValueError` left from `_normpairs`, `test_an_unrelated_capitalization_exceptions_valueerror_has_no_v1_hint` turns the filter off to keep the except-by-type clause pinned. THE SAME FOLD, ONE CHARACTER IN: an entry with a CJK full stop at its EDGE (not only full stops) passed the filter, and the set algebra compared it raw while `Lexicon` folded it, so `c.suffix_not_acronyms.add('ma。')` (beside the ambiguous `ma`) raised the gate-bypass check and a bound `'zed。'` beside a never-given particle `zed` raised the contradiction check, both from 2.3.0 on; measured accepted on 1.4.0 and 2.2.0 by the review of this change. `_build_snapshot` now compares `_normalize`d spellings in exactly the two computations `Lexicon` re-checks after its own fold — dropping a `suffix_not_acronyms` entry whose fold is an ambiguous acronym's, and taking into `particles_ambiguous` every particle whose fold is a bound given name's — so such an entry reads exactly as its folded spelling (`test_an_edge_full_stop_collision_reads_as_its_folded_spelling`; the fold-off control is `_UNFOLDED_OUTCOME`). That reading is the `dean。` widening below, not v1's: 1.4.0 matched `ma。` only on a `ma。` token. DECLINED, after being written and reviewed: folding EVERY kept entry before the set algebra. `non_first_name_prefixes` and `suffix_acronyms_ambiguous` reach `Lexicon` only through raw set algebra, so folding them is a NEW match, not the 2.3.0 one: `prefixes zz` with `non_first_name_prefixes 'zz。'` moved `zz smith` from first `zz` to last `zz smith`, an NFD/NFC pair across those fields moved the same way, and `suffix_acronyms_ambiguous 'zq。'` beside `suffix_acronyms zq` moved `john zq` from suffix to last — the particle readings the same on 1.4.0, 2.2.0, 2.3.0 and the pre-fold tree, the `zq` one on 2.2.0, 2.3.0 and the pre-fold tree (1.4.0 gives last `zq` there by its own two-word reading, not by matching the entry). They are pinned at those readings by `test_an_edge_full_stop_entry_outside_the_two_checks_reads_as_before`. ACCEPTED WIDENING, left as it is (Derek): since 2.3.0 a `titles` entry `'dean。'` also matches a bare `dean`, where 1.4.0 matched only `dean。`. The same fold makes it, and dropping the entry would lose the `dean。` match too, so it sits beside the capitalized-key widening above. -- 2026-10-02 #542 — A DECOMPOSED WORD IS REPAIRED AS ITS COMPOSED TWIN, by reading a combining mark (Unicode category M) as part of the letter before it, in every clause of case repair, and never by composing the text. A name typed NFD (macOS file names, some databases) had been split at each mark by `_WORD`'s `(\w|\.)+`, `\w` matching no mark, so `josé garcía` repaired to `José GarcíA` on every release from 1.4.0. Python's `re` has no `\p{M}`, and a hand-written class of the combining blocks would be a copy of Unicode data that drifts (it would already miss Adlam, a cased script whose marks sit outside them), so the test is `unicodedata.category(ch)[0] == "M"`, asked by `_render._sub_words` after each `_WORD` match and by `_render._beside` for a letter's neighbours. DECLINED: NFC-composing before repair, the issue's option 2. It fixes the split but hands back text in a form the writer did not use, which is a change beyond case under rules.md#R4 — `tests/v2/test_properties.py::test_case_repair_changes_case_and_nothing_else` compares with `casefold()`, which does not normalize, and would report it. THE ISSUE'S CLAIM THAT THE MASK NEEDED NO CHANGE WAS MEASURED FALSE, in two shapes it did not try: `_apply_mask`'s split-off-initial override and `_letter_run_ge2` both read a letter's neighbour by raw index, which in NFD is the mark rather than the full stop or the letter past it. Under a caller's `('éx', 'éx')` mask, `john smith é.x.` forced gave `É.X.` composed and `é.X.` decomposed; under `('éex', 'éeX')`, `john smith ée.x` gave `ée.X` against `éE.X`. Both read past marks now. The Mac/Mc clause had the same split one level down, `_MAC`'s `\w{2,}` failing at the mark (`macée` gave `MacÉe` composed and `Macée` decomposed), and is now DECIDED on the composed spelling and applied to the word as written. AMENDS the 2026-09-24 branch-review bullet (a): its SPLITTING half is retired. `parse("ǰo smith").capitalized(force=True)` gives `J̌o`, and a second forced pass now gives `J̌o` back rather than `J̌O`, the mark staying with its letter (measured 2026-10-02); the LENGTHENING half (`ß`, `ʼn`) stands. A mark with no letter before it heads no word: `str.capitalize()` on a word beginning with one would upper-case the mark and leave the letter after it lowercase. Out of reach and left there: unspaced hangul, whose NFD parse differs from its NFC parse because segmentation matches the census list as written (docs/usage.rst, "Decomposed text"), so repair follows a different parse; hangul jamo are letters, not marks, and nothing here touches them. VERIFICATION: the differential gate cannot see this — `capitalized()` is not a compared surface (this entry, 2026-08-29) and no corpus name is written NFD. Standing in its place: rules.md#R4's two NFD example lines, `tests/v2/test_properties.py::test_a_decomposed_name_repairs_as_its_composed_twin`, which decomposes every corpus and case-table text that decomposes, compares its repair with the composed text's on both surfaces plain and forced, and records as its control 144 disagreements over 137 texts with 97af1f02's `_render.py` over this change's corpus (measured 2026-10-02), and `tests/v2/test_render.py::test_a_decomposed_word_repairs_as_its_composed_twin` for the shapes no corpus name reaches, whose two mask rows were mutation-checked by restoring raw-index neighbours. +- 2026-10-02 #542 — A DECOMPOSED WORD IS REPAIRED AS ITS COMPOSED TWIN, by reading a combining mark (Unicode category M) as part of the letter before it, in every clause of case repair, and never by composing the text. A name typed NFD (macOS file names, some databases) had been split at each mark by `_WORD`'s `(\w|\.)+`, `\w` matching no mark, so `josé garcía` repaired to `José GarcíA` on every release from 1.4.0. Python's `re` has no `\p{M}`, and a hand-written class of the combining blocks would be a copy of Unicode data that drifts (it would already miss Adlam, a cased script whose marks sit outside them), so the test is `unicodedata.category(ch)[0] == "M"`, asked by `_render._sub_words` after each `_WORD` match and by `_render._beside` for a letter's neighbours. DECLINED: NFC-composing before repair, the issue's option 2. It fixes the split but hands back text in a form the writer did not use, which is a change beyond case under rules.md#R4 — `tests/v2/test_properties.py::test_case_repair_changes_case_and_nothing_else` compares with `casefold()`, which does not normalize, and would report it. THE ISSUE'S CLAIM THAT THE MASK NEEDED NO CHANGE ONCE THE WORD REACHED IT WHOLE WAS MEASURED FALSE, and so was the same assumption about three other clauses; each was found by fixing the splitter alone and comparing again. `_apply_mask`'s split-off-initial override and `_letter_run_ge2` read a letter's neighbour by raw index, which in NFD is the mark rather than the full stop or the letter past it: with the splitter alone, under a caller's `('éx', 'éx')` mask `john smith é.x.` forced gave `É.X.` composed and `é.X.` decomposed — a shape that AGREED before the fix (`É.X.` both ways) only because the old splitter cut the word at its mark before the mask was looked up — and under `('éex', 'éeX')` `john smith ée.x` gave `ée.X` against `éE.X` (`ÉE.X` before the fix). Both read past marks now. The Mac/Mc clause's `\w{2,}` fails at the mark, so with the splitter alone `macée` gave `Macée` decomposed against `MacÉe` composed (`MacéE` before the fix, the split having made the last `e` a word of its own); the clause is now DECIDED on the composed spelling and applied to the word as written. The two initial-shape tests, `_DOTTED_INITIAL` in the hyphen clause and `_INITIAL` in the unclassified-text fallback, matched the raw word and now match its composed spelling, so a decomposed `й.` — `й` is the one default conjunction that decomposes — is the initial its composed twin is: `ivan petrov-й.-sidorov` repaired to `Ivan Petrov-й.-Sidorov` decomposed against `Ivan Petrov-Й.-Sidorov` composed, and a middle spliced in as decomposed `й.` kept `й.` against `Й.`, at 97af1f02 and at this change's first commit alike (found by the docs review, 2026-10-02). AMENDS the 2026-09-24 branch-review bullet (a): its SPLITTING half is retired. `parse("ǰo smith").capitalized(force=True)` gives `J̌o`, and a second forced pass now gives `J̌o` back rather than `J̌O`, the mark staying with its letter (measured 2026-10-02); the LENGTHENING half (`ß`, `ʼn`) stands. A mark with no letter before it heads no word: `str.capitalize()` on a word beginning with one would upper-case the mark and leave the letter after it lowercase. Out of reach and left there, two of them: `initials()`, a rules.md#R3 view rather than case repair, takes a token's first character, which for a decomposed initial letter is the base letter without its accent (`parse("émile zola")` initials `é. z.` composed and `e. z.` decomposed, measured 2026-10-02); and unspaced hangul, whose NFD parse differs from its NFC parse because segmentation matches the census list as written (docs/usage.rst, "Decomposed text"), so repair follows a different parse; hangul jamo are letters, not marks, and nothing here touches them. VERIFICATION: the differential gate cannot see this — `capitalized()` is not a compared surface (this entry, 2026-08-29); the two NFD rows this change adds to the rules corpus reach the gate as parses only. Standing in its place: rules.md#R4's two NFD example lines, `tests/v2/test_properties.py::test_a_decomposed_name_repairs_as_its_composed_twin`, which decomposes every corpus and case-table text that decomposes, compares its repair with the composed text's on both surfaces plain and forced, and records as its control 144 disagreements over 137 texts with 97af1f02's `_render.py` over this change's corpus (measured 2026-10-02), and `tests/v2/test_render.py::test_a_decomposed_word_repairs_as_its_composed_twin` and `test_a_spliced_decomposed_initial_reads_as_its_composed_twin` for the shapes no corpus name reaches, whose two mask rows were mutation-checked by restoring raw-index neighbours. Excluded (CAPITALIZATION_EXCEPTIONS — meng, edd, lac, ded, left out of the 2026-09-24 masks, #459): diff --git a/nameparser/_lexicon.py b/nameparser/_lexicon.py index 0d6ace01..518d0398 100644 --- a/nameparser/_lexicon.py +++ b/nameparser/_lexicon.py @@ -482,9 +482,11 @@ def _normpairs( f"nothing)" ) # Stored NFC-composed, case kept (unicodedata, not _normalize): - # _apply_mask reads the mask one character at a time, and a - # decomposed letter would read as a base letter split off - # beside a non-alpha combining mark. The mask check below + # one stored spelling per value, whichever form the caller + # wrote. _apply_mask reads both forms alike since #542 + # (_render._beside looks past combining marks); before it, a + # decomposed letter read as a base letter split off beside a + # non-alpha combining mark. The mask check below # passes either spelling, _normalize composing both of its # sides. `written` keeps the caller's own spelling -- composed # or not -- for the mismatch error below: a decomposed value diff --git a/nameparser/_render.py b/nameparser/_render.py index 140c7098..14b6c36f 100644 --- a/nameparser/_render.py +++ b/nameparser/_render.py @@ -129,8 +129,10 @@ def _reads_as_conjunction(word: str, lex: Lexicon) -> bool: NAME -- rules.md#P3's one-case fork is the live example -- which is why it is the fallback and the tags are the rule. """ + # the shape on the composed spelling, as _normalize composes the + # lookup: a decomposed 'й.' is the initial its composed twin is (#542) return bool(_normalize(word) in lex.conjunctions - and not _INITIAL.fullmatch(word)) + and not _INITIAL.fullmatch(unicodedata.normalize("NFC", word))) def _collapse(rendered: str) -> str: @@ -540,7 +542,9 @@ def cap(word: str) -> str: first, last = named[0], named[-1] return "-".join( part.lower() - if (first < at < last and not _DOTTED_INITIAL.fullmatch(part) + if (first < at < last + and not _DOTTED_INITIAL.fullmatch( + unicodedata.normalize("NFC", part)) and _normalize(part) in lex.conjunctions) else _sub_words(cap, part) for at, part in enumerate(parts)) diff --git a/tests/v2/test_render.py b/tests/v2/test_render.py index 8a51f2bb..ea878fcb 100644 --- a/tests/v2/test_render.py +++ b/tests/v2/test_render.py @@ -930,6 +930,9 @@ def _nfc_texts(name: ParsedName) -> list[str]: ((("éx", "éx"),), "john smith é.x."), # a letter after a mark still has the letter before it as neighbour ((("éex", "éeX"),), "john smith ée.x"), + # the hyphen clause's initial test, on the composed spelling: 'й' + # is a default conjunction, and 'й.' between hyphens an initial + ((), "ivan petrov-й.-sidorov"), ]) def test_a_decomposed_word_repairs_as_its_composed_twin( pairs: tuple[tuple[str, str], ...], text: str) -> None: @@ -938,7 +941,10 @@ def test_a_decomposed_word_repairs_as_its_composed_twin( composed one -- the word splitter, the Mac/Mc clause, and the mask's neighbour tests -- and the output keeps the decomposed form. These are the shapes the corpus walk in test_properties.py does not - reach; each disagreed before the fix.""" + reach. The two mask rows guard _beside: 'é.x.' agreed before the + fix only because the old splitter cut the word at its mark before + the mask was consulted, and both rows disagree with _beside reading + raw neighbours.""" composed = unicodedata.normalize("NFC", text) decomposed = unicodedata.normalize("NFD", text) assert decomposed != composed @@ -949,6 +955,17 @@ def test_a_decomposed_word_repairs_as_its_composed_twin( for t in got.tokens) +def test_a_spliced_decomposed_initial_reads_as_its_composed_twin() -> None: + """#542, the unclassified-text fallback (_reads_as_conjunction): + a middle spliced in as decomposed 'й.' is the initial its composed + spelling is, not the conjunction 'й'.""" + for form in ("NFC", "NFD"): + name = parse("ivan petrov").replace( + middle=unicodedata.normalize(form, "й.")) + assert unicodedata.normalize( + "NFC", name.capitalized(force=True).middle) == "Й." + + def test_a_letter_that_uppercases_to_a_combining_mark_is_a_fixpoint( ) -> None: """'ǰ' upper-cases to 'J' + a combining caron. Before #542 the From 8eabc2f4323b101bcf9e737365366b2c4ab70edd Mon Sep 17 00:00:00 2001 From: Derek Gulbranson Date: Fri, 2 Oct 2026 15:24:28 -0700 Subject: [PATCH 4/5] docs(#542): describe both mark mechanisms and what the gate compares Second docs review: the R4 bullet and capitalized()'s docstring said every clause reads a mark as part of its letter, while the three shape clauses ask about the composed spelling instead; both now name the two mechanisms and the one case where they differ. The clause count and the gate's compared surfaces are corrected. Co-Authored-By: Claude Opus 5.5 --- docs/design/decisions.md | 2 +- nameparser/_render.py | 7 ++++--- 2 files changed, 5 insertions(+), 4 deletions(-) diff --git a/docs/design/decisions.md b/docs/design/decisions.md index f3685fe7..43076f83 100644 --- a/docs/design/decisions.md +++ b/docs/design/decisions.md @@ -1442,7 +1442,7 @@ Accepted costs, deferred to the rescoped #459 rather than relitigated here: the - 2026-09-24 #478 — DECIDED (Derek, 2026-09-24): the hyphen clause keeps its reading in a name written wholly in one case, and the case where that disagrees with rules.md#P3 is a recorded boundary, not a change. The two collide on a marked letter. Spaced, a letter the vocabulary marks as reading both ways reads as an initial in a one-case name (P3), so `maria silva e sousa` repairs to `Maria Silva E Sousa`; hyphenated, the interior part reads as the connective whatever the name's case, so `maria silva-e-sousa` repairs to `Maria Silva-e-Sousa`. Derek's reasoning: hyphenating `Silva-e-Sousa` is the writer joining the surname on purpose, so the interior word is a connective by that act, whatever the one-case rule would say of the spaced form. THE COST, measured 2026-09-24: a one-case name whose hyphenated bare initials happen to spell a connective — `J-E-P DUPONT` gives `J-e-P Dupont`, where every release 1.4.0 through 2.3.0 and the parent 4d0680e6 gave `J-E-P Dupont`, and `JOHN A-Y-B SMITH` gives `John A-y-B Smith` (every release: `John A-Y-B Smith`). The period-marked spelling is unaffected (`j.-e.-p. dupont` keeps `J.-E.-P. Dupont`, the 2026-09-23 #478 bullet). `conjunctions_ambiguous`, P3's knob, does NOT reach a hyphenated word: under `Lexicon.default().add(conjunctions_ambiguous={"y"})` the spaced `JOSE ORTEGA Y GASSET` repairs to `Jose Ortega Y Gasset` (default lexicon: `Jose Ortega y Gasset`) while `JOSE ORTEGA-Y-GASSET` stays `Jose Ortega-y-Gasset` under both. rules.md#R4's hyphen sentence said the hyphens join the name "as the spaced connective would", which claimed an equality the one-case case breaks; it now says the hyphens are the writer's join and states the split, with an Accepted paragraph and the `J-E-P DUPONT` boundary line. Pinned by `tests/v2/test_render.py::test_the_hyphen_is_the_writers_join_even_in_a_one_case_name`. - 2026-10-02 #541 — THE "LEFT FOR THE ORCHESTRATOR TO FILE" FINDING of the 2026-09-24 branch-review bullet (d) above is #541, and it was wider than that bullet saw: besides the two raises it names, an entry carrying whitespace v1 never matched SILENTLY ACTIVATED in every set field the shim copies through and in `capitalization_exceptions` keys — measured 2026-10-02 against 1.4.0 (run outside the worktree, `PYTHONSAFEPATH=1`): `titles ' dean '` gave title `dean` on `dean john smith` where 1.4.0 gave first `dean`, and likewise `prefixes`, `suffix_acronyms`, `suffix_not_acronyms`, `conjunctions`, `bound_first_names` and a `' zzc '` key; the raise (`suffix_not_acronyms 'ma '`) and the activation (`titles ' dean '`) both reproduce on the 2.0.0, 2.2.0 and 2.3.0 wheels. DECIDED (Derek): `Constants._snapshot()` drops every entry `_config_shim._v1_matchable` rejects — empty after `lc()`, or not equal to its own single-spaced re-join — from all nine set fields and the keys, BEFORE its set algebra, and names them in one `UserWarning` whose remedy runs (`tests/v2/test_config_shim.py::test_the_offered_remedy_runs_and_silences_the_warning`). The test is exact because a v1 piece comes from a whitespace split re-joined only with single spaces; it generalizes the filter `given_name_titles` already had, and `tests/v2/test_config_shim.py::_UNFILTERED_OUTCOME` records what each swept row does with it off. One exception, decided when the 1.4 pickle test caught it: every release from at least 0.5.8 through 1.4.0 SHIPPED two such entries in `TITLES` (`'actor '`, `'television '`), so a restored 1.4 pickle carries them and its user never wrote them; they are dropped WITHOUT the warning (`_V14_SHIPPED_UNMATCHABLE`, held to the pickle's own unmatchable set by `test_the_1_4_shipped_unmatchable_roster_is_exactly_the_pickles`), keeping that pickle warning-free. The exemption is keyed by entry, not by provenance, so a user who writes `'actor '` or `'television '` themselves is not warned either — accepted, since the two strings are 1.x's own typos. The drop is a reading change for that pickle: `Actor John Smith` gave title `Actor` on 2.0.0 through 2.3.0 and gives first `Actor` now, as 1.4.0 did. DECLINED: stripping whitespace in `SetManager` (activates every v1-inert entry above), and a silent drop for user entries (Derek: the entry is certainly a typo and certainly dead, so say so). ACCEPTED WIDENING, deliberately left: a key written with capitals or edge periods (`'McDonald'`, `'phd.'`) never matched on 1.4.0, whose lookup is `lc(word)` against keys stored as written, but has matched since 2.0 through `Lexicon`'s fold; dropping it would break a working 2.x config to restore an inertness nobody wanted. Also left: a set entry `Lexicon` folds to empty that `lc()` does not, a lone non-ASCII full stop (`'。'`), still raises at the first parse in every field `Lexicon` receives directly — `titles`, `prefixes`, `suffix_acronyms`, `suffix_not_acronyms`, `conjunctions`, `bound_first_names` and a `capitalization_exceptions` key; `first_name_titles` drops it quietly (`if t`), and `non_first_name_prefixes` and `suffix_acronyms_ambiguous` never pass it to `Lexicon` at all — v1-matchable, so outside this rule. - 2026-10-02 #582 — SUPERSEDES the #541 bullet's "Also left" sentence above. DECIDED (Derek): an entry `Lexicon._normalize` folds to empty is dropped by `_config_shim._v1_matchable` with the #541 warning, in every set field and key. The case is a lone CJK full stop (`'。'`, `'.'`, `'。'`, or a run of them), which 2.3.0's `FULL_STOPS` fold (#322/#323) empties while v1's `lc()` strips only `.`. Measured 2026-10-02 on released wheels run outside the checkout with `PYTHONSAFEPATH=1`: 1.4.0 through 2.2.0 accept `c.titles.add('。')` and read `。 john smith` with title `。`, while 2.3.0 raises `ValueError` at the first parse, as did this tree before the change, in `titles`, `prefixes`, `suffix_acronyms`, `suffix_not_acronyms`, `conjunctions`, `bound_first_names` and a key. ACCEPTED DEPARTURE from 1.4.0, the only one this filter makes: such an entry used to act on a token that is nothing but that full stop. Since 2.3.0 `Lexicon` can neither hold the entry nor match the token, its lookup fold emptying both, so reproducing v1 is not on offer and the choice was drop or raise. The warning's wording moved with it, from "nameparser 1.x never matched" to "match no name word". `given_name_titles`' `if t` filter was deleted as unreachable: `_title_key` is empty only when every word folds away, and then the whole entry folds to empty and was already dropped. With no shim-reachable plain `ValueError` left from `_normpairs`, `test_an_unrelated_capitalization_exceptions_valueerror_has_no_v1_hint` turns the filter off to keep the except-by-type clause pinned. THE SAME FOLD, ONE CHARACTER IN: an entry with a CJK full stop at its EDGE (not only full stops) passed the filter, and the set algebra compared it raw while `Lexicon` folded it, so `c.suffix_not_acronyms.add('ma。')` (beside the ambiguous `ma`) raised the gate-bypass check and a bound `'zed。'` beside a never-given particle `zed` raised the contradiction check, both from 2.3.0 on; measured accepted on 1.4.0 and 2.2.0 by the review of this change. `_build_snapshot` now compares `_normalize`d spellings in exactly the two computations `Lexicon` re-checks after its own fold — dropping a `suffix_not_acronyms` entry whose fold is an ambiguous acronym's, and taking into `particles_ambiguous` every particle whose fold is a bound given name's — so such an entry reads exactly as its folded spelling (`test_an_edge_full_stop_collision_reads_as_its_folded_spelling`; the fold-off control is `_UNFOLDED_OUTCOME`). That reading is the `dean。` widening below, not v1's: 1.4.0 matched `ma。` only on a `ma。` token. DECLINED, after being written and reviewed: folding EVERY kept entry before the set algebra. `non_first_name_prefixes` and `suffix_acronyms_ambiguous` reach `Lexicon` only through raw set algebra, so folding them is a NEW match, not the 2.3.0 one: `prefixes zz` with `non_first_name_prefixes 'zz。'` moved `zz smith` from first `zz` to last `zz smith`, an NFD/NFC pair across those fields moved the same way, and `suffix_acronyms_ambiguous 'zq。'` beside `suffix_acronyms zq` moved `john zq` from suffix to last — the particle readings the same on 1.4.0, 2.2.0, 2.3.0 and the pre-fold tree, the `zq` one on 2.2.0, 2.3.0 and the pre-fold tree (1.4.0 gives last `zq` there by its own two-word reading, not by matching the entry). They are pinned at those readings by `test_an_edge_full_stop_entry_outside_the_two_checks_reads_as_before`. ACCEPTED WIDENING, left as it is (Derek): since 2.3.0 a `titles` entry `'dean。'` also matches a bare `dean`, where 1.4.0 matched only `dean。`. The same fold makes it, and dropping the entry would lose the `dean。` match too, so it sits beside the capitalized-key widening above. -- 2026-10-02 #542 — A DECOMPOSED WORD IS REPAIRED AS ITS COMPOSED TWIN, by reading a combining mark (Unicode category M) as part of the letter before it, in every clause of case repair, and never by composing the text. A name typed NFD (macOS file names, some databases) had been split at each mark by `_WORD`'s `(\w|\.)+`, `\w` matching no mark, so `josé garcía` repaired to `José GarcíA` on every release from 1.4.0. Python's `re` has no `\p{M}`, and a hand-written class of the combining blocks would be a copy of Unicode data that drifts (it would already miss Adlam, a cased script whose marks sit outside them), so the test is `unicodedata.category(ch)[0] == "M"`, asked by `_render._sub_words` after each `_WORD` match and by `_render._beside` for a letter's neighbours. DECLINED: NFC-composing before repair, the issue's option 2. It fixes the split but hands back text in a form the writer did not use, which is a change beyond case under rules.md#R4 — `tests/v2/test_properties.py::test_case_repair_changes_case_and_nothing_else` compares with `casefold()`, which does not normalize, and would report it. THE ISSUE'S CLAIM THAT THE MASK NEEDED NO CHANGE ONCE THE WORD REACHED IT WHOLE WAS MEASURED FALSE, and so was the same assumption about three other clauses; each was found by fixing the splitter alone and comparing again. `_apply_mask`'s split-off-initial override and `_letter_run_ge2` read a letter's neighbour by raw index, which in NFD is the mark rather than the full stop or the letter past it: with the splitter alone, under a caller's `('éx', 'éx')` mask `john smith é.x.` forced gave `É.X.` composed and `é.X.` decomposed — a shape that AGREED before the fix (`É.X.` both ways) only because the old splitter cut the word at its mark before the mask was looked up — and under `('éex', 'éeX')` `john smith ée.x` gave `ée.X` against `éE.X` (`ÉE.X` before the fix). Both read past marks now. The Mac/Mc clause's `\w{2,}` fails at the mark, so with the splitter alone `macée` gave `Macée` decomposed against `MacÉe` composed (`MacéE` before the fix, the split having made the last `e` a word of its own); the clause is now DECIDED on the composed spelling and applied to the word as written. The two initial-shape tests, `_DOTTED_INITIAL` in the hyphen clause and `_INITIAL` in the unclassified-text fallback, matched the raw word and now match its composed spelling, so a decomposed `й.` — `й` is the one default conjunction that decomposes — is the initial its composed twin is: `ivan petrov-й.-sidorov` repaired to `Ivan Petrov-й.-Sidorov` decomposed against `Ivan Petrov-Й.-Sidorov` composed, and a middle spliced in as decomposed `й.` kept `й.` against `Й.`, at 97af1f02 and at this change's first commit alike (found by the docs review, 2026-10-02). AMENDS the 2026-09-24 branch-review bullet (a): its SPLITTING half is retired. `parse("ǰo smith").capitalized(force=True)` gives `J̌o`, and a second forced pass now gives `J̌o` back rather than `J̌O`, the mark staying with its letter (measured 2026-10-02); the LENGTHENING half (`ß`, `ʼn`) stands. A mark with no letter before it heads no word: `str.capitalize()` on a word beginning with one would upper-case the mark and leave the letter after it lowercase. Out of reach and left there, two of them: `initials()`, a rules.md#R3 view rather than case repair, takes a token's first character, which for a decomposed initial letter is the base letter without its accent (`parse("émile zola")` initials `é. z.` composed and `e. z.` decomposed, measured 2026-10-02); and unspaced hangul, whose NFD parse differs from its NFC parse because segmentation matches the census list as written (docs/usage.rst, "Decomposed text"), so repair follows a different parse; hangul jamo are letters, not marks, and nothing here touches them. VERIFICATION: the differential gate cannot see this — `capitalized()` is not a compared surface (this entry, 2026-08-29); the two NFD rows this change adds to the rules corpus reach the gate as parses only. Standing in its place: rules.md#R4's two NFD example lines, `tests/v2/test_properties.py::test_a_decomposed_name_repairs_as_its_composed_twin`, which decomposes every corpus and case-table text that decomposes, compares its repair with the composed text's on both surfaces plain and forced, and records as its control 144 disagreements over 137 texts with 97af1f02's `_render.py` over this change's corpus (measured 2026-10-02), and `tests/v2/test_render.py::test_a_decomposed_word_repairs_as_its_composed_twin` and `test_a_spliced_decomposed_initial_reads_as_its_composed_twin` for the shapes no corpus name reaches, whose two mask rows were mutation-checked by restoring raw-index neighbours. +- 2026-10-02 #542 — A DECOMPOSED WORD IS REPAIRED AS ITS COMPOSED TWIN, and the repaired text keeps the form it was typed in: no clause composes its OUTPUT. Two mechanisms do it. The word splitter and the mask's neighbour tests read a combining mark (Unicode category M) as part of the letter before it; the three clauses that ask a regex about a whole word's SHAPE — Mac/Mc and the two initial tests — ask it of the word's composed spelling. They differ only for a mark with no precomposed form, where composing leaves the word as written and so agrees with the parse: under a caller-added conjunction `q́`, `ivan q́. petrov` parses `q́.` as the conjunction, and the hyphen clause keeps `Petrov-q́.-Sidorov` lowercase as it does (measured by the docs review, 2026-10-02). A name typed NFD (macOS file names, some databases) had been split at each mark by `_WORD`'s `(\w|\.)+`, `\w` matching no mark, so `josé garcía` repaired to `José GarcíA` on every release from 1.4.0. Python's `re` has no `\p{M}`, and a hand-written class of the combining blocks would be a copy of Unicode data that drifts (it would already miss Adlam, a cased script whose marks sit outside them), so the test is `unicodedata.category(ch)[0] == "M"`, asked by `_render._sub_words` after each `_WORD` match and by `_render._beside` for a letter's neighbours. DECLINED: NFC-composing before repair, the issue's option 2. It fixes the split but hands back text in a form the writer did not use, which is a change beyond case under rules.md#R4 — `tests/v2/test_properties.py::test_case_repair_changes_case_and_nothing_else` compares with `casefold()`, which does not normalize, and would report it. THE ISSUE'S CLAIM THAT THE MASK NEEDED NO CHANGE ONCE THE WORD REACHED IT WHOLE WAS MEASURED FALSE, and the Mac/Mc clause needed one too; both were found by fixing the splitter alone and comparing again. `_apply_mask`'s split-off-initial override and `_letter_run_ge2` read a letter's neighbour by raw index, which in NFD is the mark rather than the full stop or the letter past it: with the splitter alone, under a caller's `('éx', 'éx')` mask `john smith é.x.` forced gave `É.X.` composed and `é.X.` decomposed — a shape that AGREED before the fix (`É.X.` both ways) only because the old splitter cut the word at its mark before the mask was looked up — and under `('éex', 'éeX')` `john smith ée.x` gave `ée.X` against `éE.X` (`ÉE.X` before the fix). Both read past marks now. The Mac/Mc clause's `\w{2,}` fails at the mark, so with the splitter alone `macée` gave `Macée` decomposed against `MacÉe` composed (`MacéE` before the fix, the split having made the last `e` a word of its own); the clause is now DECIDED on the composed spelling and applied to the word as written. The two initial-shape tests, `_DOTTED_INITIAL` in the hyphen clause and `_INITIAL` in the unclassified-text fallback, matched the raw word and now match its composed spelling, so a decomposed `й.` — `й` is the one default conjunction that decomposes — is the initial its composed twin is: `ivan petrov-й.-sidorov` repaired to `Ivan Petrov-й.-Sidorov` decomposed against `Ivan Petrov-Й.-Sidorov` composed, and a middle spliced in as decomposed `й.` kept `й.` against `Й.`, at 97af1f02 and at this change's first commit alike (found by the docs review, 2026-10-02). AMENDS the 2026-09-24 branch-review bullet (a): its SPLITTING half is retired. `parse("ǰo smith").capitalized(force=True)` gives `J̌o`, and a second forced pass now gives `J̌o` back rather than `J̌O`, the mark staying with its letter (measured 2026-10-02); the LENGTHENING half (`ß`, `ʼn`) stands. A mark with no letter before it heads no word: `str.capitalize()` on a word beginning with one would upper-case the mark and leave the letter after it lowercase. Out of reach and left there, two of them: `initials()`, a rules.md#R3 view rather than case repair, takes a token's first character, which for a decomposed initial letter is the base letter without its accent (`parse("émile zola")` initials `é. z.` composed and `e. z.` decomposed, measured 2026-10-02); and unspaced hangul, whose NFD parse differs from its NFC parse because segmentation matches the census list as written (docs/usage.rst, "Decomposed text"), so repair follows a different parse; hangul jamo are letters, not marks, and nothing here touches them. VERIFICATION: the differential gate cannot see this — `capitalized()` is not a compared surface (this entry, 2026-08-29); the two NFD rows this change adds to the rules corpus reach the gate through the role fields, `_ambiguities` and `_initials` only. Standing in its place: rules.md#R4's two NFD example lines, `tests/v2/test_properties.py::test_a_decomposed_name_repairs_as_its_composed_twin`, which decomposes every corpus and case-table text that decomposes, compares its repair with the composed text's on both surfaces plain and forced, and records as its control 144 disagreements over 137 texts with 97af1f02's `_render.py` over this change's corpus (measured 2026-10-02), and `tests/v2/test_render.py::test_a_decomposed_word_repairs_as_its_composed_twin` and `test_a_spliced_decomposed_initial_reads_as_its_composed_twin` for the shapes no corpus name reaches, whose two mask rows were mutation-checked by restoring raw-index neighbours. Excluded (CAPITALIZATION_EXCEPTIONS — meng, edd, lac, ded, left out of the 2026-09-24 masks, #459): diff --git a/nameparser/_render.py b/nameparser/_render.py index 14b6c36f..9c3d5330 100644 --- a/nameparser/_render.py +++ b/nameparser/_render.py @@ -609,9 +609,10 @@ def capitalized(name: ParsedName, lexicon: Lexicon | None, *, ('ǰ' to 'J' + a combining caron) is not such a case since #542: a word runs on through its marks (_sub_words), so 'J̌o' is one word on the second pass and comes back unchanged. - A name typed decomposed (NFD) repairs as its composed twin does, - every clause reading a mark as part of the letter before it, and - keeps the form it was typed in.""" + A name typed decomposed (NFD) repairs as its composed twin does + and keeps the form it was typed in: the splitter and the mask read + a mark as part of the letter before it, and the shape tests (Mac/Mc, + the initial tests) ask about the composed spelling.""" if lexicon is not None and not isinstance(lexicon, Lexicon): # eager, before the gate: a garbage argument must not become a # silent no-op on mixed-case input or a deep AttributeError From 81b04814a937eff56b2d129001e3267c5cb3e63a Mon Sep 17 00:00:00 2001 From: Derek Gulbranson Date: Fri, 2 Oct 2026 16:35:53 -0700 Subject: [PATCH 5/5] docs(#542): link the initials() follow-up, #585 Co-Authored-By: Claude Opus 5.5 --- docs/design/decisions.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/design/decisions.md b/docs/design/decisions.md index 43076f83..20bf129c 100644 --- a/docs/design/decisions.md +++ b/docs/design/decisions.md @@ -1442,7 +1442,7 @@ Accepted costs, deferred to the rescoped #459 rather than relitigated here: the - 2026-09-24 #478 — DECIDED (Derek, 2026-09-24): the hyphen clause keeps its reading in a name written wholly in one case, and the case where that disagrees with rules.md#P3 is a recorded boundary, not a change. The two collide on a marked letter. Spaced, a letter the vocabulary marks as reading both ways reads as an initial in a one-case name (P3), so `maria silva e sousa` repairs to `Maria Silva E Sousa`; hyphenated, the interior part reads as the connective whatever the name's case, so `maria silva-e-sousa` repairs to `Maria Silva-e-Sousa`. Derek's reasoning: hyphenating `Silva-e-Sousa` is the writer joining the surname on purpose, so the interior word is a connective by that act, whatever the one-case rule would say of the spaced form. THE COST, measured 2026-09-24: a one-case name whose hyphenated bare initials happen to spell a connective — `J-E-P DUPONT` gives `J-e-P Dupont`, where every release 1.4.0 through 2.3.0 and the parent 4d0680e6 gave `J-E-P Dupont`, and `JOHN A-Y-B SMITH` gives `John A-y-B Smith` (every release: `John A-Y-B Smith`). The period-marked spelling is unaffected (`j.-e.-p. dupont` keeps `J.-E.-P. Dupont`, the 2026-09-23 #478 bullet). `conjunctions_ambiguous`, P3's knob, does NOT reach a hyphenated word: under `Lexicon.default().add(conjunctions_ambiguous={"y"})` the spaced `JOSE ORTEGA Y GASSET` repairs to `Jose Ortega Y Gasset` (default lexicon: `Jose Ortega y Gasset`) while `JOSE ORTEGA-Y-GASSET` stays `Jose Ortega-y-Gasset` under both. rules.md#R4's hyphen sentence said the hyphens join the name "as the spaced connective would", which claimed an equality the one-case case breaks; it now says the hyphens are the writer's join and states the split, with an Accepted paragraph and the `J-E-P DUPONT` boundary line. Pinned by `tests/v2/test_render.py::test_the_hyphen_is_the_writers_join_even_in_a_one_case_name`. - 2026-10-02 #541 — THE "LEFT FOR THE ORCHESTRATOR TO FILE" FINDING of the 2026-09-24 branch-review bullet (d) above is #541, and it was wider than that bullet saw: besides the two raises it names, an entry carrying whitespace v1 never matched SILENTLY ACTIVATED in every set field the shim copies through and in `capitalization_exceptions` keys — measured 2026-10-02 against 1.4.0 (run outside the worktree, `PYTHONSAFEPATH=1`): `titles ' dean '` gave title `dean` on `dean john smith` where 1.4.0 gave first `dean`, and likewise `prefixes`, `suffix_acronyms`, `suffix_not_acronyms`, `conjunctions`, `bound_first_names` and a `' zzc '` key; the raise (`suffix_not_acronyms 'ma '`) and the activation (`titles ' dean '`) both reproduce on the 2.0.0, 2.2.0 and 2.3.0 wheels. DECIDED (Derek): `Constants._snapshot()` drops every entry `_config_shim._v1_matchable` rejects — empty after `lc()`, or not equal to its own single-spaced re-join — from all nine set fields and the keys, BEFORE its set algebra, and names them in one `UserWarning` whose remedy runs (`tests/v2/test_config_shim.py::test_the_offered_remedy_runs_and_silences_the_warning`). The test is exact because a v1 piece comes from a whitespace split re-joined only with single spaces; it generalizes the filter `given_name_titles` already had, and `tests/v2/test_config_shim.py::_UNFILTERED_OUTCOME` records what each swept row does with it off. One exception, decided when the 1.4 pickle test caught it: every release from at least 0.5.8 through 1.4.0 SHIPPED two such entries in `TITLES` (`'actor '`, `'television '`), so a restored 1.4 pickle carries them and its user never wrote them; they are dropped WITHOUT the warning (`_V14_SHIPPED_UNMATCHABLE`, held to the pickle's own unmatchable set by `test_the_1_4_shipped_unmatchable_roster_is_exactly_the_pickles`), keeping that pickle warning-free. The exemption is keyed by entry, not by provenance, so a user who writes `'actor '` or `'television '` themselves is not warned either — accepted, since the two strings are 1.x's own typos. The drop is a reading change for that pickle: `Actor John Smith` gave title `Actor` on 2.0.0 through 2.3.0 and gives first `Actor` now, as 1.4.0 did. DECLINED: stripping whitespace in `SetManager` (activates every v1-inert entry above), and a silent drop for user entries (Derek: the entry is certainly a typo and certainly dead, so say so). ACCEPTED WIDENING, deliberately left: a key written with capitals or edge periods (`'McDonald'`, `'phd.'`) never matched on 1.4.0, whose lookup is `lc(word)` against keys stored as written, but has matched since 2.0 through `Lexicon`'s fold; dropping it would break a working 2.x config to restore an inertness nobody wanted. Also left: a set entry `Lexicon` folds to empty that `lc()` does not, a lone non-ASCII full stop (`'。'`), still raises at the first parse in every field `Lexicon` receives directly — `titles`, `prefixes`, `suffix_acronyms`, `suffix_not_acronyms`, `conjunctions`, `bound_first_names` and a `capitalization_exceptions` key; `first_name_titles` drops it quietly (`if t`), and `non_first_name_prefixes` and `suffix_acronyms_ambiguous` never pass it to `Lexicon` at all — v1-matchable, so outside this rule. - 2026-10-02 #582 — SUPERSEDES the #541 bullet's "Also left" sentence above. DECIDED (Derek): an entry `Lexicon._normalize` folds to empty is dropped by `_config_shim._v1_matchable` with the #541 warning, in every set field and key. The case is a lone CJK full stop (`'。'`, `'.'`, `'。'`, or a run of them), which 2.3.0's `FULL_STOPS` fold (#322/#323) empties while v1's `lc()` strips only `.`. Measured 2026-10-02 on released wheels run outside the checkout with `PYTHONSAFEPATH=1`: 1.4.0 through 2.2.0 accept `c.titles.add('。')` and read `。 john smith` with title `。`, while 2.3.0 raises `ValueError` at the first parse, as did this tree before the change, in `titles`, `prefixes`, `suffix_acronyms`, `suffix_not_acronyms`, `conjunctions`, `bound_first_names` and a key. ACCEPTED DEPARTURE from 1.4.0, the only one this filter makes: such an entry used to act on a token that is nothing but that full stop. Since 2.3.0 `Lexicon` can neither hold the entry nor match the token, its lookup fold emptying both, so reproducing v1 is not on offer and the choice was drop or raise. The warning's wording moved with it, from "nameparser 1.x never matched" to "match no name word". `given_name_titles`' `if t` filter was deleted as unreachable: `_title_key` is empty only when every word folds away, and then the whole entry folds to empty and was already dropped. With no shim-reachable plain `ValueError` left from `_normpairs`, `test_an_unrelated_capitalization_exceptions_valueerror_has_no_v1_hint` turns the filter off to keep the except-by-type clause pinned. THE SAME FOLD, ONE CHARACTER IN: an entry with a CJK full stop at its EDGE (not only full stops) passed the filter, and the set algebra compared it raw while `Lexicon` folded it, so `c.suffix_not_acronyms.add('ma。')` (beside the ambiguous `ma`) raised the gate-bypass check and a bound `'zed。'` beside a never-given particle `zed` raised the contradiction check, both from 2.3.0 on; measured accepted on 1.4.0 and 2.2.0 by the review of this change. `_build_snapshot` now compares `_normalize`d spellings in exactly the two computations `Lexicon` re-checks after its own fold — dropping a `suffix_not_acronyms` entry whose fold is an ambiguous acronym's, and taking into `particles_ambiguous` every particle whose fold is a bound given name's — so such an entry reads exactly as its folded spelling (`test_an_edge_full_stop_collision_reads_as_its_folded_spelling`; the fold-off control is `_UNFOLDED_OUTCOME`). That reading is the `dean。` widening below, not v1's: 1.4.0 matched `ma。` only on a `ma。` token. DECLINED, after being written and reviewed: folding EVERY kept entry before the set algebra. `non_first_name_prefixes` and `suffix_acronyms_ambiguous` reach `Lexicon` only through raw set algebra, so folding them is a NEW match, not the 2.3.0 one: `prefixes zz` with `non_first_name_prefixes 'zz。'` moved `zz smith` from first `zz` to last `zz smith`, an NFD/NFC pair across those fields moved the same way, and `suffix_acronyms_ambiguous 'zq。'` beside `suffix_acronyms zq` moved `john zq` from suffix to last — the particle readings the same on 1.4.0, 2.2.0, 2.3.0 and the pre-fold tree, the `zq` one on 2.2.0, 2.3.0 and the pre-fold tree (1.4.0 gives last `zq` there by its own two-word reading, not by matching the entry). They are pinned at those readings by `test_an_edge_full_stop_entry_outside_the_two_checks_reads_as_before`. ACCEPTED WIDENING, left as it is (Derek): since 2.3.0 a `titles` entry `'dean。'` also matches a bare `dean`, where 1.4.0 matched only `dean。`. The same fold makes it, and dropping the entry would lose the `dean。` match too, so it sits beside the capitalized-key widening above. -- 2026-10-02 #542 — A DECOMPOSED WORD IS REPAIRED AS ITS COMPOSED TWIN, and the repaired text keeps the form it was typed in: no clause composes its OUTPUT. Two mechanisms do it. The word splitter and the mask's neighbour tests read a combining mark (Unicode category M) as part of the letter before it; the three clauses that ask a regex about a whole word's SHAPE — Mac/Mc and the two initial tests — ask it of the word's composed spelling. They differ only for a mark with no precomposed form, where composing leaves the word as written and so agrees with the parse: under a caller-added conjunction `q́`, `ivan q́. petrov` parses `q́.` as the conjunction, and the hyphen clause keeps `Petrov-q́.-Sidorov` lowercase as it does (measured by the docs review, 2026-10-02). A name typed NFD (macOS file names, some databases) had been split at each mark by `_WORD`'s `(\w|\.)+`, `\w` matching no mark, so `josé garcía` repaired to `José GarcíA` on every release from 1.4.0. Python's `re` has no `\p{M}`, and a hand-written class of the combining blocks would be a copy of Unicode data that drifts (it would already miss Adlam, a cased script whose marks sit outside them), so the test is `unicodedata.category(ch)[0] == "M"`, asked by `_render._sub_words` after each `_WORD` match and by `_render._beside` for a letter's neighbours. DECLINED: NFC-composing before repair, the issue's option 2. It fixes the split but hands back text in a form the writer did not use, which is a change beyond case under rules.md#R4 — `tests/v2/test_properties.py::test_case_repair_changes_case_and_nothing_else` compares with `casefold()`, which does not normalize, and would report it. THE ISSUE'S CLAIM THAT THE MASK NEEDED NO CHANGE ONCE THE WORD REACHED IT WHOLE WAS MEASURED FALSE, and the Mac/Mc clause needed one too; both were found by fixing the splitter alone and comparing again. `_apply_mask`'s split-off-initial override and `_letter_run_ge2` read a letter's neighbour by raw index, which in NFD is the mark rather than the full stop or the letter past it: with the splitter alone, under a caller's `('éx', 'éx')` mask `john smith é.x.` forced gave `É.X.` composed and `é.X.` decomposed — a shape that AGREED before the fix (`É.X.` both ways) only because the old splitter cut the word at its mark before the mask was looked up — and under `('éex', 'éeX')` `john smith ée.x` gave `ée.X` against `éE.X` (`ÉE.X` before the fix). Both read past marks now. The Mac/Mc clause's `\w{2,}` fails at the mark, so with the splitter alone `macée` gave `Macée` decomposed against `MacÉe` composed (`MacéE` before the fix, the split having made the last `e` a word of its own); the clause is now DECIDED on the composed spelling and applied to the word as written. The two initial-shape tests, `_DOTTED_INITIAL` in the hyphen clause and `_INITIAL` in the unclassified-text fallback, matched the raw word and now match its composed spelling, so a decomposed `й.` — `й` is the one default conjunction that decomposes — is the initial its composed twin is: `ivan petrov-й.-sidorov` repaired to `Ivan Petrov-й.-Sidorov` decomposed against `Ivan Petrov-Й.-Sidorov` composed, and a middle spliced in as decomposed `й.` kept `й.` against `Й.`, at 97af1f02 and at this change's first commit alike (found by the docs review, 2026-10-02). AMENDS the 2026-09-24 branch-review bullet (a): its SPLITTING half is retired. `parse("ǰo smith").capitalized(force=True)` gives `J̌o`, and a second forced pass now gives `J̌o` back rather than `J̌O`, the mark staying with its letter (measured 2026-10-02); the LENGTHENING half (`ß`, `ʼn`) stands. A mark with no letter before it heads no word: `str.capitalize()` on a word beginning with one would upper-case the mark and leave the letter after it lowercase. Out of reach and left there, two of them: `initials()`, a rules.md#R3 view rather than case repair, takes a token's first character, which for a decomposed initial letter is the base letter without its accent (`parse("émile zola")` initials `é. z.` composed and `e. z.` decomposed, measured 2026-10-02); and unspaced hangul, whose NFD parse differs from its NFC parse because segmentation matches the census list as written (docs/usage.rst, "Decomposed text"), so repair follows a different parse; hangul jamo are letters, not marks, and nothing here touches them. VERIFICATION: the differential gate cannot see this — `capitalized()` is not a compared surface (this entry, 2026-08-29); the two NFD rows this change adds to the rules corpus reach the gate through the role fields, `_ambiguities` and `_initials` only. Standing in its place: rules.md#R4's two NFD example lines, `tests/v2/test_properties.py::test_a_decomposed_name_repairs_as_its_composed_twin`, which decomposes every corpus and case-table text that decomposes, compares its repair with the composed text's on both surfaces plain and forced, and records as its control 144 disagreements over 137 texts with 97af1f02's `_render.py` over this change's corpus (measured 2026-10-02), and `tests/v2/test_render.py::test_a_decomposed_word_repairs_as_its_composed_twin` and `test_a_spliced_decomposed_initial_reads_as_its_composed_twin` for the shapes no corpus name reaches, whose two mask rows were mutation-checked by restoring raw-index neighbours. +- 2026-10-02 #542 — A DECOMPOSED WORD IS REPAIRED AS ITS COMPOSED TWIN, and the repaired text keeps the form it was typed in: no clause composes its OUTPUT. Two mechanisms do it. The word splitter and the mask's neighbour tests read a combining mark (Unicode category M) as part of the letter before it; the three clauses that ask a regex about a whole word's SHAPE — Mac/Mc and the two initial tests — ask it of the word's composed spelling. They differ only for a mark with no precomposed form, where composing leaves the word as written and so agrees with the parse: under a caller-added conjunction `q́`, `ivan q́. petrov` parses `q́.` as the conjunction, and the hyphen clause keeps `Petrov-q́.-Sidorov` lowercase as it does (measured by the docs review, 2026-10-02). A name typed NFD (macOS file names, some databases) had been split at each mark by `_WORD`'s `(\w|\.)+`, `\w` matching no mark, so `josé garcía` repaired to `José GarcíA` on every release from 1.4.0. Python's `re` has no `\p{M}`, and a hand-written class of the combining blocks would be a copy of Unicode data that drifts (it would already miss Adlam, a cased script whose marks sit outside them), so the test is `unicodedata.category(ch)[0] == "M"`, asked by `_render._sub_words` after each `_WORD` match and by `_render._beside` for a letter's neighbours. DECLINED: NFC-composing before repair, the issue's option 2. It fixes the split but hands back text in a form the writer did not use, which is a change beyond case under rules.md#R4 — `tests/v2/test_properties.py::test_case_repair_changes_case_and_nothing_else` compares with `casefold()`, which does not normalize, and would report it. THE ISSUE'S CLAIM THAT THE MASK NEEDED NO CHANGE ONCE THE WORD REACHED IT WHOLE WAS MEASURED FALSE, and the Mac/Mc clause needed one too; both were found by fixing the splitter alone and comparing again. `_apply_mask`'s split-off-initial override and `_letter_run_ge2` read a letter's neighbour by raw index, which in NFD is the mark rather than the full stop or the letter past it: with the splitter alone, under a caller's `('éx', 'éx')` mask `john smith é.x.` forced gave `É.X.` composed and `é.X.` decomposed — a shape that AGREED before the fix (`É.X.` both ways) only because the old splitter cut the word at its mark before the mask was looked up — and under `('éex', 'éeX')` `john smith ée.x` gave `ée.X` against `éE.X` (`ÉE.X` before the fix). Both read past marks now. The Mac/Mc clause's `\w{2,}` fails at the mark, so with the splitter alone `macée` gave `Macée` decomposed against `MacÉe` composed (`MacéE` before the fix, the split having made the last `e` a word of its own); the clause is now DECIDED on the composed spelling and applied to the word as written. The two initial-shape tests, `_DOTTED_INITIAL` in the hyphen clause and `_INITIAL` in the unclassified-text fallback, matched the raw word and now match its composed spelling, so a decomposed `й.` — `й` is the one default conjunction that decomposes — is the initial its composed twin is: `ivan petrov-й.-sidorov` repaired to `Ivan Petrov-й.-Sidorov` decomposed against `Ivan Petrov-Й.-Sidorov` composed, and a middle spliced in as decomposed `й.` kept `й.` against `Й.`, at 97af1f02 and at this change's first commit alike (found by the docs review, 2026-10-02). AMENDS the 2026-09-24 branch-review bullet (a): its SPLITTING half is retired. `parse("ǰo smith").capitalized(force=True)` gives `J̌o`, and a second forced pass now gives `J̌o` back rather than `J̌O`, the mark staying with its letter (measured 2026-10-02); the LENGTHENING half (`ß`, `ʼn`) stands. A mark with no letter before it heads no word: `str.capitalize()` on a word beginning with one would upper-case the mark and leave the letter after it lowercase. Out of reach and left there, two of them: `initials()`, a rules.md#R3 view rather than case repair, takes a token's first character, which for a decomposed initial letter is the base letter without its accent (`parse("émile zola")` initials `é. z.` composed and `e. z.` decomposed, measured 2026-10-02; open as #585); and unspaced hangul, whose NFD parse differs from its NFC parse because segmentation matches the census list as written (docs/usage.rst, "Decomposed text"), so repair follows a different parse; hangul jamo are letters, not marks, and nothing here touches them. VERIFICATION: the differential gate cannot see this — `capitalized()` is not a compared surface (this entry, 2026-08-29); the two NFD rows this change adds to the rules corpus reach the gate through the role fields, `_ambiguities` and `_initials` only. Standing in its place: rules.md#R4's two NFD example lines, `tests/v2/test_properties.py::test_a_decomposed_name_repairs_as_its_composed_twin`, which decomposes every corpus and case-table text that decomposes, compares its repair with the composed text's on both surfaces plain and forced, and records as its control 144 disagreements over 137 texts with 97af1f02's `_render.py` over this change's corpus (measured 2026-10-02), and `tests/v2/test_render.py::test_a_decomposed_word_repairs_as_its_composed_twin` and `test_a_spliced_decomposed_initial_reads_as_its_composed_twin` for the shapes no corpus name reaches, whose two mask rows were mutation-checked by restoring raw-index neighbours. Excluded (CAPITALIZATION_EXCEPTIONS — meng, edd, lac, ded, left out of the 2026-09-24 masks, #459):