From 076a28ade9c61acd76cfd3d67d976968a5dcb56c Mon Sep 17 00:00:00 2001 From: Derek Gulbranson Date: Sun, 27 Sep 2026 15:04:30 -0700 Subject: [PATCH 01/11] fix(#544): a listed member written in period-closed chunks passes S2's period gate S2's period gate counted an ambiguous acronym as unambiguous only when written with a period after each letter, so the chunked spelling of the two members #540 marked -- 'M.Eng.', 'L.Ac.' -- read as the bare word does and 'Wang M.Eng.' lost the suffix 2.0 through 2.3 read. _dotted now also accepts two or more letter chunks each closed by a period; one trailing period ('Ma.') stays outside the gate. is_single_letter_numeral is added for the run rule and the anchor that follow. rules.md#S2 states the gate's second spelling, with the two code excerpts that quote it. Co-Authored-By: Claude Opus 5.5 --- docs/design/rules.md | 7 +++--- nameparser/_pipeline/_classify.py | 5 +++-- nameparser/_pipeline/_pieces.py | 9 ++++---- nameparser/_pipeline/_vocab.py | 37 +++++++++++++++++++++++++------ tests/v2/cases.py | 29 ++++++++++++------------ tests/v2/pipeline/test_vocab.py | 30 +++++++++++++++++++++++++ tests/v2/test_regex_sync.py | 3 +++ 7 files changed, 89 insertions(+), 31 deletions(-) diff --git a/docs/design/rules.md b/docs/design/rules.md index 4b3a0e71..252fc795 100644 --- a/docs/design/rules.md +++ b/docs/design/rules.md @@ -958,8 +958,9 @@ S2. Rationale: generational suffixes and credentials are recognized A trailing word of the suffix vocabulary reads as a suffix — generational forms and credential acronyms alike, and an ambiguous acronym written with its periods, one after each - letter, counts unambiguously; a single trailing period is the - abbreviation shape any word can wear and does not. A + letter or one after each of two or more letter chunks, counts + unambiguously; a single trailing period is the abbreviation + shape any word can wear and does not. A BARE ambiguous acronym is consumed only when the name has words to spare — as the second of two words it stays the family name — and at the slots that report, either reading carries the @@ -1051,7 +1052,7 @@ S2. Rationale: generational suffixes and credentials are recognized "john smith meng" → suffix="meng" "john smith MEng" → family="MEng" "Nguyen Van Lac" → family="Van Lac" - "Wang M.Eng." → family="M.Eng." + "Wang M.Eng." → suffix="M.Eng." "Smith, MA" → suffix="MA" "Smith, Ma" → given="Ma" "Doe, John MA" → suffix="MA" diff --git a/nameparser/_pipeline/_classify.py b/nameparser/_pipeline/_classify.py index 8b02070c..76ae08a6 100644 --- a/nameparser/_pipeline/_classify.py +++ b/nameparser/_pipeline/_classify.py @@ -68,8 +68,9 @@ # rules.md#S2: "a trailing word of the suffix vocabulary reads as a # suffix — generational forms and credential acronyms alike, and an # ambiguous acronym written with its periods, one after each -# letter, counts unambiguously; a single trailing period is the -# abbreviation shape any word can wear and does not. A +# letter or one after each of two or more letter chunks, counts +# unambiguously; a single trailing period is the abbreviation shape +# any word can wear and does not. A # bare ambiguous acronym is consumed only when the name has words to # spare" def _tags_for(token: WorkToken, n: str, state: ParseState, diff --git a/nameparser/_pipeline/_pieces.py b/nameparser/_pipeline/_pieces.py index 6af347a9..d72d1a76 100644 --- a/nameparser/_pipeline/_pieces.py +++ b/nameparser/_pipeline/_pieces.py @@ -383,10 +383,11 @@ class Peel(NamedTuple): # rules.md#S2: "a trailing word of the suffix vocabulary reads as a # suffix — generational forms and credential acronyms alike, and an -# ambiguous acronym written with its periods, one after each letter, -# counts unambiguously; a single trailing period is the abbreviation -# shape any word can wear and does not. A BARE ambiguous acronym is -# consumed only when the name has words to spare" +# ambiguous acronym written with its periods, one after each letter or +# one after each of two or more letter chunks, counts unambiguously; a +# single trailing period is the abbreviation shape any word can wear +# and does not. A BARE ambiguous acronym is consumed only when the name +# has words to spare" # (v1's are_suffixes tail rule, with the roman-numeral special) def peel_walk(start: int, ptags: Sequence[Set[str]], skip: Set[int] = frozenset()) -> list[int]: diff --git a/nameparser/_pipeline/_vocab.py b/nameparser/_pipeline/_vocab.py index ec0035a7..021ae33a 100644 --- a/nameparser/_pipeline/_vocab.py +++ b/nameparser/_pipeline/_vocab.py @@ -207,6 +207,19 @@ def is_trailing_numeral_suffix(text: str, preceding: str) -> bool: and not is_initial_shaped(preceding)) +# #544: a single-letter roman numeral ('V', 'v', 'I.') is about a +# GENERATION, not a credential -- rules.md#S3 retires the same class +# from the chunk rule for the same reason -- so it neither anchors an +# ambiguous member behind it nor counts toward C1's multi-word +# credential run. It still peels exactly as before; this answers only +# those two questions. +def is_single_letter_numeral(text: str) -> bool: + """One letter, optionally followed by periods, that is a roman + numeral ('V', 'v', 'I.', 'X').""" + letters = text.rstrip(".") + return len(letters) == 1 and _ROMAN.match(letters) is not None + + def is_initial(text: str) -> bool: """'A.' / 'j.' / bare capital -- v1's is_an_initial, narrowed to scripts that HAVE initials (#320). v1's \\w is Unicode-aware and @@ -315,16 +328,25 @@ def ambiguous_lean(text: str, one_case: bool) -> Lean | None: _DOTTED = re.compile(r"(?:[^\W\d_]\.)+") +# #544: the CHUNKED spelling -- two or more runs of letters, each +# closed by a period ('M.Eng.', 'L.Ac.'). Both callers compare the +# period-free letters against the listed member, so the chunks must +# spell it; one chunk ('Ma.', 'Ed.') is the single trailing period and +# stays outside the gate. +_CHUNKED = re.compile(r"(?:[^\W\d_]+\.){2,}") def _dotted(text: str) -> bool: """Written with its periods: one after each letter ('M.A.', - 'J.D.'), the acronym's own spelling. A single trailing period - ('Ma.', 'Ed.', 'Ms.') is the abbreviation shape any word can wear - -- the honorific's, a name's -- and is not the gate's "written - with periods" (rules.md#S2). Until #296's review the gate was - "any period", and 'Smith, Ms.' passed it as the degree.""" - return _DOTTED.fullmatch(text) is not None + 'J.D.'), the acronym's own spelling, or -- since #544 -- two or + more letter chunks each closed by a period ('M.Eng.', 'L.Ac.'), the + spelling a member with a lower-case tail is written in. A single + trailing period ('Ma.', 'Ed.', 'Ms.') is the abbreviation shape any + word can wear -- the honorific's, a name's -- and is not the gate's + "written with periods" (rules.md#S2). Until #296's review the gate + was "any period", and 'Smith, Ms.' passed it as the degree.""" + return (_DOTTED.fullmatch(text) is not None + or _CHUNKED.fullmatch(text) is not None) def suffix_as_written(n: str, text: str, lexicon: Lexicon) -> bool: @@ -508,7 +530,8 @@ class at all). clause). The '.' gate here is DELIBERATELY STRICTER than S2's own - dotted-form test (`_dotted`, a period after EACH letter): '.' + dotted-form test (`_dotted`, a period after EACH letter or after + each of two or more letter chunks): '.' anywhere excludes membership, so a single TRAILING period ('MA.', 'Ed.') is excluded here even though the LEAN still reads it as the bare acronym's case ('Smith, MA.' -> suffix 'MA.', measured). diff --git a/tests/v2/cases.py b/tests/v2/cases.py index 16cc3657..2eb4edec 100644 --- a/tests/v2/cases.py +++ b/tests/v2/cases.py @@ -741,22 +741,21 @@ def _check_cjk_shape_purity(self) -> None: "every release read suffix 'MEng PhD' ('MEng, PhD' " "through 2.2)", shape=1), - Case("dotted_meng_with_nothing_to_spare_is_the_name", + Case("dotted_meng_passes_the_period_gate", "Wang M.Eng.", - {"given": "Wang", "family": "M.Eng."}, - ambiguities=("suffix-or-name",), - notes="S2's period gate counts a member as unambiguous " - "only written one period per letter ('M.A.'), and " - "MEng and LAc are the first members whose " - "conventional dotted spelling is chunked, so " - "'M.Eng.' with nothing to spare is the family name; " - "accepted and recorded (Derek, 2026-09-25). 1.4.0 " - "read the same family, unflagged, for a different " - "reason -- its two-piece rule (a lone word after the " - "given name is the family), not S2's gate -- so this " - "row is parity. 'John Smith M.Eng.' keeps the suffix " - "by the count and reports; the period-gate question " - "goes to #544", + {"given": "Wang", "suffix": "M.Eng."}, + classification="fix(#544)", + ambiguities=("given-or-family",), + notes="S2's period gate counts a listed member written in " + "two or more letter chunks, each closed by a period, " + "as it counts one written with a period after each " + "letter: 'M.Eng.' reads as 'M.A.' does, a suffix even " + "with nothing to spare, and the one-word name reports " + "the given-or-family fork 'Wang M.A.' reports. #540 " + "had accepted family 'M.Eng.' pending this question; " + "1.4.0 read the family too, by its two-piece rule, and " + "2.3.0 the suffix. A single trailing period ('Wang " + "Ma.') stays outside the gate", shape=1), Case("leading_meng_is_a_given_name", "meng li", {"given": "meng", "family": "li"}, diff --git a/tests/v2/pipeline/test_vocab.py b/tests/v2/pipeline/test_vocab.py index 053f0f10..17c5c156 100644 --- a/tests/v2/pipeline/test_vocab.py +++ b/tests/v2/pipeline/test_vocab.py @@ -10,6 +10,7 @@ ambiguous_class_candidate, ambiguous_class_member, ambiguous_lean, caps_shape_candidate, effective_script, is_initial, is_initial_shaped, is_one_case, + is_single_letter_numeral, is_suffix_lenient, is_suffix_strict, is_title_shaped, is_wholly_suffix, maiden_marker_run, name_word_count, period_joined_vocab, resolve_script_set, single_script, @@ -784,3 +785,32 @@ def test_ambiguous_lean_reads_the_written_case() -> None: # neither way even where the name around it is mixed assert ambiguous_lean("씨", one_case=False) is None assert ambiguous_lean("毛", one_case=False) is None + + +def test_a_listed_member_written_in_period_closed_chunks_is_dotted( +) -> None: + """#544: S2's period gate counts a listed member written in two or + more letter chunks, each closed by a period ('M.Eng.'), as it + counts one written with a period after each letter ('M.A.'). One + chunk is the single trailing period any word can wear and stays + outside the gate, and the chunks must still spell the member: the + gate reads the spelling, the membership test the letters.""" + lex = Lexicon( + suffix_acronyms=frozenset({"meng", "lac", "ma", "phd"}), + suffix_acronyms_ambiguous=frozenset({"meng", "lac", "ma"}), + ) + for text in ("M.Eng.", "m.eng.", "L.Ac.", "M.A."): + assert is_suffix_strict(text, lex), text + # one chunk, a missing closing period, the bare word, and chunks + # that spell nothing listed + for text in ("Meng.", "Ma.", "M.Eng", "MEng", "X.Eng."): + assert not is_suffix_strict(text, lex), text + + +def test_is_single_letter_numeral() -> None: + """#544: the generation class C1's run and the anchor both leave + out -- one letter, periods allowed, that is a roman numeral.""" + for text in ("V", "v", "I", "I.", "x", "X."): + assert is_single_letter_numeral(text), text + for text in ("II", "IV", "Jr", "B", "", ".", "Ma"): + assert not is_single_letter_numeral(text), text diff --git a/tests/v2/test_regex_sync.py b/tests/v2/test_regex_sync.py index bef90910..fd27f345 100644 --- a/tests/v2/test_regex_sync.py +++ b/tests/v2/test_regex_sync.py @@ -131,6 +131,9 @@ def test_dotted_initial_is_the_period_alternative_of_initial() -> None: ("_vocab", "_PERIOD_ABBREV"): "period_abbreviation", ("_group", "_D"): None, ("_vocab", "_DOTTED"): None, + # #544: the chunked sibling of _DOTTED ('M.Eng.'), S2's period gate + # for a member with a lower-case tail; no config key to mirror + ("_vocab", "_CHUNKED"): None, ("_group", "_PH"): None, ("_vocab", "_ROMAN"): "roman_numeral", ("_post_rules", "_EAST_SLAVIC"): "east_slavic_patronymic", From 460e2aaf208dd6f1f26d042059e9c7529de34c32 Mon Sep 17 00:00:00 2001 From: Derek Gulbranson Date: Sun, 27 Sep 2026 15:11:23 -0700 Subject: [PATCH 02/11] fix(#544): C1's name-word count reads a credential run as it reads one word C1 flipped a part after the comma to the suffix comma by the NAME-word count only when the part was one token, so a run ending in an ambiguous member fell through to the family comma: 'John Smith, PhD MEng' read given 'PhD', middle 'MEng', and 'Jane Doe, MS LAc' title 'MS', given 'LAc'. A part of two or more words, each suffix vocabulary or a class member, at least one a member and none a single-letter roman numeral, now reads by the same count, and the flip reports once over the whole part. A title/suffix dual opening it counts as suffix vocabulary after a full name; a run whose every listed member is in capitals is left to the family-comma path, which already reads it whole with the report it always made. The test is skipped unless two words precede the comma, and an inline fold answers the common token without a Python frame, so a comma name with no run pays nothing. The fold is a named, tested predicate (_vocab.run_word_fold) rather than a hand-copied condition at the call site, kept over removing it outright because an ordinary comma name that enters the run loop and breaks on its first token still costs 6 frames more without it. Co-Authored-By: Claude Opus 5.5 --- docs/design/rules.md | 18 ++++- nameparser/_pipeline/_segment.py | 106 +++++++++++++++++++++++++++--- nameparser/_pipeline/_vocab.py | 87 ++++++++++++++++++++++++ tests/v2/cases.py | 55 ++++++++-------- tests/v2/pipeline/test_segment.py | 96 +++++++++++++++++++++++++++ tests/v2/pipeline/test_vocab.py | 76 ++++++++++++++++++++- 6 files changed, 396 insertions(+), 42 deletions(-) diff --git a/docs/design/rules.md b/docs/design/rules.md index 252fc795..c658bab3 100644 --- a/docs/design/rules.md +++ b/docs/design/rules.md @@ -1596,9 +1596,21 @@ C1. Rationale: a credential run after the comma means the name is in credential run however that part is written, and only where the count leaves the word a name — one name word before the comma — is the case read, capitals in a mixed-case name making it the - credential there too (S2). A decision either way at this comma - is reported. It is one of TWO places the comma's own decision is - reported, the other being the word trailing the given part after + credential there too (S2). The same count reads a part of two or + more words as the credential run when every word of it is a + suffix word or a word of this class, at least one of them of this + class, and none of them a single-letter roman numeral, which is a + generation rather than a credential. A word of both the title and + the suffix vocabulary opening such a part counts as a suffix word + there, the name before the comma being complete. Where every word + of this class in the part is written in capitals in a mixed-case + name, the writing has already made each of them the credential + (S2), and the part reads as the credential run on that evidence + rather than on the count. A decision either way at this comma + is reported; for a run of words the decision is the flip to the + credential run, and a run the count leaves in the listing form + reports only as S2 reads the words in it. It is one of TWO + places the comma's own decision is reported, the other being the word trailing the given part after it (S2), which is a second decision about a second word and never the same fork twice; an attachment decided after a family comma (P6) reports on its own. C2's comma-structure flag reports what the diff --git a/nameparser/_pipeline/_segment.py b/nameparser/_pipeline/_segment.py index fbce8749..4debccfc 100644 --- a/nameparser/_pipeline/_segment.py +++ b/nameparser/_pipeline/_segment.py @@ -7,7 +7,8 @@ one_case where the comma form asked for it, COMMA_STRUCTURE ambiguities for unrecognized extra segments, and SUFFIX_OR_NAME where the comma FLIPPED the structure for a member of the ambiguous -credential class. The flip and nothing else: where the structure did +credential class, or for a run of suffix words holding one (#544). +The flip and nothing else: where the structure did not move, the word's reading is still open and `assign` takes it on the family-comma path, so it reports there. Reads: Lexicon suffix vocabulary and Policy, both through @@ -30,7 +31,17 @@ token joins the ambiguous credential class by SHAPE at this stage's own candidate tests the same way a listed member does; the caps half additionally needs `one_case` to decide membership at all, which this -stage's own lazy gate supplies. +stage's own lazy gate supplies. The #544 run test reads +_vocab.run_word_fold, _vocab.ambiguous_class_candidate (its "ask" +fallback), _vocab.ambiguous_class_member (the settled-lean check for +a fold-deferred member), _vocab.is_single_letter_numeral, +_vocab.ambiguous_lean and _vocab.is_wholly_suffix (once, over the +leftovers) -- six calls, not the two `ambiguous_class_candidate`/ +`is_wholly_suffix` a reader of the earlier, un-named fold might +expect. `run_word_fold` ANSWERS membership for a simple token +outright (its own docstring proves it agrees with +`ambiguous_class_candidate` there); it is a shortcut around a real +predicate, not a condition gating a second call to one. Implements rules C1 and C2 of docs/design/rules.md, cited at the decision site below; history in decisions.md#C1. @@ -42,8 +53,9 @@ ParseState, PendingAmbiguity, Structure, comma_bucket, copy_with, ) from nameparser._pipeline._vocab import ( - ambiguous_class_candidate, ambiguous_class_member, caps_shape_candidate, - is_one_case, is_wholly_suffix, name_word_count, + ambiguous_class_candidate, ambiguous_class_member, ambiguous_lean, + caps_shape_candidate, is_one_case, is_single_letter_numeral, + is_wholly_suffix, name_word_count, run_word_fold, ) from nameparser._types import AmbiguityKind @@ -167,10 +179,11 @@ def class_run(seg: tuple[int, ...]) -> bool: # And for the AMBIGUOUS class the count is of NAME words, not of # words: 'Smith Jr., MA' is two tokens and one name, and the token # count hands its family to `given` (#289/#516). One rule for the - # whole class: the listed and dotted halves are asked only of a - # single-token part, a member of either being always exactly one - # token (a bare word, or one glued acronym), while the caps half - # may come as a RUN ('LEED AP', below). + # whole class: the listed and dotted halves are asked of a + # single-token part here, a member of either being always exactly + # one token (a bare word, or one glued acronym), and of a RUN of + # such tokens among suffix words by the #544 run test below, while + # the caps half comes as a run of its own ('LEED AP', below). # # Membership is tested CASE-FREE first (`ambiguous_class_candidate`, # the listed set OR -- since 2.4 -- a by-shape member Policy @@ -202,7 +215,10 @@ def class_run(seg: tuple[int, ...]) -> bool: # THEM asks a question the design never posed. Measured, the # un-narrowed call per token moved `'John Smith, Ed Ma'`, `'John # Smith, ma do'` and `'John Smith, X.Y.Z. A.B.'` to a credential - # run neither the spec nor any case row wants. + # run through the CAPS switch. Since #544 all three are credential + # runs at every policy, by the listed-and-dotted run test below and + # its name-word count -- a different question, asked with the + # switch off too, so this narrowing still stands. # # `one_case=False` asks the case-free question -- "if this name # turned out mixed, would EVERY token in the run join the CAPS @@ -219,6 +235,78 @@ def class_run(seg: tuple[int, ...]) -> bool: one_case=False) for i in groups[1])): candidate = case_class() is False + # rules.md#C1: "The same count reads a part of two or more words as + # the credential run when every word of it is a suffix word or a + # word of this class, at least one of them of this class" -- the + # single-token rule above generalized to RUNS (#544): 'John Smith, + # PhD MEng' is the credential run the single-token 'John Smith, + # MEng' already is. A single-letter roman numeral voids the run (a + # generation is not a credential, `is_single_letter_numeral`), + # while a title/suffix DUAL opening the part counts as the suffix + # vocabulary it is: with a full name before the comma the name is + # complete, and the legacy disjunct below already counts a dual + # that way ('John Smith, MS MA'). The given part's head is where a + # dual reads as a title ('Smith, Ms Ma'), and that exclusion is + # `_pieces.segment_suffix_reading`'s, one name word before the + # comma never reaching the flip. + # + # Asked only where the flip is possible at all: two or more words + # before the comma is a necessary condition for two NAME words + # there, so 'Smith, J. Q.' and 'Smith, PhD MEng' never enter the + # loop. Inside it, class membership is asked first and the suffix + # predicate only of what is left, once, as a run. `run_word_fold` + # answers a SIMPLE token -- ASCII, no interior period -- from the + # vocabulary sets directly, at less cost than either real + # predicate below; a member or a definite reject is settled + # without calling them at all, and only a non-simple token, or one + # `run_word_fold` defers, still takes the real predicate -- + # `ambiguous_class_candidate` for membership, `is_wholly_suffix` + # once over the leftovers. A part holding a name word ('Doe, Jane + # Q. Public') breaks on that word's "reject" without entering a + # Python frame for either predicate. + if not candidate and len(groups[1]) >= 2 and len(groups[0]) >= 2: + members: list[str] = [] + rest: list[str] = [] + settled = True + lexicon = state.lexicon + for i in groups[1]: + text = state.tokens[i].text + fold = run_word_fold(text, lexicon, state.policy) + is_member = (fold == "member" + or (fold == "ask" and ambiguous_class_candidate( + text, lexicon, state.policy))) + if is_member: + members.append(text) + # the lean is the LISTED set's alone (S2): a member + # admitted by shape ('X.Y.Z.') is read by the count + settled = (settled and text.isupper() + and (fold == "member" + or ambiguous_class_member(text, lexicon))) + elif is_single_letter_numeral(text): + break + elif fold == "reject": + break + else: + rest.append(text) + else: + # Every word is a member or left to the suffix predicate. + # A run whose every member the WRITING already settles as a + # credential (capitals in a mixed-case name, S2's lean) is + # not this rule's decision: the listing form reads such a + # part wholly as the credential run and reports it as it + # always did ('John Smith, PhD MA'), so it is left there -- + # no flip, no new report. `settled` (every member listed + # and in capitals) is a necessary condition for that lean, + # asked inline so a run with a Title-case member never + # forces the case fact here; `ambiguous_lean` is what + # answers. + candidate = (bool(members) + and (not rest or is_wholly_suffix( + rest, lexicon, state.policy)) + and not (settled and case_class() is False + and all(ambiguous_lean(t, False) + == "credential" + for t in members))) # Computed only where `candidate` is true, alongside `case_class()` # -- the same lazy gate: a non-candidate comma name never counts # its pre-comma words either. Hoisted to a local because the diff --git a/nameparser/_pipeline/_vocab.py b/nameparser/_pipeline/_vocab.py index 021ae33a..692b5dfb 100644 --- a/nameparser/_pipeline/_vocab.py +++ b/nameparser/_pipeline/_vocab.py @@ -547,6 +547,93 @@ class at all). return _normalize(text) in lexicon.suffix_acronyms_ambiguous +# #544's inline gate for the comma-run test's common token, named and +# tested here rather than hand-copied at the call site (quality-review +# finding on the first cut): a SIMPLE token -- ASCII, no INTERIOR +# period -- is fully resolved from the vocabulary sets directly, at +# less cost than either real predicate it stands in for; a non-simple +# token is left to them ("ask"). +def run_word_fold( + text: str, lexicon: Lexicon, + policy: Policy) -> Literal["member", "reject", "defer", "ask"]: + """The #544 comma-run test's per-token gate (`_segment.py`). + + "member": TEXT, with its trailing periods stripped, matches + `suffix_acronyms_ambiguous` exactly and carries NO period at all + -- `ambiguous_class_member`'s own test, reached here without + paying for that call. Provably the same answer for a simple token: + `ambiguous_class_member` is exactly "no period, and the fold is a + listed ambiguous acronym", which this branch tests directly. + + "reject": no suffix vocabulary set, the Ph./D. halves ("ph", "d"), + or a configured delimiter core could ever accept this token -- + `is_wholly_suffix([text], lexicon, policy)` is False at EVERY + `Policy`, so the run test may stop without asking it. Universal + because `is_wholly_suffix` reads only two `Policy` fields, + `lenient_comma_suffixes` and `extra_suffix_delimiters`, and this + branch's own guard (`not policy.extra_suffix_delimiters`) already + requires the second to be empty -- a delimiter policy turns what + would have been "reject" into "defer" instead, never leaving this + branch's verdict to answer for one. The first selects between + `is_suffix_lenient` and `is_suffix_strict`, and both are + membership in the same three vocabulary sets this branch has + already excluded the fold from (plus `period_joined_vocab`, + which checks the identical two sets chunk-wise and finds no + interior period to chunk on a simple token) -- so + `lenient_comma_suffixes` cannot move the answer either. + + "defer": simple, not a member, not rejected either -- vocabulary- + eligible but not the ambiguous set, so the token is left for + `is_wholly_suffix` to count as a RUN member later, and calling + `ambiguous_class_candidate` here would only confirm False: for a + simple token that fold's `"." in text` gate already reads False + (no interior period) or, with a lone trailing period, finds no + "shape" verdict (`period_joined_vocab` requires a period that is + NOT at the end) -- so "defer" is `ambiguous_class_candidate`'s + answer too, paid for with zero calls instead of one. + + "ask": every NON-simple token (an interior period, or non-ASCII): + the caller falls back to `ambiguous_class_candidate` for + membership, exactly as it always did. + + All four cases are checked, over `Lexicon.default()`'s whole + suffix vocabulary plus name-word controls, mixed case, a trailing + period, both `Policy.lenient_comma_suffixes` settings and a + delimiter-core policy, by + `tests/v2/pipeline/test_vocab.py::test_run_word_fold_agrees_with_the_real_predicates` + (its docstring carries a negative control: with one acceptance + path dropped, the sweep fails). + + Measured (2026-09-27, #544, by a + profiler-frame count over one parse), for an ORDINARY comma name + that enters the run loop and breaks on its very first token -- + 'Doe Smith, Jane Q.', 'Garcia Lopez, Maria Jose': asking + `ambiguous_class_candidate` of that token directly, with this + whole gate skipped, costs 325 and 338 frames; this named function + costs 319 and 332; the ORIGINAL hand-inlined gate (before it was + a named function at all) cost 317 and 330. So the gate itself + saves 8 frames against asking the predicate directly; making it a + named, testable call gives back 2 of those 8 (the call's own + frame); net 6 -- still the frame a comma name with no credential + in it pays less than it would with no gate at all, and the price + of a gate a test can reach on its own rather than one hand-copied + at the call site. + """ + core = text.rstrip(".") + if not (text.isascii() and "." not in core): + return "ask" + folded = core.lower() + if folded in lexicon.suffix_acronyms_ambiguous and core == text: + return "member" + if (not policy.extra_suffix_delimiters + and folded not in lexicon.suffix_acronyms + and folded not in lexicon.suffix_words + and folded not in lexicon.suffix_acronyms_ambiguous + and folded not in ("ph", "d")): + return "reject" + return "defer" + + # #516's all-caps half, ONE PREDICATE for the shape test and its # WHOLE-VOCABULARY exclusion, shared by the three sites that each # needed the identical question answered (classify's tag emission, diff --git a/tests/v2/cases.py b/tests/v2/cases.py index 2eb4edec..0c80152f 100644 --- a/tests/v2/cases.py +++ b/tests/v2/cases.py @@ -696,29 +696,27 @@ def _check_cjk_shape_purity(self) -> None: "it as a middle name. 1.4.0 read given 'MEng' too, so " "this row also restores 1.4.0's reading", shape=2), - Case("comma_credential_run_ending_in_meng_re_reads_the_comma", + Case("comma_credential_run_ending_in_meng_keeps_the_suffix_comma", "John Smith, PhD MEng", - {"given": "PhD", "middle": "MEng", "family": "John Smith"}, - classification="fix(#540)", - ambiguities=("suffix-or-name",), - notes="the run behind a suffix comma is no longer wholly " - "suffix-shaped once its last word is a bare ambiguous " - "member, so C1 reads the comma as a family comma and " - "the first credential as the given name. Every " - "release read suffix 'PhD MEng' -- the pre-existing " - "'john smith, phd ma' path, which #540 routes two " - "more words into; #544 asks whether to keep the comma", + {"given": "John", "family": "Smith", "suffix": "PhD MEng"}, + ambiguities=("suffix-or-name",), + notes="C1's name-word count reads a RUN as it reads one word " + "(#544): every word after the comma is suffix " + "vocabulary or a class member, and two name words " + "stand before it, so the part is the credential run " + "the lone 'John Smith, MEng' already is, and the flip " + "reports. 1.4.0 read the same suffix; #540 had re-read " + "the comma as a family comma (given 'PhD', middle " + "'MEng'), the cost #544 reverses", shape=3), - Case("comma_lower_credential_run_ending_in_meng_re_reads_the_comma", + Case("comma_lower_credential_run_ending_in_meng_keeps_the_suffix_comma", "john smith, phd meng", - {"given": "phd", "family": "john smith", "suffix": "meng"}, - classification="fix(#540)", + {"given": "john", "family": "smith", "suffix": "phd meng"}, ambiguities=("suffix-or-name",), - notes="the same re-read in one case -- the count then " - "peels 'meng' with a word to spare, so the writing's " - "case does not save it, and neither does the " - "all-caps 'JOHN SMITH, PHD MENG' (given 'PHD', " - "family 'JOHN SMITH', suffix 'MENG')", + notes="the same run in one case: the count decides " + "whatever case the name is written in, so 'JOHN " + "SMITH, PHD MENG' reads the suffix too. 1.4.0 read " + "the same; #540 had read given 'phd', suffix 'meng'", shape=3), Case("title_case_lac_behind_a_particle_is_the_name", "Nguyen Van Lac", @@ -2362,20 +2360,19 @@ def _check_cjk_shape_purity(self) -> None: # caps run test's `all()`. Identical to the default reading. Case("caps_switch_run_test_declines_a_pure_listed_run", "John Smith, Ed Ma", - {"given": "Ed", "middle": "Ma", "family": "John Smith"}, + {"given": "John", "family": "Smith", "suffix": "Ed Ma"}, policy=Policy(unlisted_caps_suffixes=True), - classification="fix(#531)", - ambiguities=("suffix-or-name", "suffix-or-name"), + ambiguities=("suffix-or-name",), notes="'Ed' and 'Ma' are both LISTED ambiguous members, " "Title-case (leans NAME, #289) -- the caps run test " "must not admit a run the listed class already reads " - "on its own. #531 adds the SECOND report and moves no " - "field: this is a family-comma name whose given part " - "now ends in a class member, and 'Ma' is Title-cased, " - "so the slot consults the fork and declines it " - "exactly as 'Doe, John Ma' does. The report tracks " - "the fork CONSULTED (#530), which is why a declined " - "reading still says so"), + "on its own, and it does not: what reads this part is " + "C1's listed run (#544), which counts the two name " + "words before the comma exactly as it does with the " + "switch off, so the part is the credential run and " + "the flip reports once. 1.4.0 read the same suffix; " + "before #544 the part was the given name 'Ed', " + "middle 'Ma'"), # #516 review round (second finding): the caps branch read # `one_case_own` -- true only for a token INSIDE the maiden # clause's own-words span -- where it needed the bare NAME-level diff --git a/tests/v2/pipeline/test_segment.py b/tests/v2/pipeline/test_segment.py index b95710e0..4e0c12cc 100644 --- a/tests/v2/pipeline/test_segment.py +++ b/tests/v2/pipeline/test_segment.py @@ -207,3 +207,99 @@ def test_the_structure_flip_report_counts_the_real_pre_comma_words() -> None: (amb,) = [a for a in state.ambiguities if a.kind is AmbiguityKind.SUFFIX_OR_NAME] assert "holds 3 name words" in amb.detail + + +def _flip_reports(state: ParseState) -> list[list[str]]: + return [_texts(state, a.indices) for a in state.ambiguities + if a.kind is AmbiguityKind.SUFFIX_OR_NAME] + + +def test_the_name_word_count_reads_a_run_as_it_reads_one_word() -> None: + # rules.md#C1 (#544): a part of two or more words, every one of + # them suffix vocabulary or a class member and at least one a + # member, reads by the same NAME-word count as the single token, + # and the flip reports once, over the whole part. + for text in ("John Smith, PhD Ma", "John Smith, Ed Ma", + "John Smith, Ma PhD", "john smith, phd ma", + "JOHN SMITH, PHD MA", "John Smith, MD Ma", + "John Smith, A.B. PhD", "John Smith, Ph. D. Ma"): + out = _segmented(text) + assert out.structure is Structure.SUFFIX_COMMA, text + assert _flip_reports(out) == [_texts(out, out.segments[1])], text + # one name word before the comma: the count leaves the listing form + # and the run test is not even asked, so the case fact stays unasked + out = _segmented("Smith, PhD Ma") + assert out.structure is Structure.FAMILY_COMMA + assert out.one_case is None + assert _segmented("Smith Jr., PhD Ma").structure \ + is Structure.FAMILY_COMMA + + +def test_the_run_test_declines_a_name_word_and_a_numeral() -> None: + # a name word anywhere in the part, or a word that is neither + # vocabulary nor a member, keeps the listing form -- and so does a + # single-letter roman numeral, a generation rather than a credential + for text in ("John Smith, Jones Ma", "John Smith, PhD Jones Ma", + "John Smith, V Ma", "John Smith, PhD v Ma", + "John Smith, PhD Ma.", "John Smith, J. Ma"): + out = _segmented(text) + assert out.structure is Structure.FAMILY_COMMA, text + assert not _flip_reports(out), text + + +def test_a_run_the_writing_settles_is_left_to_the_listing_form() -> None: + # every member written in capitals in a mixed-case name leans + # credential (S2), so the part is already the credential run on + # that evidence: no flip here and no report from this stage + out = _segmented("John Smith, PhD MA") + assert out.structure is Structure.FAMILY_COMMA + assert not _flip_reports(out) + assert out.one_case is False + # the lean is the LISTED set's alone: a member admitted by SHAPE + # is read by the count, capitals or not + out = _segmented("John Smith, X.Y.Z. MA") + assert out.structure is Structure.SUFFIX_COMMA + # and one Title-case member is enough to make it the count's call + assert _segmented("John Smith, MA Ed").structure \ + is Structure.SUFFIX_COMMA + + +def test_the_run_test_reads_a_delimiter_core_transparently() -> None: + # a configured delimiter core is _vocab.run_word_fold's "defer" + # (#544): without a delimiter policy + # the bare token is a definite "reject" (no suffix set, no core) + # and the run test breaks on it before it ever reaches + # `is_wholly_suffix`; with the core configured the SAME token is + # left for `is_wholly_suffix`'s own `text in cores` disjunct, + # which admits it, so the two members either side of it still + # complete the run. + text = "John Smith, Ma - Ed" + assert _segmented(text).structure is Structure.FAMILY_COMMA + state = ParseState( + original=text, lexicon=_LEX, + policy=dataclasses.replace( + Policy(), extra_suffix_delimiters=frozenset({" - "}))) + out = segment(tokenize(extract_delimited(state))) + assert out.structure is Structure.SUFFIX_COMMA + + +def test_the_run_test_reads_lenient_comma_suffixes() -> None: + # rules.md#C1's leftovers ask `is_wholly_suffix`'s own POLICY- + # selected predicate (#544): an + # initial-shaped suffix WORD ('B.') that is not a single-letter + # roman numeral is a member of neither the ambiguous class nor + # `is_single_letter_numeral`'s carve-out, so it reaches `rest` and + # is read by `is_wholly_suffix` alone -- lenient by default (v1's + # is_suffix_lenient bypasses the initial veto), strict under + # Policy(lenient_comma_suffixes=False) (v1's is_suffix, which the + # veto reads as a middle initial instead). + lex = _LEX.add(suffix_words={"b"}) + text = "John Smith, Ma B." + state = ParseState(original=text, lexicon=lex, policy=Policy()) + out = segment(tokenize(extract_delimited(state))) + assert out.structure is Structure.SUFFIX_COMMA + state = ParseState( + original=text, lexicon=lex, + policy=dataclasses.replace(Policy(), lenient_comma_suffixes=False)) + out = segment(tokenize(extract_delimited(state))) + assert out.structure is Structure.FAMILY_COMMA diff --git a/tests/v2/pipeline/test_vocab.py b/tests/v2/pipeline/test_vocab.py index 17c5c156..cfe220f1 100644 --- a/tests/v2/pipeline/test_vocab.py +++ b/tests/v2/pipeline/test_vocab.py @@ -13,7 +13,7 @@ is_single_letter_numeral, is_suffix_lenient, is_suffix_strict, is_title_shaped, is_wholly_suffix, maiden_marker_run, name_word_count, period_joined_vocab, - resolve_script_set, single_script, + resolve_script_set, run_word_fold, single_script, ) from nameparser._policy import (Policy, Script, _NO_INITIALS, _SCRIPT_RANGES) @@ -386,6 +386,80 @@ def test_a_listed_dotted_entry_is_not_read_by_shape() -> None: assert ambiguous_class_candidate("A.B.", Lexicon.default(), Policy()) +def test_run_word_fold_agrees_with_the_real_predicates() -> None: + """#544: the comma + run test's inline gate is a named, testable function precisely so + this sweep can hold it to the real predicates it stands in for, + rather than trusting a hand-copied condition at the call site + (mechanisms.md#ONE-PREDICATE-PER-QUESTION). + + Every token this sweep builds -- the whole shipped suffix + vocabulary plus a few name-word controls, in lower/Title/UPPER + case, bare and with one trailing period, under both + `Policy.lenient_comma_suffixes` settings and one delimiter-core + policy (11,880 built cases) -- is asked, with NO skip of its own: + whichever verdict `run_word_fold` returns is checked against the + real predicates, or, for "ask", left unchecked here and pinned + instead by the fixed non-simple list below. "member" must agree + with `ambiguous_class_candidate`; "defer" must NOT be a member by + that predicate either (it differs from "reject" only in being + left to `is_wholly_suffix` rather than a shortcut break); and + "reject" must mean `is_wholly_suffix` rejects the bare token too, + at every policy this sweep tries -- the two predicates the run + test would otherwise call directly. A vocabulary entry that is + itself non-simple (non-ASCII, or carrying an interior period -- + 576 of the 11,880 built cases, every Hebrew/Devanagari/Bengali/ + CJK honorific in the shipped vocabulary among them, times three + policies) settles to "ask" here like any other non-simple token, + checked by nothing but its own verdict -- no longer skipped + before `run_word_fold` is even called, which is what let a drift + in the SIMPLE-token gate go unseen (11,304 of the 11,880 cases + are actually asserted against the real predicates; the rest are + "ask" and pinned only by the fixed list below). + + A fixed list of non-simple tokens pins "ask" directly, since the + built sweep above never asserts it: an interior-period acronym + ('A.B.', 'Ph.D.'), one with no trailing period ('M.D'), and two + non-ASCII scripts ('씨', a bare CJK honorific; 'María', a Latin + name carrying a diacritic). + + Negative control, from a COLLECTING variant of this test (every + built case checked, not just the first failure) run by hand: drop + the `suffix_words` disjunct from `run_word_fold`'s reject test, so + a bare suffix WORD with no acronym or ambiguous listing (e.g. + 'Jr') folds to "reject" instead of "defer". 179 of the 11,880 + built cases fail (every case built from a shipped suffix_words + entry that is in no other suffix set, across every case/period/ + policy combination) -- the sweep is not vacuously green. + """ + lex = Lexicon.default() + vocab = sorted(lex.suffix_acronyms | lex.suffix_words + | lex.suffix_acronyms_ambiguous) + controls = ["Smith", "Jane", "van", "de"] + policies = [Policy(), + Policy(lenient_comma_suffixes=False), + Policy(extra_suffix_delimiters=frozenset({" - "}))] + for word in vocab + controls: + for case in (str.lower, str.title, str.upper): + for trailing in ("", "."): + text = case(word) + trailing + for policy in policies: + fold = run_word_fold(text, lex, policy) + if fold == "ask": + continue + real_member = ambiguous_class_candidate(text, lex, policy) + if fold == "member": + assert real_member, (text, policy) + else: + assert not real_member, (text, policy) + if fold == "reject": + assert not is_wholly_suffix([text], lex, policy), \ + (text, policy) + for text in ("A.B.", "Ph.D.", "M.D", "씨", "María"): + for policy in policies: + assert run_word_fold(text, lex, policy) == "ask", (text, policy) + + def test_a_callers_own_conjunction_marker_keeps_its_word_a_name() -> None: # The exclusion end to end through a CALLER's vocabulary rather # than the shipped one: `conjunctions_ambiguous` has no subset From 4b708ca2aed41c2e32fc305a0a7e4334f986e338 Mon Sep 17 00:00:00 2001 From: Derek Gulbranson Date: Sun, 27 Sep 2026 15:57:22 -0700 Subject: [PATCH 03/11] fix(#544): an unambiguous credential in front of an ambiguous member speaks for it The trailing readers met a listed member first, took its name lean and stopped, so the credential in front of it was never reached: 'John Smith PhD MEng' read middle 'Smith PhD', family 'MEng', 'Doe, Jane PhD MEng' middle 'MEng', and 'abdul Smith Jr Ma' middle 'Jr' -- the cost decisions.md#S2 had accepted as #289's item (ii). A member standing behind an unambiguous credential in one run of suffix words now reads as the credential whatever its writing: on the no-comma peel, in the comma part read wholly as credentials, and at the given slot, where it outranks P6's attachment as capitals do ('doe, jane v phd do'). Only a credential in FRONT speaks ('Wang Ma PhD' keeps family 'Ma'); a connective, a single-letter numeral and a dual opening the given part speak for nothing, and the reading does not reach across a maiden clause. The walk's own leading piece never anchors: it is the name a reserve keeps whatever its vocabulary, so 'Om Ma' and 'PhD Ma' keep family 'Ma' (with it anchoring, the family was lost outright), while 'John Om Ma' reads suffix 'Om Ma'. A single-letter roman numeral is excluded for being initial-shaped ('V' can be a middle initial); a multi-letter one ('Jr', 'III') anchors like any other suffix word. _pieces.credential_anchors is the one forward pass, linear in a run; credential_at_the_given_slot takes it as a thunk asked last, so a member the writing settles never pays for it. The M2 clause-agreement walk now also asks every parse it takes two questions that share one walk-back (`_qualifying_front`): whether a listed member behind an unambiguous credential reads as one (48 failures with the anchor off, 0 here), and the converse, whether a member read as a credential because of the word in front has that word in the suffix too (30 failures with the leading piece allowed to anchor, 0 here). It pins the pairs a maiden clause makes differ as its anchored_head class. test_benchmark gains the credential_run shape (15.8x for 4x the input with a per-member look-behind, 4.2x here). The reserve the company check exempts is the name's leading piece in a name with no comma before it (H4's carve-out, where 'PhD Ma' keeps family 'Ma'); a family-comma given part's first piece reads through `segment_suffix_reading` instead, so a comma before a candidate front exempts nothing, and the walk continues through a maiden clause to the name past it. The converse forces its extra parse only for a front outside both TITLE and SUFFIX, which keeps the walk at 2.5s, and each control trips its own assert, ordered ahead of the pair-agreement one so neither masks the other. Co-Authored-By: Claude Opus 5.5 --- docs/design/rules.md | 35 +++- nameparser/_pipeline/_assign.py | 30 ++- nameparser/_pipeline/_group.py | 39 +++- nameparser/_pipeline/_pieces.py | 146 +++++++++++++- tests/v2/cases.py | 22 ++- tests/v2/pipeline/test_group.py | 15 +- tests/v2/pipeline/test_pieces.py | 182 +++++++++++++++++- tests/v2/test_benchmark.py | 17 ++ tests/v2/test_parser.py | 29 ++- tests/v2/test_properties.py | 317 +++++++++++++++++++++++++++++-- 10 files changed, 758 insertions(+), 74 deletions(-) diff --git a/docs/design/rules.md b/docs/design/rules.md index c658bab3..f9426d96 100644 --- a/docs/design/rules.md +++ b/docs/design/rules.md @@ -727,8 +727,8 @@ P5. Rationale: some given-name words are incomplete alone — "abdul" "abdul Smith V" → family="Smith" "abdul Smith V" → suffix="V" "abdul Smith Jr V" → family="Smith" - "abdul Smith Jr Ma" → given="abdul Smith" - "abdul Smith Jr Ma" → middle="Jr" + "abdul Smith Jr Ma" → given="abdul" + "abdul Smith Jr Ma" → suffix="Jr Ma" "abdul Smith Ma" → given="abdul Smith" "abdul Smith Berg Ma" → middle="Berg" · boundary "abdul Sir Smith Berg" → given="abdul Sir" @@ -1024,6 +1024,22 @@ S2. Rationale: generational suffixes and credentials are recognized count leaves the word a name. At the trailing slot of the given part the comma has already settled the count, so the writing is the only evidence there is. + Company is evidence that outranks both. A member of the ambiguous + set standing BEHIND an unambiguous credential in one run of + suffix words — members of the class between them passing the run + on — reads as the credential whatever its writing and whatever + the count, at every trailing slot this rule names: the degree in + front says what the run is. Only a credential in FRONT speaks; one + behind the member says nothing about it. A connective (P3) and a + single-letter roman numeral — initial-shaped, as a bare middle + initial is, unlike a multi-letter one such as 'III' — speak for + nothing and end the run; so does any name word. A word of both + the title and the suffix vocabulary opening + the given part is a title there, and speaks for nothing either, + though it speaks wherever else it stands. At the trailing slot of + the given part this company outranks P6's attachment, as the + capitals do. It does not reach across a maiden marker's clause + (M2), whose name words stand between. An unlisted word joins this same ambiguous class by SHAPE where the caller asks for it. Two or more period-separated chunks is one such shape, admitted by default (S3); an unlisted all-caps @@ -1090,13 +1106,18 @@ S2. Rationale: generational suffixes and credentials are recognized Accepted: an unambiguous suffix is consumed even when that leaves no family name at all. "Smith Jr." → family="" - Accepted: the case signal costs a genuine suffix standing behind - a name-leaning acronym. The walk stops at the declined pick - rather than continuing past it, so a suffix word in front of one - is never reached and reads as a name word. + Accepted: a name-leaning member with nothing in front of it to + speak for it ends the name, words to spare or not, and whatever + stands in front of it is name text. Where a credential does + stand in front, the company above decides instead, except in + three shapes that keep it out of reach, each of which reads the + member as the name: a credential written split across two words + with no comma after the name, a title standing between the + credential and the member, and a member a particle chain (P2) + has already taken. "Jack Wei Ma" → family="Ma" "Jack Wei Ma" → ambiguities=("suffix-or-name",) - "abdul Smith Jr Ma" → middle="Jr" + "abdul Smith Jr Ma" → suffix="Jr Ma" Accepted: the title chain no longer takes the word this rule needs, and the argument a descriptive note here asked for is made. A title run leaves one NAME word standing and a diff --git a/nameparser/_pipeline/_assign.py b/nameparser/_pipeline/_assign.py index a9aacd4d..d6f99538 100644 --- a/nameparser/_pipeline/_assign.py +++ b/nameparser/_pipeline/_assign.py @@ -71,7 +71,7 @@ effective_script, is_suffix_lenient, resolve_script_set, ) from nameparser._pipeline._pieces import ( - credential_at_the_given_slot, + credential_anchors, credential_at_the_given_slot, is_suffix_piece, leading_titles, peel_walk, segment_suffix_reading, tail_reading, trailing_titles, ) @@ -674,6 +674,27 @@ def previous_kept(m: int, titled: tuple[int, ...]) -> int: #: which is a `copy_with(role=...)` and leaves #: text and tags identical. floors: dict[tuple[int, ...], tuple[int, bool]] = {} + #: #544's anchors, per `titled` value like `floors` and for + #: the same reason: `credential_anchors` over the pieces the + #: chain kept, computed once, the first time a member's own + #: writing declines, so a run of members is read in one + #: forward pass rather than one look-behind per member. + anchor_memo: dict[tuple[int, ...], dict[int, bool]] = {} + + def anchored(m: int, titled: tuple[int, ...]) -> bool: + memo = anchor_memo.get(titled) + if memo is None: + # from past the leading title run: a title/suffix + # dual opening the part is a TITLE there and + # anchors nothing ('Smith, MD MA Ma') + order = [q for q in range( + leading_titles(pieces, ptags, tokens), + len(pieces)) + if q not in titled] + memo = dict(zip(order, credential_anchors( + order, pieces, ptags, tokens))) + anchor_memo[titled] = memo + return memo.get(m, False) def trailing_floor(m: int, titled: tuple[int, ...]) -> int: """Where the trailing suffix run starts, walked as far @@ -833,8 +854,13 @@ def reads_as_a_suffix(m: int, titled: tuple[int, ...]) -> bool: # unchanged). Measured 2026-09-19 per # `Parser.parse`; Derek took that trade # deliberately. + # #544: an unambiguous credential in front + # of it in the same run anchors it -- asked + # through a thunk, so only a member the + # writing declines pays for the pass if credential_at_the_given_slot( - tok, state.one_case): + tok, state.one_case, + lambda: anchored(m, titled)): return True prev = previous_kept(m, titled) # trailing piece of a two-part name is unambiguously diff --git a/nameparser/_pipeline/_group.py b/nameparser/_pipeline/_group.py index 31b64488..0d6a8b41 100644 --- a/nameparser/_pipeline/_group.py +++ b/nameparser/_pipeline/_group.py @@ -44,7 +44,7 @@ from nameparser._lexicon import _run_addresses_by_given from nameparser._pipeline._pieces import ( - credential_at_the_given_slot, + credential_anchors, credential_at_the_given_slot, is_leading_title, is_suffix_piece, is_title_piece, is_trailing_title_word, Peel, leading_titles, peel_trailing, peel_walk, tail_reading, @@ -410,6 +410,21 @@ def _release_reads_off(view: Sequence[Sequence[int]], last = len(view) - 1 chain_ok = [False] * len(view) members_ok = running = True + # #544: the anchors `credential_at_the_given_slot` may ask for, + # over the view as the take would leave it -- the reading + # assign's given slot makes of the same pieces, from past the + # leading title run. One forward pass for both loops below, + # run only the first time a member's writing leaves the + # question open; each call site hands over a lambda, so no + # frame is spent building the question either. + anchor_cell: list[list[bool]] = [] + + def anchored_at(q: int) -> bool: + if not anchor_cell: + lead = leading_titles(view, view_tags, tokens) + anchor_cell.append([False] * lead + credential_anchors( + range(lead, len(view)), view, view_tags, tokens)) + return anchor_cell[0][q] for q in range(last, -1, -1): piece = view[q] if (is_suffix_piece(piece, view_tags[q], tokens) @@ -420,8 +435,9 @@ def _release_reads_off(view: Sequence[Sequence[int]], continue if (members_ok and len(piece) == 1 and AMBIGUOUS_ACRONYM_TAG in tokens[piece[0]].tags - and credential_at_the_given_slot(tokens[piece[0]], - one_case)): + and credential_at_the_given_slot( + tokens[piece[0]], one_case, + lambda: anchored_at(q))): chain_ok[q] = running continue members_ok = False @@ -448,8 +464,9 @@ def _release_reads_off(view: Sequence[Sequence[int]], return False if (len(piece) == 1 and AMBIGUOUS_ACRONYM_TAG in tokens[piece[0]].tags - and credential_at_the_given_slot(tokens[piece[0]], - one_case)): + and credential_at_the_given_slot( + tokens[piece[0]], one_case, + lambda: anchored_at(q))): continue return False elif reader is TailReader.TRAILING: @@ -844,9 +861,15 @@ def _maiden_take(pieces: Sequence[Sequence[int]], # 'Doe, Jane MA do' does with them, middle 'MA' and # family 'do Doe'). The member is asked here, the span # behind it and the name word ahead of it by the - # shared check below. - takes = credential_at_the_given_slot(tokens[head[0]], - one_case) + # shared check below. Anchored (#544) as assign's given + # slot anchors it: over the view up to the member, from + # past the leading title run. + takes = credential_at_the_given_slot( + tokens[head[0]], one_case, + lambda: credential_anchors( + range(min(leading_titles(view, view_tags, tokens), + at), at + 1), + view, view_tags, tokens)[-1]) start = at + 1 else: # TRAILING: the peel over the view IS the member's diff --git a/nameparser/_pipeline/_pieces.py b/nameparser/_pipeline/_pieces.py index d72d1a76..a12b4d1e 100644 --- a/nameparser/_pipeline/_pieces.py +++ b/nameparser/_pipeline/_pieces.py @@ -52,7 +52,7 @@ """ from __future__ import annotations -from collections.abc import Mapping, Sequence, Set +from collections.abc import Callable, Mapping, Sequence, Set from typing import NamedTuple from nameparser._pipeline._state import ( @@ -60,7 +60,7 @@ ) from nameparser._pipeline._vocab import ( _PERIOD_ABBREV, Lean, ambiguous_lean, in_initialless_script, - is_trailing_numeral_suffix, tag_marker_runs, + is_single_letter_numeral, is_trailing_numeral_suffix, tag_marker_runs, ) @@ -285,6 +285,79 @@ def _numeral_behind_the_initial_veto(piece: Sequence[int], return "vocab:suffix" in tags and "initial" in tags +def _anchors(piece: Sequence[int], tokens: Sequence[WorkToken]) -> bool: + """Whether a suffix piece may ANCHOR an ambiguous member behind it + (#544): not a connective, and not a single-letter roman numeral. + + A connective anchors nothing because between two name words it is + a link -- the generational 'i' is also Catalan's 'i' (rules.md#P3) + -- and 'Jane Doe nee Puig i Ma' must keep its clause. A + single-letter numeral anchors nothing because it is + INITIAL-SHAPED, the same shape a middle initial writes in ('V' can + be one) -- not because a numeral itself is no credential, since a + MULTI-letter one ('Jr', 'III') anchors like any other suffix piece + (`is_single_letter_numeral`). A title/suffix + DUAL ('ms', 'md', 'sr') does anchor, except at the head of the + given part, where it stands in title position ('Smith, Ms Ma' is + Ms. Ma Smith); that exclusion is the callers', which start their + walks past the leading title run or test the head themselves + (`segment_suffix_reading`). + + Asked only of a piece `is_suffix_piece` accepted, and only once a + member's own writing has declined, so an ordinary name never pays + for it.""" + if len(piece) != 1: + return True + tok = tokens[piece[0]] + return ("conjunction" not in tok.tags + and not is_single_letter_numeral(tok.text)) + + +# #544: an ambiguous member standing BEHIND an unambiguous credential +# in one run is described by that credential's company, not by its +# own writing -- 'PhD MEng' is a list of degrees whatever case 'MEng' +# is written in. ONE forward pass per question, so a run of members +# is read in linear time: a look-behind per member is quadratic in +# the run, which test_benchmark's frame-ratio guard is built to catch. +def credential_anchors(order: Sequence[int], + pieces: Sequence[Sequence[int]], + ptags: Sequence[Set[str]], + tokens: Sequence[WorkToken]) -> list[bool]: + """For each position of `order` (piece indices in text order), + whether a lone listed member standing there is ANCHORED: the + contiguous run of pieces in FRONT of it that are suffix pieces or + lone ambiguous members ends in -- counting back past the members -- + a suffix piece that may anchor (`_anchors`). Any other piece ends + the run. The anchor is in front only: 'Wang Ma PhD' leaves 'Ma' + unanchored. + + The caller decides where the walk starts, and starts it past the + leading title run, where a title/suffix dual reads as a title + ('Smith, MD MA Ma' keeps its middle name). `order`'s own leading + position is always the name the reserve keeps -- `rest[0]` at the + no-comma peel, the given part's own first piece at the other call + sites -- so ITS writing never anchors what stands behind it, + whatever vocabulary it carries: 'PhD Ma' keeps its parent reading, + family 'Ma', rather than reading as a credential list headed by + the given name itself.""" + out: list[bool] = [] + anchor = False + for pos, idx in enumerate(order): + out.append(anchor) + if pos == 0: + continue + piece = pieces[idx] + if is_suffix_piece(piece, ptags[idx], tokens): + # a suffix piece that may not anchor still ENDS the run it + # would have anchored: 'PhD v Ma' is not a credential list + # through the numeral + anchor = _anchors(piece, tokens) + elif not (len(piece) == 1 + and AMBIGUOUS_ACRONYM_TAG in tokens[piece[0]].tags): + anchor = False + return out + + def segment_suffix_reading(pieces: Sequence[Sequence[int]], ptags: Sequence[Set[str]], tokens: Sequence[WorkToken], @@ -303,6 +376,12 @@ def segment_suffix_reading(pieces: Sequence[Sequence[int]], class by SHAPE takes the count instead, which is decided at the comma and not in this walk ('Smith, A.B.' -> given 'A.B.'). + A listed member ANCHORED by an unambiguous credential in front of + it in the same run reads as a credential too, whatever its writing + (#544, `credential_anchors`): 'Smith, PhD MEng' is family 'Smith' + with two degrees. A dual opening the part is a title there and + anchors nothing ('Smith, Ms Ma' keeps its given name). + ONE answer for two readers, both in _assign.py -- the no-name gate and the router -- because they must agree piece for piece. #429 shipped the inverse of its own fix by deriving that agreement twice @@ -343,6 +422,13 @@ class by SHAPE takes the count instead, which is decided at the if not pieces: return None out: list[bool] = [] + # #544: `credential_anchors`' own reading, carried inline because + # this walk is already the forward pass it would make, and a call + # per family-comma segment 1 is a frame every comma name pays. The + # most recent suffix piece of the current run, kept through lone + # members and cleared by anything else; asked `_anchors` only when + # a member's writing has declined. Keep the two in step. + anchor: Sequence[int] | None = None for piece, tags in zip(pieces, ptags): # the verdict just recorded IS "stands behind a suffix" -- keeping # a separate flag meant maintaining that equality by hand at three @@ -350,11 +436,24 @@ class by SHAPE takes the count instead, which is decided at the # have diverged silently after_suffix = bool(out) and out[-1] if is_suffix_piece(piece, tags, tokens): + # a title/suffix dual with nothing read as a suffix ahead of + # it stands in the part's title position and anchors nothing + # ('Smith, Ms Ma'); `any` is a builtin, not a frame + anchor = (None if (len(piece) == 1 + and "vocab:title" in tokens[piece[0]].tags + and not any(out)) + else piece) out.append(True) - elif (len(piece) == 1 - and AMBIGUOUS_ACRONYM_TAG in tokens[piece[0]].tags - and listed_lean(tokens[piece[0]], one_case) - == "credential"): + continue + member = (len(piece) == 1 + and AMBIGUOUS_ACRONYM_TAG in tokens[piece[0]].tags) + if not member: + anchor = None + if member and (listed_lean(tokens[piece[0]], one_case) + == "credential" + or (anchor is not None + and SHAPE_ACRONYM_TAG not in tokens[piece[0]].tags + and _anchors(anchor, tokens))): out.append(True) elif (lenient and after_suffix and _numeral_behind_the_initial_veto(piece, tokens)): @@ -470,8 +569,9 @@ def listed_lean(token: WorkToken, one_case: bool | None) -> Lean | None: return ambiguous_lean(token.text, one_case) -def credential_at_the_given_slot(token: WorkToken, - one_case: bool | None) -> bool: +def credential_at_the_given_slot( + token: WorkToken, one_case: bool | None, + anchored: Callable[[], bool] | None = None) -> bool: """#531's reading of a class MEMBER ending the given part after a family comma: the credential unless the writing says otherwise. The caller decides membership and that the piece ends that part. @@ -481,6 +581,15 @@ def credential_at_the_given_slot(token: WorkToken, particle vocabulary reads as the credential on a POSITIVE lean alone, P6's attachment keeping every other spelling. + Or where the member is ANCHORED (#544): an unambiguous credential + in front of it in the same run (`credential_anchors`) makes it the + credential whatever its writing says, a particle member included + -- a degree in front outranks P6's attachment, as the capitals do + ('doe, jane v phd do' reads suffix 'v phd do'). Only the caller + knows the run, so it computes the anchor; `anchored` is a THUNK, + asked last, so a member the writing already settles never pays + for the walk. + Called from assign's walk over the given part, from `_release_reads_off`'s GIVEN_SLOT branch (the shared release check every maiden-walk stop asks, rules.md#M2), and from the acronym @@ -507,8 +616,10 @@ def credential_at_the_given_slot(token: WorkToken, f"and is not one. The caller decides membership -- test " f"AMBIGUOUS_ACRONYM_TAG before calling") lean = listed_lean(token, one_case) - return lean == "credential" or (lean is None - and "particle" not in token.tags) + if lean == "credential" or (lean is None + and "particle" not in token.tags): + return True + return anchored is not None and anchored() def peel_trailing(rest: Sequence[int], pieces: Sequence[Sequence[int]], @@ -529,6 +640,7 @@ def peel_trailing(rest: Sequence[int], pieces: Sequence[Sequence[int]], """ picks: list[tuple[int, ...]] = [] numeral: tuple[int, ...] | None = None + anchors: list[bool] | None = None k = len(rest) while k > 0: piece = pieces[rest[k - 1]] @@ -594,6 +706,20 @@ def peel_trailing(rest: Sequence[int], pieces: Sequence[Sequence[int]], if lean == "credential" or (lean is None and k >= 3): k -= 1 continue + # #544: or an unambiguous credential stands IN FRONT of it + # in the same run -- 'John Smith PhD MEng' is a list of + # degrees, not a family name 'MEng'. Computed once per + # walk, and only when a member has declined, so an ordinary + # name never pays for it; a LISTED member only, the + # by-shape half taking the count as above. Still a pick: + # the fork is reported as a counted member's is. + if SHAPE_ACRONYM_TAG not in tokens[piece[0]].tags: + if anchors is None: + anchors = credential_anchors(rest, pieces, ptags, + tokens) + if anchors[k - 1]: + k -= 1 + continue break return Peel(k, numeral, tuple(picks)) diff --git a/tests/v2/cases.py b/tests/v2/cases.py index 0c80152f..1fa9fbf9 100644 --- a/tests/v2/cases.py +++ b/tests/v2/cases.py @@ -1655,18 +1655,20 @@ def _check_cjk_shape_purity(self) -> None: "row is unchanged in every release. The pair above and " "below it is what makes the count visible", shape=2), - Case("a_declined_ambiguous_pick_stops_the_walk", + Case("a_credential_in_front_anchors_a_declined_pick", "abdul Smith Jr Ma", - {"given": "abdul Smith", "middle": "Jr", "family": "Ma"}, - classification="fix(#289)", + {"given": "abdul", "family": "Smith", "suffix": "Jr Ma"}, + classification="fix(#436/#437)", ambiguities=("suffix-or-name",), - notes="an ACCEPTED cost, pinned rather than repaired " - "(decisions.md#S2): the surname lean breaks the peel " - "AT 'Ma', so the unambiguous 'Jr' in front of it is " - "never reached and becomes a name word. The walk stops " - "at the declined pick rather than continuing past it, " - "and a name-leaning acronym blocking a genuine suffix " - "behind it is the shape that costs", + notes="the cost #289 had accepted, reversed by #544: the " + "surname lean declines 'Ma', but the unambiguous 'Jr' " + "in FRONT of it anchors it, so the peel takes both and " + "P5's reserve, seeing the family the join would take, " + "declines the join. The roles are 1.4.0's and 2.3.0's; " + "what the classification records is the suffix STRING, " + "which 1.4.0 wrote 'Jr, Ma'. The walk still stops at a " + "declined pick with nothing in front of it ('Jack Wei " + "Ma')", shape=1), Case("a_tail_segment_of_leaning_credentials_is_a_run", "Steven Hardman, MD, DO, DDS", diff --git a/tests/v2/pipeline/test_group.py b/tests/v2/pipeline/test_group.py index e01fc463..ccfcfdc6 100644 --- a/tests/v2/pipeline/test_group.py +++ b/tests/v2/pipeline/test_group.py @@ -785,15 +785,14 @@ def test_the_reserve_mirrors_the_bare_acronym_fork() -> None: # now runs that same peel over the view and declines (#425); it # used to count 'Ma' as a name word and join. # - # MOVED by #289, not deleted: 'Ma' is Title-case in a mixed-case - # name, so it leans SURNAME and the peel declines it even with - # words to spare -- the walk stops at the declined pick, 'jr' - # never reached behind it, so the reserve now sees the SAME - # suffixes on both sides of the join (none) and the join stands - # (the accepted cost decisions.md#S2 records for - # 'abdul Smith Jr Ma'). + # MOVED by #289, then back by #544: 'Ma' is Title-case in a + # mixed-case name, so its writing leans SURNAME, but the + # unambiguous 'jr' IN FRONT of it anchors it (the peel's + # `credential_anchors`), so the peel takes both unjoined. Joined, + # the view would be two name pieces and the family would be gone, + # so the reserve declines the join again, as it did before #289. out = _grouped("abdul Smith jr Ma", lexicon=_AMBIGUOUS_LEX) - assert _piece_texts(out) == [["abdul Smith", "jr", "Ma"]] + assert _piece_texts(out) == [["abdul", "Smith", "jr", "Ma"]] def test_the_join_never_turns_a_suffix_into_a_name() -> None: diff --git a/tests/v2/pipeline/test_pieces.py b/tests/v2/pipeline/test_pieces.py index c7d4b5ef..a6876646 100644 --- a/tests/v2/pipeline/test_pieces.py +++ b/tests/v2/pipeline/test_pieces.py @@ -5,17 +5,20 @@ These pin the two contracts that shape cannot reach: a defensive branch no parse can produce, and the stability its readers rest on. """ +import dataclasses from collections.abc import Sequence, Set import pytest +from nameparser import parse from nameparser._lexicon import Lexicon, _normalize from nameparser._pipeline import STAGES from nameparser._pipeline._assign import assign from nameparser._pipeline._classify import classify from nameparser._pipeline._group import group from nameparser._pipeline._pieces import ( - _numeral_behind_the_initial_veto, is_leading_title, leading_titles, + _anchors, _numeral_behind_the_initial_veto, credential_anchors, + credential_at_the_given_slot, is_leading_title, leading_titles, own_words, peel_trailing, peel_walk, segment_suffix_reading, trailing_titles, ) @@ -312,14 +315,179 @@ def test_peel_trailing_keeps_its_two_piece_floor_under_a_lean() -> None: assert peeled.picks == () -def test_peel_trailing_stops_at_a_declined_ambiguous_pick() -> None: - # The accepted cost (decisions.md#S2): the surname lean breaks the - # walk AT 'Ma', so the unambiguous 'Jr' in front of it is never - # reached and becomes a name word. The walk stops at the declined - # pick rather than continuing past it. +def test_peel_trailing_takes_a_declined_pick_an_anchor_stands_before( +) -> None: + # #544 reverses the cost decisions.md#S2 had accepted: the surname + # lean declines 'Ma', but the unambiguous 'Jr' IN FRONT of it + # anchors it, so the walk takes it and reaches 'Jr' too -- and the + # member is still a pick, reported as a counted one is. rest, pieces, ptags, tokens = _peel_inputs("abdul Smith Jr Ma") peeled = peel_trailing(rest, pieces, ptags, tokens, one_case=False) - assert peeled.names == len(rest) + assert peeled.names == len(rest) - 2 + assert len(peeled.picks) == 1 + # a credential BEHIND the member anchors nothing: the walk peels + # 'PhD' and stops at the declined 'Ma', a name word + rest, pieces, ptags, tokens = _peel_inputs("Wang Ma PhD") + peeled = peel_trailing(rest, pieces, ptags, tokens, one_case=False) + assert peeled.names == len(rest) - 1 + + +def test_the_walks_own_leading_piece_never_anchors_what_follows_it( +) -> None: + # #544: a + # bare unambiguous suffix word standing in the walk's OWN leading + # position -- 'PhD'/'Om'/'Jr' are each unambiguous suffix + # vocabulary and also plausible given names/surnames -- must never + # anchor a member behind it, because that leading position is + # always the name H4's carve-out keeps. Anchoring it let the run + # collapse entirely: 'PhD Ma' read given 'PhD', suffix 'Ma', + # losing the family outright, rather than 'PhD Ma' keeping its + # parent reading (given 'PhD', family 'Ma', decided by the count + # alone, as it was before this commit existed). + for text, family in (("Om Ma", "Ma"), ("PhD Ma", "Ma"), + ("Jr Ma", "Ma")): + rest, pieces, ptags, tokens = _peel_inputs(text) + peeled = peel_trailing(rest, pieces, ptags, tokens, one_case=None) + assert peeled.names == len(rest), text + n = parse("Om Ma") + assert (n.given, n.family, n.suffix) == ("Om", "Ma", "") + n = parse("Om Ma Jr") + assert (n.given, n.family, n.suffix) == ("Om", "Ma", "Jr") + n = parse("PhD Ma") + assert (n.given, n.family, n.suffix) == ("PhD", "Ma", "") + # a genuine given name ahead of it frees the SAME word to anchor: + # 'Om' is no longer the walk's own leading piece once 'John' is, + # so it anchors 'Ma' exactly as any other unambiguous credential + # standing in front does + n = parse("John Om Ma") + assert (n.family, n.suffix) == ("", "Om Ma") + + +def test_a_reserve_kept_leading_piece_beside_a_genuine_family_loss( +) -> None: + # decisions.md#S2's Accepted boundary ("an unambiguous suffix is + # consumed even when that leaves no family name at all", 'Smith + # Jr.' -> family='') applies just the same when the LEADING piece + # is itself listed suffix vocabulary rather than an ordinary name: + # 'Om' is the walk's own leading piece and stays given (the + # reserve), but 'Jr' -- NOT the leading piece -- is still consumed + # unconditionally, leaving no family at all. + n = parse("Om Jr") + assert (n.given, n.family, n.suffix) == ("Om", "", "Jr") + # and 'Jr', standing in front of 'Ma' but not itself the leading + # piece, anchors it exactly as any other unambiguous credential + # does -- the boundary costs the family, not the anchor + n = parse("Om Jr Ma") + assert (n.given, n.family, n.suffix) == ("Om", "", "Jr Ma") + + +def _anchored_words(text: str) -> list[str]: + rest, pieces, ptags, tokens = _peel_inputs(text) + return [" ".join(tokens[t].text for t in pieces[i]) + for i, anchored in zip( + rest, credential_anchors(rest, pieces, ptags, tokens)) + if anchored] + + +def test_credential_anchors_reads_in_front_and_through_members() -> None: + # #544: a position is anchored when the run in FRONT of it holds a + # suffix piece that may anchor, and a lone member passes the + # anchor on to the member behind it + assert _anchored_words("John Smith PhD Ma") == ["Ma"] + assert _anchored_words("John Smith PhD Ed Ma") == ["Ed", "Ma"] + # the credential behind anchors nothing + assert _anchored_words("John Smith Ma PhD") == [] + assert _anchored_words("Wang Ma PhD") == [] + + +def test_credential_anchors_stops_at_a_numeral_or_a_name_word() -> None: + # the piece straight behind the credential is marked whatever it + # is; what matters is the member past it, which a single-letter + # numeral (initial-shaped, like a bare middle initial) or a name + # word leaves unanchored. A multi-letter numeral is not this + # shape and anchors like any other suffix piece (pinned below). + assert _anchored_words("John Smith PhD v Ma") == ["v"] + assert _anchored_words("John Smith PhD Jones Ma") == ["Jones"] + assert _anchored_words("John Smith PhD III Ma") == ["III", "Ma"] + + +def test_a_connective_or_a_numeral_suffix_word_anchors_nothing() -> None: + # the generational 'i' is also Catalan's conjunction, and between + # two name words it is a link (rules.md#P3); a single-letter roman + # numeral is INITIAL-SHAPED, the same shape a middle initial + # writes in -- not excluded for being "no credential", since a + # multi-letter numeral ('III') anchors like any other suffix word + # (pinned in `test_credential_anchors_stops_at_a_numeral...` and + # `test_a_multi_letter_numeral_anchors_like_any_suffix_word` + # below). Neither a connective nor a single-letter numeral may + # anchor. + state = _through_group("John Smith Jr Ma") + tokens = list(state.tokens) + jr = next(i for i, t in enumerate(tokens) if t.text == "Jr") + assert _anchors((jr,), tokens) + tokens[jr] = dataclasses.replace( + tokens[jr], tags=tokens[jr].tags | {"conjunction"}) + assert not _anchors((jr,), tokens) + state = _through_group("John Smith v Ma") + v = next(i for i, t in enumerate(state.tokens) if t.text == "v") + assert not _anchors((v,), state.tokens) + # a merged split credential is ONE suffix piece of two tokens, and + # it anchors as the unsplit spelling does wherever it is in the + # walk -- after a comma ('John Smith Ph. D. MEng' is the no-comma + # peel's Accepted limit, the merged piece standing outside its walk) + assert parse("Doe, Jane Ph. D. MEng").suffix == "Ph. D. MEng" + assert parse("Smith, Ph. D. MEng").suffix == "Ph. D. MEng" + + +def test_a_multi_letter_numeral_anchors_like_any_suffix_word() -> None: + # #544: the exclusion is about SHAPE (a single letter + # reads as a middle initial, 'John Smith PhD V Ma' keeping 'V' a + # name word), not about numerals being no credential -- 'III' is + # unambiguous suffix vocabulary of more than one letter and + # anchors exactly as 'Jr' does. + n = parse("John Smith PhD III Ma") + assert (n.family, n.suffix) == ("Smith", "PhD III Ma") + n = parse("John Smith PhD V Ma") + assert (n.middle, n.family, n.suffix) == ("Smith PhD V", "Ma", "") + + +def test_the_anchor_pass_starts_past_the_leading_title_run() -> None: + # a title/suffix dual opening the given part is a title there and + # anchors nothing, at the given slot and in the part read whole + assert parse("Smith, MD MA Ma").middle == "Ma" + assert parse("Smith, Ms Ma").given == "Ma" + # past that position the same dual anchors + assert parse("John Smith MD MEng").suffix == "MD MEng" + + +def test_the_given_slot_asks_the_anchor_only_when_the_writing_declines( +) -> None: + # #544: `credential_at_the_given_slot` takes the anchor as a THUNK + # and asks it last, so a member the writing already settles never + # pays for the anchor pass + calls: list[str] = [] + + def anchored() -> bool: + calls.append("asked") + return True + + tokens = _through_group("Doe, John MA Ma do").tokens + caps, title, particle = ( + next(t for t in tokens if t.text == text) + for text in ("MA", "Ma", "do")) + # the capitals lean credential, and one case leans nothing (the + # count's reading): neither asks + assert credential_at_the_given_slot(caps, False, anchored) + assert credential_at_the_given_slot(title, True, anchored) + assert calls == [] + # a Title-case member and a particle member with no lean decline, + # so the anchor answers -- once each + assert credential_at_the_given_slot(title, False, anchored) + assert credential_at_the_given_slot(particle, True, anchored) + assert calls == ["asked", "asked"] + # with no anchor to ask, the declined reading stands + assert not credential_at_the_given_slot(title, False) + assert not credential_at_the_given_slot(particle, True) def test_a_shape_only_token_reports_without_being_taken() -> None: diff --git a/tests/v2/test_benchmark.py b/tests/v2/test_benchmark.py index ab416f69..f6105823 100644 --- a/tests/v2/test_benchmark.py +++ b/tests/v2/test_benchmark.py @@ -233,6 +233,15 @@ def test_a_thousand_names_still_parse_in_reasonable_time( # frozen loop declines it on the tag # ('i und ' reaches nothing, 'i Und ' # reaches everything) +# S2 ANCHOR pass credential_run ONLY -- a Title-case +# member in a mixed-case name declines on +# its writing, so the peel asks +# `credential_anchors` (#544), and every +# 'Ma' stands behind a 'PhD' that anchors +# it: 39 of the 40 words of 'PhD Ma ' x20 +# read as the suffix. 'PhD MA ' reads the +# same and never asks the pass at all, the +# capitals deciding each member first _SHAPES = { "delimiter_pairs": "(a) ", # extract: matched pairs -> masked spans "quote_pairs": '"a" ', # extract: the open==close path @@ -247,6 +256,7 @@ def test_a_thousand_names_still_parse_in_reasonable_time( "bound_given": "abdul ", # group: the P5 reserve over every piece "maiden_clause": "nee MA ", # group: M2's view over the segment "link_run": "i Und ", # group: P3's both-sides walk (#397) + "credential_run": "PhD Ma ", # pieces: S2's anchor pass (#544) } _BASE = 800 @@ -304,6 +314,13 @@ def test_a_thousand_names_still_parse_in_reasonable_time( # neither number moved. The absolute cost is the shape's own price # and is paid at the top of the clean range: 11.6ms at base 800 # against 48.7ms at 3200. +# The fourteenth (credential_run, #544) was measured against the +# per-member look-behind the one-pass anchor replaces -- each position +# walking back over the whole run in front of it, behavior-identical +# -- which reads 14.8 at base 200, 15.1 at 400 and 15.8 at 800, the +# strongest signal on record. The shape reads 4.13-4.21 at every one +# of the three bases on this tree (py3.11, 2026-09-27), inside the +# clean column; neither number moved. _MAX_RATIO = 6.0 diff --git a/tests/v2/test_parser.py b/tests/v2/test_parser.py index 4225f94b..b0006e4a 100644 --- a/tests/v2/test_parser.py +++ b/tests/v2/test_parser.py @@ -574,13 +574,14 @@ def test_the_reserve_spares_the_family_the_acronym_fork_would_take() -> None: n, m = parse(bound), parse(plain) assert (n.family, n.suffix) == (m.family, m.suffix) assert n.family != "" - # MOVED by #289, not deleted: 'Ed' is Title-case in a mixed-case - # name, so it now leans SURNAME and the peel declines it with - # words to spare -- the walk stops at the declined pick, 'Jr' - # never reached behind it, and both join as name words - # (decisions.md#S2's accepted cost, the 'abdul Smith Jr Ma' shape). - n = parse("abu Bakar Jr Ed") - assert (n.family, n.suffix) == ("Ed", "") + # MOVED by #289, then back by #544: 'Ed' is Title-case in a + # mixed-case name, so its writing leans SURNAME, but the + # unambiguous 'Jr' in front of it anchors it, so the peel takes + # both and the reserve spares the family as for 'abdul Smith Jr + # Ma' above -- the ordinary-given twin reads the same + n, m = parse("abu Bakar Jr Ed"), parse("John Bakar Jr Ed") + assert (n.family, n.suffix) == (m.family, m.suffix) == ("Bakar", + "Jr Ed") # and the join never turns a suffix into a name: unjoined, the # acronym is a credential with words to spare, so 'abdul Smith # Ma' reads as 'John Smith Ma' does (1.4.0 parity restored) @@ -592,6 +593,20 @@ def test_the_reserve_spares_the_family_the_acronym_fork_would_take() -> None: assert (n.family, n.suffix) == (m.family, m.suffix) == ("Ma", "") +def test_the_leading_piece_never_anchors_under_any_name_order() -> None: + # #544: the walk's leading position is the piece the + # H4 carve-out keeps regardless of which ROLE it ends up in, so + # the fix must hold under every name_order -- FAMILY_FIRST puts + # the same 'PhD'/'Om' word in the FAMILY slot instead, and it + # still must not anchor 'Ma' behind it. + for order in (FAMILY_FIRST, FAMILY_FIRST_GIVEN_LAST): + p = Parser(policy=Policy(name_order=order)) + n = p.parse("Om Ma") + assert (n.given, n.family, n.suffix) == ("Ma", "Om", "") + n = p.parse("PhD Ma") + assert (n.given, n.family, n.suffix) == ("Ma", "PhD", "") + + def test_a_joined_pair_is_never_peeled_as_a_title() -> None: # The conjunction merge derives a title tag for 'Sheikh and Ahmad'; # the bound join takes the piece (a mid-name title word is a name diff --git a/tests/v2/test_properties.py b/tests/v2/test_properties.py index a31a51b8..20dd528c 100644 --- a/tests/v2/test_properties.py +++ b/tests/v2/test_properties.py @@ -12,6 +12,7 @@ import itertools import re import warnings +from collections.abc import Callable import pytest from hypothesis import given, settings @@ -27,7 +28,7 @@ from nameparser._pipeline import run from nameparser._pipeline._state import (AMBIGUOUS_ACRONYM_TAG, ParseState) -from nameparser._pipeline._vocab import effective_script +from nameparser._pipeline._vocab import ambiguous_lean, effective_script from nameparser._types import (UNJOINED_CONJUNCTION_TAG, UNJOINED_TAG, AmbiguityKind, ParsedName, Role, Token) @@ -274,6 +275,20 @@ def test_the_comma_agreement_exceptions_are_all_still_exceptions( #: single-policy grid held -- three markers x three policies, and the #: class is indifferent to both, which is the finding. (Before #533 #: the same grid had 186 allowlisted and 984 failing.) +#: +#: The ANCHORED-HEAD class beside it (#544), pinned the same way: a +#: head ending in an unambiguous credential anchors a member written +#: straight after it and not one written after a clause (rules.md#M2's +#: boundary). Recorded 2026-09-27 on the #544 tree: 810 of the 20,412 +#: pairs, 270 for each head ending in a credential ('Jane Doe Jr.', +#: 'Jane Doe PhD', 'Doe, Jane PhD') and every one a Title-case member +#: (a caps member leans credential on both sides and a lower one is +#: counted on both, so neither disagrees); 0 with the anchor off, which +#: is the recorded negative control. The one-case class's 1,026 is +#: unmoved by the two heads #544 added, both being mixed case. +_MAIDEN_ANCHORED_HEAD_EXCEPTIONS = 810 +_MAIDEN_ANCHORED_HEAD_DIGEST = ( + "c5a8f5277e49fe583e0edc834346f623dd02459845ccd6ab9325ecfe7a167c7c") _MAIDEN_AGREEMENT_EXCEPTIONS = 1026 _MAIDEN_AGREEMENT_DIGEST = ( "4b70727a2633fea1a9c219173d48b223cc0866a5d340b69a9eb4f14fc386ae6a") @@ -287,6 +302,186 @@ def _one_case(text: str) -> bool | None: policy=Policy())).one_case +def _qualifying_front(toks: list[tuple[Token, tuple[int, int]]], i: int, + original: str) -> tuple[Token, int] | None: + """rules.md#S2's company clause: walk back from `toks[i]` (a + member of the ambiguous class) through lone members to the piece + that would speak for their company, and return it -- with its OWN + index, which a caller's role question needs -- if it structurally + qualifies: unambiguous suffix vocabulary, not a connective, not a + single-letter roman numeral, and nothing but a comma or a maiden + marker stands between it and `toks[i]` (the company does not + reach across a clause, rules.md#M2). None if no such piece exists + or it fails one of those tests. + + Says NOTHING about either token's ROLE -- not `toks[i]`'s, and not + the front's. That is deliberate: a caller checking whether the + front SPEAKS asks its own role question afterward (a title heading + the part speaks for nothing, 'Smith, Ms Ma'; the piece a name + reserve keeps speaks for nothing either, `_reserve_kept` below), + and the two properties this feeds ask DIFFERENT things of the + SAME front's role -- shared here only where they agree.""" + tok, span = toks[i] + j = i - 1 + while (j >= 0 and AMBIGUOUS_ACRONYM_TAG in toks[j][0].tags + and "shape:acronym" not in toks[j][0].tags): + j -= 1 + if j < 0: + return None + front, front_span = toks[j] + between = original[front_span[1]:span[0]] + letters = front.text.rstrip(".") + if ("," in between or _MARKER_RE.search(between) + or "vocab:suffix" not in front.tags + or "initial" in front.tags + or "conjunction" in front.tags + or (len(letters) == 1 and letters.lower() in "ivx")): + return None + return front, j + + +def _reserve_kept(toks: list[tuple[Token, tuple[int, int]]], j: int, + original: str) -> bool: + """Whether `toks[j]` is the reserve rules.md#H4's no-comma + carve-out keeps regardless of its own vocabulary: the name's + first non-TITLE piece, in a name with NO COMMA anywhere before + it, and no GIVEN/MIDDLE/FAMILY token before it either. 'PhD Ma' + is this shape (the reserve keeps 'PhD'); 'Smith, PhD Ma' is NOT -- + a family already exists from before the comma, so nothing here + is H4's carve-out, whatever `segment_suffix_reading` does with + the given part on its own account (not modelled by this walk; see + the docstring below on where that shape is pinned instead). A + comma anywhere before `toks[j]` therefore answers False outright. + + A maiden clause crossed on the way answers nothing on its own: a + MAIDEN-role token is simply not a name role, so the walk continues + through it exactly as it does through a TITLE, to whatever real + name may stand on the far side of the marker -- 'Jane nee Doe PhD + Ma' finds 'Jane' (GIVEN) past the clause and answers False, PhD + not being the reserve there either.""" + for k in range(j - 1, -1, -1): + tok, span = toks[k] + _, next_span = toks[k + 1] + between = original[span[1]:next_span[0]] + if "," in between: + return False + if tok.role in (Role.GIVEN, Role.MIDDLE, Role.FAMILY): + return False + return True + + +def _outside_its_company(name: ParsedName) -> list[str]: + """rules.md#S2's company clause, read off a finished parse (#544): + a LISTED member of the ambiguous class written behind an + unambiguous credential -- with nothing but listed members between + them, in the same comma part and on the same side of a maiden + marker -- is read as a credential. The word in front must + structurally qualify (`_qualifying_front`) and speak for itself: + not read as a TITLE, which is how the parse reads a dual opening + the given part ('Smith, Ms Ma'), and not the piece a name reserve + keeps regardless of its own vocabulary (`_reserve_kept`: 'PhD Ma' + reads family 'Ma', because 'PhD' is the name the reserve kept, not + a credential run's own member). + + Deliberately does NOT require the front to already be in the + SUFFIX role: that was tried and it blinded the check to exactly + the no-comma half of the defect this exists for -- with the + anchor off, a lost anchor costs the credential IN FRONT its + SUFFIX role too ('Jane Doe Jr. Ma' read middle 'Doe Jr.' before + #544), and a check requiring `front.role is Role.SUFFIX` would + have passed on that very defect by exempting every front it broke. + + The walk this test runs never puts a bare, real family name + before a comma with nothing but a credential run after it + ('Smith, PhD Ma') -- every head either has no comma at all or + already carries a given name across it. That comma-side shape is + pinned separately: `tests/v2/cases.py`'s `family_comma_lone_degree` + and `family_comma_two_credentials` rows, and rules.md#S2's + 'Smith, PhD' boundary example. + + DELIBERATELY A SECOND IMPLEMENTATION of `_pieces.credential_anchors`, + over the finished parse's tokens, tags and roles rather than over + pieces.""" + toks = _placed(name) + out = [] + for i, (tok, span) in enumerate(toks): + if (tok.role is Role.SUFFIX + or AMBIGUOUS_ACRONYM_TAG not in tok.tags + or "shape:acronym" in tok.tags): + continue + qualifies = _qualifying_front(toks, i, name.original) + if qualifies is None: + continue + front, j = qualifies + if front.role is Role.TITLE or _reserve_kept(toks, j, name.original): + continue + out.append(f"{tok.text!r} -> {tok.role.value} behind " + f"{front.text!r}") + return out + + +def _credential_without_suffix_role( + name: ParsedName, + one_case_thunk: Callable[[], bool | None]) -> list[str]: + """The converse of `_outside_its_company` (#544): a + member read as a credential (role SUFFIX) because a qualifying + word stands in front of it must have that FRONT word in the + SUFFIX role too -- the front cannot itself be reserved as the + given or family name while lending its credential-ness to the + member behind it. 'PhD Ma' read given 'PhD', suffix 'Ma', family + '' before this fix: 'Ma' inherited PhD's company while PhD itself + was handed back to the given slot the walk had to reserve, losing + the family entirely. Same `_qualifying_front` walk-back as + `_outside_its_company`, checked the other way; the two properties + then ask DIFFERENT role questions of the front on purpose (see + `_qualifying_front`'s docstring) -- this one does NOT exempt a + front the reserve kept, because that is exactly the shape the + defect took: the reserve-kept front is the one whose SUFFIX role + went missing. + + `one_case_thunk` computes `_one_case(name.original)` -- a full + extra pipeline run -- and is forced only for a SURVIVING + candidate: a member with a qualifying front that is itself + outside both TITLE and SUFFIX, i.e. exactly the shape that is + about to be reported. Every other member is decided by role tests + alone (cheap), so the vast majority of parses -- which have no + such front at all -- never force the thunk; a rare text with more + than one surviving candidate forces it once per candidate rather + than caching, since that shape is itself the exception. + + Skips a member whose OWN case-based lean (rules.md#S2, #289 -- an + ALL-CAPS member of a mixed-case name) already reads it as a + credential with no company at all: that class predates #544 and + reaches the very same H4 carve-out on its own ('PhD MA' -> given + 'PhD', suffix 'MA' on 1.4.0), so it is not this property's + question.""" + toks = _placed(name) + out = [] + for i, (tok, span) in enumerate(toks): + if tok.role is not Role.SUFFIX or AMBIGUOUS_ACRONYM_TAG not in tok.tags: + continue + qualifies = _qualifying_front(toks, i, name.original) + if qualifies is None: + continue + front, _ = qualifies + if front.role is Role.TITLE or front.role is Role.SUFFIX: + continue + # a surviving candidate: only now is the thunk worth its cost + one_case = one_case_thunk() + # inlined `_pieces.listed_lean`'s own gate, over the public + # `Token` rather than the pipeline's `WorkToken` -- the two + # types share `.text`/`.tags`, and this is a second + # implementation on purpose, like the rest of this function + own_lean = (None if (one_case is None + or "shape:acronym" in tok.tags) + else ambiguous_lean(tok.text, one_case)) + if own_lean == "credential": + continue + out.append(f"{front.text!r} -> {front.role.value} in " + f"front of {tok.text!r}, which reads suffix") + return out + + def test_a_maiden_clause_does_not_change_how_a_trailing_word_reads( ) -> None: """#533's invariant: appending a maiden clause to a name must not @@ -305,25 +500,56 @@ def test_a_maiden_clause_does_not_change_how_a_trailing_word_reads( Measured on this tree before #533: 984 of 2016 pairs disagreed outside the allowlist. After: 0, and the allowlist holds exactly its recorded size. + + #544 joins this walk rather than opening a grid: every parse it + takes is also asked `_outside_its_company` and + `_credential_without_suffix_role` (rules.md#S2's company clause, + both directions), and the two heads ending in 'PhD' were added so + that the given slot holds members behind a credential beside the + comma-less trailing slot 'Jane Doe Jr.' already held. A third + head, the bare 'PhD', was added so `_credential_without_suffix_role` + has a shape to catch at all -- none of the other heads is itself a + bare listed-suffix word, so none reaches the walk's own leading + position, which is exactly the shape the converse question is + about. Each text is parsed ONCE -- the plain form once per head + and member, not once per body and marker -- which took the walk + from 3.25s to 2.53s on 3.11 while the grid grew from 18,144 pairs + to 20,412 (measured 2026-09-27, `--durations`; `_credential_without_suffix_role` + forces its own extra parse, `_one_case`, only for a surviving + candidate a qualifying front's role has not already decided, + rather than unconditionally for every parse). + + The `company` and `orphaned` asserts below run BEFORE `failures`, + so each control here trips its OWN assert rather than one masking + another: `assert not company` is the company check's recorded + negative control, `assert not orphaned` the converse's. + RECORDED NEGATIVE CONTROL for `assert not company`: with + `credential_anchors` answering False and `segment_suffix_reading`'s + inline anchor off, it fails on 48 of the walk's parses ('Jane Doe + Jr. Ma' reading family 'Ma'); 0 here. RECORDED NEGATIVE CONTROL + for `assert not orphaned`: with only `credential_anchors`' + leading-position exclusion removed (the position-0 defect this + commit fixes), it fails on 30 of the walk's plain-form parses + ('PhD Ma' reading given 'PhD', suffix 'Ma'); 0 here. """ members = ("ba", "do", "ed", "jd", "ma", "x.y.z.", "r.a.i.") heads = ("Jane Doe", "Doe, Jane", "John", "J.", "Dr.", "Jane", "Jane van der Berg", "JANE DOE", "jane doe", "DOE, JANE", "doe, jane", "Jane Q. Doe", "Doe, Dr. Jane", "Doe, J.", - "Smith, Jane", "Jane Doe Jr.") + "Smith, Jane", "Jane Doe Jr.", "Jane Doe PhD", + "Doe, Jane PhD", "PhD") bodies = ("Smith", "Yo-Yo", "van der Berg", "Jones Smith", "MA", "Ma") # three markers and the two 2.4 switches beside the default: the # switches change WHICH tokens are in the class, and the marker # spellings are what `own_words` stops at, so both are dimensions # the allowlist's structural argument rests on. - markers = ("nee", "n\u00e9e", "geb.") + markers = ("nee", "née", "geb.") policies = (("default", Policy()), ("caps", Policy(unlisted_caps_suffixes=True)), ("nodot", Policy(unlisted_dotted_suffixes=False))) - def side(parser: Parser, text: str, word: str) -> str: - name = parser.parse(text) + def side(name: ParsedName, word: str) -> str: hits = [t for t in name.tokens if t.text == word] if not hits: return "gone" @@ -332,33 +558,94 @@ def side(parser: Parser, text: str, word: str) -> str: pairs = 0 allowed: list[str] = [] + # #544: the ANCHORED-HEAD class. A head that ENDS in an unambiguous + # credential ('Jane Doe Jr.') anchors a member written straight + # after it ('Jane Doe Jr. Ma' reads suffix 'Jr. Ma'), while the + # clause puts its name words between the two ('Jane Doe Jr. nee + # Smith Ma'), and the anchor never reaches across a clause + # (rules.md#M2, decided, a boundary). So the pair differs by + # construction, and in ONE direction only: the clause reads a name, + # the plain form a credential. Admitted only where the parser + # itself gives the head's last token the suffix role and the pair + # disagrees in that direction, and pinned by count and digest like + # the class above. + anchored_head: list[str] = [] + company: list[str] = [] + # #544: the converse of `company` -- a member reading + # SUFFIX because of its company must have that company ALSO in the + # SUFFIX role, never handed back to a given/family reserve while + # still lending its credential-ness behind it + orphaned: list[str] = [] failures = [] for label, policy in policies: parser = Parser(policy=policy) for head in heads: - for body in bodies: - for base in members: - for word in (base.lower(), base.title(), - base.upper()): + head_tokens = parser.parse(head).tokens + head_ends_in_credential = bool(head_tokens) and ( + head_tokens[-1].role is Role.SUFFIX) + for base in members: + for word in (base.lower(), base.title(), base.upper()): + plain = f"{head} {word}" + plain_name = parser.parse(plain) + company += [f"[{label}] {plain!r}: {bad}" for bad + in _outside_its_company(plain_name)] + orphaned += [f"[{label}] {plain!r}: {bad}" for bad + in _credential_without_suffix_role( + plain_name, + lambda: _one_case(plain))] + plain_side = side(plain_name, word) + for body in bodies: for marker in markers: clause = f"{head} {marker} {body} {word}" - plain = f"{head} {word}" + clause_name = parser.parse(clause) + company += [ + f"[{label}] {clause!r}: {bad}" for bad + in _outside_its_company(clause_name)] + orphaned += [ + f"[{label}] {clause!r}: {bad}" for bad + in _credential_without_suffix_role( + clause_name, + lambda: _one_case(clause))] + clause_side = side(clause_name, word) pairs += 1 - if side(parser, clause, word) == side( - parser, plain, word): + if clause_side == plain_side: continue if _one_case(clause) and not _one_case(plain): allowed.append( f"{label}|{marker}|{clause}") continue + if (head_ends_in_credential + and clause_side == "name" + and plain_side == "credential"): + anchored_head.append( + f"{label}|{marker}|{clause}") + continue failures.append( f"[{label}] {clause!r} reads " - f"{side(parser, clause, word)} but " - f"{plain!r} reads " - f"{side(parser, plain, word)}") + f"{clause_side} but {plain!r} reads " + f"{plain_side}") + # company and orphaned assert BEFORE failures: each is its own + # control (the anchor-off/reserve-regression negative controls + # this test's docstring records), and a pair-agreement failure + # elsewhere in the grid must never mask either one + assert not company, ( + f"{len(company)} member(s) read as a name behind an " + f"unambiguous credential:\n" + "\n".join(company[:15])) + assert not orphaned, ( + f"{len(orphaned)} member(s) read as a credential behind a " + f"word not itself in the suffix role:\n" + + "\n".join(orphaned[:15])) assert not failures, ( f"{len(failures)} of {pairs} pair(s) disagree outside the " f"one-case-head class:\n" + "\n".join(failures[:15])) + anchored_digest = hashlib.sha256( + "\n".join(sorted(anchored_head)).encode()).hexdigest() + assert (len(anchored_head), anchored_digest) == ( + _MAIDEN_ANCHORED_HEAD_EXCEPTIONS, _MAIDEN_ANCHORED_HEAD_DIGEST), ( + f"the anchored-head class holds {len(anchored_head)} of {pairs} " + f"pairs with digest {anchored_digest}. Re-record both " + f"deliberately, saying why. Members:\n" + + "\n".join(sorted(anchored_head)[:10])) digest = hashlib.sha256( "\n".join(sorted(allowed)).encode()).hexdigest() assert (len(allowed), digest) == ( From 9e37aaddd58038a1707f764bb2ac87f5107b526f Mon Sep 17 00:00:00 2001 From: Derek Gulbranson Date: Sun, 27 Sep 2026 17:36:37 -0700 Subject: [PATCH 04/11] test(#544): case rows, rules.md examples, corpora and ledgers Twenty-four case rows: the issue's targets in each comma shape, and the boundaries each half keeps -- a credential BEHIND the member ('Wang Ma PhD'), a dual opening the given part ('Smith, Ms Ma', 'Smith, MD MEng'), a numeral in front ('smith, v ed'), the run the capitals settle ('John Smith, PhD MA'), a by-shape member ('John Smith, X.Y.Z. MA'), the maiden-clause boundary ('Jane Doe Jr. nee Smith Ma') and the three limits rules.md#S2 accepts. rules.md gains the C1, S2 and M2 example lines; the corpora are rebuilt. Ledgers: a rule is named for the change that moves the name against its baseline. fix(#544) rules classify the names #544 moves; #540's comma-run and chunked-dotted rules are retired, the comma-run names now the fix(#544) rules' and 'Wang M.Eng.', whose roles every release from 2.0 read, feat(#449)'s report rule at 2.0-2.2 and the dotted fix(suffix-routing) rule beside 'Jack M.A.' at 1.4.0. Names whose remaining diff is R1's spacing join the fix(#436/#437) alternations, and where the spacing rides with another change's diff the label is joint: the anchor rule is fix(#436/#437/#544) at 2.0-2.2, and fix(#425), taking `given` back at 2.0/2.1 as 'abdul Smith Jr Ma' leaves the fix(#289) ones, is fix(#425/#436/#437). At 1.4.0 the maiden-clause names #544 moves no role on join the rules for the changes that do (fix(#274/#436/#437) for the two PhD MEng clauses, fix(#274/#424) for 'Jane Doe Jr. nee Smith Ma', which joins fix(#533)'s report rule at 2.x). 'Smith, PhD MEng' is fix(#325)'s at 1.4.0, pinned beside 'Smith, PhD Jr.' with the same argument. Gate: exit 0 at all five baselines, radar counts unchanged. Co-Authored-By: Claude Opus 5.5 --- docs/design/rules.md | 27 +- tests/v2/cases.py | 252 +++++++++++ tests/v2/test_ledger_guards.py | 451 ++++++++++++++----- tools/differential/compare.py | 4 + tools/differential/corpus_rules.jsonl | 12 + tools/differential/corpus_shapes.jsonl | 18 + tools/differential/expected_since_1.4.0.toml | 228 ++++++---- tools/differential/expected_since_2.0.0.toml | 221 ++++++--- tools/differential/expected_since_2.1.0.toml | 221 ++++++--- tools/differential/expected_since_2.2.0.toml | 200 +++++--- tools/differential/expected_since_2.3.0.toml | 169 ++++--- 11 files changed, 1341 insertions(+), 462 deletions(-) diff --git a/docs/design/rules.md b/docs/design/rules.md index f9426d96..cbbd41d7 100644 --- a/docs/design/rules.md +++ b/docs/design/rules.md @@ -1069,6 +1069,13 @@ S2. Rationale: generational suffixes and credentials are recognized "john smith MEng" → family="MEng" "Nguyen Van Lac" → family="Van Lac" "Wang M.Eng." → suffix="M.Eng." + "John Smith PhD MEng" → suffix="PhD MEng" + "John Smith PhD Ed Ma" → suffix="PhD Ed Ma" + "Wang Ma PhD" → family="Ma" · boundary + "Smith, PhD MEng" → suffix="PhD MEng" + "Smith, Ms Ma" → given="Ma" · boundary + "Doe, Jane PhD MEng" → suffix="PhD MEng" + "doe, jane v phd do" → suffix="v phd do" "Smith, MA" → suffix="MA" "Smith, Ma" → given="Ma" "Doe, John MA" → suffix="MA" @@ -1114,7 +1121,14 @@ S2. Rationale: generational suffixes and credentials are recognized member as the name: a credential written split across two words with no comma after the name, a title standing between the credential and the member, and a member a particle chain (P2) - has already taken. + has already taken. Each reading of the three is older than the + company clause, so they are witnessed by tests/v2/cases.py's + a_split_degree_in_front_is_out_of_the_walk, + a_title_between_degree_and_member_keeps_it_a_name and + a_member_the_particle_chain_took_is_out_of_reach rather than by + lines here, which would bring each name into the corpus that + enforces them at released baselines for a reading this clause + did not make. "Jack Wei Ma" → family="Ma" "Jack Wei Ma" → ambiguities=("suffix-or-name",) "abdul Smith Jr Ma" → suffix="Jr Ma" @@ -1521,6 +1535,13 @@ M2. Rationale: a maiden marker announces that what follows it is the there goes to the family. Open: #548. "Doe nee Smith Jr. Prof., Jane" → family="Doe Prof." "Doe nee Smith V, Jane" → family="Doe V" + Accepted: a credential written in front of the marker speaks for + no word of the clause (S2's company), the clause's own words + standing between the two, so the clause keeps a member its + writing declines even where the same name written without the + clause reads that member as the credential. + "Jane Doe Jr. nee Smith Ma" → maiden="Smith Ma" + "Jane Doe Jr. Ma" → suffix="Jr. Ma" · boundary history: decisions.md#M2 · interacts: P2, P3, P5, P6, R1, R2, M1, S1, S2, H1, H5 · implemented: nameparser/_pipeline/_group.py M3. Rationale: an enclosure says nothing about whether it means @@ -1699,6 +1720,10 @@ C1. Rationale: a credential run after the comma means the name is in "Royce, Ed" → given="Ed" · boundary "Smith Jr., MA" → suffix="Jr., MA" "Smith Jr., Ma" → given="Ma" · boundary + "John Smith, PhD MEng" → suffix="PhD MEng" + "John Smith, Ed Ma" → suffix="Ed Ma" + "Jane Doe, MS LAc" → suffix="MS LAc" + "Smith, PhD MEng" → family="Smith" · boundary "John Smith, A.B." → suffix="A.B." "John Smith, A.B." unlisted_dotted_suffixes-off → given="A.B." "Smith, A.B." → given="A.B." · boundary diff --git a/tests/v2/cases.py b/tests/v2/cases.py index 1fa9fbf9..1a7e4bf7 100644 --- a/tests/v2/cases.py +++ b/tests/v2/cases.py @@ -761,6 +761,258 @@ def _check_cjk_shape_purity(self) -> None: "reports from, and a leading word is none of them. No " "report, and every release read it this way", shape=1), + # ---- #544: a credential run keeps its ambiguous members --------- + # C1's name-word count reads a RUN after the comma as it reads one + # word, and S2's company clause lets an unambiguous credential IN + # FRONT of a member speak for it at every trailing slot. The rows + # below are the issue's names and the boundaries each half keeps. + Case("comma_run_of_title_case_members_is_the_credential_run", + "John Smith, Ed Ma", + {"given": "John", "family": "Smith", "suffix": "Ed Ma"}, + ambiguities=("suffix-or-name",), + notes="C1's run rule with no unambiguous word in it: two " + "members, both Title case, and two name words before " + "the comma, so the count reads the part as the " + "credential run and the flip reports once over the " + "whole part. 1.4.0 read the same; 2.0.0 through 2.3.0 " + "read given 'Ed', middle 'Ma'", + shape=3), + Case("comma_run_led_by_a_member_is_the_credential_run", + "John Smith, MEng PhD", + {"given": "John", "family": "Smith", "suffix": "MEng PhD"}, + ambiguities=("suffix-or-name",), + notes="a member OPENING the run: the count, not the company, " + "decides after a comma, so the order of the words does " + "not matter. 1.4.0 and 2.3.0 read the same; #540 had " + "read given 'MEng', family 'John Smith'", + shape=3), + Case("comma_run_opened_by_a_dual_is_the_credential_run", + "Jane Doe, MS LAc", + {"given": "Jane", "family": "Doe", "suffix": "MS LAc"}, + ambiguities=("suffix-or-name",), + notes="the issue's headline: 'MS' is title and suffix " + "vocabulary both, and opening the part after a FULL " + "name it counts as the suffix word it is, as the " + "legacy test already counts it in 'Jane Doe, MS PhD'. " + "1.4.0 and 2.3.0 read the same; #540 had read title " + "'MS', given 'LAc', family 'Jane Doe', silently", + shape=3), + Case("comma_run_opened_by_a_dual_before_a_meng", + "John Smith, MD MEng", + {"given": "John", "family": "Smith", "suffix": "MD MEng"}, + ambiguities=("suffix-or-name",), + notes="the same dual after a full name, before the other " + "chunked member. 1.4.0 and 2.3.0 read the same suffix; " + "#540 had read title 'MD', given 'MEng'", + shape=3), + Case("comma_run_in_one_case_is_the_credential_run", + "john smith, md ma", + {"given": "john", "family": "smith", "suffix": "md ma"}, + ambiguities=("suffix-or-name",), + notes="one case says nothing, and the count decides as it " + "does in mixed case. 1.4.0 read the same; 2.3.0 read " + "title 'md', given 'ma', family 'john smith', silently", + shape=3), + Case("comma_run_opened_by_an_honorific_dual_is_the_accepted_cost", + "John Smith, Ms Ma", + {"given": "John", "family": "Smith", "suffix": "Ms Ma"}, + ambiguities=("suffix-or-name",), + notes="the accepted cost of reading a dual opening the part " + "as its suffix: 'Ms' is the honorific here, read as " + "'John Smith, Ms' alone already reads it, and the flip " + "reports so a caller can see the call (Derek, " + "2026-09-27). 1.4.0 read the same; 2.3.0 read title " + "'Ms', given 'Ma'. With ONE name word before the comma " + "the dual is a title ('Smith, Ms Ma')", + shape=3), + Case("comma_run_the_capitals_already_settle_is_not_flipped", + "John Smith, PhD MA", + {"given": "John", "family": "Smith", "suffix": "PhD MA"}, + notes="every member written in capitals in a mixed-case name " + "leans credential, so the listing form already reads " + "the part as the credential run and the run rule stands " + "down: no flip, and no report the part did not already " + "make. Pinned because the reading now rests on the " + "family-comma path alone. 1.4.0 read the same; 2.3.0 " + "read given 'PhD', middle 'MA'"), + Case("comma_run_by_shape_member_takes_the_count", + "John Smith, X.Y.Z. MA", + {"given": "John", "family": "Smith", "suffix": "X.Y.Z. MA"}, + classification="fix(#544)", + ambiguities=("suffix-or-name",), + notes="the lean that lets a settled run stand down is the " + "LISTED set's alone (S2): 'X.Y.Z.' joins the class by " + "shape and carries no writing to read, so the run is " + "the count's and flips though both words are in " + "capitals. 1.4.0 read given 'X.Y.Z.', family 'John " + "Smith', suffix 'MA'", + shape=3), + Case("a_degree_in_front_anchors_a_trailing_meng", + "John Smith PhD MEng", + {"given": "John", "family": "Smith", "suffix": "PhD MEng"}, + classification="fix(#436/#437)", + ambiguities=("suffix-or-name",), + notes="S2's company clause: 'MEng' is written the way the " + "name lean reads, but the unambiguous 'PhD' in front " + "of it anchors it, so the peel takes both and reports " + "the member it picked. The roles are 1.4.0's and " + "2.3.0's; the classification records the suffix " + "STRING, 1.4.0's 'PhD, MEng'. #540 had read middle " + "'Smith PhD', family 'MEng'", + shape=1), + Case("a_degree_in_front_anchors_a_trailing_ma", + "John Smith PhD Ma", + {"given": "John", "family": "Smith", "suffix": "PhD Ma"}, + classification="fix(#436/#437)", + ambiguities=("suffix-or-name",), + notes="the same company for a member marked since 2.0: the " + "#289 lean had read middle 'Smith PhD', family 'Ma' " + "(bisected to dfb31709). 1.4.0 and 2.3.0 read the " + "roles; 1.4.0 wrote the suffix 'PhD, Ma'", + shape=1), + Case("the_anchor_passes_through_a_run_of_members", + "John Smith PhD Ed Ma", + {"given": "John", "family": "Smith", "suffix": "PhD Ed Ma"}, + classification="fix(#436/#437)", + ambiguities=("suffix-or-name", "suffix-or-name"), + notes="a member behind a member behind the degree: the " + "first passes the anchor on, and each reports as a " + "pick. 1.4.0 and 2.3.0 read the roles; 1.4.0 wrote " + "'PhD, Ed, Ma'", + shape=1), + Case("a_degree_behind_the_member_anchors_nothing", + "Wang Ma PhD", + {"given": "Wang", "family": "Ma", "suffix": "PhD"}, + ambiguities=("suffix-or-name",), + notes="the company is IN FRONT only: 'PhD' behind 'Ma' says " + "nothing about it, so the Title-case member is the " + "family name and the degree is peeled on its own. " + "1.4.0 and 2.3.0 read the same roles", + shape=1), + Case("the_anchor_reads_inside_the_maiden_clause", + "Jane Doe nee Smith PhD MEng", + {"given": "Jane", "family": "Doe", "suffix": "PhD MEng", + "maiden": "Smith"}, + classification="fix(#544)", + ambiguities=("suffix-or-name",), + notes="the maiden walk reads the end of the clause through " + "the same peel, so the anchored member ends it too and " + "the clause keeps 'Smith'. 2.3.0 read the same; #540 " + "had read middle 'Doe PhD', family 'MEng'. 1.4.0 had " + "no maiden routing", + shape=1), + Case("a_title_between_degree_and_member_keeps_it_a_name", + "John Smith PhD Prof. Ma", + {"given": "John", "middle": "Smith PhD Prof.", "family": "Ma"}, + classification="fix(#289)", + ambiguities=("suffix-or-name",), + notes="rules.md#S2's Accepted limit: a title standing between " + "the credential and the member ends the run, so the " + "member's writing decides and the degree and title are " + "name text. Unchanged by #544; 2.3.0 read title 'Prof.', " + "suffix 'PhD Ma', and 1.4.0 middle 'Smith PhD', last " + "'Prof.', suffix 'Ma'"), + Case("a_split_degree_in_front_is_out_of_the_walk", + "John Smith Ph. D. MEng", + {"given": "John", "middle": "Smith", "family": "MEng", + "suffix": "Ph. D."}, + classification="fix(#540)", + ambiguities=("suffix-or-name",), + notes="rules.md#S2's Accepted limit: the merged 'Ph. D.' is " + "read as a suffix at any position and so is not in the " + "no-comma walk the company is read over, and the " + "member's writing decides. The comma and given-slot " + "spellings do reach it ('Doe, Jane Ph. D. MEng' reads " + "suffix 'Ph. D. MEng'). Unchanged by #544; 2.3.0 read " + "suffix 'Ph. D. MEng', and 1.4.0 'Ph. D., MEng'"), + Case("a_member_the_particle_chain_took_is_out_of_reach", + "John Smith PhD Do Do", + {"given": "John", "middle": "Smith PhD", "family": "Do Do"}, + notes="rules.md#S2's Accepted limit: 'Do' is particle " + "vocabulary too, and P2's chain joins 'Do Do' into one " + "piece before the peel, so no lone member stands behind " + "the degree. Unchanged by #544; 1.4.0 and 2.3.0 read " + "the same"), + Case("the_anchor_does_not_reach_across_a_maiden_clause", + "Jane Doe Jr. nee Smith Ma", + {"given": "Jane", "family": "Doe", "suffix": "Jr.", + "maiden": "Smith Ma"}, + classification="fix(#533)", + ambiguities=("suffix-or-name",), + notes="rules.md#M2's boundary: the clause's name words stand " + "between 'Jr.' and 'Ma', so the credential in front " + "speaks for nothing past the marker and the clause keeps " + "its member, where 'Jane Doe Jr. Ma' reads suffix 'Jr. " + "Ma'. Unchanged by #544; 2.3.0 read the same fields " + "without the report, and 1.4.0 had no maiden routing", + shape=1), + Case("a_degree_after_a_one_word_family_anchors_the_member", + "Smith, PhD MEng", + {"family": "Smith", "suffix": "PhD MEng"}, + classification="fix(#544)", + notes="one name word before the comma, so C1 reads the " + "listing form, and the part holds no name word once " + "'PhD' anchors 'MEng': the credential run, whole, with " + "no report, 'PhD' not being a member. 2.3.0 read the " + "same; #540 had read given 'PhD', middle 'MEng', and " + "1.4.0 title 'PhD', first 'MEng'", + shape=2), + Case("a_degree_in_the_given_part_anchors_the_member", + "Doe, Jane PhD MEng", + {"given": "Jane", "family": "Doe", "suffix": "PhD MEng"}, + classification="fix(#436/#437)", + ambiguities=("suffix-or-name",), + notes="the given part's trailing slot (#531) asks the " + "company once the member's writing declines. 2.3.0 read " + "the same fields; #540 had read middle 'MEng', and " + "1.4.0 wrote the suffix 'PhD, MEng'", + shape=2), + Case("the_anchor_reads_inside_a_clause_after_a_comma", + "Doe, Jane nee Smith PhD MEng", + {"given": "Jane", "family": "Doe", "suffix": "PhD MEng", + "maiden": "Smith"}, + classification="fix(#544)", + ambiguities=("suffix-or-name",), + notes="the maiden walk's given-slot reader asks the same " + "company over the name the take would leave. 2.3.0 " + "read the same; #540 had read middle 'MEng', and 1.4.0 " + "had no maiden routing", + shape=2), + Case("a_dual_opening_the_given_part_is_a_title_there", + "Smith, Ms Ma", + {"title": "Ms", "given": "Ma", "family": "Smith"}, + notes="the dual exclusion's scope: with ONE name word before " + "the comma 'Ms' opens the given part, reads as the " + "title there, and speaks for nothing, so 'Ma' stays the " + "given name. 1.4.0 and 2.3.0 read the same", + shape=2), + Case("a_dual_opening_the_given_part_anchors_nothing", + "Smith, MD MEng", + {"title": "MD", "given": "MEng", "family": "Smith"}, + notes="the same exclusion for the other chunked member. " + "1.4.0 read the same; 2.3.0 read suffix 'MD MEng', " + "the reading #540's marking moved"), + Case("a_numeral_in_front_anchors_nothing", + "smith, v ed", + {"given": "v", "family": "smith", "suffix": "ed"}, + ambiguities=("suffix-or-name",), + notes="a single-letter roman numeral is a generation, not a " + "credential, so it speaks for nothing: 'ed' is read by " + "the given slot's own rule. 1.4.0 read the same; 2.3.0 " + "read middle 'ed'"), + Case("an_anchored_particle_member_outranks_p6", + "doe, jane v phd do", + {"given": "jane", "family": "doe", "suffix": "v phd do"}, + classification="fix(#436/#437)", + ambiguities=("suffix-or-name",), + notes="'do' is a member AND a particle, and written in one " + "case it leans nothing, so P6's attachment would take " + "it (as in 'NASCIMENTO, EDSON ARANTES DO'); the degree " + "in front outranks the attachment as the capitals do " + "(Derek, 2026-09-27). The roles are 1.4.0's; 2.3.0 " + "read family 'do doe', and 1.4.0 wrote the suffix " + "'v, phd, do'", + shape=2), Case("by_design_trailing_mc_reads_as_a_credential", "Donald Mc", {"given": "Donald", "suffix": "Mc"}, classification="fix(suffix-routing)", diff --git a/tests/v2/test_ledger_guards.py b/tests/v2/test_ledger_guards.py index 6bfdefeb..ad82b5cd 100644 --- a/tests/v2/test_ledger_guards.py +++ b/tests/v2/test_ledger_guards.py @@ -1000,8 +1000,15 @@ def test_case_shape_ids_exist_in_the_inventory() -> None: # the name is claimed by `fix(#289) a written case contrast # decides a bare ambiguous acronym`, and this rule, anchored to # the DOTTED spelling, must still never reach it. - "fix(suffix-routing) the dotted M.A. spelling reads as a credential (ma-do)": - ("Jack Ma", "Jack MA", "John Smith M.A."), + # 2026-09-27 (#544): 'Wang M.Eng.' joined the 1.4.0 rule, anchored on + # 'wang' for the reason 'John Smith M.A.' is a probe: the three-token + # 'John Smith M.Eng.' does not diff there. 'Wang Ma.' is the single + # trailing period the gate still does not count. Relabelled the same + # day for the shape both names share, from 'the dotted M.A. + # spelling reads as a credential (ma-do)'. + "fix(suffix-routing) a two-token name ending in a dotted credential spelling keeps it in `suffix`": + ("Jack Ma", "Jack MA", "John Smith M.A.", "John Smith M.Eng.", + "Wang Ma.", "Dr. Wang M.Eng."), # #484: the connective rule is case-sensitive on the single letters # so that the #462 shapes -- a capital or dotted E that is an # INITIAL the facade drops -- are never claimed as the per-word @@ -1130,7 +1137,7 @@ def test_case_shape_ids_exist_in_the_inventory() -> None: # against it. "feat(#449) a lone name word reports given-or-family": ("Dr. Andrew", "Andrew Smith", "Smith, Andrew", "abdul", "de", - "Dr. Smith Jr.", "J"), + "Dr. Smith Jr.", "J", "Dr. Wang M.Eng."), "feat(#449) a wholly-katakana name keeps the declared order, so the convention decides it": ("マイケル ジャクソン", "マイケル・ジャクソン", "Dr. マイケル"), "feat(#449) an interpunct transcription declines the script order, so the convention decides it": @@ -1209,8 +1216,11 @@ def test_case_shape_ids_exist_in_the_inventory() -> None: "change(suffix-acronym-collisions) ph leaves the acronym set": ("John Smith Ph. D.", "Smith, Ph. D.", "john smith phd", "John Smith Ph.D."), - # #540's eight rules, keyed on the full issue since all eight - # carry `fix(#540)`. Each wall is the other rules' names plus the + # #540's rules, keyed on the full issue since all of them carry + # `fix(#540)` -- eight when written, six since #544 (2026-09-27) + # read the comma run and the chunked dotted spelling again, which + # retired the comma-cost rule the sentences below still name. + # Each wall is the other rules' names plus the # spellings a case-blind widening would reach: 'meng li' leads # with the word and nothing moved; the one-case 'john smith meng' # and 'JOHN SMITH MENG' keep the credential, so the cost rule has @@ -1236,12 +1246,31 @@ def test_case_shape_ids_exist_in_the_inventory() -> None: ("John Smith, MEng", "Smith, John meng", "Dr. Smith, John MEng"), "fix(#540) a lone meng or MEng after a family comma reads as the given name": ("Smith, MENG", "Smith, John MEng", "Dr. Smith, meng"), - "fix(#540) accepted: a credential run behind a suffix comma that ends in meng re-reads the comma as a family comma": - ("John Smith, MEng", "John Smith, PhD", "Dr. John Smith, PhD MEng"), "fix(#540) a Title-case Lac behind a particle is the family name": ("nguyen van lac", "Dr. Nguyen Van Lac"), - "fix(#540) accepted: a chunked dotted M.Eng. with nothing to spare is the family name": - ("John Smith M.Eng.", "Wang M.A.", "Dr. Wang M.Eng."), + # #544's rules (2026-09-27), which retired #540's comma-run and + # chunked-dotted pair, keyed the same way. The chunked-dotted name, + # 'Wang M.Eng.', is feat(#449)'s at the 2.x baselines and + # fix(suffix-routing)'s at 1.4.0, #544 returning the reading every + # release had, so no #544 rule claims it. Each wall is the + # boundary its rule's comment argues -- the capitals-settled run + # that reports nothing new, one name word before the comma, the + # dual opening the given part, the credential BEHIND the member -- + # plus one superstring probe per literal rule, for the + # anchor-dropping widening the #540 block above records. The + # anchor rule's key drops its tag: it is fix(#436/#437/#544) at + # 2.0.0 through 2.2.0 and fix(#544) at 2.3.0. + "fix(#544) a credential run after a suffix comma keeps the comma, and the flip reports": + ("John Smith, PhD MA", "Smith, PhD MEng", + "Dr. John Smith, PhD MEng"), + "fix(#544) a run of ambiguous members after a suffix comma reads by the name-word count": + ("Smith, Ed Ma", "Smith, Ms Ma", "Dr. John Smith, Ed Ma", + "Dr. John Smith, X.Y.Z. MA"), + "#544) an unambiguous credential in front anchors the member behind it": + ("Wang Ma PhD", "John Smith MEng PhD", "Dr. John Smith PhD MEng"), + "fix(#544) a degree in front outranks P6's attachment for a particle member": + ("doe, jane do", "NASCIMENTO, EDSON ARANTES DO", + "Dr. doe, jane v phd do"), # The esq boundary is every spelling SUFFIX_WORDS still carries, # in each of the three positions the corpora write it in. "change(suffix-acronym-collisions) esq leaves the acronym set": @@ -1444,12 +1473,14 @@ def test_case_shape_ids_exist_in_the_inventory() -> None: "fix(#274/#424) accepted: a maiden clause keeps a trailing credential": ("Jane Doe nee Smith MA", "Doe, Jane nee Smith MA", "Jane Doe nee MA Smith", "Jane Doe nee Smith V", - "John Smith MA"), + "John Smith MA", "Jane Doe Jr. Ma", + "Dr. Jane Doe Jr. nee Smith Ma"), # ...and the name whose clause an UNAMBIGUOUS credential ended # must not reach the ambiguous ones, in either word order. "fix(#274/#436/#437) a clause the unambiguous credential ended": ("Jane Doe nee Smith MA PhD", "Jane Doe nee Smith MA", - "Jane Doe nee Smith PhD", "John Doe PhD MA"), + "Jane Doe nee Smith PhD", "John Doe PhD MA", + "Dr. Jane Doe nee Smith PhD MEng"), # 1.4.0's #533 half must not reach the pre-existing one, and the # capitals are the evidence: 'Jane Doe nee Smith Ma JD' is in the # rule and 'Jane Doe nee Smith Ma' is not. @@ -2395,23 +2426,19 @@ class _LatinCopy(NamedTuple): # dotted shape would reach 'Jack M.A.', 'John Smith Ph.D.' and # 'Doe, John Msc.Ed.', which the vocabulary answers for and this # change does not touch. _MUST_NOT_MATCH carries both rosters. - frozenset({"Davis Royce, Ed", "Freiherr von Berg MA", - "JOHN SMITH, MA", "Jack MA", r"Jack MA\.", "Jack Wei Ma", + frozenset({"Davis Royce, Ed", "Freiherr von Berg MA", "JOHN SMITH, MA", + "Jack MA", r"Jack MA\.", "Jack Wei Ma", "John Smith Ma", + "John Smith, Ed", "John Smith, MA", "John Smith, Ma", + "John de Ma", "John van der Berg Ma", r"Smith Jr\., MA", + r"Smith Jr\., Ma", "Smith, MA", "abdul Smith Berg Ma", + "abdul Smith Ma", "john smith, ma"}), + frozenset({"Davis Royce, Ed", r"Doe, Dr\. MA", "Doe, MA", "Doe, MA PhD", + r"Doe, Mr\. MA PhD", "Freiherr von Berg MA", "JOHN SMITH, MA", + "Jack MA", r"Jack MA\.", "Jack Wei Ma", r"John Prof\. MA", "John Smith Ma", "John Smith, Ed", "John Smith, MA", "John Smith, Ma", "John de Ma", "John van der Berg Ma", r"Smith Jr\., MA", r"Smith Jr\., Ma", "Smith, MA", - "abdul Smith Berg Ma", "abdul Smith Jr Ma", - "abdul Smith Ma", "john smith, ma"}), - frozenset({"Davis Royce, Ed", r"Doe, Dr\. MA", "Doe, MA", - "Doe, MA PhD", r"Doe, Mr\. MA PhD", - "Freiherr von Berg MA", - "JOHN SMITH, MA", "Jack MA", r"Jack MA\.", "Jack Wei Ma", - r"John Prof\. MA", "John Smith Ma", "John Smith, Ed", - "John Smith, MA", "John Smith, Ma", "John de Ma", - "John van der Berg Ma", - r"Smith Jr\., MA", r"Smith Jr\., Ma", "Smith, MA", - "abdul Smith Berg Ma", "abdul Smith Jr Ma", - "abdul Smith Ma", "john smith, ma"}), + "abdul Smith Berg Ma", "abdul Smith Ma", "john smith, ma"}), frozenset({r"Jack X\.Y\.I\.", r"John Smith B\.Tech\.", r"John Smith C\.H\.A\.", r"John Smith E\.S\.Q\.", r"John Smith Q\.W\.E\.R\.T\.", r"John Smith X\.Y\.Z\.", @@ -2480,9 +2507,10 @@ class _LatinCopy(NamedTuple): # vocabulary only participates in negatively (a lone name word NO # vocabulary claimed), and a member copying any wordlist would # reach names that report nothing: 'abdul' is bound-given - # vocabulary and 'de' a particle, and neither moves. Seven members + # vocabulary and 'de' a particle, and neither moves. Eight members # carry a trailing suffix -- 'Smith Jr.', "'Smitty' Jones Jr.", - # 'John V', 'Jack M.A.', 'Carod i', 'Donald mc', 'Mohamad X' -- + # 'John V', 'Jack M.A.', 'Carod i', 'Donald mc', 'Mohamad X', and + # 'Wang M.Eng.' (joined 2026-09-27, #544) -- # where the peel took the suffix and the convention placed the ONE # name word left, so a suffix wordlist is not what this list copies # either. One set, identical in all three 2.x ledgers. @@ -2491,7 +2519,7 @@ class _LatinCopy(NamedTuple): "Duke of Wellington", "Garcia", r"Jack M\.A\.", "John & Jane", "John V", "John of the Doe", "Juan & Garcia", "Juan and Garcia", "Mohamad X", - "Smith", r"Smith Jr\.", "e and e", + "Smith", r"Smith Jr\.", r"Wang M\.Eng\.", "e and e", "part1 of The part2 of the part3 and part4", "part1 of and The part2 of the part3 And part4", "test", "سلمان،"}), @@ -2607,7 +2635,6 @@ class _LatinCopy(NamedTuple): # lists above). Neither copies SUFFIX_ACRONYMS_AMBIGUOUS -- same # reason as the rest of this roster. frozenset({"John Smith MEng PhD", "john smith MEng"}), - frozenset({"John Smith, PhD MEng", "john smith, phd meng"}), # fix(#462)'s letter shape: a bare capital E/Y or a dotted E./Y. # It is the initial SHAPE (v1's `initial` regex, _render._INITIAL) # intersected with the single-letter conjunctions, not a copy of @@ -2798,10 +2825,10 @@ class _LatinCopy(NamedTuple): # Accepted example of a title in front of a kept credential, # which #535 moves nothing on, frozenset({"Berg, Jane van der nee Smith DO", "Berg, abdul nee Jones MA", - r"Doe nee Smith Prof\. ba", - r"Doe, Dr\. nee Smith MA", "Doe, Jane nee Smith Do", - "Doe, Jane nee Smith MA do", "Doe, Jane nee Smith Ma", - "Doe, Jane nee Smith do", "JOHN NEE JONES SMITH MA PHD", + r"Doe nee Smith Prof\. ba", r"Doe, Dr\. nee Smith MA", + "Doe, Jane nee Smith Do", "Doe, Jane nee Smith MA do", + "Doe, Jane nee Smith Ma", "Doe, Jane nee Smith do", + "JOHN NEE JONES SMITH MA PHD", r"Jane Doe Jr\. nee Smith Ma", "Jane Doe nee MA", "Jane Doe nee MA PhD", "Jane Doe nee Smith DO DO", "Jane Doe nee Smith Ma", "Jane Doe nee Yo-Yo Ma", "John née Jones Smith Ma"}), @@ -2812,31 +2839,33 @@ class _LatinCopy(NamedTuple): r"Doe, Dr\. nee Smith MA", "Doe, Jane nee Smith Do", "Doe, Jane nee Smith MA do", "Doe, Jane nee Smith Ma", "Doe, Jane nee Smith do", "JOHN NEE JONES SMITH MA PHD", - r"Jane Doe nee King\. ba", + r"Jane Doe Jr\. nee Smith Ma", r"Jane Doe nee King\. ba", "Jane Doe nee MA", "Jane Doe nee MA PhD", "Jane Doe nee Smith DO DO", "Jane Doe nee Smith Ma", "Jane Doe nee Yo-Yo Ma", "John née Jones Smith Ma"}), # and at 2.0.0 and 2.1.0 (nine): the review added 'Jane Doe nee # King. ba' beside 'Doe, J. nee MA ba', the same FLOOR boundary. frozenset({r"Doe, Dr\. nee Smith MA", r"Doe, J\. nee MA ba", - "Doe, Jane nee Smith Ma", r"Jane Doe nee King\. ba", - "Jane Doe nee MA", "Jane Doe nee MA PhD", - "Jane Doe nee Smith DO DO", "Jane Doe nee Smith Ma", - "Jane Doe nee Yo-Yo Ma"}), + "Doe, Jane nee Smith Ma", r"Jane Doe Jr\. nee Smith Ma", + r"Jane Doe nee King\. ba", "Jane Doe nee MA", + "Jane Doe nee MA PhD", "Jane Doe nee Smith DO DO", + "Jane Doe nee Smith Ma", "Jane Doe nee Yo-Yo Ma"}), # The two names where the released member lands somewhere other # than the trailing peel. One set, shared by the 1.4.0 rule and # the 2.2.0/2.3.0 one; at 2.0.0 and 2.1.0 the rule holds one name # and has no alternation to declare. # The 1.4.0 pair and the 1.4.0 halves. The clause KEEPS the member - # in six of them, which at that baseline is not a report but the + # in seven of them ('Jane Doe Jr. nee Smith Ma' joined 2026-09-27, + # #544), which at that baseline is not a report but the # `suffix` v1 read emptying; it gives the member up in four, whose # runs v1 also wrote with commas. The two-member set is the # narrowed fix(#424/#445) anchor, which holds the two spellings # whose reading that rule describes and no longer reaches the # third by (?i). frozenset({"Doe, Jane nee Smith MA do", "Doe, Jane nee Smith Ma", - "Doe, Jane nee Smith do", "Jane Doe nee MA", - "Jane Doe nee Smith Ma", "Jane Doe nee Yo-Yo Ma"}), + "Doe, Jane nee Smith do", r"Jane Doe Jr\. nee Smith Ma", + "Jane Doe nee MA", "Jane Doe nee Smith Ma", + "Jane Doe nee Yo-Yo Ma"}), frozenset({r"Doe, Dr\. nee Smith MA", r"Doe, J\. nee MA ba", "Doe, Jane nee Smith Ma", "Jane Doe nee MA", "Jane Doe nee MA PhD", "Jane Doe nee Smith DO DO", @@ -2856,27 +2885,27 @@ class _LatinCopy(NamedTuple): # a jr, so a member copying SUFFIX_WORDS would reach every # generation-bearing name in the corpora, most of which do not # move. - frozenset({"JOHN DOE PHD MD", "John Doe MD PhD", - "John Smith MD PhD", "John Smith Mc V", - "Josep Lluis Carod i III", - "Kenneth Clarke QC MP", "Smith, John PhD I\\.", + frozenset({"Doe, Jane PhD MEng", "JOHN DOE PHD MD", r"Jane Doe Jr\. Ma", + "John Doe MD PhD", "John Smith MD PhD", "John Smith Mc V", + "John Smith PhD Ed Ma", "John Smith PhD MEng", + "John Smith PhD Ma", "Josep Lluis Carod i III", + "Kenneth Clarke QC MP", r"Smith, John PhD I\.", "The Rt Hon Kenneth Clarke QC MP, HMG", - "Washington Jr\\. MD, Franklin", "abdul Smith Jr Ma", - "abdul Smith Jr V"}), + r"Washington Jr\. MD, Franklin", "abdul Smith Jr Ma", + "abdul Smith Jr V", "doe, jane v phd do"}), # 2026-09-20, #397 review: one more name joins the 2.x set and not # the 1.4.0 one, for the reason 'Jane Doe nee Smith PhD MA' did -- # 'Jane Doe nee Puig i III' is a comma-written generation run # those baselines share with the tree's spaced one, while at 1.4.0 # its ROLES move too and a rule of that ledger's own owns it. - frozenset({"JOHN DOE PHD MD", "Jane Doe nee Puig i III", - "Jane Doe nee Smith PhD MA", - "John Doe MD PhD", - "John Smith MD PhD", "John Smith Mc V", + frozenset({"JOHN DOE PHD MD", r"Jane Doe Jr\. Ma", + "Jane Doe nee Puig i III", "Jane Doe nee Smith PhD MA", + "John Doe MD PhD", "John Smith MD PhD", "John Smith Mc V", + "John Smith PhD Ed Ma", "John Smith PhD Ma", "Josep Lluis Carod i III", "Josep Lluis Carod i V", - "Kenneth Clarke QC MP", "Rovira, Josep Carod i Jr\\.", - "Smith, John PhD I\\.", - "The Rt Hon Kenneth Clarke QC MP, HMG", - "Washington Jr\\. MD, Franklin", "abdul Smith Jr Ma", + "Kenneth Clarke QC MP", r"Rovira, Josep Carod i Jr\.", + r"Smith, John PhD I\.", "The Rt Hon Kenneth Clarke QC MP, HMG", + r"Washington Jr\. MD, Franklin", "abdul Smith Jr Ma", "abdul Smith Jr V"}), # #397's maiden-clause rule, one corpus name per alternative -- a # list of names, not a copy of any wordlist. What selects the @@ -2971,6 +3000,26 @@ class _LatinCopy(NamedTuple): "part1 of and The part2 of the part3 And part4", "the and Jon Dough", "ХОСЕ И МАРИЯ САНТОС", "Хосе И Мария Сантос", "хосе и мария сантос"}), + # #544 (2026-09-27): its literal rules, one alternative per + # corpus name -- #544's own case rows and rules.md examples, + # selected by the SHAPE each rule reads (a comma run, a + # credential in front of a member, a clause boundary), which + # no wordlist decides. Two of the sets are 1.4.0 rules #544's + # names joined rather than #544's own, #544 moving no role on + # them against that baseline: the clause an unambiguous credential + # ended (fix(#274/#436/#437)), and the dotted spelling of a + # two-token name (fix(suffix-routing)), whose members are + # anchored on the name word and so copy no vocabulary. + frozenset({"Doe, Jane nee Smith PhD MEng", "Jane Doe nee Smith PhD MA", + "Jane Doe nee Smith PhD MEng"}), + frozenset({r"jack\s+m\.a\.", r"wang\s+m\.eng\."}), + frozenset({"John Smith, Ed Ma", "John Smith, Ms Ma", + r"John Smith, X\.Y\.Z\. MA", "john smith, md ma"}), + frozenset({"Doe, Jane PhD MEng", "Doe, Jane nee Smith PhD MEng", + "Jane Doe nee Smith PhD MEng", "John Smith PhD MEng"}), + frozenset({"Jane Doe, MS LAc", "John Smith, MD MEng", + "John Smith, MEng PhD", "John Smith, PhD MEng", + "john smith, phd meng"}), }) def _unjustified_reach(name_regex: str, members: set[str]) -> list[str]: @@ -3392,7 +3441,9 @@ def _claim(rule: dict) -> _Claim: # zero role movers at review time. "change(suffix-acronym-collisions) ph leaves the acronym set": _Claim(1, ('family', 'middle', 'suffix'), '8a2e1dbb972d', None), - # #540's four rules, the only #540 rules this baseline needs + # #540's rules, four at landing (2026-09-25) and three since + # #544 (2026-09-27) retired the comma-run one, the only #540 + # rules this baseline needs # (every one of them either an accepted cost or the gain that # is its mirror): the mixed-case credential read as the name # even with words to spare, no-comma (now also the credential @@ -3404,8 +3455,9 @@ def _claim(rule: dict) -> _Claim: _Claim(2, ('family', 'middle', 'suffix'), "28537bf159a3", ('DEFAULT',)), "fix(#540) accepted: a mixed-case MEng after a family comma reads as a middle name": _Claim(1, ('middle', 'suffix'), "84dcae6d2ff5", ('DEFAULT',)), - "fix(#540) accepted: a credential run behind a suffix comma that ends in meng re-reads the comma as a family comma": - _Claim(2, ('family', 'given', 'middle', 'suffix'), "17e8a418c176", ('DEFAULT',)), + # 2026-09-27, #544: the rule recorded here is retired, #544 + # reading its names as the rules that now claim them say; its + # names were 'John Smith, PhD MEng', 'john smith, phd meng'. "fix(#540) a Title-case Lac behind a particle is the family name": _Claim(1, ('family', 'suffix'), "828c38e0abaa", ('DEFAULT',)), # #436/#437's Latin alternation, first in every ledger. @@ -3417,8 +3469,11 @@ def _claim(rule: dict) -> _Claim: # leaves byte-identical -- measured against the parent # 46651750 -- and which arrived with that change's own case # rows. Roles unmoved, `suffix` alone as before. + # 2026-09-27, #544: 11 -> 17; gains 'Doe, Jane PhD MEng', + # 'Jane Doe Jr. Ma', 'John Smith PhD Ed Ma', 'John Smith PhD + # MEng' and 2 more. "fix(#436/#437) a space-separated post-nominal run renders with spaces, not commas": - _Claim(11, ('suffix',), "eaece748b3c9", None), + _Claim(17, ('suffix',), "dd1fbf116ab6", None), # #346's alternation. Four corpus names, `family` and # `given` together: the fold moves both roles at once, so a # widening taking one alone would change the roles here @@ -3494,8 +3549,11 @@ def _claim(rule: dict) -> _Claim: # King. ba', 'Doe, Jane nee Smith V, PhD'). Reach again -- # both carry a marker -- and verified name by name; no role # joined the list. + # 2026-09-27, #544: 105 -> 108; gains 'Doe, Jane nee Smith PhD + # MEng', 'Jane Doe Jr. nee Smith Ma', 'Jane Doe nee Smith PhD + # MEng'. "fix(#274) maiden markers consumed": - _Claim(105, ('family', 'maiden', 'middle'), 'c48af99f8184', None), + _Claim(108, ('family', 'maiden', 'middle'), "19733e2db435", None), # 2026-09-19, #533: 5 -> 6, the same one new corpus name # '田中 太郎 旧姓 佐藤 MA' as the CJK rule above. "fix(cjk-maiden-marker) maiden marker consumed, compounding with the CJK order flip": @@ -3527,8 +3585,9 @@ def _claim(rule: dict) -> _Claim: # do' and 'Doe, Jane nee Smith do', the do pair this change # added. Reach, not explanation: all three are the contest # fix(#274) is now declared to outrank. + # 2026-09-27, #544: 26 -> 27; gains 'doe, jane v phd do'. "fix(#379) a tussenvoegsel after a family comma attaches to the family": - _Claim(26, ('family', 'middle'), '22c55d325d9d', None), + _Claim(27, ('family', 'middle'), "48aafe72402e", None), "fix(#380) a trailing vd after a family comma is the tussenvoegsel, not a post-nominal": _Claim(2, ('family', 'suffix'), "ec0d45289dc1", None), # 279 -> 280 with #371, and the growth is corpus, not behavior: @@ -3625,8 +3684,11 @@ def _claim(rule: dict) -> _Claim: # PhD', rules.md#M2's third-comma-part example, its comma # landing in this rule's reach too. Reach again, verified name # by name. + # 2026-09-27, #544: 380 -> 392; gains 'Doe, Jane PhD MEng', + # 'Doe, Jane nee Smith PhD MEng', 'Jane Doe, MS LAc', 'John + # Smith, Ed Ma' and 8 more. "fix(comma-family) lone post-comma piece routes to suffix/title, not first": - _Claim(380, ('given', 'suffix', 'title'), '658cb8403f4b', None), + _Claim(392, ('given', 'suffix', 'title'), "86492f9f87ba", None), "fix(comma-family) a comma followed only by titles keeps the given/family split": _Claim(2, ('family', 'given'), "5bd9c6d96c38", None), "fix(comma-family) a comma followed only by titles keeps the given/family split, the C1 example": @@ -3663,8 +3725,9 @@ def _claim(rule: dict) -> _Claim: # for both, so there is no diff here to explain. "fix(#296) a lone post-comma credential is a suffix": _Claim(23, ('family', 'given', 'suffix', 'title'), "54c1ae9911e1", None), + # 2026-09-27, #544: 6 -> 7; gains 'Smith, PhD MEng'. "fix(#325) a split credential followed by another suffix after a one-word family comma reads as suffixes": - _Claim(6, ('given', 'suffix', 'title'), "7911e0158337", None), + _Claim(7, ('given', 'suffix', 'title'), "6ab773e59aeb", None), "fix(#325) a credential run across a second comma reads as suffixes": _Claim(1, ('suffix', 'title'), "f025c5f70a4e", None), "fix(#367) an inferred title no longer displaces a leading particle either": @@ -3709,8 +3772,11 @@ def _claim(rule: dict) -> _Claim: # names as the rule above and for the same reason. # 2026-09-26, #535 review: 376 -> 377, the same one new comma # name as the rule above and for the same reason. + # 2026-09-27, #544: 380 -> 392; gains 'Doe, Jane PhD MEng', + # 'Doe, Jane nee Smith PhD MEng', 'Jane Doe, MS LAc', 'John + # Smith, Ed Ma' and 8 more. "fix(comma-precomma-family) pre-comma run reads as family, not given": - _Claim(380, ('family', 'given'), '658cb8403f4b', None), + _Claim(392, ('family', 'given'), "86492f9f87ba", None), # 2026-09-20, #397: retitled in place, reach and digest # unchanged -- the rule keeps 'Carod i', which the landing # leaves byte-identical. @@ -3883,8 +3949,12 @@ def _claim(rule: dict) -> _Claim: # boundary line. Reach, verified name by name. "fix(suffix-routing) a two-token name ending in a credential acronym keeps it in `suffix`": _Claim(3, ('family', 'suffix'), "07c5470c399e", None), - "fix(suffix-routing) the dotted M.A. spelling reads as a credential (ma-do)": - _Claim(1, ('family', 'suffix'), "17379620526b", None), + # 2026-09-27, #544: 1 -> 2; gains 'Wang M.Eng.', the 2.0 + # reading #544 returns to, and relabelled for the shape the two + # names share (it was 'the dotted M.A. spelling reads as a + # credential (ma-do)'). + "fix(suffix-routing) a two-token name ending in a dotted credential spelling keeps it in `suffix`": + _Claim(2, ('family', 'suffix'), "5d450ee3cadb", None), # #484's six `_initials` rules. Four of them reach far more # than they explain, and the gap is the pseudo-field's own # doing rather than a widening: `_initials` enters a diff ONLY @@ -4074,12 +4144,13 @@ def _claim(rule: dict) -> _Claim: # 2026-09-18, second round: two more names, 'John Smith, Ma' # and 'Smith Jr., Ma', the Title-case halves of two minimal # pairs rules.md#C1 now states. No role joined the list. + # 2026-09-27, #544: 19 -> 18; loses 'abdul Smith Jr Ma'. "fix(#289) a written case contrast decides a bare ambiguous acronym": # 2026-09-18, review round: 'John de Ma' joins, the # one-particle spelling of 'John van der Berg Ma'. Its diff # is the restored chain report at the 2.x baselines and the # role move at 1.4.0 and 2.2/2.3; no role joined the list. - _Claim(19, ('family', 'given', 'middle', 'suffix'), "7465eef956d4", ('DEFAULT',)), + _Claim(18, ('family', 'given', 'middle', 'suffix'), "b1b2b5419b22", ('DEFAULT',)), # #516's alternation. Literal-anchored to the by-shape movers, # `orders` DEFAULT. Same reasoning as the rule above: the # class is a shape the vocabulary does not spell, so the @@ -4137,14 +4208,19 @@ def _claim(rule: dict) -> _Claim: # A number that grows here without a corpus row growing with # it is this rule reaching a name whose clause DOES give the # word up, which is the rule below. + # 2026-09-27, #544: 6 -> 7; gains 'Jane Doe Jr. nee Smith Ma', + # a case row #544 added whose roles every 2.x release reads. "fix(#274/#424) accepted: a maiden clause keeps a trailing credential v1 read as a post-nominal": - _Claim(6, ('family', 'maiden', 'middle', 'suffix'), "447e065d4852", None), + _Claim(7, ('family', 'maiden', 'middle', 'suffix'), "2d5c58b81dba", None), # One corpus name, four roles. The rule is a compound of two # released changes and #533 moves nothing on it, so a growth # here is this rule reaching a name whose clause the # ambiguous member ended instead. + # 2026-09-27, #544: 1 -> 3; gains 'Doe, Jane nee Smith PhD + # MEng' and 'Jane Doe nee Smith PhD MEng', case rows #544 added + # whose roles every 2.x release reads -- the same two changes. "fix(#274/#436/#437) a clause the unambiguous credential ended, and the run it left renders with spaces": - _Claim(1, ('family', 'maiden', 'middle', 'suffix'), "51b39568664f", None), + _Claim(3, ('family', 'maiden', 'middle', 'suffix'), "60850ce87c18", None), # Four corpus names, four roles. The `suffix` is where the # released member lands and `maiden` is what it left, so a # widening that took one without the other would change the @@ -4310,6 +4386,9 @@ def _claim(rule: dict) -> _Claim: "fix(#274/#397/#535) a link the clause stops at keeps a run that would not read off": _Claim(1, ('family', 'maiden', 'middle', 'title'), '1b74e094fed9', None), + # 2026-09-27, #544: new, 1; gains 'John Smith, X.Y.Z. MA'. + "fix(#544) a run of ambiguous members after a suffix comma reads by the name-word count": + _Claim(1, ('family', 'given', 'suffix'), "23c792cefe18", ('DEFAULT',)), }, "expected_since_2.0.0.toml": { # The ph removal (#459/#521): one literal name, the cases.py @@ -4318,8 +4397,10 @@ def _claim(rule: dict) -> _Claim: # zero role movers at review time. "change(suffix-acronym-collisions) ph leaves the acronym set": _Claim(1, ('family', 'middle', 'suffix'), '8a2e1dbb972d', None), - # #540's eight rules: five literal alternations and three - # literal names, every name one of #540's own case rows. The + # #540's rules, eight at landing (2026-09-25) -- five literal + # alternations and three literal names -- and six since #544 + # (2026-09-27) retired two, every name one of #540's own case + # rows. The # report rule's `_ambiguities`-only roles are its point: a # widening that took a role would change them here first. "fix(#540) meng and lac are ambiguous credential acronyms, so a bare trailing Meng or Lac with no words to spare is the family name": @@ -4332,12 +4413,14 @@ def _claim(rule: dict) -> _Claim: _Claim(2, ('_ambiguities', 'given', 'suffix'), "96bf15fae64b", ('DEFAULT',)), "fix(#540) a bare trailing meng or lac with words to spare stays the credential and reports the fork": _Claim(2, ('_ambiguities',), "4a72b3bde603", ('DEFAULT',)), - "fix(#540) accepted: a credential run behind a suffix comma that ends in meng re-reads the comma as a family comma": - _Claim(2, ('_ambiguities', 'family', 'given', 'middle', 'suffix'), "17e8a418c176", ('DEFAULT',)), + # 2026-09-27, #544: the rule recorded here is retired, #544 + # reading its names as the rules that now claim them say; its + # names were 'John Smith, PhD MEng', 'john smith, phd meng'. "fix(#540) a Title-case Lac behind a particle is the family name": _Claim(1, ('_ambiguities', 'family', 'suffix'), "828c38e0abaa", ('DEFAULT',)), - "fix(#540) accepted: a chunked dotted M.Eng. with nothing to spare is the family name": - _Claim(1, ('_ambiguities', 'family', 'suffix'), "1d95a4b7cfc9", ('DEFAULT',)), + # 2026-09-27, #544: the rule recorded here is retired, #544 + # reading its names as the rules that now claim them say; its + # names were 'Wang M.Eng.'. # #436/#437's Latin alternation, first in every ledger. # Ten corpus names, `suffix` alone: the rule moves the # SEPARATOR and no role, so a widening that took a role would @@ -4348,10 +4431,13 @@ def _claim(rule: dict) -> _Claim: # leaves byte-identical -- measured against the parent # 46651750 -- and each of which arrived with that change's own # case rows. Roles unmoved, `suffix` alone as before. + # 2026-09-27, #544: 15 -> 18; gains 'Jane Doe Jr. Ma', 'John + # Smith PhD Ed Ma', 'John Smith PhD Ma'. "fix(#436/#437) a space-separated post-nominal run renders with spaces, not commas": - _Claim(15, ('suffix',), "b610147f490b", None), + _Claim(18, ('suffix',), "60ec2103c480", None), # #449's six rules, second in every 2.x ledger. The - # alternation reaches twenty-two corpus names and + # alternation reached twenty-two corpus names at landing + # (twenty-three since #544, 2026-09-27) and # `_ambiguities` alone: no role moves anywhere in this change, # so a widening that took a role would change the row here # before it reached the gate, and a member reaching a name @@ -4366,8 +4452,10 @@ def _claim(rule: dict) -> _Claim: # honorific rules are why -- only at this baseline is #308's # peel still in the diff beside the report, and the gate # refuses a rule declaring a role no diff it explains moves. + # 2026-09-27, #544: 22 -> 23; gains 'Wang M.Eng.', whose diff + # here is the one-word name's report alone. "feat(#449) a lone name word reports given-or-family": - _Claim(22, ('_ambiguities',), "0ef8cf9a9272", None), + _Claim(23, ('_ambiguities',), "ea61ec446eef", None), "feat(#449) a wholly-katakana name keeps the declared order, so the convention decides it": _Claim(1, ('_ambiguities',), "80777383a11a", None), "feat(#449) an interpunct transcription declines the script order, so the convention decides it": @@ -4449,6 +4537,7 @@ def _claim(rule: dict) -> _Claim: # comma names as the 1.4 twin, whose entry carries the roster. # 2026-09-19, #531 review round: 19 -> 20, the same one new # name as that twin ('Doe, John van DO'). + # 2026-09-27, #544: 26 -> 27; gains 'doe, jane v phd do'. "fix(#379) a tussenvoegsel after a family comma attaches to the family": # 2026-09-19, #531 fix round: 20 -> 21, the same one new # corpus name as the 1.4.0 copy ('Doe, John MA do'), reached @@ -4456,7 +4545,7 @@ def _claim(rule: dict) -> _Claim: # 2026-09-19, #533: 21 -> 24, the same three new corpus # names as the 1.4.0 copy -- the do pair this change added # after a family comma. - _Claim(26, ('_ambiguities', 'family', 'middle'), '22c55d325d9d', None), + _Claim(27, ('_ambiguities', 'family', 'middle'), "48aafe72402e", None), # 2026-09-18: 126 -> 131. Five corpus names arrived with # #289/#516's own case rows -- 'J.씨', 'John Smith 田.中.', # '毛泽东, MA', '田中 太郎, MA', '마틴 킹, MA' -- all of them @@ -4532,8 +4621,11 @@ def _claim(rule: dict) -> _Claim: # stops diffing at this baseline altogether and # 'abdul Smith Jr Ma' now reads middle 'Jr'; the ledger's # own dated paragraph carries the argument. - "fix(#425) the bound-given reserve runs assign's peel over the joined view": - _Claim(2, ('family', 'middle', 'suffix'), "ef1ab03b617e", None), + # 2026-09-27, #544: 2 -> 2; roles ['family', 'middle', + # 'suffix'] -> ['family', 'given', 'suffix'], and relabelled + # fix(#425/#436/#437), the `suffix` being R1's spacing alone. + "fix(#425/#436/#437) the bound-given reserve runs assign's peel over the joined view": + _Claim(2, ('family', 'given', 'suffix'), "ef1ab03b617e", None), "fix(#424) the particle chain stops before the trailing numeral": _Claim(1, ('_ambiguities', 'family', 'suffix'), "2c99162bc9cf", None), # 2026-09-18 (#289): 1 -> 2. 'john van der berg ma' entered @@ -4572,8 +4664,9 @@ def _claim(rule: dict) -> _Claim: # `given`, a field outside this rule's own ('suffix', 'title'). "fix(#296) a lone post-comma credential is a suffix": _Claim(23, ('suffix', 'title'), "54c1ae9911e1", None), + # 2026-09-27, #544: 6 -> 7; gains 'Smith, PhD MEng'. "fix(#325) a split credential followed by another suffix after a one-word family comma reads as suffixes": - _Claim(6, ('given', 'suffix', 'title'), "7911e0158337", None), + _Claim(7, ('given', 'suffix', 'title'), "6ab773e59aeb", None), "fix(#325) a credential run across a second comma reads as suffixes": _Claim(1, ('suffix', 'title'), "f025c5f70a4e", None), "fix(#296) a glued honorific before a lone credential: the credential is the postnominal": @@ -4724,6 +4817,7 @@ def _claim(rule: dict) -> _Claim: # 2026-09-18, second round: two more names, 'John Smith, Ma' # and 'Smith Jr., Ma', the Title-case halves of two minimal # pairs rules.md#C1 now states. No role joined the list. + # 2026-09-27, #544: 24 -> 23; loses 'abdul Smith Jr Ma'. "fix(#289) a written case contrast decides a bare ambiguous acronym": # 2026-09-18, review round: 'John de Ma' joins, the # one-particle spelling of 'John van der Berg Ma'. Its diff @@ -4739,7 +4833,7 @@ def _claim(rule: dict) -> _Claim: # 'Doe, Mr. MA PhD' -- 'Doe, Dr. MA' with a credential run # behind the member. Verified to be that name and no other; # no role joined the list. - _Claim(24, ('_ambiguities', 'family', 'given', 'middle', 'suffix'), "9af04fd5b07d", ('DEFAULT',)), + _Claim(23, ('_ambiguities', 'family', 'given', 'middle', 'suffix'), "1f21a81740ea", ('DEFAULT',)), # #516's alternation. Literal-anchored to the by-shape movers, # `orders` DEFAULT. Same reasoning as the rule above: the # class is a shape the vocabulary does not spell, so the @@ -4806,8 +4900,10 @@ def _claim(rule: dict) -> _Claim: # baseline already splits suffix 'ba' from maiden 'King.', so # the whole diff is #533's; #535 moves no role here either, # only the report. + # 2026-09-27, #544: 9 -> 10; gains 'Jane Doe Jr. nee Smith + # Ma'. "fix(#533) the maiden clause reports the credential it keeps": - _Claim(9, ('_ambiguities',), 'f999caeb63dc', None), + _Claim(10, ('_ambiguities',), "7c30ed523806", None), # One corpus name. The by-shape member has no lean to read, # so a growth here is the rule reaching a LISTED member -- # a different reading under this rule's sentence. @@ -4968,6 +5064,25 @@ def _claim(rule: dict) -> _Claim: "fix(#397/#535) a link the clause stops at keeps a run that would not read off": _Claim(1, ('_ambiguities', 'family', 'maiden', 'middle', 'title'), '1b74e094fed9', None), + # 2026-09-27, #544: new, 5; gains 'Jane Doe, MS LAc', 'John + # Smith, MD MEng', 'John Smith, MEng PhD', 'John Smith, PhD + # MEng' and 1 more. + "fix(#544) a credential run after a suffix comma keeps the comma, and the flip reports": + _Claim(5, ('_ambiguities',), "bac52af569e4", ('DEFAULT',)), + # 2026-09-27, #544: new, 4; gains 'John Smith, Ed Ma', 'John + # Smith, Ms Ma', 'John Smith, X.Y.Z. MA', 'john smith, md ma'. + "fix(#544) a run of ambiguous members after a suffix comma reads by the name-word count": + _Claim(4, ('_ambiguities', 'family', 'given', 'middle', 'suffix', 'title'), "65222cc6fe3d", ('DEFAULT',)), + # 2026-09-27, #544: new, 4; gains 'Doe, Jane PhD MEng', 'Doe, + # Jane nee Smith PhD MEng', 'Jane Doe nee Smith PhD MEng', + # 'John Smith PhD MEng'. Relabelled the same day + # fix(#436/#437/#544), the `suffix` it claims being R1's + # spacing alone; reach, roles and digest unchanged. + "fix(#436/#437/#544) an unambiguous credential in front anchors the member behind it": + _Claim(4, ('_ambiguities', 'suffix'), "c9d1d53dee20", ('DEFAULT',)), + # 2026-09-27, #544: new, 1; gains 'doe, jane v phd do'. + "fix(#544) a degree in front outranks P6's attachment for a particle member": + _Claim(1, ('_ambiguities', 'middle', 'suffix'), "a2b8cea490a7", ('DEFAULT',)), }, # The 2.3 cycle's first rule, and a facade-only render fix: every # role is identical, so `_initials` alone. Reach and digest as in @@ -4980,8 +5095,10 @@ def _claim(rule: dict) -> _Claim: # zero role movers at review time. "change(suffix-acronym-collisions) ph leaves the acronym set": _Claim(1, ('family', 'middle', 'suffix'), '8a2e1dbb972d', None), - # #540's eight rules: five literal alternations and three - # literal names, every name one of #540's own case rows. The + # #540's rules, eight at landing (2026-09-25) -- five literal + # alternations and three literal names -- and six since #544 + # (2026-09-27) retired two, every name one of #540's own case + # rows. The # report rule's `_ambiguities`-only roles are its point: a # widening that took a role would change them here first. "fix(#540) meng and lac are ambiguous credential acronyms, so a bare trailing Meng or Lac with no words to spare is the family name": @@ -4994,12 +5111,14 @@ def _claim(rule: dict) -> _Claim: _Claim(2, ('_ambiguities', 'given', 'suffix'), "96bf15fae64b", ('DEFAULT',)), "fix(#540) a bare trailing meng or lac with words to spare stays the credential and reports the fork": _Claim(2, ('_ambiguities',), "4a72b3bde603", ('DEFAULT',)), - "fix(#540) accepted: a credential run behind a suffix comma that ends in meng re-reads the comma as a family comma": - _Claim(2, ('_ambiguities', 'family', 'given', 'middle', 'suffix'), "17e8a418c176", ('DEFAULT',)), + # 2026-09-27, #544: the rule recorded here is retired, #544 + # reading its names as the rules that now claim them say; its + # names were 'John Smith, PhD MEng', 'john smith, phd meng'. "fix(#540) a Title-case Lac behind a particle is the family name": _Claim(1, ('_ambiguities', 'family', 'suffix'), "828c38e0abaa", ('DEFAULT',)), - "fix(#540) accepted: a chunked dotted M.Eng. with nothing to spare is the family name": - _Claim(1, ('_ambiguities', 'family', 'suffix'), "1d95a4b7cfc9", ('DEFAULT',)), + # 2026-09-27, #544: the rule recorded here is retired, #544 + # reading its names as the rules that now claim them say; its + # names were 'Wang M.Eng.'. # #436/#437's Latin alternation, first in every ledger. # Ten corpus names, `suffix` alone: the rule moves the # SEPARATOR and no role, so a widening that took a role would @@ -5010,10 +5129,13 @@ def _claim(rule: dict) -> _Claim: # leaves byte-identical -- measured against the parent # 46651750 -- and each of which arrived with that change's own # case rows. Roles unmoved, `suffix` alone as before. + # 2026-09-27, #544: 15 -> 18; gains 'Jane Doe Jr. Ma', 'John + # Smith PhD Ed Ma', 'John Smith PhD Ma'. "fix(#436/#437) a space-separated post-nominal run renders with spaces, not commas": - _Claim(15, ('suffix',), "b610147f490b", None), + _Claim(18, ('suffix',), "60ec2103c480", None), # #449's six rules, second in every 2.x ledger. The - # alternation reaches twenty-two corpus names and + # alternation reached twenty-two corpus names at landing + # (twenty-three since #544, 2026-09-27) and # `_ambiguities` alone: no role moves anywhere in this change, # so a widening that took a role would change the row here # before it reached the gate, and a member reaching a name @@ -5025,8 +5147,10 @@ def _claim(rule: dict) -> _Claim: # beside them. The digests are the same in all three 2.x # ledgers because the regexes are the same strings and the # corpora are one set. + # 2026-09-27, #544: 22 -> 23; gains 'Wang M.Eng.', whose diff + # here is the one-word name's report alone. "feat(#449) a lone name word reports given-or-family": - _Claim(22, ('_ambiguities',), "0ef8cf9a9272", None), + _Claim(23, ('_ambiguities',), "ea61ec446eef", None), "feat(#449) a wholly-katakana name keeps the declared order, so the convention decides it": _Claim(1, ('_ambiguities',), "80777383a11a", None), "feat(#449) an interpunct transcription declines the script order, so the convention decides it": @@ -5167,6 +5291,7 @@ def _claim(rule: dict) -> _Claim: # 2026-09-18, second round: two more names, 'John Smith, Ma' # and 'Smith Jr., Ma', the Title-case halves of two minimal # pairs rules.md#C1 now states. No role joined the list. + # 2026-09-27, #544: 24 -> 23; loses 'abdul Smith Jr Ma'. "fix(#289) a written case contrast decides a bare ambiguous acronym": # 2026-09-18, review round: 'John de Ma' joins, the # one-particle spelling of 'John van der Berg Ma'. Its diff @@ -5182,7 +5307,7 @@ def _claim(rule: dict) -> _Claim: # 'Doe, Mr. MA PhD' -- 'Doe, Dr. MA' with a credential run # behind the member. Verified to be that name and no other; # no role joined the list. - _Claim(24, ('_ambiguities', 'family', 'given', 'middle', 'suffix'), "9af04fd5b07d", ('DEFAULT',)), + _Claim(23, ('_ambiguities', 'family', 'given', 'middle', 'suffix'), "1f21a81740ea", ('DEFAULT',)), # #516's alternation. Literal-anchored to the by-shape movers, # `orders` DEFAULT. Same reasoning as the rule above: the # class is a shape the vocabulary does not spell, so the @@ -5243,8 +5368,10 @@ def _claim(rule: dict) -> _Claim: # baseline already splits suffix 'ba' from maiden 'King.', so # the whole diff is #533's; #535 moves no role here either, # only the report. + # 2026-09-27, #544: 15 -> 16; gains 'Jane Doe Jr. nee Smith + # Ma'. "fix(#533) the maiden clause reports the credential it keeps": - _Claim(15, ('_ambiguities',), '410b3a9f7f62', None), + _Claim(16, ('_ambiguities',), "de33054cd9e5", None), # New rule (#535): one corpus name, 'Doe, Jane nee Smith V, # PhD' -- after a family comma the given slot reads a lone # numeral as a suffix only where the given part is the LAST @@ -5365,6 +5492,25 @@ def _claim(rule: dict) -> _Claim: "fix(#397/#535) a link the clause stops at keeps a run that would not read off": _Claim(1, ('_ambiguities', 'family', 'maiden', 'middle', 'title'), '1b74e094fed9', None), + # 2026-09-27, #544: new, 5; gains 'Jane Doe, MS LAc', 'John + # Smith, MD MEng', 'John Smith, MEng PhD', 'John Smith, PhD + # MEng' and 1 more. + "fix(#544) a credential run after a suffix comma keeps the comma, and the flip reports": + _Claim(5, ('_ambiguities',), "bac52af569e4", ('DEFAULT',)), + # 2026-09-27, #544: new, 4; gains 'John Smith, Ed Ma', 'John + # Smith, Ms Ma', 'John Smith, X.Y.Z. MA', 'john smith, md ma'. + "fix(#544) a run of ambiguous members after a suffix comma reads by the name-word count": + _Claim(4, ('_ambiguities', 'family', 'given', 'middle', 'suffix', 'title'), "65222cc6fe3d", ('DEFAULT',)), + # 2026-09-27, #544: new, 4; gains 'Doe, Jane PhD MEng', 'Doe, + # Jane nee Smith PhD MEng', 'Jane Doe nee Smith PhD MEng', + # 'John Smith PhD MEng'. Relabelled the same day + # fix(#436/#437/#544), the `suffix` it claims being R1's + # spacing alone; reach, roles and digest unchanged. + "fix(#436/#437/#544) an unambiguous credential in front anchors the member behind it": + _Claim(4, ('_ambiguities', 'suffix'), "c9d1d53dee20", ('DEFAULT',)), + # 2026-09-27, #544: new, 1; gains 'doe, jane v phd do'. + "fix(#544) a degree in front outranks P6's attachment for a particle member": + _Claim(1, ('_ambiguities', 'family', 'suffix'), "a2b8cea490a7", ('DEFAULT',)), }, "expected_since_2.1.0.toml": { # The ph removal (#459/#521): one literal name, the cases.py @@ -5373,8 +5519,10 @@ def _claim(rule: dict) -> _Claim: # zero role movers at review time. "change(suffix-acronym-collisions) ph leaves the acronym set": _Claim(1, ('family', 'middle', 'suffix'), '8a2e1dbb972d', None), - # #540's eight rules: five literal alternations and three - # literal names, every name one of #540's own case rows. The + # #540's rules, eight at landing (2026-09-25) -- five literal + # alternations and three literal names -- and six since #544 + # (2026-09-27) retired two, every name one of #540's own case + # rows. The # report rule's `_ambiguities`-only roles are its point: a # widening that took a role would change them here first. "fix(#540) meng and lac are ambiguous credential acronyms, so a bare trailing Meng or Lac with no words to spare is the family name": @@ -5387,12 +5535,14 @@ def _claim(rule: dict) -> _Claim: _Claim(2, ('_ambiguities', 'given', 'suffix'), "96bf15fae64b", ('DEFAULT',)), "fix(#540) a bare trailing meng or lac with words to spare stays the credential and reports the fork": _Claim(2, ('_ambiguities',), "4a72b3bde603", ('DEFAULT',)), - "fix(#540) accepted: a credential run behind a suffix comma that ends in meng re-reads the comma as a family comma": - _Claim(2, ('_ambiguities', 'family', 'given', 'middle', 'suffix'), "17e8a418c176", ('DEFAULT',)), + # 2026-09-27, #544: the rule recorded here is retired, #544 + # reading its names as the rules that now claim them say; its + # names were 'John Smith, PhD MEng', 'john smith, phd meng'. "fix(#540) a Title-case Lac behind a particle is the family name": _Claim(1, ('_ambiguities', 'family', 'suffix'), "828c38e0abaa", ('DEFAULT',)), - "fix(#540) accepted: a chunked dotted M.Eng. with nothing to spare is the family name": - _Claim(1, ('_ambiguities', 'family', 'suffix'), "1d95a4b7cfc9", ('DEFAULT',)), + # 2026-09-27, #544: the rule recorded here is retired, #544 + # reading its names as the rules that now claim them say; its + # names were 'Wang M.Eng.'. # #436/#437's Latin alternation, first in every ledger. # Ten corpus names, `suffix` alone: the rule moves the # SEPARATOR and no role, so a widening that took a role would @@ -5403,10 +5553,13 @@ def _claim(rule: dict) -> _Claim: # leaves byte-identical -- measured against the parent # 46651750 -- and each of which arrived with that change's own # case rows. Roles unmoved, `suffix` alone as before. + # 2026-09-27, #544: 15 -> 18; gains 'Jane Doe Jr. Ma', 'John + # Smith PhD Ed Ma', 'John Smith PhD Ma'. "fix(#436/#437) a space-separated post-nominal run renders with spaces, not commas": - _Claim(15, ('suffix',), "b610147f490b", None), + _Claim(18, ('suffix',), "60ec2103c480", None), # #449's six rules, second in every 2.x ledger. The - # alternation reaches twenty-two corpus names and + # alternation reached twenty-two corpus names at landing + # (twenty-three since #544, 2026-09-27) and # `_ambiguities` alone: no role moves anywhere in this change, # so a widening that took a role would change the row here # before it reached the gate, and a member reaching a name @@ -5418,8 +5571,10 @@ def _claim(rule: dict) -> _Claim: # beside them. The digests are the same in all three 2.x # ledgers because the regexes are the same strings and the # corpora are one set. + # 2026-09-27, #544: 22 -> 23; gains 'Wang M.Eng.', whose diff + # here is the one-word name's report alone. "feat(#449) a lone name word reports given-or-family": - _Claim(22, ('_ambiguities',), "0ef8cf9a9272", None), + _Claim(23, ('_ambiguities',), "ea61ec446eef", None), "feat(#449) a wholly-katakana name keeps the declared order, so the convention decides it": _Claim(1, ('_ambiguities',), "80777383a11a", None), "feat(#449) an interpunct transcription declines the script order, so the convention decides it": @@ -5512,6 +5667,7 @@ def _claim(rule: dict) -> _Claim: # comma names as the 1.4 twin, whose entry carries the roster. # 2026-09-19, #531 review round: 19 -> 20, the same one new # name as that twin ('Doe, John van DO'). + # 2026-09-27, #544: 26 -> 27; gains 'doe, jane v phd do'. "fix(#379) a tussenvoegsel after a family comma attaches to the family": # 2026-09-19, #531 fix round: 20 -> 21, the same one new # corpus name as the 1.4.0 copy ('Doe, John MA do'), reached @@ -5519,7 +5675,7 @@ def _claim(rule: dict) -> _Claim: # 2026-09-19, #533: 21 -> 24, the same three new corpus # names as the 1.4.0 copy -- the do pair this change added # after a family comma. - _Claim(26, ('_ambiguities', 'family', 'middle'), '22c55d325d9d', None), + _Claim(27, ('_ambiguities', 'family', 'middle'), "48aafe72402e", None), "fix(#424) an unlisted abbreviation is as transparent as a listed title to the leading particle": _Claim(1, ('_ambiguities', 'family', 'given'), "ca7b37af6cf8", None), "fix(#367) a title no longer displaces a leading particle out of the leading position": @@ -5563,8 +5719,11 @@ def _claim(rule: dict) -> _Claim: # stops diffing at this baseline altogether and # 'abdul Smith Jr Ma' now reads middle 'Jr'; the ledger's # own dated paragraph carries the argument. - "fix(#425) the bound-given reserve runs assign's peel over the joined view": - _Claim(2, ('family', 'middle', 'suffix'), "ef1ab03b617e", None), + # 2026-09-27, #544: 2 -> 2; roles ['family', 'middle', + # 'suffix'] -> ['family', 'given', 'suffix'], and relabelled + # fix(#425/#436/#437), the `suffix` being R1's spacing alone. + "fix(#425/#436/#437) the bound-given reserve runs assign's peel over the joined view": + _Claim(2, ('family', 'given', 'suffix'), "ef1ab03b617e", None), "fix(#424) the particle chain stops before the trailing numeral": _Claim(1, ('_ambiguities', 'family', 'suffix'), "2c99162bc9cf", None), # 2026-09-18 (#289): 1 -> 2. 'john van der berg ma' entered @@ -5603,8 +5762,9 @@ def _claim(rule: dict) -> _Claim: # `given`, a field outside this rule's own ('suffix', 'title'). "fix(#296) a lone post-comma credential is a suffix": _Claim(23, ('suffix', 'title'), "54c1ae9911e1", None), + # 2026-09-27, #544: 6 -> 7; gains 'Smith, PhD MEng'. "fix(#325) a split credential followed by another suffix after a one-word family comma reads as suffixes": - _Claim(6, ('given', 'suffix', 'title'), "7911e0158337", None), + _Claim(7, ('given', 'suffix', 'title'), "6ab773e59aeb", None), "fix(#325) a credential run across a second comma reads as suffixes": _Claim(1, ('suffix', 'title'), "f025c5f70a4e", None), "fix(#296) a glued honorific before a lone credential: the credential is the postnominal": @@ -5740,6 +5900,7 @@ def _claim(rule: dict) -> _Claim: # 2026-09-18, second round: two more names, 'John Smith, Ma' # and 'Smith Jr., Ma', the Title-case halves of two minimal # pairs rules.md#C1 now states. No role joined the list. + # 2026-09-27, #544: 24 -> 23; loses 'abdul Smith Jr Ma'. "fix(#289) a written case contrast decides a bare ambiguous acronym": # 2026-09-18, review round: 'John de Ma' joins, the # one-particle spelling of 'John van der Berg Ma'. Its diff @@ -5755,7 +5916,7 @@ def _claim(rule: dict) -> _Claim: # 'Doe, Mr. MA PhD' -- 'Doe, Dr. MA' with a credential run # behind the member. Verified to be that name and no other; # no role joined the list. - _Claim(24, ('_ambiguities', 'family', 'given', 'middle', 'suffix'), "9af04fd5b07d", ('DEFAULT',)), + _Claim(23, ('_ambiguities', 'family', 'given', 'middle', 'suffix'), "1f21a81740ea", ('DEFAULT',)), # #516's alternation. Literal-anchored to the by-shape movers, # `orders` DEFAULT. Same reasoning as the rule above: the # class is a shape the vocabulary does not spell, so the @@ -5822,8 +5983,10 @@ def _claim(rule: dict) -> _Claim: # baseline already splits suffix 'ba' from maiden 'King.', so # the whole diff is #533's; #535 moves no role here either, # only the report. + # 2026-09-27, #544: 9 -> 10; gains 'Jane Doe Jr. nee Smith + # Ma'. "fix(#533) the maiden clause reports the credential it keeps": - _Claim(9, ('_ambiguities',), 'f999caeb63dc', None), + _Claim(10, ('_ambiguities',), "7c30ed523806", None), # One corpus name. The by-shape member has no lean to read, # so a growth here is the rule reaching a LISTED member -- # a different reading under this rule's sentence. @@ -5990,6 +6153,25 @@ def _claim(rule: dict) -> _Claim: "fix(#397/#535) a link the clause stops at keeps a run that would not read off": _Claim(1, ('_ambiguities', 'family', 'maiden', 'middle', 'title'), '1b74e094fed9', None), + # 2026-09-27, #544: new, 5; gains 'Jane Doe, MS LAc', 'John + # Smith, MD MEng', 'John Smith, MEng PhD', 'John Smith, PhD + # MEng' and 1 more. + "fix(#544) a credential run after a suffix comma keeps the comma, and the flip reports": + _Claim(5, ('_ambiguities',), "bac52af569e4", ('DEFAULT',)), + # 2026-09-27, #544: new, 4; gains 'John Smith, Ed Ma', 'John + # Smith, Ms Ma', 'John Smith, X.Y.Z. MA', 'john smith, md ma'. + "fix(#544) a run of ambiguous members after a suffix comma reads by the name-word count": + _Claim(4, ('_ambiguities', 'family', 'given', 'middle', 'suffix', 'title'), "65222cc6fe3d", ('DEFAULT',)), + # 2026-09-27, #544: new, 4; gains 'Doe, Jane PhD MEng', 'Doe, + # Jane nee Smith PhD MEng', 'Jane Doe nee Smith PhD MEng', + # 'John Smith PhD MEng'. Relabelled the same day + # fix(#436/#437/#544), the `suffix` it claims being R1's + # spacing alone; reach, roles and digest unchanged. + "fix(#436/#437/#544) an unambiguous credential in front anchors the member behind it": + _Claim(4, ('_ambiguities', 'suffix'), "c9d1d53dee20", ('DEFAULT',)), + # 2026-09-27, #544: new, 1; gains 'doe, jane v phd do'. + "fix(#544) a degree in front outranks P6's attachment for a particle member": + _Claim(1, ('_ambiguities', 'middle', 'suffix'), "a2b8cea490a7", ('DEFAULT',)), }, "expected_since_2.3.0.toml": { # The ph removal (#459/#521): one literal name, the cases.py @@ -5998,8 +6180,10 @@ def _claim(rule: dict) -> _Claim: # zero role movers at review time. "change(suffix-acronym-collisions) ph leaves the acronym set": _Claim(1, ('family', 'middle', 'suffix'), '8a2e1dbb972d', None), - # #540's eight rules: five literal alternations and three - # literal names, every name one of #540's own case rows. The + # #540's rules, eight at landing (2026-09-25) -- five literal + # alternations and three literal names -- and six since #544 + # (2026-09-27) retired two, every name one of #540's own case + # rows. The # report rule's `_ambiguities`-only roles are its point: a # widening that took a role would change them here first. "fix(#540) meng and lac are ambiguous credential acronyms, so a bare trailing Meng or Lac with no words to spare is the family name": @@ -6012,12 +6196,14 @@ def _claim(rule: dict) -> _Claim: _Claim(2, ('_ambiguities', 'given', 'suffix'), "96bf15fae64b", ('DEFAULT',)), "fix(#540) a bare trailing meng or lac with words to spare stays the credential and reports the fork": _Claim(2, ('_ambiguities',), "4a72b3bde603", ('DEFAULT',)), - "fix(#540) accepted: a credential run behind a suffix comma that ends in meng re-reads the comma as a family comma": - _Claim(2, ('_ambiguities', 'family', 'given', 'middle', 'suffix'), "17e8a418c176", ('DEFAULT',)), + # 2026-09-27, #544: the rule recorded here is retired, #544 + # reading its names as the rules that now claim them say; its + # names were 'John Smith, PhD MEng', 'john smith, phd meng'. "fix(#540) a Title-case Lac behind a particle is the family name": _Claim(1, ('_ambiguities', 'family', 'suffix'), "828c38e0abaa", ('DEFAULT',)), - "fix(#540) accepted: a chunked dotted M.Eng. with nothing to spare is the family name": - _Claim(1, ('_ambiguities', 'family', 'suffix'), "1d95a4b7cfc9", ('DEFAULT',)), + # 2026-09-27, #544: the rule recorded here is retired, #544 + # reading its names as the rules that now claim them say; its + # names were 'Wang M.Eng.'. # #383/#479's three rules, the first this ledger carries. The # role rule is the 2.x shape of the 1.4.0 rule of the same # name -- two corpus names, the union of two disjoint role @@ -6054,6 +6240,7 @@ def _claim(rule: dict) -> _Claim: # 2026-09-18, second round: two more names, 'John Smith, Ma' # and 'Smith Jr., Ma', the Title-case halves of two minimal # pairs rules.md#C1 now states. No role joined the list. + # 2026-09-27, #544: 24 -> 23; loses 'abdul Smith Jr Ma'. "fix(#289) a written case contrast decides a bare ambiguous acronym": # 2026-09-18, review round: 'John de Ma' joins, the # one-particle spelling of 'John van der Berg Ma'. Its diff @@ -6069,7 +6256,7 @@ def _claim(rule: dict) -> _Claim: # 'Doe, Mr. MA PhD' -- 'Doe, Dr. MA' with a credential run # behind the member. Verified to be that name and no other; # no role joined the list. - _Claim(24, ('_ambiguities', 'family', 'given', 'middle', 'suffix'), "9af04fd5b07d", ('DEFAULT',)), + _Claim(23, ('_ambiguities', 'family', 'given', 'middle', 'suffix'), "1f21a81740ea", ('DEFAULT',)), # #516's alternation. Literal-anchored to the by-shape movers, # `orders` DEFAULT. Same reasoning as the rule above: the # class is a shape the vocabulary does not spell, so the @@ -6130,8 +6317,10 @@ def _claim(rule: dict) -> _Claim: # eight at 2.0.0 and 2.1.0. `_ambiguities` alone, so a role # appearing here is this rule reaching a name whose clause # gave the member up. + # 2026-09-27, #544: 15 -> 16; gains 'Jane Doe Jr. nee Smith + # Ma'. "fix(#533) the maiden clause reports the credential it keeps": - _Claim(15, ('_ambiguities',), 'e30f2d27ccde', None), + _Claim(16, ('_ambiguities',), "b34e76bea15b", None), # New rule (#535): one corpus name, 'Doe, Jane nee Smith V, # PhD' -- after a family comma the given slot reads a lone # numeral as a suffix only where the given part is the LAST @@ -6241,6 +6430,23 @@ def _claim(rule: dict) -> _Claim: "fix(#397/#535) a link the clause stops at keeps a run that would not read off": _Claim(1, ('_ambiguities', 'family', 'maiden', 'middle', 'title'), '1b74e094fed9', None), + # 2026-09-27, #544: new, 5; gains 'Jane Doe, MS LAc', 'John + # Smith, MD MEng', 'John Smith, MEng PhD', 'John Smith, PhD + # MEng' and 1 more. + "fix(#544) a credential run after a suffix comma keeps the comma, and the flip reports": + _Claim(5, ('_ambiguities',), "bac52af569e4", ('DEFAULT',)), + # 2026-09-27, #544: new, 4; gains 'John Smith, Ed Ma', 'John + # Smith, Ms Ma', 'John Smith, X.Y.Z. MA', 'john smith, md ma'. + "fix(#544) a run of ambiguous members after a suffix comma reads by the name-word count": + _Claim(4, ('_ambiguities', 'family', 'given', 'middle', 'suffix', 'title'), "65222cc6fe3d", ('DEFAULT',)), + # 2026-09-27, #544: new, 4; gains 'Doe, Jane PhD MEng', 'Doe, + # Jane nee Smith PhD MEng', 'Jane Doe nee Smith PhD MEng', + # 'John Smith PhD MEng'. + "fix(#544) an unambiguous credential in front anchors the member behind it": + _Claim(4, ('_ambiguities',), "c9d1d53dee20", ('DEFAULT',)), + # 2026-09-27, #544: new, 1; gains 'doe, jane v phd do'. + "fix(#544) a degree in front outranks P6's attachment for a particle member": + _Claim(1, ('_ambiguities', 'family', 'suffix'), "a2b8cea490a7", ('DEFAULT',)), }, } @@ -6693,6 +6899,15 @@ def test_every_rule_claims_the_recorded_share_of_the_corpus() -> None: "Smith, PhD Jr.": "fix(#325) a split credential followed by another suffix " "after a one-word family comma reads as suffixes", + # #544, 2026-09-27: 'Smith, PhD MEng' is 'Smith, PhD Jr.' with + # an anchored member in the second slot -- a credential run + # after a one-word family comma that collapses whole into + # `suffix` (1.4.0 title 'PhD', given 'MEng'), reached by the + # same regex and won by the same file order, so the argument + # above carries over whole. + "Smith, PhD MEng": + "fix(#325) a split credential followed by another suffix " + "after a one-word family comma reads as suffixes", # PAIR B, three names, OVERLAPPING `fields`. This is the pair # #498 was filed on: `fix(#271/#272/#298)` declares # {family, given, middle} and `fix(cjk-delimited-nickname)` diff --git a/tools/differential/compare.py b/tools/differential/compare.py index 68692eb1..2d71d222 100644 --- a/tools/differential/compare.py +++ b/tools/differential/compare.py @@ -1807,6 +1807,10 @@ class _ShapeMismatch(NamedTuple): "Smith, Ph. D. MD": ("suffix", "title"), "Smith, Ph.D. Jr.": ("given", "suffix"), "Smith, PhD Jr.": ("given", "suffix", "title"), + # #544, 2026-09-27: the same contest and the same argument as + # 'Smith, PhD Jr.' above -- the unambiguous PhD anchors the + # MEng behind it, so the run collapses whole into `suffix`. + "Smith, PhD MEng": ("given", "suffix", "title"), # #528's two, adjudicated 2026-09-13, and kept BELOW #498's # block so the three cohorts read down the dict in the order # the PROVENANCE note above tells them. Both are contested by diff --git a/tools/differential/corpus_rules.jsonl b/tools/differential/corpus_rules.jsonl index d4ab6602..c63db184 100644 --- a/tools/differential/corpus_rules.jsonl +++ b/tools/differential/corpus_rules.jsonl @@ -44,6 +44,7 @@ "Doe nee Smith V, Jane" "Doe nee Smith ba Prof." "Doe, Dr. nee Smith MA" +"Doe, Jane PhD MEng" "Doe, Jane nee Smith St." "Doe, Jane nee Smith V, PhD" "Doe, Jane, and Jr." @@ -114,6 +115,8 @@ "Jane Doe (nee Smith Ma)" "Jane Doe (nee Smith Prof.)" "Jane Doe (nee Smith) MA" +"Jane Doe Jr. Ma" +"Jane Doe Jr. nee Smith Ma" "Jane Doe nee King." "Jane Doe nee King. ba" "Jane Doe nee MA" @@ -135,6 +138,7 @@ "Jane Doe nee Smith do MA Prof." "Jane Doe nee Smith do Prof. MA" "Jane Doe nee Smith i DO Prof." +"Jane Doe, MS LAc" "Jane Smith (Nee)" "Jane Smith (Nee) (Jones)" "Jane Smith (née Jones)" @@ -177,6 +181,8 @@ "John Smith Mc V" "John Smith Msc.Ed." "John Smith PhD" +"John Smith PhD Ed Ma" +"John Smith PhD MEng" "John Smith Prof." "John Smith Prof. Dr." "John Smith Prof. Jr." @@ -188,6 +194,7 @@ "John Smith Xyz." "John Smith, A.B." "John Smith, Ed" +"John Smith, Ed Ma" "John Smith, Jones" "John Smith, LEED AP" "John Smith, MA" @@ -200,6 +207,7 @@ "John Smith, Mr." "John Smith, Mr. Jr." "John Smith, PhD" +"John Smith, PhD MEng" "John Smith, V." "John and Jane Smith" "John née Jones Smith MA" @@ -322,12 +330,14 @@ "Smith, MD PhD" "Smith, Ma" "Smith, Major. John" +"Smith, Ms Ma" "Smith, Ms." "Smith, Ms. Jane" "Smith, PSM I" "Smith, PSM I." "Smith, Ph. D. Jr." "Smith, PhD" +"Smith, PhD MEng" "Smith, Sr." "Smith, de Mesnil Jean" "Smith. John" @@ -337,6 +347,7 @@ "Vega, Juan de la" "Vincent van Gogh van Beethoven" "Wang M.Eng." +"Wang Ma PhD" "Xyz. John Smith" "Xyz. Smith, John" "Xyz. van Johnson" @@ -365,6 +376,7 @@ "de la Vega" "de la Vega y Santos Juan" "de los Santos" +"doe, jane v phd do" "donovan mcnabb-smith" "dr. juan garcia III" "ibn Awf abdul Rahman" diff --git a/tools/differential/corpus_shapes.jsonl b/tools/differential/corpus_shapes.jsonl index 7dcb23ed..348a9578 100644 --- a/tools/differential/corpus_shapes.jsonl +++ b/tools/differential/corpus_shapes.jsonl @@ -24,6 +24,7 @@ {"name": "Jack Ma", "shape": 1} {"name": "Jack X.Y.I.", "shape": 1} {"name": "Jack X.Y.Z.", "shape": 1} +{"name": "Jane Doe Jr. nee Smith Ma", "shape": 1} {"name": "Jane Doe geb. Smith MA", "shape": 1} {"name": "Jane Doe nee MA", "shape": 1} {"name": "Jane Doe nee MA Smith", "shape": 1} @@ -40,6 +41,7 @@ {"name": "Jane Doe nee Smith Ma JD", "shape": 1} {"name": "Jane Doe nee Smith Ma Prof.", "shape": 1} {"name": "Jane Doe nee Smith PhD MA", "shape": 1} +{"name": "Jane Doe nee Smith PhD MEng", "shape": 1} {"name": "Jane Doe nee Smith Prof.", "shape": 1} {"name": "Jane Doe nee Smith Prof. MA", "shape": 1} {"name": "Jane Doe nee Smith V", "shape": 1} @@ -64,6 +66,9 @@ {"name": "John Smith MEng PhD", "shape": 1} {"name": "John Smith Ma", "shape": 1} {"name": "John Smith Ph.", "shape": 1} +{"name": "John Smith PhD Ed Ma", "shape": 1} +{"name": "John Smith PhD MEng", "shape": 1} +{"name": "John Smith PhD Ma", "shape": 1} {"name": "John Smith Q.W.E.R.T.", "shape": 1} {"name": "John Smith R.A.I.", "shape": 1} {"name": "John Smith X.Y.Z.", "shape": 1} @@ -102,6 +107,7 @@ {"name": "Nguyen Van Lac", "shape": 1} {"name": "Sir Bob Andrew Dole", "shape": 1} {"name": "Wang M.Eng.", "shape": 1} +{"name": "Wang Ma PhD", "shape": 1} {"name": "X.Y.Z. Smith", "shape": 1} {"name": "abdul Smith Jr Ma", "shape": 1} {"name": "anh van do", "shape": 1} @@ -140,6 +146,7 @@ {"name": "Doe, J. MA", "shape": 2} {"name": "Doe, J. ba", "shape": 2} {"name": "Doe, J. nee MA ba", "shape": 2} +{"name": "Doe, Jane PhD MEng", "shape": 2} {"name": "Doe, Jane Q. nee Smith MA", "shape": 2} {"name": "Doe, Jane nee Puig i Soler", "shape": 2} {"name": "Doe, Jane nee Smith DO", "shape": 2} @@ -148,6 +155,7 @@ {"name": "Doe, Jane nee Smith MA Prof.", "shape": 2} {"name": "Doe, Jane nee Smith MA do", "shape": 2} {"name": "Doe, Jane nee Smith Ma", "shape": 2} +{"name": "Doe, Jane nee Smith PhD MEng", "shape": 2} {"name": "Doe, Jane nee Smith do", "shape": 2} {"name": "Doe, Jane nee Smith ma", "shape": 2} {"name": "Doe, John A.", "shape": 2} @@ -208,22 +216,32 @@ {"name": "Smith, MA", "shape": 2} {"name": "Smith, MEng", "shape": 2} {"name": "Smith, Ma", "shape": 2} +{"name": "Smith, Ms Ma", "shape": 2} +{"name": "Smith, PhD MEng", "shape": 2} {"name": "Smith, meng", "shape": 2} {"name": "de la Vega, Juan", "shape": 2} +{"name": "doe, jane v phd do", "shape": 2} {"name": "doe, john ma", "shape": 2} {"name": "Davis Royce, Ed", "shape": 3} {"name": "Dr. John P. Doe-Ray, CLU, CFP, LUTC", "shape": 3} {"name": "JOHN SMITH, MA", "shape": 3} +{"name": "Jane Doe, MS LAc", "shape": 3} {"name": "John Smith Jr., PhD", "shape": 3} {"name": "John Smith, A.B.", "shape": 3} {"name": "John Smith, Ed", "shape": 3} +{"name": "John Smith, Ed Ma", "shape": 3} {"name": "John Smith, MA", "shape": 3} +{"name": "John Smith, MD MEng", "shape": 3} {"name": "John Smith, MD, Ma", "shape": 3} {"name": "John Smith, MD, R.A.I.", "shape": 3} +{"name": "John Smith, MEng PhD", "shape": 3} +{"name": "John Smith, Ms Ma", "shape": 3} {"name": "John Smith, PhD", "shape": 3} {"name": "John Smith, PhD MEng", "shape": 3} +{"name": "John Smith, X.Y.Z. MA", "shape": 3} {"name": "Steven Hardman, MD, DO, DDS", "shape": 3} {"name": "john smith, ma", "shape": 3} +{"name": "john smith, md ma", "shape": 3} {"name": "john smith, phd meng", "shape": 3} {"name": "John Smith, Dr.", "shape": 4} {"name": "de Mesnil Jean, Dr.", "shape": 4} diff --git a/tools/differential/expected_since_1.4.0.toml b/tools/differential/expected_since_1.4.0.toml index ed55acb5..9eada5d4 100644 --- a/tools/differential/expected_since_1.4.0.toml +++ b/tools/differential/expected_since_1.4.0.toml @@ -90,7 +90,13 @@ issue = "fix(#436/#437) a space-separated post-nominal run renders with spaces, # 'i III'), the link having no name word to its right. What grew is # the CORPUS -- the name arrived with #397's own case rows -- and the # run rendering is what the diff is about. -name_regex = "^(?:JOHN DOE PHD MD|John Doe MD PhD|John Smith MD PhD|John Smith Mc V|Josep Lluis Carod i III|Kenneth Clarke QC MP|Smith, John PhD I\\.|The Rt Hon Kenneth Clarke QC MP, HMG|Washington Jr\\. MD, Franklin|abdul Smith Jr Ma|abdul Smith Jr V)$" +# 2026-09-27, #544: Six names join the alternation, each diffing here on +# the rendering alone -- the roles are this baseline's, and #544 is what +# put them back ('Doe, Jane PhD MEng', 'Jane Doe Jr. Ma', 'John Smith PhD Ed Ma', 'John Smith PhD MEng', 'John Smith PhD Ma', 'doe, jane v phd do'). The tree before #544 read the Title-case +# member behind the credential as a name word, the case lean stopping +# the peel at it; the company clause (rules.md#S2) reads it as the +# credential, so the run is v1's again and only R1's spacing differs. +name_regex = "^(?:Doe, Jane PhD MEng|JOHN DOE PHD MD|Jane Doe Jr\\. Ma|John Doe MD PhD|John Smith MD PhD|John Smith Mc V|John Smith PhD Ed Ma|John Smith PhD MEng|John Smith PhD Ma|Josep Lluis Carod i III|Kenneth Clarke QC MP|Smith, John PhD I\\.|The Rt Hon Kenneth Clarke QC MP, HMG|Washington Jr\\. MD, Franklin|abdul Smith Jr Ma|abdul Smith Jr V|doe, jane v phd do)$" fields = ["suffix"] # #346: swami, guru, baba and lama moved from the TITLES-only block @@ -742,6 +748,11 @@ issue = "fix(#325) a split credential followed by another suffix after a one-wor # whole run is suffixes now. Literal rather than the shape: a shape # regex absorbed 'Smith, Dr. Jr.', which is the comma-family rule's # (the comment review). +# 2026-09-27, #544: 'Smith, PhD MEng' is reached too, as 'Smith, PhD +# Jr.' is -- the unambiguous PhD anchors the MEng behind it (rules.md#S2's +# company clause), so the run after the one-word family comma collapses +# whole into `suffix` where the tree before #544 read given 'PhD', +# middle 'MEng'. name_regex = "(?i)^smith,\\s*ph\\.?\\s?d\\.?\\s" fields = ["title", "given", "suffix"] @@ -2965,24 +2976,31 @@ name_regex = "(?i)^\\S+\\s+(mc|mp)$" fields = ["family", "suffix"] [[change]] -issue = "fix(suffix-routing) the dotted M.A. spelling reads as a credential (ma-do)" -# 'Jack M.A.': v1 read the trailing token as a name part and 2.x reads -# it as a credential. Whether 1.4 itself read that token as `family` is -# outside what this ledger's worker can check directly -# (tools/differential/README.md's warning against reading a v1 release -# from a cached environment); the diff this rule explains at this -# baseline moves exactly {family, suffix}, which is consistent with -# that reading and is what the fields below claim. -# -# This is a DECIDED reading, unlike the fix(#342) and fix(#397) rules -# above, which classify a diff because its cause is known and not -# because the reading is wanted. decisions.md#ma-do decides it: "ma and -# do joined the ambiguous acronym set because both are common surnames; -# the two-word 'Jack Ma' is kept intact by S2's words-to-spare guard, -# while the periods gate governs the dotted spellings ('M.A.' counts -# unambiguously)". The bare and dotted spellings of one name therefore -# read differently ON PURPOSE, which is why both spellings of the bare -# one are _MUST_NOT_MATCH probes here. +issue = "fix(suffix-routing) a two-token name ending in a dotted credential spelling keeps it in `suffix`" +# 'Jack M.A.' and 'Wang M.Eng.': v1 read the trailing token as a name +# part and 2.x reads it as a credential. Whether 1.4 itself read that +# token as `family` is outside what this ledger's worker can check +# directly (tools/differential/README.md's warning against reading a +# v1 release from a cached environment); the diff this rule explains +# at this baseline moves exactly {family, suffix}, which is consistent +# with that reading and is what the fields below claim. +# +# The label names the SHAPE the two share -- one name word and a +# dotted credential behind it -- because no single decision covers +# both. Until 2026-09-27 the rule held 'Jack M.A.' alone and was +# labelled for decisions.md#ma-do, which decides that name and not +# the other. +# +# 'Jack M.A.' is a DECIDED reading, unlike the fix(#342) and fix(#397) +# rules above, which classify a diff because its cause is known and +# not because the reading is wanted. decisions.md#ma-do decides it: +# "ma and do joined the ambiguous acronym set because both are common +# surnames; the two-word 'Jack Ma' is kept intact by S2's +# words-to-spare guard, while the periods gate governs the dotted +# spellings ('M.A.' counts unambiguously)". The bare and dotted +# spellings of one name therefore read differently ON PURPOSE, which +# is why both spellings of the bare one are _MUST_NOT_MATCH probes +# here. # # 2026-09-18 (#289): the quoted sentence holds for the spelling it # names and not for its all-caps twin. 'Jack Ma' is still kept intact @@ -2993,24 +3011,39 @@ issue = "fix(suffix-routing) the dotted M.A. spelling reads as a credential (ma- # rule's own roster at the bottom of this file, where the roster # comment says why. decisions.md#ma-do carries the amendment. # -# Literal-anchored, and it could not be anything else. Measured: the -# member `m\.?a\.?` matches the fragment 'M.A.' in the corpus, and -# _normalize leaves that as 'm.a', which is not a SUFFIX_ACRONYMS entry -# -- so test_latin_alternations_mean_something_the_vocabulary_ships -# rejects it as an alternation member -- its helper -# _reaches_non_vocabulary puts the rule as "every fragment a member -# matches must BE an entry". The ambiguous-surname-acronym rule above -# records the same finding and answers it the same way: keep the -# periods out of the members. There is no alternation here at all, so -# nothing is owed to _LATIN_ALTERNATION_SOURCES. -# -# Anchored on 'jack' as well as on the spelling: 'John Smith M.A.' is -# also a corpus name (it arrived with corpus_rules.jsonl, #414), it is -# three tokens rather than two, and it does not diff against this -# baseline at all -- so a rule keyed on the dotted spelling alone would -# stand ready to explain a future regression on it. It is a -# _MUST_NOT_MATCH probe for that reason. -name_regex = "(?i)^jack\\s+m\\.a\\.$" +# 'Wang M.Eng.' (joined 2026-09-27, #544) reads suffix 'M.Eng.' at +# 2.0.0 through 2.3.0 for a different reason: 'meng' was an +# unambiguous acronym there, and rules.md#S2 consumes an unambiguous +# suffix "even when that leaves no family name at all". #540 made +# 'meng' ambiguous in the 2.4 cycle, which read the family for the +# chunked spelling, and the tree reads the suffix again by S2's period +# gate, which counts an ambiguous acronym "written with its periods, +# one after each letter or one after each of two or more letter +# chunks" unambiguously. Against this baseline the diff is the same +# {family, suffix} either way, and 2.0 is where it arrived. +# +# Literal-anchored, one whole name per alternative, and the members +# could not be spellings alone. Measured: a member `m\.?a\.?` matches +# the fragment 'M.A.' in the corpus, and _normalize leaves that as +# 'm.a', which is not a SUFFIX_ACRONYMS entry -- so +# test_latin_alternations_mean_something_the_vocabulary_ships rejects +# it as an alternation member -- its helper _reaches_non_vocabulary +# puts the rule as "every fragment a member matches must BE an entry". +# The ambiguous-surname-acronym rule above records the same finding +# and answers it the same way: keep the periods out of the members. +# The alternation here is of two anchored NAMES, not vocabulary, so it +# is declared in _NOT_A_VOCABULARY_COPY and owes +# _LATIN_ALTERNATION_SOURCES nothing. +# +# Anchored on the name word as well as on the spelling: 'John Smith +# M.A.' and 'John Smith M.Eng.' are three tokens rather than two and +# do not diff against this baseline -- 'John Smith M.A.' is a corpus +# name (it arrived with corpus_rules.jsonl, #414) -- so a rule keyed on +# the dotted spelling alone would stand ready to explain a future +# regression on either. Both are _MUST_NOT_MATCH probes for that +# reason, with 'Wang Ma.' (the single trailing period S2's gate does +# not count) and the superstring 'Dr. Wang M.Eng.'. +name_regex = "(?i)^(?:jack\\s+m\\.a\\.|wang\\s+m\\.eng\\.)$" fields = ["family", "suffix"] # --------------------------------------------------------------------- @@ -3913,7 +3946,14 @@ issue = "fix(#289) a written case contrast decides a bare ambiguous acronym" # two rules and leave file order to arbitrate, which is the shape the # harness refuses to let go undeclared; so the 2.x copies carry them # and this one does not. -name_regex = "^(?:Davis Royce, Ed|Freiherr von Berg MA|JOHN SMITH, MA|Jack MA|Jack MA\\.|Jack Wei Ma|John Smith Ma|John Smith, Ed|John Smith, MA|John Smith, Ma|John de Ma|John van der Berg Ma|Smith Jr\\., MA|Smith Jr\\., Ma|Smith, MA|abdul Smith Berg Ma|abdul Smith Jr Ma|abdul Smith Ma|john smith, ma)$" +# 2026-09-27, #544: 'abdul Smith Jr Ma' leaves this alternation. The +# unambiguous 'Jr' in front of 'Ma' anchors it (rules.md#S2's company +# clause), the peel takes both, and P5's reserve declines the join, so +# the accepted cost this rule recorded for it -- middle 'Jr', family +# 'Ma' -- is gone; what the name still diffs on here is claimed by the +# rule that explains it, and a literal left in this one would stand +# ready to explain that cost coming back. +name_regex = "^(?:Davis Royce, Ed|Freiherr von Berg MA|JOHN SMITH, MA|Jack MA|Jack MA\\.|Jack Wei Ma|John Smith Ma|John Smith, Ed|John Smith, MA|John Smith, Ma|John de Ma|John van der Berg Ma|Smith Jr\\., MA|Smith Jr\\., Ma|Smith, MA|abdul Smith Berg Ma|abdul Smith Ma|john smith, ma)$" fields = ["family", "given", "middle", "suffix"] orders = ["DEFAULT"] @@ -4159,7 +4199,8 @@ orders = ["DEFAULT"] [[change]] issue = "fix(#274/#424) accepted: a maiden clause keeps a trailing credential v1 read as a post-nominal" -# Six names whose reading this change does not touch: each reads on +# Six names at landing (seven since #544, below) whose reading this +# change does not touch: each reads on # this tree exactly as it read at 2f57ff21, measured name by name, # and the whole of the 1.4.0 diff is v1 having no maiden field. The # marker takes the words after it, the trailing member among them -- @@ -4170,13 +4211,13 @@ issue = "fix(#274/#424) accepted: a maiden clause keeps a trailing credential v1 # the member is the only word the marker would leave ('Jane Doe nee # MA'), and wherever P6's attachment claims the word instead ('Doe, # Jane nee Smith do', 'Doe, Jane nee Smith MA do'). What #533 adds on -# these six is a REPORT, and 1.4.0 has no surface to compare one +# these names is a REPORT, and 1.4.0 has no surface to compare one # against -- so nothing of this change is visible from here, and the # rule is named for the change that is. # # Its own rule rather than a widening of fix(#274) above, and the # reason is that rule's `fields`: it stops at maiden/middle/family -# because a Latin marker moves only those, while these six also move +# because a Latin marker moves only those, while these names also move # the `suffix` v1 read. Folding them in would have fix(#274) stand # ready to explain a suffix regression on every name carrying a # marker -- the trade 'fix(#424/#445) accepted: the maiden walk keeps @@ -4185,12 +4226,22 @@ issue = "fix(#274/#424) accepted: a maiden clause keeps a trailing credential v1 # Literal-anchored: the subject is a SLOT, and a regex for it would # claim the names whose writing gives the word UP as readily as # these. -name_regex = "^(?:Doe, Jane nee Smith MA do|Doe, Jane nee Smith Ma|Doe, Jane nee Smith do|Jane Doe nee MA|Jane Doe nee Smith Ma|Jane Doe nee Yo-Yo Ma)$" +# +# 2026-09-27, #544: a seventh joins, 'Jane Doe Jr. nee Smith Ma' -- +# maiden 'Smith Ma', suffix 'Jr.', where v1 read middle 'Doe Jr. nee', +# last 'Smith', suffix 'Ma'. It reads so at 2.0.0 through 2.3.0, at the +# tree before #544 and at the tree, the report aside: rules.md#M2's +# Accepted boundary, "a credential written in front of the marker speaks +# for no word of the clause". #544 decided that boundary and moves no role +# on the name, so the 1.4.0 diff is this rule's, the Title-case member's +# writing declining it as it does for 'Jane Doe nee Smith Ma'. +name_regex = "^(?:Doe, Jane nee Smith MA do|Doe, Jane nee Smith Ma|Doe, Jane nee Smith do|Jane Doe Jr\\. nee Smith Ma|Jane Doe nee MA|Jane Doe nee Smith Ma|Jane Doe nee Yo-Yo Ma)$" fields = ["family", "maiden", "middle", "suffix"] [[change]] issue = "fix(#274/#436/#437) a clause the unambiguous credential ended, and the run it left renders with spaces" -# 'Jane Doe nee Smith PhD MA', and #533 moves nothing on it either: +# 'Jane Doe nee Smith PhD MA' (the two #544 names below read the same +# way), and #533 moves nothing on it either: # the unambiguous PhD already ended the walk before this change, so # the clause gave the MA up then as it does now -- maiden 'Smith', # suffix 'PhD MA', the same reading as at 2f57ff21. Two changes meet @@ -4201,7 +4252,14 @@ issue = "fix(#274/#436/#437) a clause the unambiguous credential ended, and the # The name is the order-sensitivity #533 was filed about, seen from # the side that never moved: 'Jane Doe nee Smith MA PhD' reads the # same way now and did not before, and it is in the rule below. -name_regex = "^Jane Doe nee Smith PhD MA$" +# +# 2026-09-27, #544: 'Jane Doe nee Smith PhD MEng' and 'Doe, Jane nee +# Smith PhD MEng' join, the same two changes and nothing else: maiden +# 'Smith', suffix 'PhD MEng', the roles 2.0.0 through 2.3.0 read. The +# tree before #544 read the Title-case MEng as a name word, and #544's +# company clause (rules.md#S2) put the run back, so against this +# baseline #544 moves no role on either name. +name_regex = "^(?:Doe, Jane nee Smith PhD MEng|Jane Doe nee Smith PhD MA|Jane Doe nee Smith PhD MEng)$" fields = ["family", "maiden", "middle", "suffix"] [[change]] @@ -4595,24 +4653,31 @@ name_regex = "^John Smith Ph\\.$" fields = ["family", "middle", "suffix"] # --------------------------------------------------------------- -# #540: MENG AND LAC ARE AMBIGUOUS CREDENTIAL ACRONYMS. Four rules here -# where the 2.x ledgers carry eight. 'wang meng' and 'tran lac' read the -# family at 1.4.0 as the tree does, 'Smith, meng' and 'Smith, MEng' read -# the given name the same way, and 'john smith meng' and 'nguyen van lac' -# keep 1.4.0's suffix -- their only new thing is a report, and -# `_ambiguities` is a v2 surface this baseline does not compare. ('nguyen -# van lac' does diff here, on `_initials`, for fix(#385/#402)'s reason, -# and that rule's list carries it; the Title-case 'Nguyen Van Lac' -# reaches that same rule too, case-insensitively, but its role move is -# explained by the Title-case Lac rule below instead.) 'Wang M.Eng.' -# also reads the family here, for an unrelated reason -- its two-piece -# rule (a lone word after the given name is the family), not S2's -# gate -- which is why that rule is 2.x only. What is left is four -# rules: the marking's cost, paid twice -# (once bare behind a full name, once after a family comma), the same -# cost arriving as a credential run behind a suffix comma, and the -# Title-case Lac gain that is the cost's mirror. +# #540: MENG AND LAC ARE AMBIGUOUS CREDENTIAL ACRONYMS. Only some of +# the 2.x ledgers' #540 rules are needed here. 'wang meng' and 'tran +# lac' read the family at 1.4.0 as the tree does, 'Smith, meng' and +# 'Smith, MEng' read the given name the same way, and 'john smith +# meng' and 'nguyen van lac' keep 1.4.0's suffix -- their only new +# thing is a report, and `_ambiguities` is a v2 surface this baseline +# does not compare. ('nguyen van lac' does diff here, on `_initials`, +# for fix(#385/#402)'s reason, and that rule's list carries it; the +# Title-case 'Nguyen Van Lac' reaches that same rule too, +# case-insensitively, but its role move is explained by the +# Title-case Lac rule below instead.) What is left is the marking's +# cost, paid twice (once bare behind a full name, once after a family +# comma), and the Title-case Lac gain that is the cost's mirror. # decisions.md#suffix-acronym-collisions records the decision. +# +# At landing (2026-09-25) there were four rules here: the fourth was +# the same cost arriving as a credential run behind a suffix comma, +# and 'Wang M.Eng.' had no rule, reading the family here as 1.4.0 did +# (by its two-piece rule, a lone word after the given name being the +# family, not by S2's gate). #544 (2026-09-27) reads the run whole +# again, so 'John Smith, PhD MEng' and 'john smith, phd meng' read as +# this baseline read them and that rule is retired; and it reads +# 'Wang M.Eng.' as the suffix 2.0.0 through 2.3.0 read, so the name +# diffs here once more and the fix(suffix-routing) rule for a +# two-token dotted credential spelling claims it beside 'Jack M.A.'. # --------------------------------------------------------------- [[change]] @@ -4650,27 +4715,26 @@ fields = ["middle", "suffix"] orders = ["DEFAULT"] [[change]] -issue = "fix(#540) accepted: a credential run behind a suffix comma that ends in meng re-reads the comma as a family comma" -# 'John Smith, PhD MEng' and 'john smith, phd meng': a credential run -# behind a suffix comma is no longer wholly suffix-shaped once its last -# word is a bare ambiguous member, so C1 reads the comma as a family -# comma instead and the run's own words fall into the usual post-comma -# name slots. 'John Smith, PhD MEng' gives given 'PhD', middle 'MEng', -# family 'John Smith'; 'john smith, phd meng' gives given 'phd', -# family 'john smith', suffix 'meng' (the count then peels 'meng' -# with a word to spare). 1.4.0 read the whole run as a suffix. The -# pre-existing 'john smith, phd ma' path, which #540 routes two more -# words into; #544 asks whether the run should keep its comma. -# -# Literal, two names, fields the union of what each moves ('middle' is -# the first name's alone -- the OVER-DECLARED check accepts a field -# that at least one explained name moves). Probes: 'John Smith, MEng' -# (a lone credential after the comma keeps the suffix, C1's count) and -# 'John Smith, PhD' (a lone credential of the other shape, also kept) -# are _MUST_NOT_MATCH, along with the superstring 'Dr. John Smith, PhD -# MEng'. -name_regex = "^(?:John Smith, PhD MEng|john smith, phd meng)$" -fields = ["given", "middle", "family", "suffix"] +issue = "fix(#544) a run of ambiguous members after a suffix comma reads by the name-word count" +# 'John Smith, X.Y.Z. MA': given 'John', family 'Smith', suffix 'X.Y.Z. +# MA', where this baseline read the comma as a family comma -- first +# 'X.Y.Z.', last 'John Smith', suffix 'MA' -- and so did the tree before +# #544. The part after the comma is the credential run on the count of +# the two name words before it, rules.md#C1: "The same count reads a +# part of two or more words as the credential run when every word of it +# is a suffix word or a word of this class, at least one of them of this +# class". 'X.Y.Z.' is a member by shape, whose capitals lean nothing -- +# rules.md#S2 reads the writing of a LISTED member only -- so the run +# is not one the capitals settle. +# +# The one name of the 2.x ledgers' rule that diffs here: 'John Smith, +# Ed Ma', 'John Smith, Ms Ma' and 'john smith, md ma' read the run as a +# suffix at this baseline as well, rendered as the tree renders it, and +# do not diff. Literal; `fields` is the union this name moves. Probes: +# 'Smith, Ed Ma', 'Smith, Ms Ma' and the superstrings 'Dr. John Smith, +# Ed Ma' and 'Dr. John Smith, X.Y.Z. MA' are _MUST_NOT_MATCH. +name_regex = "^John Smith, X\\.Y\\.Z\\. MA$" +fields = ["given", "family", "suffix"] orders = ["DEFAULT"] [[change]] diff --git a/tools/differential/expected_since_2.0.0.toml b/tools/differential/expected_since_2.0.0.toml index 11f376da..beeacd85 100644 --- a/tools/differential/expected_since_2.0.0.toml +++ b/tools/differential/expected_since_2.0.0.toml @@ -82,7 +82,13 @@ issue = "fix(#436/#437) a space-separated post-nominal run renders with spaces, # before and after (given 'Jane', family 'Doe', suffix 'i III', # maiden 'Puig'), the link having a generation and not a name word # to its right. What grew is the CORPUS. -name_regex = "^(?:JOHN DOE PHD MD|Jane Doe nee Puig i III|Jane Doe nee Smith PhD MA|John Doe MD PhD|John Smith MD PhD|John Smith Mc V|Josep Lluis Carod i III|Josep Lluis Carod i V|Kenneth Clarke QC MP|Rovira, Josep Carod i Jr\\.|Smith, John PhD I\\.|The Rt Hon Kenneth Clarke QC MP, HMG|Washington Jr\\. MD, Franklin|abdul Smith Jr Ma|abdul Smith Jr V)$" +# 2026-09-27, #544: Three names join the alternation, each diffing here on +# the rendering alone -- the roles are this baseline's, and #544 is what +# put them back ('Jane Doe Jr. Ma', 'John Smith PhD Ed Ma', 'John Smith PhD Ma'). The tree before #544 read the Title-case +# member behind the credential as a name word, the case lean stopping +# the peel at it; the company clause (rules.md#S2) reads it as the +# credential, so the run is v1's again and only R1's spacing differs. +name_regex = "^(?:JOHN DOE PHD MD|Jane Doe Jr\\. Ma|Jane Doe nee Puig i III|Jane Doe nee Smith PhD MA|John Doe MD PhD|John Smith MD PhD|John Smith Mc V|John Smith PhD Ed Ma|John Smith PhD Ma|Josep Lluis Carod i III|Josep Lluis Carod i V|Kenneth Clarke QC MP|Rovira, Josep Carod i Jr\\.|Smith, John PhD I\\.|The Rt Hon Kenneth Clarke QC MP, HMG|Washington Jr\\. MD, Franklin|abdul Smith Jr Ma|abdul Smith Jr V)$" fields = ["suffix"] # The six #449 rules go SECOND, not first: the rule above @@ -94,7 +100,8 @@ fields = ["suffix"] # default (#382). [[change]] issue = "feat(#449) a lone name word reports given-or-family" -# rules.md#O5's convention, now reported. Twenty-two corpus names, and +# rules.md#O5's convention, now reported. Twenty-two corpus names at +# landing (twenty-three since #544, 2026-09-27), and # the alternation is a list of NAMES rather than a copy of any # wordlist -- what selects them is the SHAPE (one name word, and no # title, maiden name, comma, script order or vocabulary claim deciding @@ -108,7 +115,18 @@ issue = "feat(#449) a lone name word reports given-or-family" # GLUED_HONORIFICS. #322/#323 opened the one escape from that pin -- # a member set declared in _NOT_A_VOCABULARY_COPY -- and nothing here # declares one, so the five rules stand. -name_regex = "(?i)^(?:'Smitty' Jones Jr\\.|Andrew|Carod i|Dean of Chemistry|Donald mc|Duke of Edinburgh|Duke of Wellington|Garcia|Jack M\\.A\\.|John & Jane|John V|John of the Doe|Juan & Garcia|Juan and Garcia|Mohamad X|Smith|Smith Jr\\.|e and e|part1 of The part2 of the part3 and part4|part1 of and The part2 of the part3 And part4|test|سلمان،)$" +# 2026-09-27, #544: 'Wang M.Eng.' joins the alternation. Its roles are +# this baseline's, suffix 'M.Eng.', and the whole of its diff here is +# the report this rule describes, as for 'Jack M.A.'. The two readings +# of the suffix have different grounds: this baseline read it because +# 'meng' was an unambiguous acronym, which rules.md#S2 consumes "even +# when that leaves no family name at all"; #540 made 'meng' ambiguous +# in the 2.4 cycle, which read the family, and the tree reads the +# suffix again because S2 counts an ambiguous acronym "written with its +# periods, one after each letter or one after each of two or more +# letter chunks" unambiguously. Neither change is what the name diffs +# on against this baseline. +name_regex = "(?i)^(?:'Smitty' Jones Jr\\.|Andrew|Carod i|Dean of Chemistry|Donald mc|Duke of Edinburgh|Duke of Wellington|Garcia|Jack M\\.A\\.|John & Jane|John V|John of the Doe|Juan & Garcia|Juan and Garcia|Mohamad X|Smith|Smith Jr\\.|Wang M\\.Eng\\.|e and e|part1 of The part2 of the part3 and part4|part1 of and The part2 of the part3 And part4|test|سلمان،)$" fields = ["_ambiguities"] [[change]] @@ -784,7 +802,7 @@ name_regex = "(?i)^abdul\\s+ph\\.\\s+d\\.\\s+smith\\s+berg$" fields = ["given", "middle", "suffix"] [[change]] -issue = "fix(#425) the bound-given reserve runs assign's peel over the joined view" +issue = "fix(#425/#436/#437) the bound-given reserve runs assign's peel over the joined view" # 'abdul Smith Jr Ma', 'abdul Smith Ma': rules.md#P5 -- "the join is # tried on the pieces as it would leave them, assign's trailing peel # (S2) is read over that, and the name words it leaves are the words @@ -812,8 +830,20 @@ issue = "fix(#425) the bound-given reserve runs assign's peel over the joined vi # never reached. `fields` is narrowed to the union the run measures # here -- `given` no longer moves on the one name left diffing, and # a role nothing moves is a standing claim (#452). +# 2026-09-27, #544: the paragraph above is true again, `given` included. +# The company clause (rules.md#S2) reads the 'Ma' behind 'Jr' as the +# credential, so the peel takes both and the reserve declines the join: +# 'abdul Smith Jr Ma' reads given 'abdul', family 'Smith', suffix 'Jr +# Ma', the diff here {given, family, suffix}, and `middle` moves on +# nothing any more. `fields` follows the run. +# Of that diff, {given, family} is #425's -- 2.2.0 and 2.3.0 read those +# roles too -- and `suffix` is R1's spacing alone, the 'Jr, Ma' this +# baseline wrote (#436/#437). #544 moves nothing on it against this +# baseline: it takes back a reading the 2.4 cycle had broken. One rule +# has to explain the whole of the diff, so the label is joint, as the +# 1.4.0 ledger's fix(#274/#436/#437) rules are. name_regex = "(?i)^abdul\\s+smith\\s+(jr\\s+)?ma$" -fields = ["family", "middle", "suffix"] +fields = ["given", "family", "suffix"] [[change]] issue = "fix(#424) an unlisted abbreviation is as transparent as a listed title to the leading particle, the P4 example" @@ -1045,6 +1075,11 @@ issue = "fix(#325) a split credential followed by another suffix after a one-wor # whole run is suffixes now. Literal rather than the shape: a shape # regex absorbed 'Smith, Dr. Jr.', which is the comma-family rule's # (the comment review). +# 2026-09-27, #544: 'Smith, PhD MEng' is reached too, as 'Smith, PhD +# Jr.' is -- the unambiguous PhD anchors the MEng behind it (rules.md#S2's +# company clause), so the run after the one-word family comma collapses +# whole into `suffix` where the tree before #544 read given 'PhD', +# middle 'MEng'. name_regex = "(?i)^smith,\\s*ph\\.?\\s?d\\.?\\s" fields = ["title", "given", "suffix"] @@ -2478,7 +2513,14 @@ issue = "fix(#289) a written case contrast decides a bare ambiguous acronym" # given baseline is simply one the reading already agreed with there. # `fields` is per-baseline, the union the run at THAT baseline # measures (#452). -name_regex = "^(?:Davis Royce, Ed|Doe, Dr\\. MA|Doe, MA|Doe, MA PhD|Doe, Mr\\. MA PhD|Freiherr von Berg MA|JOHN SMITH, MA|Jack MA|Jack MA\\.|Jack Wei Ma|John Prof\\. MA|John Smith Ma|John Smith, Ed|John Smith, MA|John Smith, Ma|John de Ma|John van der Berg Ma|Smith Jr\\., MA|Smith Jr\\., Ma|Smith, MA|abdul Smith Berg Ma|abdul Smith Jr Ma|abdul Smith Ma|john smith, ma)$" +# 2026-09-27, #544: 'abdul Smith Jr Ma' leaves this alternation. The +# unambiguous 'Jr' in front of 'Ma' anchors it (rules.md#S2's company +# clause), the peel takes both, and P5's reserve declines the join, so +# the accepted cost this rule recorded for it -- middle 'Jr', family +# 'Ma' -- is gone; what the name still diffs on here is claimed by the +# rule that explains it, and a literal left in this one would stand +# ready to explain that cost coming back. +name_regex = "^(?:Davis Royce, Ed|Doe, Dr\\. MA|Doe, MA|Doe, MA PhD|Doe, Mr\\. MA PhD|Freiherr von Berg MA|JOHN SMITH, MA|Jack MA|Jack MA\\.|Jack Wei Ma|John Prof\\. MA|John Smith Ma|John Smith, Ed|John Smith, MA|John Smith, Ma|John de Ma|John van der Berg Ma|Smith Jr\\., MA|Smith Jr\\., Ma|Smith, MA|abdul Smith Berg Ma|abdul Smith Ma|john smith, ma)$" fields = ["family", "given", "middle", "suffix", "_ambiguities"] orders = ["DEFAULT"] @@ -2860,7 +2902,11 @@ issue = "fix(#533) the maiden clause reports the credential it keeps" # rule of its own rather than a widening of the one above: `fields` # is `_ambiguities` alone, so it cannot absorb a role diff on any of # the six. -name_regex = "^(?:Doe, Dr\\. nee Smith MA|Doe, J\\. nee MA ba|Doe, Jane nee Smith Ma|Jane Doe nee MA|Jane Doe nee MA PhD|Jane Doe nee King\\. ba|Jane Doe nee Smith DO DO|Jane Doe nee Smith Ma|Jane Doe nee Yo-Yo Ma)$" +# 2026-09-27, #544: 'Jane Doe Jr. nee Smith Ma' joins, the declining +# half at rules.md#M2's new Accepted boundary: the credential in front +# of the marker speaks for no member of the clause, the Title-case 'Ma' +# is kept, and the clause reports it. +name_regex = "^(?:Doe, Dr\\. nee Smith MA|Doe, J\\. nee MA ba|Doe, Jane nee Smith Ma|Jane Doe Jr\\. nee Smith Ma|Jane Doe nee King\\. ba|Jane Doe nee MA|Jane Doe nee MA PhD|Jane Doe nee Smith DO DO|Jane Doe nee Smith Ma|Jane Doe nee Yo-Yo Ma)$" fields = ["_ambiguities"] [[change]] @@ -3510,28 +3556,34 @@ name_regex = "^John Smith Ph\\.$" fields = ["family", "middle", "suffix"] # --------------------------------------------------------------- -# #540: MENG AND LAC ARE AMBIGUOUS CREDENTIAL ACRONYMS. Eight rules -# -- the two names whose family name comes back, the marking's +# #540: MENG AND LAC ARE AMBIGUOUS CREDENTIAL ACRONYMS. The rules +# below: the two names whose family name comes back, the marking's # cost behind a full name, the same cost after a family comma, the # lone word after a family comma (given name back, cost for the -# conventional spelling), the same cost arriving as a credential -# run behind a suffix comma, the chunked dotted M.Eng. cost, the -# Title-case Lac gain that is the cost's mirror, and the two that -# keep the credential and gain a report. No rule of the parser -# changed: the two words joined suffix_acronyms_ambiguous, and -# rules.md#S2 reads them as it reads 'ma' and 'ba' -- "A BARE -# ambiguous acronym is consumed only when the name has words to -# spare". decisions.md#suffix-acronym-collisions records the -# decision, and the removal and the masks it declined. +# conventional spelling), the Title-case Lac gain that is the cost's +# mirror, and the two that keep the credential and gain a report. No +# rule of the parser changed: the two words joined +# suffix_acronyms_ambiguous, and rules.md#S2 reads them as it reads +# 'ma' and 'ba' -- "A BARE ambiguous acronym is consumed only when the +# name has words to spare". decisions.md#suffix-acronym-collisions +# records the decision, and the removal and the masks it declined. # # Every name here is one of #540's own shape-tagged case rows. No # corpus name written before the change carried a bare trailing # 'meng' or 'lac' (measured 2026-09-25 over the corpus glob at -# 120033b5), so the gate saw nothing until the rows admitted it. -# One set of eight rules for the four 2.x ledgers; the 1.4.0 -# ledger carries four of them (every rule but the family-comes-back -# one, the lone-word one, the dotted one and the report one), 1.4.0 -# having read the other seven names' roles as the tree does. +# 120033b5), so the gate saw nothing until the rows admitted it. The +# same rules stand in the four 2.x ledgers; the 1.4.0 ledger carries +# the two cost rules and the Title-case Lac rule, 1.4.0 having read +# the other names' roles as the tree does. +# +# At landing (2026-09-25) there were eight rules here, and two more +# costs among them: the credential run behind a suffix comma re-read +# as a family comma, and the chunked dotted 'Wang M.Eng.' read as the +# family. #544 (2026-09-27) reads both shapes again and retired both +# rules. The comma-run names are the fix(#544) rules' at the +# bottom of this file; 'Wang M.Eng.' reads the roles this baseline +# read, and its report is the feat(#449) lone-name-word rule's near +# the top. # --------------------------------------------------------------- [[change]] @@ -3625,28 +3677,85 @@ fields = ["_ambiguities"] orders = ["DEFAULT"] [[change]] -issue = "fix(#540) accepted: a credential run behind a suffix comma that ends in meng re-reads the comma as a family comma" -# 'John Smith, PhD MEng' and 'john smith, phd meng': a credential run -# behind a suffix comma is no longer wholly suffix-shaped once its last -# word is a bare ambiguous member, so C1 reads the comma as a family -# comma instead and the run's own words fall into the usual post-comma -# name slots. 'John Smith, PhD MEng' gives given 'PhD', middle 'MEng', -# family 'John Smith'; 'john smith, phd meng' gives given 'phd', -# family 'john smith', suffix 'meng' (the count then peels 'meng' -# with a word to spare). Every release read the whole run as a suffix. -# The pre-existing 'john smith, phd ma' path, which #540 routes two -# more words into; #544 asks whether the run should keep its -# comma. -# -# Literal, two names, fields the union of what each moves ('middle' is -# the first name's alone -- the OVER-DECLARED check accepts a field -# that at least one explained name moves). Probes: 'John Smith, MEng' -# (a lone credential after the comma keeps the suffix, C1's count) and -# 'John Smith, PhD' (a lone credential of the other shape, also kept) -# are _MUST_NOT_MATCH, along with the superstring 'Dr. John Smith, PhD -# MEng'. -name_regex = "^(?:John Smith, PhD MEng|john smith, phd meng)$" -fields = ["given", "middle", "family", "suffix", "_ambiguities"] +issue = "fix(#544) a credential run after a suffix comma keeps the comma, and the flip reports" +# Every role is what this baseline read -- the whole run after the +# comma a suffix -- and what is new is the report. rules.md#C1: "The +# same count reads a part of two or more words as the credential run +# when every word of it is a suffix word or a word of this class, at +# least one of them of this class", and "A decision either way at this +# comma is reported". 'John Smith, PhD MEng' and 'john smith, phd meng' +# come back here from #540's accepted cost, which had read the comma as +# a family comma; 'Jane Doe, MS LAc' and 'John Smith, MD MEng' open the +# part with a title-and-suffix word, which counts as the suffix word it +# is after a full name, where #540 had read it as a title. +# +# `_ambiguities` alone, so this rule cannot absorb a ROLE diff on any of +# them -- the names whose roles move are the rule below. Literal; the +# probes 'John Smith, PhD MA' (capitals in a mixed-case name already +# settle the run, so it reports nothing new) and the superstring 'Dr. +# John Smith, PhD MEng' are _MUST_NOT_MATCH. +name_regex = "^(?:Jane Doe, MS LAc|John Smith, MD MEng|John Smith, MEng PhD|John Smith, PhD MEng|john smith, phd meng)$" +fields = ["_ambiguities"] +orders = ["DEFAULT"] + +[[change]] +issue = "fix(#544) a run of ambiguous members after a suffix comma reads by the name-word count" +# The part after the comma is the credential run on C1's count of the +# two name words before it, where this baseline read it as the listing +# form ('John Smith, Ed Ma' read given 'Ed', middle 'Ma'). A +# title-and-suffix word opening the part counts as the suffix word it is +# after a full name ('John Smith, Ms Ma', the accepted cost, Derek +# 2026-09-27; 'john smith, md ma'), and 'X.Y.Z.' is a member by shape, +# whose capitals lean nothing -- rules.md#S2 reads the writing of a +# LISTED member only -- so its run flips too. 1.4.0 read the listed- +# member names as suffixes as well. +# +# Literal; `fields` is the union the names move ('title' is the duals'). +# Probes: 'Smith, Ed Ma' (one name word before the comma keeps the +# listing form), 'Smith, Ms Ma' (a dual opening the given part is a +# title) and the superstring 'Dr. John Smith, Ed Ma' are _MUST_NOT_MATCH. +name_regex = "^(?:John Smith, Ed Ma|John Smith, Ms Ma|John Smith, X\\.Y\\.Z\\. MA|john smith, md ma)$" +fields = ["given", "middle", "family", "suffix", "title", "_ambiguities"] +orders = ["DEFAULT"] + +[[change]] +issue = "fix(#436/#437/#544) an unambiguous credential in front anchors the member behind it" +# rules.md#S2's company clause: a member written behind an unambiguous +# credential in one run "reads as the credential whatever its writing +# and whatever the count", where the tree before #544 read the Title-case +# MEng as a name word ('John Smith PhD MEng' family 'MEng'; 'Doe, Jane +# PhD MEng' middle 'MEng'), in the maiden clause's reading of the name +# too. The diff against this baseline is what remains once the roles +# come back: the report of the member the peel picked, and -- through +# 2.2 -- the run 1.4.0 and 2.x wrote 'PhD, MEng', which rules.md#R1 +# writes as the writer spaced it (#436/#437), so one rule explains the +# whole of each diff. +# +# The label is joint for that reason, as the 1.4.0 ledger's +# fix(#274/#436/#437) rules are: the `suffix` this rule claims moves on +# the spacing alone, which is #436/#437's, and `_ambiguities` is the +# report #544 adds. The 2.3.0 ledger, whose baseline already spaces the +# run, carries the same rule as #544's alone. +# +# Literal. Probes: 'Wang Ma PhD' (the credential BEHIND the member speaks +# for nothing) and the superstring 'Dr. John Smith PhD MEng' are +# _MUST_NOT_MATCH. +name_regex = "^(?:Doe, Jane PhD MEng|Doe, Jane nee Smith PhD MEng|Jane Doe nee Smith PhD MEng|John Smith PhD MEng)$" +fields = ["suffix", "_ambiguities"] +orders = ["DEFAULT"] + +[[change]] +issue = "fix(#544) a degree in front outranks P6's attachment for a particle member" +# 'doe, jane v phd do': the one-case 'do' leans nothing and is particle +# vocabulary, so P6's attachment would take it into the family, and the +# unambiguous 'phd' in front of it outranks that attachment as capitals +# do (rules.md#S2, decided 2026-09-27) -- suffix 'v phd do', reporting +# suffix-or-name where this baseline read the particle into a name part. +# 'NASCIMENTO, EDSON ARANTES DO', with nothing in front, keeps P6's +# reading. Literal; the superstring 'Dr. doe, jane v phd do' and 'doe, +# jane do' are _MUST_NOT_MATCH. +name_regex = "^doe, jane v phd do$" +fields = ["middle", "suffix", "_ambiguities"] orders = ["DEFAULT"] [[change]] @@ -3664,25 +3773,3 @@ issue = "fix(#540) a Title-case Lac behind a particle is the family name" name_regex = "^Nguyen Van Lac$" fields = ["family", "suffix", "_ambiguities"] orders = ["DEFAULT"] - -[[change]] -issue = "fix(#540) accepted: a chunked dotted M.Eng. with nothing to spare is the family name" -# 'Wang M.Eng.': given 'Wang', family 'M.Eng.', with a suffix-or-name -# report, where every release from 2.0.0 read suffix 'M.Eng.' with no family name. -# S2's period gate counts a member as unambiguous only when it is -# written one period per letter ('M.A.'), and MEng and LAc are the -# first members whose conventional dotted spelling is chunked instead, -# so 'M.Eng.' with nothing to spare reads as the ambiguous bare word -# does and the family name comes back. Accepted and recorded rather -# than widening the period gate (Derek, 2026-09-25); 1.4.0 already read -# this family, unflagged, for an unrelated reason (its two-piece rule: -# a lone word after the given name is the family), so that ledger -# carries no copy of this rule. -# -# Literal, one name; 'John Smith M.Eng.' keeps the suffix by the count -# (words to spare) and 'Wang M.A.' keeps the suffix too (M.A. passes -# the period gate, unaffected by this change) -- both are -# _MUST_NOT_MATCH probes, along with the superstring 'Dr. Wang M.Eng.'. -name_regex = "^Wang M\\.Eng\\.$" -fields = ["family", "suffix", "_ambiguities"] -orders = ["DEFAULT"] diff --git a/tools/differential/expected_since_2.1.0.toml b/tools/differential/expected_since_2.1.0.toml index 1c95431e..44417b48 100644 --- a/tools/differential/expected_since_2.1.0.toml +++ b/tools/differential/expected_since_2.1.0.toml @@ -106,7 +106,13 @@ issue = "fix(#436/#437) a space-separated post-nominal run renders with spaces, # before and after (given 'Jane', family 'Doe', suffix 'i III', # maiden 'Puig'), the link having a generation and not a name word # to its right. What grew is the CORPUS. -name_regex = "^(?:JOHN DOE PHD MD|Jane Doe nee Puig i III|Jane Doe nee Smith PhD MA|John Doe MD PhD|John Smith MD PhD|John Smith Mc V|Josep Lluis Carod i III|Josep Lluis Carod i V|Kenneth Clarke QC MP|Rovira, Josep Carod i Jr\\.|Smith, John PhD I\\.|The Rt Hon Kenneth Clarke QC MP, HMG|Washington Jr\\. MD, Franklin|abdul Smith Jr Ma|abdul Smith Jr V)$" +# 2026-09-27, #544: Three names join the alternation, each diffing here on +# the rendering alone -- the roles are this baseline's, and #544 is what +# put them back ('Jane Doe Jr. Ma', 'John Smith PhD Ed Ma', 'John Smith PhD Ma'). The tree before #544 read the Title-case +# member behind the credential as a name word, the case lean stopping +# the peel at it; the company clause (rules.md#S2) reads it as the +# credential, so the run is v1's again and only R1's spacing differs. +name_regex = "^(?:JOHN DOE PHD MD|Jane Doe Jr\\. Ma|Jane Doe nee Puig i III|Jane Doe nee Smith PhD MA|John Doe MD PhD|John Smith MD PhD|John Smith Mc V|John Smith PhD Ed Ma|John Smith PhD Ma|Josep Lluis Carod i III|Josep Lluis Carod i V|Kenneth Clarke QC MP|Rovira, Josep Carod i Jr\\.|Smith, John PhD I\\.|The Rt Hon Kenneth Clarke QC MP, HMG|Washington Jr\\. MD, Franklin|abdul Smith Jr Ma|abdul Smith Jr V)$" fields = ["suffix"] # The six #449 rules go SECOND, not first: the rule above @@ -118,7 +124,8 @@ fields = ["suffix"] # default (#382). [[change]] issue = "feat(#449) a lone name word reports given-or-family" -# rules.md#O5's convention, now reported. Twenty-two corpus names, and +# rules.md#O5's convention, now reported. Twenty-two corpus names at +# landing (twenty-three since #544, 2026-09-27), and # the alternation is a list of NAMES rather than a copy of any # wordlist -- what selects them is the SHAPE (one name word, and no # title, maiden name, comma, script order or vocabulary claim deciding @@ -132,7 +139,18 @@ issue = "feat(#449) a lone name word reports given-or-family" # GLUED_HONORIFICS. #322/#323 opened the one escape from that pin -- # a member set declared in _NOT_A_VOCABULARY_COPY -- and nothing here # declares one, so the five rules stand. -name_regex = "(?i)^(?:'Smitty' Jones Jr\\.|Andrew|Carod i|Dean of Chemistry|Donald mc|Duke of Edinburgh|Duke of Wellington|Garcia|Jack M\\.A\\.|John & Jane|John V|John of the Doe|Juan & Garcia|Juan and Garcia|Mohamad X|Smith|Smith Jr\\.|e and e|part1 of The part2 of the part3 and part4|part1 of and The part2 of the part3 And part4|test|سلمان،)$" +# 2026-09-27, #544: 'Wang M.Eng.' joins the alternation. Its roles are +# this baseline's, suffix 'M.Eng.', and the whole of its diff here is +# the report this rule describes, as for 'Jack M.A.'. The two readings +# of the suffix have different grounds: this baseline read it because +# 'meng' was an unambiguous acronym, which rules.md#S2 consumes "even +# when that leaves no family name at all"; #540 made 'meng' ambiguous +# in the 2.4 cycle, which read the family, and the tree reads the +# suffix again because S2 counts an ambiguous acronym "written with its +# periods, one after each letter or one after each of two or more +# letter chunks" unambiguously. Neither change is what the name diffs +# on against this baseline. +name_regex = "(?i)^(?:'Smitty' Jones Jr\\.|Andrew|Carod i|Dean of Chemistry|Donald mc|Duke of Edinburgh|Duke of Wellington|Garcia|Jack M\\.A\\.|John & Jane|John V|John of the Doe|Juan & Garcia|Juan and Garcia|Mohamad X|Smith|Smith Jr\\.|Wang M\\.Eng\\.|e and e|part1 of The part2 of the part3 and part4|part1 of and The part2 of the part3 And part4|test|سلمان،)$" fields = ["_ambiguities"] [[change]] @@ -440,7 +458,7 @@ name_regex = "(?i)^abdul\\s+ph\\.\\s+d\\.\\s+smith\\s+berg$" fields = ["given", "middle", "suffix"] [[change]] -issue = "fix(#425) the bound-given reserve runs assign's peel over the joined view" +issue = "fix(#425/#436/#437) the bound-given reserve runs assign's peel over the joined view" # 'abdul Smith Jr Ma', 'abdul Smith Ma': rules.md#P5 -- "the join is # tried on the pieces as it would leave them, assign's trailing peel # (S2) is read over that, and the name words it leaves are the words @@ -468,8 +486,20 @@ issue = "fix(#425) the bound-given reserve runs assign's peel over the joined vi # never reached. `fields` is narrowed to the union the run measures # here -- `given` no longer moves on the one name left diffing, and # a role nothing moves is a standing claim (#452). +# 2026-09-27, #544: the paragraph above is true again, `given` included. +# The company clause (rules.md#S2) reads the 'Ma' behind 'Jr' as the +# credential, so the peel takes both and the reserve declines the join: +# 'abdul Smith Jr Ma' reads given 'abdul', family 'Smith', suffix 'Jr +# Ma', the diff here {given, family, suffix}, and `middle` moves on +# nothing any more. `fields` follows the run. +# Of that diff, {given, family} is #425's -- 2.2.0 and 2.3.0 read those +# roles too -- and `suffix` is R1's spacing alone, the 'Jr, Ma' this +# baseline wrote (#436/#437). #544 moves nothing on it against this +# baseline: it takes back a reading the 2.4 cycle had broken. One rule +# has to explain the whole of the diff, so the label is joint, as the +# 1.4.0 ledger's fix(#274/#436/#437) rules are. name_regex = "(?i)^abdul\\s+smith\\s+(jr\\s+)?ma$" -fields = ["family", "middle", "suffix"] +fields = ["given", "family", "suffix"] [[change]] issue = "fix(#424) an unlisted abbreviation is as transparent as a listed title to the leading particle, the P4 example" @@ -711,6 +741,11 @@ issue = "fix(#325) a split credential followed by another suffix after a one-wor # whole run is suffixes now. Literal rather than the shape: a shape # regex absorbed 'Smith, Dr. Jr.', which is the comma-family rule's # (the comment review). +# 2026-09-27, #544: 'Smith, PhD MEng' is reached too, as 'Smith, PhD +# Jr.' is -- the unambiguous PhD anchors the MEng behind it (rules.md#S2's +# company clause), so the run after the one-word family comma collapses +# whole into `suffix` where the tree before #544 read given 'PhD', +# middle 'MEng'. name_regex = "(?i)^smith,\\s*ph\\.?\\s?d\\.?\\s" fields = ["title", "given", "suffix"] @@ -2365,7 +2400,14 @@ issue = "fix(#289) a written case contrast decides a bare ambiguous acronym" # given baseline is simply one the reading already agreed with there. # `fields` is per-baseline, the union the run at THAT baseline # measures (#452). -name_regex = "^(?:Davis Royce, Ed|Doe, Dr\\. MA|Doe, MA|Doe, MA PhD|Doe, Mr\\. MA PhD|Freiherr von Berg MA|JOHN SMITH, MA|Jack MA|Jack MA\\.|Jack Wei Ma|John Prof\\. MA|John Smith Ma|John Smith, Ed|John Smith, MA|John Smith, Ma|John de Ma|John van der Berg Ma|Smith Jr\\., MA|Smith Jr\\., Ma|Smith, MA|abdul Smith Berg Ma|abdul Smith Jr Ma|abdul Smith Ma|john smith, ma)$" +# 2026-09-27, #544: 'abdul Smith Jr Ma' leaves this alternation. The +# unambiguous 'Jr' in front of 'Ma' anchors it (rules.md#S2's company +# clause), the peel takes both, and P5's reserve declines the join, so +# the accepted cost this rule recorded for it -- middle 'Jr', family +# 'Ma' -- is gone; what the name still diffs on here is claimed by the +# rule that explains it, and a literal left in this one would stand +# ready to explain that cost coming back. +name_regex = "^(?:Davis Royce, Ed|Doe, Dr\\. MA|Doe, MA|Doe, MA PhD|Doe, Mr\\. MA PhD|Freiherr von Berg MA|JOHN SMITH, MA|Jack MA|Jack MA\\.|Jack Wei Ma|John Prof\\. MA|John Smith Ma|John Smith, Ed|John Smith, MA|John Smith, Ma|John de Ma|John van der Berg Ma|Smith Jr\\., MA|Smith Jr\\., Ma|Smith, MA|abdul Smith Berg Ma|abdul Smith Ma|john smith, ma)$" fields = ["family", "given", "middle", "suffix", "_ambiguities"] orders = ["DEFAULT"] @@ -2747,7 +2789,11 @@ issue = "fix(#533) the maiden clause reports the credential it keeps" # rule of its own rather than a widening of the one above: `fields` # is `_ambiguities` alone, so it cannot absorb a role diff on any of # the six. -name_regex = "^(?:Doe, Dr\\. nee Smith MA|Doe, J\\. nee MA ba|Doe, Jane nee Smith Ma|Jane Doe nee MA|Jane Doe nee MA PhD|Jane Doe nee King\\. ba|Jane Doe nee Smith DO DO|Jane Doe nee Smith Ma|Jane Doe nee Yo-Yo Ma)$" +# 2026-09-27, #544: 'Jane Doe Jr. nee Smith Ma' joins, the declining +# half at rules.md#M2's new Accepted boundary: the credential in front +# of the marker speaks for no member of the clause, the Title-case 'Ma' +# is kept, and the clause reports it. +name_regex = "^(?:Doe, Dr\\. nee Smith MA|Doe, J\\. nee MA ba|Doe, Jane nee Smith Ma|Jane Doe Jr\\. nee Smith Ma|Jane Doe nee King\\. ba|Jane Doe nee MA|Jane Doe nee MA PhD|Jane Doe nee Smith DO DO|Jane Doe nee Smith Ma|Jane Doe nee Yo-Yo Ma)$" fields = ["_ambiguities"] [[change]] @@ -3421,28 +3467,34 @@ name_regex = "^John Smith Ph\\.$" fields = ["family", "middle", "suffix"] # --------------------------------------------------------------- -# #540: MENG AND LAC ARE AMBIGUOUS CREDENTIAL ACRONYMS. Eight rules -# -- the two names whose family name comes back, the marking's +# #540: MENG AND LAC ARE AMBIGUOUS CREDENTIAL ACRONYMS. The rules +# below: the two names whose family name comes back, the marking's # cost behind a full name, the same cost after a family comma, the # lone word after a family comma (given name back, cost for the -# conventional spelling), the same cost arriving as a credential -# run behind a suffix comma, the chunked dotted M.Eng. cost, the -# Title-case Lac gain that is the cost's mirror, and the two that -# keep the credential and gain a report. No rule of the parser -# changed: the two words joined suffix_acronyms_ambiguous, and -# rules.md#S2 reads them as it reads 'ma' and 'ba' -- "A BARE -# ambiguous acronym is consumed only when the name has words to -# spare". decisions.md#suffix-acronym-collisions records the -# decision, and the removal and the masks it declined. +# conventional spelling), the Title-case Lac gain that is the cost's +# mirror, and the two that keep the credential and gain a report. No +# rule of the parser changed: the two words joined +# suffix_acronyms_ambiguous, and rules.md#S2 reads them as it reads +# 'ma' and 'ba' -- "A BARE ambiguous acronym is consumed only when the +# name has words to spare". decisions.md#suffix-acronym-collisions +# records the decision, and the removal and the masks it declined. # # Every name here is one of #540's own shape-tagged case rows. No # corpus name written before the change carried a bare trailing # 'meng' or 'lac' (measured 2026-09-25 over the corpus glob at -# 120033b5), so the gate saw nothing until the rows admitted it. -# One set of eight rules for the four 2.x ledgers; the 1.4.0 -# ledger carries four of them (every rule but the family-comes-back -# one, the lone-word one, the dotted one and the report one), 1.4.0 -# having read the other seven names' roles as the tree does. +# 120033b5), so the gate saw nothing until the rows admitted it. The +# same rules stand in the four 2.x ledgers; the 1.4.0 ledger carries +# the two cost rules and the Title-case Lac rule, 1.4.0 having read +# the other names' roles as the tree does. +# +# At landing (2026-09-25) there were eight rules here, and two more +# costs among them: the credential run behind a suffix comma re-read +# as a family comma, and the chunked dotted 'Wang M.Eng.' read as the +# family. #544 (2026-09-27) reads both shapes again and retired both +# rules. The comma-run names are the fix(#544) rules' at the +# bottom of this file; 'Wang M.Eng.' reads the roles this baseline +# read, and its report is the feat(#449) lone-name-word rule's near +# the top. # --------------------------------------------------------------- [[change]] @@ -3536,28 +3588,85 @@ fields = ["_ambiguities"] orders = ["DEFAULT"] [[change]] -issue = "fix(#540) accepted: a credential run behind a suffix comma that ends in meng re-reads the comma as a family comma" -# 'John Smith, PhD MEng' and 'john smith, phd meng': a credential run -# behind a suffix comma is no longer wholly suffix-shaped once its last -# word is a bare ambiguous member, so C1 reads the comma as a family -# comma instead and the run's own words fall into the usual post-comma -# name slots. 'John Smith, PhD MEng' gives given 'PhD', middle 'MEng', -# family 'John Smith'; 'john smith, phd meng' gives given 'phd', -# family 'john smith', suffix 'meng' (the count then peels 'meng' -# with a word to spare). Every release read the whole run as a suffix. -# The pre-existing 'john smith, phd ma' path, which #540 routes two -# more words into; #544 asks whether the run should keep its -# comma. -# -# Literal, two names, fields the union of what each moves ('middle' is -# the first name's alone -- the OVER-DECLARED check accepts a field -# that at least one explained name moves). Probes: 'John Smith, MEng' -# (a lone credential after the comma keeps the suffix, C1's count) and -# 'John Smith, PhD' (a lone credential of the other shape, also kept) -# are _MUST_NOT_MATCH, along with the superstring 'Dr. John Smith, PhD -# MEng'. -name_regex = "^(?:John Smith, PhD MEng|john smith, phd meng)$" -fields = ["given", "middle", "family", "suffix", "_ambiguities"] +issue = "fix(#544) a credential run after a suffix comma keeps the comma, and the flip reports" +# Every role is what this baseline read -- the whole run after the +# comma a suffix -- and what is new is the report. rules.md#C1: "The +# same count reads a part of two or more words as the credential run +# when every word of it is a suffix word or a word of this class, at +# least one of them of this class", and "A decision either way at this +# comma is reported". 'John Smith, PhD MEng' and 'john smith, phd meng' +# come back here from #540's accepted cost, which had read the comma as +# a family comma; 'Jane Doe, MS LAc' and 'John Smith, MD MEng' open the +# part with a title-and-suffix word, which counts as the suffix word it +# is after a full name, where #540 had read it as a title. +# +# `_ambiguities` alone, so this rule cannot absorb a ROLE diff on any of +# them -- the names whose roles move are the rule below. Literal; the +# probes 'John Smith, PhD MA' (capitals in a mixed-case name already +# settle the run, so it reports nothing new) and the superstring 'Dr. +# John Smith, PhD MEng' are _MUST_NOT_MATCH. +name_regex = "^(?:Jane Doe, MS LAc|John Smith, MD MEng|John Smith, MEng PhD|John Smith, PhD MEng|john smith, phd meng)$" +fields = ["_ambiguities"] +orders = ["DEFAULT"] + +[[change]] +issue = "fix(#544) a run of ambiguous members after a suffix comma reads by the name-word count" +# The part after the comma is the credential run on C1's count of the +# two name words before it, where this baseline read it as the listing +# form ('John Smith, Ed Ma' read given 'Ed', middle 'Ma'). A +# title-and-suffix word opening the part counts as the suffix word it is +# after a full name ('John Smith, Ms Ma', the accepted cost, Derek +# 2026-09-27; 'john smith, md ma'), and 'X.Y.Z.' is a member by shape, +# whose capitals lean nothing -- rules.md#S2 reads the writing of a +# LISTED member only -- so its run flips too. 1.4.0 read the listed- +# member names as suffixes as well. +# +# Literal; `fields` is the union the names move ('title' is the duals'). +# Probes: 'Smith, Ed Ma' (one name word before the comma keeps the +# listing form), 'Smith, Ms Ma' (a dual opening the given part is a +# title) and the superstring 'Dr. John Smith, Ed Ma' are _MUST_NOT_MATCH. +name_regex = "^(?:John Smith, Ed Ma|John Smith, Ms Ma|John Smith, X\\.Y\\.Z\\. MA|john smith, md ma)$" +fields = ["given", "middle", "family", "suffix", "title", "_ambiguities"] +orders = ["DEFAULT"] + +[[change]] +issue = "fix(#436/#437/#544) an unambiguous credential in front anchors the member behind it" +# rules.md#S2's company clause: a member written behind an unambiguous +# credential in one run "reads as the credential whatever its writing +# and whatever the count", where the tree before #544 read the Title-case +# MEng as a name word ('John Smith PhD MEng' family 'MEng'; 'Doe, Jane +# PhD MEng' middle 'MEng'), in the maiden clause's reading of the name +# too. The diff against this baseline is what remains once the roles +# come back: the report of the member the peel picked, and -- through +# 2.2 -- the run 1.4.0 and 2.x wrote 'PhD, MEng', which rules.md#R1 +# writes as the writer spaced it (#436/#437), so one rule explains the +# whole of each diff. +# +# The label is joint for that reason, as the 1.4.0 ledger's +# fix(#274/#436/#437) rules are: the `suffix` this rule claims moves on +# the spacing alone, which is #436/#437's, and `_ambiguities` is the +# report #544 adds. The 2.3.0 ledger, whose baseline already spaces the +# run, carries the same rule as #544's alone. +# +# Literal. Probes: 'Wang Ma PhD' (the credential BEHIND the member speaks +# for nothing) and the superstring 'Dr. John Smith PhD MEng' are +# _MUST_NOT_MATCH. +name_regex = "^(?:Doe, Jane PhD MEng|Doe, Jane nee Smith PhD MEng|Jane Doe nee Smith PhD MEng|John Smith PhD MEng)$" +fields = ["suffix", "_ambiguities"] +orders = ["DEFAULT"] + +[[change]] +issue = "fix(#544) a degree in front outranks P6's attachment for a particle member" +# 'doe, jane v phd do': the one-case 'do' leans nothing and is particle +# vocabulary, so P6's attachment would take it into the family, and the +# unambiguous 'phd' in front of it outranks that attachment as capitals +# do (rules.md#S2, decided 2026-09-27) -- suffix 'v phd do', reporting +# suffix-or-name where this baseline read the particle into a name part. +# 'NASCIMENTO, EDSON ARANTES DO', with nothing in front, keeps P6's +# reading. Literal; the superstring 'Dr. doe, jane v phd do' and 'doe, +# jane do' are _MUST_NOT_MATCH. +name_regex = "^doe, jane v phd do$" +fields = ["middle", "suffix", "_ambiguities"] orders = ["DEFAULT"] [[change]] @@ -3575,25 +3684,3 @@ issue = "fix(#540) a Title-case Lac behind a particle is the family name" name_regex = "^Nguyen Van Lac$" fields = ["family", "suffix", "_ambiguities"] orders = ["DEFAULT"] - -[[change]] -issue = "fix(#540) accepted: a chunked dotted M.Eng. with nothing to spare is the family name" -# 'Wang M.Eng.': given 'Wang', family 'M.Eng.', with a suffix-or-name -# report, where every release from 2.0.0 read suffix 'M.Eng.' with no family name. -# S2's period gate counts a member as unambiguous only when it is -# written one period per letter ('M.A.'), and MEng and LAc are the -# first members whose conventional dotted spelling is chunked instead, -# so 'M.Eng.' with nothing to spare reads as the ambiguous bare word -# does and the family name comes back. Accepted and recorded rather -# than widening the period gate (Derek, 2026-09-25); 1.4.0 already read -# this family, unflagged, for an unrelated reason (its two-piece rule: -# a lone word after the given name is the family), so that ledger -# carries no copy of this rule. -# -# Literal, one name; 'John Smith M.Eng.' keeps the suffix by the count -# (words to spare) and 'Wang M.A.' keeps the suffix too (M.A. passes -# the period gate, unaffected by this change) -- both are -# _MUST_NOT_MATCH probes, along with the superstring 'Dr. Wang M.Eng.'. -name_regex = "^Wang M\\.Eng\\.$" -fields = ["family", "suffix", "_ambiguities"] -orders = ["DEFAULT"] diff --git a/tools/differential/expected_since_2.2.0.toml b/tools/differential/expected_since_2.2.0.toml index 30901436..6cc62125 100644 --- a/tools/differential/expected_since_2.2.0.toml +++ b/tools/differential/expected_since_2.2.0.toml @@ -95,7 +95,13 @@ issue = "fix(#436/#437) a space-separated post-nominal run renders with spaces, # before and after (given 'Jane', family 'Doe', suffix 'i III', # maiden 'Puig'), the link having a generation and not a name word # to its right. What grew is the CORPUS. -name_regex = "^(?:JOHN DOE PHD MD|Jane Doe nee Puig i III|Jane Doe nee Smith PhD MA|John Doe MD PhD|John Smith MD PhD|John Smith Mc V|Josep Lluis Carod i III|Josep Lluis Carod i V|Kenneth Clarke QC MP|Rovira, Josep Carod i Jr\\.|Smith, John PhD I\\.|The Rt Hon Kenneth Clarke QC MP, HMG|Washington Jr\\. MD, Franklin|abdul Smith Jr Ma|abdul Smith Jr V)$" +# 2026-09-27, #544: Three names join the alternation, each diffing here on +# the rendering alone -- the roles are this baseline's, and #544 is what +# put them back ('Jane Doe Jr. Ma', 'John Smith PhD Ed Ma', 'John Smith PhD Ma'). The tree before #544 read the Title-case +# member behind the credential as a name word, the case lean stopping +# the peel at it; the company clause (rules.md#S2) reads it as the +# credential, so the run is v1's again and only R1's spacing differs. +name_regex = "^(?:JOHN DOE PHD MD|Jane Doe Jr\\. Ma|Jane Doe nee Puig i III|Jane Doe nee Smith PhD MA|John Doe MD PhD|John Smith MD PhD|John Smith Mc V|John Smith PhD Ed Ma|John Smith PhD Ma|Josep Lluis Carod i III|Josep Lluis Carod i V|Kenneth Clarke QC MP|Rovira, Josep Carod i Jr\\.|Smith, John PhD I\\.|The Rt Hon Kenneth Clarke QC MP, HMG|Washington Jr\\. MD, Franklin|abdul Smith Jr Ma|abdul Smith Jr V)$" fields = ["suffix"] # The six #449 rules go SECOND, not first: the rule above @@ -107,7 +113,8 @@ fields = ["suffix"] # default (#382). [[change]] issue = "feat(#449) a lone name word reports given-or-family" -# rules.md#O5's convention, now reported. Twenty-two corpus names, and +# rules.md#O5's convention, now reported. Twenty-two corpus names at +# landing (twenty-three since #544, 2026-09-27), and # the alternation is a list of NAMES rather than a copy of any # wordlist -- what selects them is the SHAPE (one name word, and no # title, maiden name, comma, script order or vocabulary claim deciding @@ -121,7 +128,18 @@ issue = "feat(#449) a lone name word reports given-or-family" # GLUED_HONORIFICS. #322/#323 opened the one escape from that pin -- # a member set declared in _NOT_A_VOCABULARY_COPY -- and nothing here # declares one, so the five rules stand. -name_regex = "(?i)^(?:'Smitty' Jones Jr\\.|Andrew|Carod i|Dean of Chemistry|Donald mc|Duke of Edinburgh|Duke of Wellington|Garcia|Jack M\\.A\\.|John & Jane|John V|John of the Doe|Juan & Garcia|Juan and Garcia|Mohamad X|Smith|Smith Jr\\.|e and e|part1 of The part2 of the part3 and part4|part1 of and The part2 of the part3 And part4|test|سلمان،)$" +# 2026-09-27, #544: 'Wang M.Eng.' joins the alternation. Its roles are +# this baseline's, suffix 'M.Eng.', and the whole of its diff here is +# the report this rule describes, as for 'Jack M.A.'. The two readings +# of the suffix have different grounds: this baseline read it because +# 'meng' was an unambiguous acronym, which rules.md#S2 consumes "even +# when that leaves no family name at all"; #540 made 'meng' ambiguous +# in the 2.4 cycle, which read the family, and the tree reads the +# suffix again because S2 counts an ambiguous acronym "written with its +# periods, one after each letter or one after each of two or more +# letter chunks" unambiguously. Neither change is what the name diffs +# on against this baseline. +name_regex = "(?i)^(?:'Smitty' Jones Jr\\.|Andrew|Carod i|Dean of Chemistry|Donald mc|Duke of Edinburgh|Duke of Wellington|Garcia|Jack M\\.A\\.|John & Jane|John V|John of the Doe|Juan & Garcia|Juan and Garcia|Mohamad X|Smith|Smith Jr\\.|Wang M\\.Eng\\.|e and e|part1 of The part2 of the part3 and part4|part1 of and The part2 of the part3 And part4|test|سلمان،)$" fields = ["_ambiguities"] [[change]] @@ -975,7 +993,14 @@ issue = "fix(#289) a written case contrast decides a bare ambiguous acronym" # given baseline is simply one the reading already agreed with there. # `fields` is per-baseline, the union the run at THAT baseline # measures (#452). -name_regex = "^(?:Davis Royce, Ed|Doe, Dr\\. MA|Doe, MA|Doe, MA PhD|Doe, Mr\\. MA PhD|Freiherr von Berg MA|JOHN SMITH, MA|Jack MA|Jack MA\\.|Jack Wei Ma|John Prof\\. MA|John Smith Ma|John Smith, Ed|John Smith, MA|John Smith, Ma|John de Ma|John van der Berg Ma|Smith Jr\\., MA|Smith Jr\\., Ma|Smith, MA|abdul Smith Berg Ma|abdul Smith Jr Ma|abdul Smith Ma|john smith, ma)$" +# 2026-09-27, #544: 'abdul Smith Jr Ma' leaves this alternation. The +# unambiguous 'Jr' in front of 'Ma' anchors it (rules.md#S2's company +# clause), the peel takes both, and P5's reserve declines the join, so +# the accepted cost this rule recorded for it -- middle 'Jr', family +# 'Ma' -- is gone; what the name still diffs on here is claimed by the +# rule that explains it, and a literal left in this one would stand +# ready to explain that cost coming back. +name_regex = "^(?:Davis Royce, Ed|Doe, Dr\\. MA|Doe, MA|Doe, MA PhD|Doe, Mr\\. MA PhD|Freiherr von Berg MA|JOHN SMITH, MA|Jack MA|Jack MA\\.|Jack Wei Ma|John Prof\\. MA|John Smith Ma|John Smith, Ed|John Smith, MA|John Smith, Ma|John de Ma|John van der Berg Ma|Smith Jr\\., MA|Smith Jr\\., Ma|Smith, MA|abdul Smith Berg Ma|abdul Smith Ma|john smith, ma)$" fields = ["family", "given", "middle", "suffix", "_ambiguities"] orders = ["DEFAULT"] @@ -1331,7 +1356,11 @@ issue = "fix(#533) the maiden clause reports the credential it keeps" # # Literal-anchored: the class is the slot's declining half, and a # regex for it would claim the seventeen movers above. -name_regex = "^(?:Berg, Jane van der nee Smith DO|Berg, abdul nee Jones MA|Doe, Dr\\. nee Smith MA|Doe, Jane nee Smith Do|Doe, Jane nee Smith MA do|Doe, Jane nee Smith Ma|Doe, Jane nee Smith do|JOHN NEE JONES SMITH MA PHD|Jane Doe nee King\\. ba|Jane Doe nee MA|Jane Doe nee MA PhD|Jane Doe nee Smith DO DO|Jane Doe nee Smith Ma|Jane Doe nee Yo-Yo Ma|John née Jones Smith Ma)$" +# 2026-09-27, #544: 'Jane Doe Jr. nee Smith Ma' joins, the declining +# half at rules.md#M2's new Accepted boundary: the credential in front +# of the marker speaks for no member of the clause, the Title-case 'Ma' +# is kept, and the clause reports it. +name_regex = "^(?:Berg, Jane van der nee Smith DO|Berg, abdul nee Jones MA|Doe, Dr\\. nee Smith MA|Doe, Jane nee Smith Do|Doe, Jane nee Smith MA do|Doe, Jane nee Smith Ma|Doe, Jane nee Smith do|JOHN NEE JONES SMITH MA PHD|Jane Doe Jr\\. nee Smith Ma|Jane Doe nee King\\. ba|Jane Doe nee MA|Jane Doe nee MA PhD|Jane Doe nee Smith DO DO|Jane Doe nee Smith Ma|Jane Doe nee Yo-Yo Ma|John née Jones Smith Ma)$" fields = ["_ambiguities"] [[change]] @@ -1833,28 +1862,34 @@ name_regex = "^John Smith Ph\\.$" fields = ["family", "middle", "suffix"] # --------------------------------------------------------------- -# #540: MENG AND LAC ARE AMBIGUOUS CREDENTIAL ACRONYMS. Eight rules -# -- the two names whose family name comes back, the marking's +# #540: MENG AND LAC ARE AMBIGUOUS CREDENTIAL ACRONYMS. The rules +# below: the two names whose family name comes back, the marking's # cost behind a full name, the same cost after a family comma, the # lone word after a family comma (given name back, cost for the -# conventional spelling), the same cost arriving as a credential -# run behind a suffix comma, the chunked dotted M.Eng. cost, the -# Title-case Lac gain that is the cost's mirror, and the two that -# keep the credential and gain a report. No rule of the parser -# changed: the two words joined suffix_acronyms_ambiguous, and -# rules.md#S2 reads them as it reads 'ma' and 'ba' -- "A BARE -# ambiguous acronym is consumed only when the name has words to -# spare". decisions.md#suffix-acronym-collisions records the -# decision, and the removal and the masks it declined. +# conventional spelling), the Title-case Lac gain that is the cost's +# mirror, and the two that keep the credential and gain a report. No +# rule of the parser changed: the two words joined +# suffix_acronyms_ambiguous, and rules.md#S2 reads them as it reads +# 'ma' and 'ba' -- "A BARE ambiguous acronym is consumed only when the +# name has words to spare". decisions.md#suffix-acronym-collisions +# records the decision, and the removal and the masks it declined. # # Every name here is one of #540's own shape-tagged case rows. No # corpus name written before the change carried a bare trailing # 'meng' or 'lac' (measured 2026-09-25 over the corpus glob at -# 120033b5), so the gate saw nothing until the rows admitted it. -# One set of eight rules for the four 2.x ledgers; the 1.4.0 -# ledger carries four of them (every rule but the family-comes-back -# one, the lone-word one, the dotted one and the report one), 1.4.0 -# having read the other seven names' roles as the tree does. +# 120033b5), so the gate saw nothing until the rows admitted it. The +# same rules stand in the four 2.x ledgers; the 1.4.0 ledger carries +# the two cost rules and the Title-case Lac rule, 1.4.0 having read +# the other names' roles as the tree does. +# +# At landing (2026-09-25) there were eight rules here, and two more +# costs among them: the credential run behind a suffix comma re-read +# as a family comma, and the chunked dotted 'Wang M.Eng.' read as the +# family. #544 (2026-09-27) reads both shapes again and retired both +# rules. The comma-run names are the fix(#544) rules' at the +# bottom of this file; 'Wang M.Eng.' reads the roles this baseline +# read, and its report is the feat(#449) lone-name-word rule's near +# the top. # --------------------------------------------------------------- [[change]] @@ -1948,28 +1983,85 @@ fields = ["_ambiguities"] orders = ["DEFAULT"] [[change]] -issue = "fix(#540) accepted: a credential run behind a suffix comma that ends in meng re-reads the comma as a family comma" -# 'John Smith, PhD MEng' and 'john smith, phd meng': a credential run -# behind a suffix comma is no longer wholly suffix-shaped once its last -# word is a bare ambiguous member, so C1 reads the comma as a family -# comma instead and the run's own words fall into the usual post-comma -# name slots. 'John Smith, PhD MEng' gives given 'PhD', middle 'MEng', -# family 'John Smith'; 'john smith, phd meng' gives given 'phd', -# family 'john smith', suffix 'meng' (the count then peels 'meng' -# with a word to spare). Every release read the whole run as a suffix. -# The pre-existing 'john smith, phd ma' path, which #540 routes two -# more words into; #544 asks whether the run should keep its -# comma. -# -# Literal, two names, fields the union of what each moves ('middle' is -# the first name's alone -- the OVER-DECLARED check accepts a field -# that at least one explained name moves). Probes: 'John Smith, MEng' -# (a lone credential after the comma keeps the suffix, C1's count) and -# 'John Smith, PhD' (a lone credential of the other shape, also kept) -# are _MUST_NOT_MATCH, along with the superstring 'Dr. John Smith, PhD -# MEng'. -name_regex = "^(?:John Smith, PhD MEng|john smith, phd meng)$" -fields = ["given", "middle", "family", "suffix", "_ambiguities"] +issue = "fix(#544) a credential run after a suffix comma keeps the comma, and the flip reports" +# Every role is what this baseline read -- the whole run after the +# comma a suffix -- and what is new is the report. rules.md#C1: "The +# same count reads a part of two or more words as the credential run +# when every word of it is a suffix word or a word of this class, at +# least one of them of this class", and "A decision either way at this +# comma is reported". 'John Smith, PhD MEng' and 'john smith, phd meng' +# come back here from #540's accepted cost, which had read the comma as +# a family comma; 'Jane Doe, MS LAc' and 'John Smith, MD MEng' open the +# part with a title-and-suffix word, which counts as the suffix word it +# is after a full name, where #540 had read it as a title. +# +# `_ambiguities` alone, so this rule cannot absorb a ROLE diff on any of +# them -- the names whose roles move are the rule below. Literal; the +# probes 'John Smith, PhD MA' (capitals in a mixed-case name already +# settle the run, so it reports nothing new) and the superstring 'Dr. +# John Smith, PhD MEng' are _MUST_NOT_MATCH. +name_regex = "^(?:Jane Doe, MS LAc|John Smith, MD MEng|John Smith, MEng PhD|John Smith, PhD MEng|john smith, phd meng)$" +fields = ["_ambiguities"] +orders = ["DEFAULT"] + +[[change]] +issue = "fix(#544) a run of ambiguous members after a suffix comma reads by the name-word count" +# The part after the comma is the credential run on C1's count of the +# two name words before it, where this baseline read it as the listing +# form ('John Smith, Ed Ma' read given 'Ed', middle 'Ma'). A +# title-and-suffix word opening the part counts as the suffix word it is +# after a full name ('John Smith, Ms Ma', the accepted cost, Derek +# 2026-09-27; 'john smith, md ma'), and 'X.Y.Z.' is a member by shape, +# whose capitals lean nothing -- rules.md#S2 reads the writing of a +# LISTED member only -- so its run flips too. 1.4.0 read the listed- +# member names as suffixes as well. +# +# Literal; `fields` is the union the names move ('title' is the duals'). +# Probes: 'Smith, Ed Ma' (one name word before the comma keeps the +# listing form), 'Smith, Ms Ma' (a dual opening the given part is a +# title) and the superstring 'Dr. John Smith, Ed Ma' are _MUST_NOT_MATCH. +name_regex = "^(?:John Smith, Ed Ma|John Smith, Ms Ma|John Smith, X\\.Y\\.Z\\. MA|john smith, md ma)$" +fields = ["given", "middle", "family", "suffix", "title", "_ambiguities"] +orders = ["DEFAULT"] + +[[change]] +issue = "fix(#436/#437/#544) an unambiguous credential in front anchors the member behind it" +# rules.md#S2's company clause: a member written behind an unambiguous +# credential in one run "reads as the credential whatever its writing +# and whatever the count", where the tree before #544 read the Title-case +# MEng as a name word ('John Smith PhD MEng' family 'MEng'; 'Doe, Jane +# PhD MEng' middle 'MEng'), in the maiden clause's reading of the name +# too. The diff against this baseline is what remains once the roles +# come back: the report of the member the peel picked, and -- through +# 2.2 -- the run 1.4.0 and 2.x wrote 'PhD, MEng', which rules.md#R1 +# writes as the writer spaced it (#436/#437), so one rule explains the +# whole of each diff. +# +# The label is joint for that reason, as the 1.4.0 ledger's +# fix(#274/#436/#437) rules are: the `suffix` this rule claims moves on +# the spacing alone, which is #436/#437's, and `_ambiguities` is the +# report #544 adds. The 2.3.0 ledger, whose baseline already spaces the +# run, carries the same rule as #544's alone. +# +# Literal. Probes: 'Wang Ma PhD' (the credential BEHIND the member speaks +# for nothing) and the superstring 'Dr. John Smith PhD MEng' are +# _MUST_NOT_MATCH. +name_regex = "^(?:Doe, Jane PhD MEng|Doe, Jane nee Smith PhD MEng|Jane Doe nee Smith PhD MEng|John Smith PhD MEng)$" +fields = ["suffix", "_ambiguities"] +orders = ["DEFAULT"] + +[[change]] +issue = "fix(#544) a degree in front outranks P6's attachment for a particle member" +# 'doe, jane v phd do': the one-case 'do' leans nothing and is particle +# vocabulary, so P6's attachment would take it into the family, and the +# unambiguous 'phd' in front of it outranks that attachment as capitals +# do (rules.md#S2, decided 2026-09-27) -- suffix 'v phd do', reporting +# suffix-or-name where this baseline read the particle into a name part. +# 'NASCIMENTO, EDSON ARANTES DO', with nothing in front, keeps P6's +# reading. Literal; the superstring 'Dr. doe, jane v phd do' and 'doe, +# jane do' are _MUST_NOT_MATCH. +name_regex = "^doe, jane v phd do$" +fields = ["family", "suffix", "_ambiguities"] orders = ["DEFAULT"] [[change]] @@ -1987,25 +2079,3 @@ issue = "fix(#540) a Title-case Lac behind a particle is the family name" name_regex = "^Nguyen Van Lac$" fields = ["family", "suffix", "_ambiguities"] orders = ["DEFAULT"] - -[[change]] -issue = "fix(#540) accepted: a chunked dotted M.Eng. with nothing to spare is the family name" -# 'Wang M.Eng.': given 'Wang', family 'M.Eng.', with a suffix-or-name -# report, where every release from 2.0.0 read suffix 'M.Eng.' with no family name. -# S2's period gate counts a member as unambiguous only when it is -# written one period per letter ('M.A.'), and MEng and LAc are the -# first members whose conventional dotted spelling is chunked instead, -# so 'M.Eng.' with nothing to spare reads as the ambiguous bare word -# does and the family name comes back. Accepted and recorded rather -# than widening the period gate (Derek, 2026-09-25); 1.4.0 already read -# this family, unflagged, for an unrelated reason (its two-piece rule: -# a lone word after the given name is the family), so that ledger -# carries no copy of this rule. -# -# Literal, one name; 'John Smith M.Eng.' keeps the suffix by the count -# (words to spare) and 'Wang M.A.' keeps the suffix too (M.A. passes -# the period gate, unaffected by this change) -- both are -# _MUST_NOT_MATCH probes, along with the superstring 'Dr. Wang M.Eng.'. -name_regex = "^Wang M\\.Eng\\.$" -fields = ["family", "suffix", "_ambiguities"] -orders = ["DEFAULT"] diff --git a/tools/differential/expected_since_2.3.0.toml b/tools/differential/expected_since_2.3.0.toml index 318204a5..7c3c43b5 100644 --- a/tools/differential/expected_since_2.3.0.toml +++ b/tools/differential/expected_since_2.3.0.toml @@ -281,7 +281,14 @@ issue = "fix(#289) a written case contrast decides a bare ambiguous acronym" # given baseline is simply one the reading already agreed with there. # `fields` is per-baseline, the union the run at THAT baseline # measures (#452). -name_regex = "^(?:Davis Royce, Ed|Doe, Dr\\. MA|Doe, MA|Doe, MA PhD|Doe, Mr\\. MA PhD|Freiherr von Berg MA|JOHN SMITH, MA|Jack MA|Jack MA\\.|Jack Wei Ma|John Prof\\. MA|John Smith Ma|John Smith, Ed|John Smith, MA|John Smith, Ma|John de Ma|John van der Berg Ma|Smith Jr\\., MA|Smith Jr\\., Ma|Smith, MA|abdul Smith Berg Ma|abdul Smith Jr Ma|abdul Smith Ma|john smith, ma)$" +# 2026-09-27, #544: 'abdul Smith Jr Ma' leaves this alternation. The +# unambiguous 'Jr' in front of 'Ma' anchors it (rules.md#S2's company +# clause), the peel takes both, and P5's reserve declines the join, so +# the accepted cost this rule recorded for it -- middle 'Jr', family +# 'Ma' -- is gone; what the name still diffs on here is claimed by the +# rule that explains it, and a literal left in this one would stand +# ready to explain that cost coming back. +name_regex = "^(?:Davis Royce, Ed|Doe, Dr\\. MA|Doe, MA|Doe, MA PhD|Doe, Mr\\. MA PhD|Freiherr von Berg MA|JOHN SMITH, MA|Jack MA|Jack MA\\.|Jack Wei Ma|John Prof\\. MA|John Smith Ma|John Smith, Ed|John Smith, MA|John Smith, Ma|John de Ma|John van der Berg Ma|Smith Jr\\., MA|Smith Jr\\., Ma|Smith, MA|abdul Smith Berg Ma|abdul Smith Ma|john smith, ma)$" fields = ["family", "given", "middle", "suffix", "_ambiguities"] orders = ["DEFAULT"] @@ -639,7 +646,11 @@ issue = "fix(#533) the maiden clause reports the credential it keeps" # does, so #535 moves nothing. # Literal-anchored: the class is the slot's declining half, and a # regex for it would claim the seventeen movers above. -name_regex = "^(?:Berg, Jane van der nee Smith DO|Berg, abdul nee Jones MA|Doe nee Smith Prof\\. ba|Doe, Dr\\. nee Smith MA|Doe, Jane nee Smith Do|Doe, Jane nee Smith MA do|Doe, Jane nee Smith Ma|Doe, Jane nee Smith do|JOHN NEE JONES SMITH MA PHD|Jane Doe nee MA|Jane Doe nee MA PhD|Jane Doe nee Smith DO DO|Jane Doe nee Smith Ma|Jane Doe nee Yo-Yo Ma|John née Jones Smith Ma)$" +# 2026-09-27, #544: 'Jane Doe Jr. nee Smith Ma' joins, the declining +# half at rules.md#M2's new Accepted boundary: the credential in front +# of the marker speaks for no member of the clause, the Title-case 'Ma' +# is kept, and the clause reports it. +name_regex = "^(?:Berg, Jane van der nee Smith DO|Berg, abdul nee Jones MA|Doe nee Smith Prof\\. ba|Doe, Dr\\. nee Smith MA|Doe, Jane nee Smith Do|Doe, Jane nee Smith MA do|Doe, Jane nee Smith Ma|Doe, Jane nee Smith do|JOHN NEE JONES SMITH MA PHD|Jane Doe Jr\\. nee Smith Ma|Jane Doe nee MA|Jane Doe nee MA PhD|Jane Doe nee Smith DO DO|Jane Doe nee Smith Ma|Jane Doe nee Yo-Yo Ma|John née Jones Smith Ma)$" fields = ["_ambiguities"] [[change]] @@ -1119,28 +1130,33 @@ name_regex = "^John Smith Ph\\.$" fields = ["family", "middle", "suffix"] # --------------------------------------------------------------- -# #540: MENG AND LAC ARE AMBIGUOUS CREDENTIAL ACRONYMS. Eight rules -# -- the two names whose family name comes back, the marking's +# #540: MENG AND LAC ARE AMBIGUOUS CREDENTIAL ACRONYMS. The rules +# below: the two names whose family name comes back, the marking's # cost behind a full name, the same cost after a family comma, the # lone word after a family comma (given name back, cost for the -# conventional spelling), the same cost arriving as a credential -# run behind a suffix comma, the chunked dotted M.Eng. cost, the -# Title-case Lac gain that is the cost's mirror, and the two that -# keep the credential and gain a report. No rule of the parser -# changed: the two words joined suffix_acronyms_ambiguous, and -# rules.md#S2 reads them as it reads 'ma' and 'ba' -- "A BARE -# ambiguous acronym is consumed only when the name has words to -# spare". decisions.md#suffix-acronym-collisions records the -# decision, and the removal and the masks it declined. +# conventional spelling), the Title-case Lac gain that is the cost's +# mirror, and the two that keep the credential and gain a report. No +# rule of the parser changed: the two words joined +# suffix_acronyms_ambiguous, and rules.md#S2 reads them as it reads +# 'ma' and 'ba' -- "A BARE ambiguous acronym is consumed only when the +# name has words to spare". decisions.md#suffix-acronym-collisions +# records the decision, and the removal and the masks it declined. # # Every name here is one of #540's own shape-tagged case rows. No # corpus name written before the change carried a bare trailing # 'meng' or 'lac' (measured 2026-09-25 over the corpus glob at -# 120033b5), so the gate saw nothing until the rows admitted it. -# One set of eight rules for the four 2.x ledgers; the 1.4.0 -# ledger carries four of them (every rule but the family-comes-back -# one, the lone-word one, the dotted one and the report one), 1.4.0 -# having read the other seven names' roles as the tree does. +# 120033b5), so the gate saw nothing until the rows admitted it. The +# same rules stand in the four 2.x ledgers; the 1.4.0 ledger carries +# the two cost rules and the Title-case Lac rule, 1.4.0 having read +# the other names' roles as the tree does. +# +# At landing (2026-09-25) there were eight rules here, and two more +# costs among them: the credential run behind a suffix comma re-read +# as a family comma, and the chunked dotted 'Wang M.Eng.' read as the +# family. #544 (2026-09-27) reads both shapes again and retired both +# rules. The comma-run names are the fix(#544) rules' at the +# bottom of this file; 'Wang M.Eng.' does not diff here, reading as +# this baseline read it, report and all. # --------------------------------------------------------------- [[change]] @@ -1234,28 +1250,79 @@ fields = ["_ambiguities"] orders = ["DEFAULT"] [[change]] -issue = "fix(#540) accepted: a credential run behind a suffix comma that ends in meng re-reads the comma as a family comma" -# 'John Smith, PhD MEng' and 'john smith, phd meng': a credential run -# behind a suffix comma is no longer wholly suffix-shaped once its last -# word is a bare ambiguous member, so C1 reads the comma as a family -# comma instead and the run's own words fall into the usual post-comma -# name slots. 'John Smith, PhD MEng' gives given 'PhD', middle 'MEng', -# family 'John Smith'; 'john smith, phd meng' gives given 'phd', -# family 'john smith', suffix 'meng' (the count then peels 'meng' -# with a word to spare). Every release read the whole run as a suffix. -# The pre-existing 'john smith, phd ma' path, which #540 routes two -# more words into; #544 asks whether the run should keep its -# comma. -# -# Literal, two names, fields the union of what each moves ('middle' is -# the first name's alone -- the OVER-DECLARED check accepts a field -# that at least one explained name moves). Probes: 'John Smith, MEng' -# (a lone credential after the comma keeps the suffix, C1's count) and -# 'John Smith, PhD' (a lone credential of the other shape, also kept) -# are _MUST_NOT_MATCH, along with the superstring 'Dr. John Smith, PhD -# MEng'. -name_regex = "^(?:John Smith, PhD MEng|john smith, phd meng)$" -fields = ["given", "middle", "family", "suffix", "_ambiguities"] +issue = "fix(#544) a credential run after a suffix comma keeps the comma, and the flip reports" +# Every role is what this baseline read -- the whole run after the +# comma a suffix -- and what is new is the report. rules.md#C1: "The +# same count reads a part of two or more words as the credential run +# when every word of it is a suffix word or a word of this class, at +# least one of them of this class", and "A decision either way at this +# comma is reported". 'John Smith, PhD MEng' and 'john smith, phd meng' +# come back here from #540's accepted cost, which had read the comma as +# a family comma; 'Jane Doe, MS LAc' and 'John Smith, MD MEng' open the +# part with a title-and-suffix word, which counts as the suffix word it +# is after a full name, where #540 had read it as a title. +# +# `_ambiguities` alone, so this rule cannot absorb a ROLE diff on any of +# them -- the names whose roles move are the rule below. Literal; the +# probes 'John Smith, PhD MA' (capitals in a mixed-case name already +# settle the run, so it reports nothing new) and the superstring 'Dr. +# John Smith, PhD MEng' are _MUST_NOT_MATCH. +name_regex = "^(?:Jane Doe, MS LAc|John Smith, MD MEng|John Smith, MEng PhD|John Smith, PhD MEng|john smith, phd meng)$" +fields = ["_ambiguities"] +orders = ["DEFAULT"] + +[[change]] +issue = "fix(#544) a run of ambiguous members after a suffix comma reads by the name-word count" +# The part after the comma is the credential run on C1's count of the +# two name words before it, where this baseline read it as the listing +# form ('John Smith, Ed Ma' read given 'Ed', middle 'Ma'). A +# title-and-suffix word opening the part counts as the suffix word it is +# after a full name ('John Smith, Ms Ma', the accepted cost, Derek +# 2026-09-27; 'john smith, md ma'), and 'X.Y.Z.' is a member by shape, +# whose capitals lean nothing -- rules.md#S2 reads the writing of a +# LISTED member only -- so its run flips too. 1.4.0 read the listed- +# member names as suffixes as well. +# +# Literal; `fields` is the union the names move ('title' is the duals'). +# Probes: 'Smith, Ed Ma' (one name word before the comma keeps the +# listing form), 'Smith, Ms Ma' (a dual opening the given part is a +# title) and the superstring 'Dr. John Smith, Ed Ma' are _MUST_NOT_MATCH. +name_regex = "^(?:John Smith, Ed Ma|John Smith, Ms Ma|John Smith, X\\.Y\\.Z\\. MA|john smith, md ma)$" +fields = ["given", "middle", "family", "suffix", "title", "_ambiguities"] +orders = ["DEFAULT"] + +[[change]] +issue = "fix(#544) an unambiguous credential in front anchors the member behind it" +# rules.md#S2's company clause: a member written behind an unambiguous +# credential in one run "reads as the credential whatever its writing +# and whatever the count", where the tree before #544 read the Title-case +# MEng as a name word ('John Smith PhD MEng' family 'MEng'; 'Doe, Jane +# PhD MEng' middle 'MEng'), in the maiden clause's reading of the name +# too. The diff against this baseline is what remains once the roles +# come back: the report of the member the peel picked, and -- through +# 2.2 -- the run 1.4.0 and 2.x wrote 'PhD, MEng', which rules.md#R1 +# writes as the writer spaced it (#436/#437), so one rule explains the +# whole of each diff. +# +# Literal. Probes: 'Wang Ma PhD' (the credential BEHIND the member speaks +# for nothing) and the superstring 'Dr. John Smith PhD MEng' are +# _MUST_NOT_MATCH. +name_regex = "^(?:Doe, Jane PhD MEng|Doe, Jane nee Smith PhD MEng|Jane Doe nee Smith PhD MEng|John Smith PhD MEng)$" +fields = ["_ambiguities"] +orders = ["DEFAULT"] + +[[change]] +issue = "fix(#544) a degree in front outranks P6's attachment for a particle member" +# 'doe, jane v phd do': the one-case 'do' leans nothing and is particle +# vocabulary, so P6's attachment would take it into the family, and the +# unambiguous 'phd' in front of it outranks that attachment as capitals +# do (rules.md#S2, decided 2026-09-27) -- suffix 'v phd do', reporting +# suffix-or-name where this baseline read the particle into a name part. +# 'NASCIMENTO, EDSON ARANTES DO', with nothing in front, keeps P6's +# reading. Literal; the superstring 'Dr. doe, jane v phd do' and 'doe, +# jane do' are _MUST_NOT_MATCH. +name_regex = "^doe, jane v phd do$" +fields = ["family", "suffix", "_ambiguities"] orders = ["DEFAULT"] [[change]] @@ -1273,25 +1340,3 @@ issue = "fix(#540) a Title-case Lac behind a particle is the family name" name_regex = "^Nguyen Van Lac$" fields = ["family", "suffix", "_ambiguities"] orders = ["DEFAULT"] - -[[change]] -issue = "fix(#540) accepted: a chunked dotted M.Eng. with nothing to spare is the family name" -# 'Wang M.Eng.': given 'Wang', family 'M.Eng.', with a suffix-or-name -# report, where every release from 2.0.0 read suffix 'M.Eng.' with no family name. -# S2's period gate counts a member as unambiguous only when it is -# written one period per letter ('M.A.'), and MEng and LAc are the -# first members whose conventional dotted spelling is chunked instead, -# so 'M.Eng.' with nothing to spare reads as the ambiguous bare word -# does and the family name comes back. Accepted and recorded rather -# than widening the period gate (Derek, 2026-09-25); 1.4.0 already read -# this family, unflagged, for an unrelated reason (its two-piece rule: -# a lone word after the given name is the family), so that ledger -# carries no copy of this rule. -# -# Literal, one name; 'John Smith M.Eng.' keeps the suffix by the count -# (words to spare) and 'Wang M.A.' keeps the suffix too (M.A. passes -# the period gate, unaffected by this change) -- both are -# _MUST_NOT_MATCH probes, along with the superstring 'Dr. Wang M.Eng.'. -name_regex = "^Wang M\\.Eng\\.$" -fields = ["family", "suffix", "_ambiguities"] -orders = ["DEFAULT"] From 0c8d333b6f1382d6bb59b8993d4769da34cc172e Mon Sep 17 00:00:00 2001 From: Derek Gulbranson Date: Sun, 27 Sep 2026 18:20:53 -0700 Subject: [PATCH 05/11] docs(#544): decision records, release notes and prose sweep decisions.md#S2 gains the #544 entry: the eight forks (the whole run at C1, every trailing path anchored in front, the chunked dotted gate, single-letter numerals, no new report where the listed members' capitals settle the run, the maiden-clause boundary, the anchor over P6 at the given slot, the dual exclusion scoped to the given part's head); the four boundaries of what may anchor (a connective, a dual opening the given part, the walk's own leading piece, a single-letter numeral for being initial-shaped); the accepted costs ('John Smith, Ms Ma', 'Om Jr Ma', the report 'doe, jane v phd do' now makes) and limits (the merged 'Ph. D.' without a comma, a title in the run, a particle member P2 has chained, 'Smith, PhD Do Ma' among them); the two-input check's recorded controls; and the grid, corpus and frame measurements re-taken on this tree against e10e83b4 with the recipe and comparator. Why no mechanisms entry is owed is stated there. Dated addenda, nothing earlier rewritten: #289's accepted item (ii) reversed; #531's P6 pairing gains the anchor as a second exception; C1 records the run rule; M2 records the clause boundary; and suffix-acronym-collisions answers #540's two pending questions, saying where 'Wang M.Eng.' is now classified (feat(#449) at 2.0-2.2, fix(suffix-routing) at 1.4.0, no diff at 2.3.0) rather than under a fix(#544) rule. rules.md#C1: the capitals settle a run only where every word of the class in it is a LISTED one; a by-shape member leaves the part to the count. rules.md#S2's interacts line gains M2. docs/release_log.rst: a #544 bullet, and the #540 bullet's comma-run, walk-stopping and dotted passages replaced by the costs that remain. _types.py's SUFFIX_OR_NAME note says which slots report an anchored pick and that the first-piece emitter stays silent on 'Smith, PhD MEng'. AGENTS.md's benchmark paragraph drops its shape count and adds the anchor-pass shape. tests/v2/cases.py: the 'smith, v ed' row's note gives the numeral's shape, not its meaning, as why it anchors nothing. Co-Authored-By: Claude Opus 5.5 --- AGENTS.md | 2 +- docs/design/decisions.md | 13 +++++++++++++ docs/design/rules.md | 12 +++++++----- docs/release_log.rst | 4 +++- nameparser/_types.py | 8 ++++++++ tests/v2/cases.py | 6 ++++-- 6 files changed, 36 insertions(+), 9 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index c63f8324..ea8c8abf 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -382,7 +382,7 @@ Add a dedicated `copy.deepcopy()` round-trip test for it too (see `test_regexes_ **`_normalize` must reach a fixed point** — storage and match-time share the one fold, and `Lexicon.__setstate__` re-validates, so a value that changes on re-normalization changes under its owner. `strip().strip(".")` alone is not idempotent (`'. a .'` → `' a '` → `'a'`). The loop is the fix; keep any new stripping inside it. **Anything built on `_normalize` must converge too** — `_fold_words` runs `_normalize` per word and DROPS the words that fold away (`_title_key` is that list space-joined, and `_run_addresses_by_given` reads the list itself, so its last-word arm is the last word of the FOLDED key by construction); keeping the empty slot stored `'lt .'` as `'lt '`, a key match-time can never rebuild (so the entry is silently inert) and `__setstate__` rejects on the next round-trip as "not written by this version". -**Perf regressions are caught by the scaling test, not the absolute-time ones** — `tests/v2/test_benchmark.py::test_parse_cost_grows_no_worse_than_linearly` times a repeated unit at n vs 4n over thirteen shapes (one per pipeline inner loop) and bounds the ratio; the `_thousand_names` tests use constant-size, delimiter-free input and are structurally blind to a complexity regression. Two rules when touching it: calibrate `_MAX_RATIO` against the WEAKEST quadratic's signal (a mixed quadratic surfaces far below the textbook 16×, so the operating point `_BASE` matters more than the bound), and confirm a planted regression fails it across REPEATED runs — one failure is a coin-flip on a timing test. The thirteen shapes cover different dimensions (segment count only via `commas`, intra-piece accumulation only via `particles`/`conjunctions`, non-ASCII input only via `honorifics` — the other twelve are pure ASCII, so `script_segment` returns at its bail and the CJK stages go unmeasured, M2's clause view only via `maiden_clause`, whose unit has to END on a class member: `MA nee ` holds the same two words, the peel stops at the trailing marker, and the shape reaches nothing — and connective RUN LENGTH only via `link_run`, whose unit has to be MIXED CASE and hold a connective of the generational class: `i und ` reads the letter as an initial and reaches nothing where `i Und ` reaches everything); measure before pruning one. **A shape whose input needs a PREFIX cannot be a `_SHAPES` row at all**, since that table repeats a unit and nothing else — a maiden clause needs a name word and a marker before the run it is about, and `"nee i Und " * n` reaches the clause rule not at all, measuring the identical ratio on a broken tree and a fixed one. That one is `test_a_clause_link_run_does_not_cost_quadratically`, which builds its own input and counts FRAMES. **A shape the CLOCK cannot reach needs a FRAME-count guard instead**, which is the second scaling test in that file (`test_a_trailing_credential_run_does_not_cost_exponentially`, #531): where the defect is an exponential rather than a quadratic, the input length that separates the curves on a timing test does not finish, so the guard counts frames over 8 units against 16 and bounds THAT ratio. One pair does not see every curve, and the fix round for #531 measured why: at 2× the input the per-member LINEAR work swamps a quadratic (2.08× for a genuine one against 1.73× clean), so that pair guards the exponential alone and a second, longer pair — 16 against 64, where the same quadratic reads 7.42× against 3.53× clean — is what can see one. Assert them in that order: an exponential never returns from the longer run, so the cheap pair has to have failed first. Frame counts do not move under load, so this shape needs no repeated-run calibration — but it does need the same reachability assertion `_POLICY_SHAPES` rows carry, since the walk under measurement runs only while every unit still reads as a credential. A stage gated on an opt-in `Policy` field needs a `_POLICY_SHAPES` entry instead, since bare `parse()` never enters it — and that table's rows carry a **reachability probe** run before the measurement, because a precedence change can quietly stop the shape reaching the stage and leave a green test measuring a no-op (`_POLICY_SHAPES` is also asserted non-empty: an empty `parametrize` is a skip, not a failure, so deleting its last row would retire the guard silently). +**Perf regressions are caught by the scaling test, not the absolute-time ones** — `tests/v2/test_benchmark.py::test_parse_cost_grows_no_worse_than_linearly` times a repeated unit at n vs 4n over a table of shapes (one per pipeline inner loop) and bounds the ratio; the `_thousand_names` tests use constant-size, delimiter-free input and are structurally blind to a complexity regression. Two rules when touching it: calibrate `_MAX_RATIO` against the WEAKEST quadratic's signal (a mixed quadratic surfaces far below the textbook 16×, so the operating point `_BASE` matters more than the bound), and confirm a planted regression fails it across REPEATED runs — one failure is a coin-flip on a timing test. The shapes cover different dimensions (segment count only via `commas`, intra-piece accumulation only via `particles`/`conjunctions`, non-ASCII input only via `honorifics` — every other unit is pure ASCII, so `script_segment` returns at its bail and the CJK stages go unmeasured, M2's clause view only via `maiden_clause`, whose unit has to END on a class member: `MA nee ` holds the same two words, the peel stops at the trailing marker, and the shape reaches nothing — and connective RUN LENGTH only via `link_run`, whose unit has to be MIXED CASE and hold a connective of the generational class: `i und ` reads the letter as an initial and reaches nothing where `i Und ` reaches everything, and S2's anchor pass only via `credential_run`, whose unit has to hold a Title-case member behind an unambiguous credential: `PhD MA ` reads the same fields and never asks the pass, the capitals deciding each member first); measure before pruning one. **A shape whose input needs a PREFIX cannot be a `_SHAPES` row at all**, since that table repeats a unit and nothing else — a maiden clause needs a name word and a marker before the run it is about, and `"nee i Und " * n` reaches the clause rule not at all, measuring the identical ratio on a broken tree and a fixed one. That one is `test_a_clause_link_run_does_not_cost_quadratically`, which builds its own input and counts FRAMES. **A shape the CLOCK cannot reach needs a FRAME-count guard instead**, which is the second scaling test in that file (`test_a_trailing_credential_run_does_not_cost_exponentially`, #531): where the defect is an exponential rather than a quadratic, the input length that separates the curves on a timing test does not finish, so the guard counts frames over 8 units against 16 and bounds THAT ratio. One pair does not see every curve, and the fix round for #531 measured why: at 2× the input the per-member LINEAR work swamps a quadratic (2.08× for a genuine one against 1.73× clean), so that pair guards the exponential alone and a second, longer pair — 16 against 64, where the same quadratic reads 7.42× against 3.53× clean — is what can see one. Assert them in that order: an exponential never returns from the longer run, so the cheap pair has to have failed first. Frame counts do not move under load, so this shape needs no repeated-run calibration — but it does need the same reachability assertion `_POLICY_SHAPES` rows carry, since the walk under measurement runs only while every unit still reads as a credential. A stage gated on an opt-in `Policy` field needs a `_POLICY_SHAPES` entry instead, since bare `parse()` never enters it — and that table's rows carry a **reachability probe** run before the measurement, because a precedence change can quietly stop the shape reaching the stage and leave a green test measuring a no-op (`_POLICY_SHAPES` is also asserted non-empty: an empty `parametrize` is a skip, not a failure, so deleting its last row would retire the guard silently). **Expected-failure tests use `@pytest.mark.xfail`** — the conftest parametrized fixture breaks `@unittest.expectedFailure`; always use `@pytest.mark.xfail` instead. diff --git a/docs/design/decisions.md b/docs/design/decisions.md index 5d3a3286..de6b7bea 100644 --- a/docs/design/decisions.md +++ b/docs/design/decisions.md @@ -222,6 +222,7 @@ the fullwidth-colon marker (旧姓:佐藤 arrives as one word; the head-peel q - 2026-09-22 #397 follow-up — A DELIMITER CORE PAST THE CLAUSE'S FIRST WORD PASSES FOR THE NAME WORD BESIDE A LINK, RECORDED AS A DEVIATION RATHER THAN REPAIRED. The link exception above wants "a connective standing between two name words of the clause", and a separator the caller declared through `Policy.extra_suffix_delimiters` is structure, not a name word — so a link with one beside it joins nothing and should end the clause like any other suffix word. Between the marker and the clause's first word that already holds, the bound refusing the core before either piece test is asked. PAST that first word it does not: the core is an ordinary index to the run walk, which steps over connectives and nothing else, so it stands in for the name word on the link's left and the clause runs on past a title it would otherwise stop at. MEASURED 2026-09-22 under `extra_suffix_delimiters=(" - ",)`: `Smith, John, PhD née Puig Mr. - i Soler` reads maiden 'Puig Mr. i Soler' where its separator-less twin `Smith, John, PhD née Puig Mr. i Soler` stops at 'Puig Mr.'. THE POPULATION is the branch's own sweep, recorded with the code it describes (2026-09-21, corpus ∪ cases.py ∪ the property grids ∪ a 50,925-name generated set with cores, under thirteen core-bearing policies): the predicate is asked about a core in 51,072 of 900,023 calls, the answer differs from a core-skipping reading in 8,094 parses over 1,278 texts, and 1,824 of those move `maiden` on 288 texts — none of the 288 reachable at the default policy, `extra_suffix_delimiters` being empty there. NOT REPAIRED HERE: the fix threads the core set through three call sites into the run walk and moves the parent's reading as well, which makes it its own change rather than a rider on a review round. Open: #538. PINNED TWICE MEANWHILE. rules.md#M2 carries the shape as a `deviates: #538` example under an `extra_suffix_delimiters-dash` annotation — the first entry `tests/v2/rules_doc.py`'s registry has had for that field, named after the Policy field and carrying the delimiter in the suffix because the field's value is a set rather than a flag. And `tests/v2/pipeline/test_group.py::test_a_core_beside_a_link_wrongly_passes_for_a_word_until_538` holds the pair at the piece level, named so nobody reads it as the contract. ONE COST OF THE DOC EXAMPLE, worth knowing before the repair lands: its string enters `corpus_rules.jsonl`, where the differential gate parses it with the DEFAULT facade — no delimiter declared, so the dash is an ordinary name word and #538's reading is off the path entirely. It moves there for the 2026-09-20 link fix instead, and is classified as that at all five baselines: added to the `fix(#397) a link inside a maiden clause stays in the birth name` alternation at the four 2.x ledgers (suffix 'PhD i Soler' → 'PhD', maiden 'Puig Mr. -' → 'Puig Mr. - i Soler', identical at each), and given its own rule at 1.4.0, where v1 had no maiden markers and read the whole suffix-comma tail as one suffix. - 2026-09-26 #538 — A DELIMITER CORE IS STEPPED OVER BY THE LINK'S NEIGHBOUR SEARCH, AS A CONNECTIVE IS (Derek, 2026-09-26). The 2026-09-22 follow-up above recorded the deviation; this resolves it. `_run_neighbours` treats a lone core like a connective, so the word on a link's side is the one past the core and a maiden clause reads exactly as the same text written without the core: `Smith, John, PhD née Puig Mr. - i Soler` under `extra_suffix_delimiters=(" - ",)` now stops at maiden 'Puig Mr.' as its separator-less twin does. INVARIANT, pinned by `tests/v2/test_properties.py::test_a_delimiter_core_reads_as_if_it_were_not_written` over 81 generated texts, maiden field: 6 disagreed at e0f1a2fa, 0 after. Declined: THE BOUNDARY READING, which the 2026-09-22 entry's wording implied ("a link with one beside it joins nothing") — a core ending the neighbour search on its side. It fixes the reported shape equally, and it moves `Smith, John, PhD née Puig - i Soler` and `… Puig i - Soler` from maiden 'Puig i Soler' to maiden 'Puig', suffix 'PhD i Soler' — a birth-name link pushed into the credentials, where the skip reading leaves both as they read. Default-unreachable either way. THE SAME STEPPING APPLIES OUTSIDE A CLAUSE: the `frozen` loop's own `_run_neighbours` call also takes `cores`, so a connective beside a core looks past it; where a credential or nothing stands beyond, it stays a lone suffix word rather than joining, and the core it stood beside -- now a lone piece with nothing joined to it -- is dropped as #206 drops any lone core, the same way it already drops one between two ordinary post-nominals: `Smith, John, PhD - i Soler` 'PhD - i Soler' -> 'PhD, i Soler', `Smith, John, PhD - i - MD` 'PhD - i - MD' -> 'PhD, i, MD', `Smith, John, - i Puig` '- i Puig' -> 'i Puig', `Smith, John, Puig i -` 'Puig i -' -> 'Puig i'. MEASURED 2026-09-26: 191 of 8,097 generated core-bearing tail texts (the three heads `Smith, John,`, `Smith, John, Jr.,` and `John Smith,`, each followed by every 2-4-word product of {PhD, MD, Puig, i, y, -, Jr., Soler, Mr.} containing a core) move `suffix` and no other field, none reachable at the default policy; the comparison is each text parsed under `extra_suffix_delimiters=(" - ",)` with `_run_neighbours` handed its cores and with it handed none. Not repaired here: a core the join MERGES into a joined piece -- interior (`… Puig Dr. i - y Soler`, still 'PhD i - y Soler') or at the edge a link joins across, a name word standing beyond it rather than a credential or nothing (`Smith, John, Puig - i Soler` 'Puig - i Soler', its separator-less twin 'Puig i Soler'; `Smith, John, Puig i - Soler` 'Puig i - Soler', measured 2026-09-26) -- is not dropped from the suffix text; the join is right, the surviving separator is the gap. Also in that family: an ORDINARY connective's join still takes a core as its neighbour, the stepping above being for generational vocabulary alone (rules.md#P3) -- `Smith, John, PhD - and MD` keeps suffix 'PhD - and MD' (measured 2026-09-26), pre-existing and unchanged here. Open: #549, which carries both the surviving core and this one. - 2026-09-26 #535 — THE MAIDEN WALK READS THE TRAILING TITLE CHAIN, AND EVERY STOP ASKS ONE RELEASE QUESTION OF THE SPAN IT GIVES UP (Derek, 2026-09-26). This resolves the NOT TRANSPARENT HERE clause of the 2026-09-19 #533 entry above, which recorded `Jane Doe nee Smith MA Prof.` and `Jane Doe nee Smith Prof. MA` disagreeing and deferred the pair as one decision about what "trailing" means inside a clause. BOTH HALVES WERE TAKEN TOGETHER: a trailing title ends the clause (`Jane Doe nee Smith Prof.` reads title 'Prof.', maiden 'Smith', where 2.3.0 read maiden 'Smith Prof.'), and a credential or numeral standing in front of that title gets the stop it gets with the title absent (`… Smith MA Prof.` gives suffix 'MA', `… Smith V Prof.` suffix 'V'). Only the first half would have left the pair disagreeing in the other direction; H5's transparency is a statement about the peel and the chain read together to their fixed point, so the walk now reads the end of the clause the way assign reads the end of the name — `tail_reading`, the shared predicate mechanisms.md#ONE-PREDICATE-PER-QUESTION names — and over the forms the transparency property test below exercises, the two spellings give one answer. They split where something in or ahead of the clause stands to take the title — for example a particle chain (ahead of the clause or inside it), a bound-given join, a name left with no name word, or, after a family comma, a title word in front of a class member with a title behind it, which the given part's chain stops short of exactly as it does with no marker (`Doe, Jane nee Smith Rev. MA Prof.` reads maiden 'Smith Rev.', title 'Prof.', suffix 'MA', where `… Rev. Prof. MA` reads title 'Rev. Prof.', maiden 'Smith', suffix 'MA'; bare `Doe, Jane Rev. MA Prof.` reads middle 'Rev.') — and the Accepted pairs recorded here and in rules.md#M2 are examples of that, not a complete list. `tests/v2/test_properties.py::test_a_trailing_title_is_transparent_to_the_maiden_clause` holds that as an invariant over the two inputs, over heads and runs where the credential is given up and nothing ahead would take the title, with its negative control at e0f1a2fa in its docstring. ACCEPTED, THE TRANSPARENCY BOUNDARY: where the clause KEEPS the credential the spellings differ, because a clause is one contiguous run and a title inside the kept text cannot leave without the words behind it — `Doe nee Smith ba Prof.` reads title 'Prof.', maiden 'Smith ba', and `Doe nee Smith Prof. ba` maiden 'Smith Prof. ba'; `abdul nee Smith MA Prof.` reads title 'Prof.', family 'abdul', maiden 'Smith MA', and `abdul nee Smith Prof. MA` given 'abdul', maiden 'Smith Prof. MA' (2.3.0 kept every word in all four). The report differs with them: `Doe, Dr. nee Smith MA` and `… Prof. MA` report `suffix-or-name` on the kept MA, `… MA Prof.` reports nothing, the kept member no longer ENDING the clause, which is what M2's report is asked of. rules.md#M2 carries the first pair as an Accepted example. ACCEPTED, FURTHER SPLITS, measured 2026-09-26. Where something ahead of the clause would take the title, the spellings differ even with the credential given up in both, because the first-suffix-word stop ends the clause before the title check runs and the title check alone declines: `Jane van der Berg nee Smith PhD Prof.` reads title 'Prof.', maiden 'Smith', suffix 'PhD', and `… Prof. PhD` maiden 'Smith Prof.', suffix 'PhD' — H5's accepted particle-chain boundary inherited, the bare `Jane van der Berg Smith PhD Prof.` / `… Prof. PhD` splitting the same way — and `Berg, abdul nee Smith PhD Prof.` (bound-given join) and `Doe, Dr. nee Smith PhD Prof.` (no name word left) split likewise; the parent d9d80492 and 2.3.0 read all of these as the tree does, so only the claim is new. And a released particle with a title behind it is withdrawn: `Jane Doe nee Smith DO Prof.` reads title 'Prof.', maiden 'Smith DO', where `… Prof. DO` reads title 'Prof.', maiden 'Smith', suffix 'DO' (2.3.0 maiden 'Smith DO Prof.' and 'Smith Prof. DO'); `do` likewise. A particle INSIDE the clause does it from the other side, its chain able to take the credential behind it: `Jane Doe nee Smith do MA Prof.` reads title 'Prof.', maiden 'Smith do MA', and `… do Prof. MA` title 'Prof.', maiden 'Smith do', suffix 'MA' (2.3.0 kept every word in both). rules.md#M2 carries a pair of each shape as Accepted examples. ACCEPTED, H5'S REACH: `Jane Doe nee Smith King.` reads title 'King.', maiden 'Smith' (2.3.0 maiden 'Smith King.'), H5's accepted `Mary Jane King.` cost now reaching the last word of a birth name as it reaches the last word of a current one; rules.md#M2 carries it as an Accepted example. READER SCOPE IS UNCHANGED: the chain is consulted only where a trailing rule reads the clause's words (no comma, the part before a suffix comma, the given part after a family comma); before a family comma and in a tail segment the walk reads the peel alone and the clause keeps the title (`Doe nee Smith Prof., Jane` keeps maiden 'Smith Prof.'). THE FIRST-WORD FLOOR MOVED INTO THE CHAIN. `trailing_titles` and `tail_reading` take a `floor` (1 everywhere else), and the walk passes the position just past the marker's first word, so the chain never takes that word: `Jane Doe nee King.` keeps its maiden name (TITLES holds borne surnames and the marker announced one), and `Jane Doe nee Prof. Dr.` keeps 'Prof.' and gives up 'Dr.'. The first draft applied the floor as a clamp on the title stop after the chain had run, and that is measurably wrong rather than merely inelegant: a chain allowed to take the first word has already spliced it out of the count the re-peel reads, so `Jane Doe nee King. ba` kept 'ba' in the clause where `Jane Doe nee Smith ba` gives it up. `test_a_title_first_word_counts_as_a_word` is the invariant (a title-vocabulary first word against an ordinary one, the credential behind it read alike) and its docstring records the clamp's failure count. ONE RELEASE CHECK FOR THREE STOPS. The numeral, the credential and the title stop each ask `_release_reads_off` whether the name the take would leave reads what they give up as titles or suffixes with no join below the take absorbing it — rules.md#M2's invariant, "a word the clause gives up reads as a post-nominal or the clause keeps it", which the title stop now answers to as well. It is asked of the whole SPAN a stop gives up, not of the stop's own word: a first draft that checked the word alone broke the invariant on 513 of a 5,198-parse sweep (measured 2026-09-26 on that draft, which is not in the tree to recompute), because a stop gives up every word behind it — `DOE NEE SMITH PROF. MA` stopped at the title and handed the MA to the family, where the name left standing reads MA as a name; it keeps maiden 'SMITH PROF. MA' now. The question is asked per reader, the way that reader reads: the TRAILING reader is assign's own `tail_reading` over the view; the GIVEN_SLOT reader has its count settled by the comma, so it is the writing alone with a name word ahead. The name-word-ahead half is what keeps `Dr. nee Jones Smith Prof.` whole — the take would leave `Dr. Prof.`, whose Prof. would be the family name. THE JOINS. A released span holding a title with a particle ahead of it declines, because P2's chain runs on over a trailing title (rules.md#H5's Accepted `John van der Berg Prof.`): `Jane van der Berg nee Smith Prof.` keeps maiden 'Smith Prof.', and a released particle with a title behind it is withdrawn for the same reason (`Jane Doe nee Smith MA do Prof.` keeps maiden 'Smith MA do'). The bound-given (P5) half of `_join_takes_the_member` is asked for the GIVEN_SLOT reader alone: P5 is `BoundJoin.LENIENT` only after a family comma, and before one its STRICT reserve already refuses a join that would change a suffix reading, so asking there over-declined — `abdul nee Smith V` kept 'V' in the clause, where the scoped check lets it go to suffix as 2.3.0 did. The consequence is recorded by a case row rather than prevented: `abdul nee Smith Dr.` reads title 'Dr.', family 'abdul', maiden 'Smith', which is how `abdul Dr.` reads bare (2.3.0 read given 'abdul', maiden 'Smith Dr.'). THREE NUMERAL-STOP READINGS MOVE WITH IT, decided in rather than deferred, since each is the same invariant broken at the stop this change was already rewriting. The numeral stop never asked the join question: `Berg, abdul nee Smith V` read given 'abdul V' at 2.2.0 and 2.3.0 (given 'abdul nee', suffix 'V' at 2.0.0 and 2.1.0, #411's reserve differing between those pairs before the walk runs) and reads maiden 'Smith V' now. After a family comma the given slot reads a lone numeral as a suffix only where the given part is the LAST comma part — assign's own two-segment condition from #144 — and the walk now asks it too (`tail_follows`): `Doe, Jane nee Smith V, PhD` read middle 'V' at 2.2.0, 2.3.0 and e0f1a2fa, an M2 violation predating this change that its first draft had extended to `… V Prof., PhD`, and it reads maiden 'Smith V' now, as 2.0.0 read it. The withdrawal reaches through a title behind the numeral, so `Doe, Jane nee Smith Prof. V, PhD` keeps maiden 'Smith Prof. V' where `Doe, Jane nee Smith V Prof., PhD` gives up the title — the same asymmetry the bare given slot has with no marker in it (`Doe, Jane Prof. V, PhD` against `Doe, Jane V Prof., PhD`). And the numeral stop reads FROM the marker, unlike the other two, so a numeral straight after the marker is not held to the first-word floor, and it may decline the clause; examples, not a rule over every shape: `Dr. nee V` and `Doe, J. nee V` keep maiden 'V' as 2.3.0 did, though `Doe, J. V` reads suffix 'V', and `Doe nee V, Jane` and `Smith, John, PhD nee V` read no clause at all (family 'Doe nee V', suffix 'PhD nee V', as at 2.3.0) — `Jane Doe nee V Prof.` read maiden 'V Prof.' at 2.3.0 and reads family 'nee', suffix 'V', title 'Prof.' now, as `Jane Doe nee V` reads plus the title. After a family comma with another comma part behind the given one, the same #144 condition that keeps `Doe, Jane nee Smith V, PhD` whole now keeps a lone numeral in the clause as well: `Doe, Jane nee V, PhD` read given 'Jane', middle 'nee V', suffix 'PhD' at 2.3.0 and at the parent d9d80492 and reads maiden 'V', suffix 'PhD' now, as 2.0.0 and 2.1.0 read it, `Doe, Jane nee V, Jr.` likewise — the marker had been read as a name word there at 2.2.0 and 2.3.0, and a case row records it. A LINK THE WALK STOPS AT ASKS THE RELEASE QUESTION TOO. Reading the end of the clause through the title chain makes the link exception refuse a link it used to join — in `Jane Doe nee Smith i DO Prof.` the DO is the peel's once the title is chained — and a stop at a link gives up the words behind it. Unchecked, that put 'Doe i' in the middle name and 'DO Prof.' in the family (e0f1a2fa read maiden 'Smith i DO Prof.'), and `Berg, abdul nee Smith i V Prof., MD` read middle 'V'. So a link the exception refuses asks `_release_reads_off` of the run it would give up — only where the exception, bounded by the peel over the words as WRITTEN, would have joined it (the refusal is the title chain's), where a trailing rule reads the clause, and where the link is not the first word after the marker (a first-word stop declines the clause and gives nothing up); where that fails the clause keeps the link and walks on. Scoped that narrowly because a first version asked it of every refused link and kept links the name left standing reads off: `Doe, Jane nee Smith i V` kept maiden 'Smith i' where every release gives up suffix 'i V' (23,472 of a 411,936-parse link grid moved clean readings, measured 2026-09-26). With it, the given-part model reads the lenient numeral in both of assign's passes — the literal last piece first, then the last one standing once the chain has taken the titles behind it — and only where no comma part follows (`Doe, Jane nee Smith i V Prof.` gives up 'i V' and the title, as bare `Doe, Jane i V Prof.` reads them; `… i V Prof., PhD` keeps maiden 'Smith i V'). The same model moves the credential stop where a lenient numeral follows the credential: `Doe, Jane nee Smith MA V` now gives up 'MA V' (suffix 'MA V', maiden 'Smith') as bare `Doe, Jane MA V` reads it, where d9d80492 and 2.3.0 read maiden 'Smith MA', suffix 'V'; `… MA V, PhD` keeps maiden 'Smith MA V'. Over that link grid (links i, y, e, and, i y, y i, and i × 8 heads × bodies × `, MD` tail × 3 cases × 3 name orders), against the tree before the link check: 0 new violations, 6,195 fixed, and the clean readings that move now match the bare given part's (`DOE, JANE NEE SMITH MA I` gives up 'MA I', as `DOE, JANE MA I` reads suffix 'MA I'). `Jane Doe nee Smith i DO Prof.` now reads title 'Prof.', maiden 'Smith i DO' and reports the kept DO, and `Berg, abdul nee Smith i V Prof., MD` maiden 'Smith i V Prof.', suffix 'MD'. A RELEASED TITLE THAT IS ALSO A PARTICLE IS KEPT AFTER A FAMILY COMMA, because P6 attaches a particle trailing the given part to the family: released by the title stop, 'St.' in `Doe, Jane nee Smith St.` reads family 'St. Doe', so `_join_takes_the_member` declines it and the name reads maiden 'Smith St.', as e0f1a2fa and 2.3.0 read it, and so does `Doe, Jane nee Smith MA St.`. Titles only — a credential that is also a particle ('DO') is the given slot's own lean, which the credential stop has already asked (#533). A FOURTH MEASUREMENT, against e0f1a2fa rather than d9d80492: the M2 invariant over a fuzz of twelve heads ({`Jane Doe`, `Doe, Jane`, `Doe, Prof.`, `Jane Doe, PhD`, `Doe, J.`, `Berg, Jane van der`, `Jane van der Berg`, `Berg, abdul`, `abdul Berg`, `J. Doe`, `Prof. Jane Doe`, `Dr.`}), ` nee Smith ` and every one-to-three-word sequence over {MA, Ma, V, Prof., St., King., PhD, ba, do, DO, Jr., i, van, y} holding at least one of Prof., St., King., then nothing or `, MD`, as written, upper- and lower-cased (107,352 parses, measured 2026-09-26 on the narrowed tree; the intermediate tree above read 2,228): 588 texts that violated it at e0f1a2fa no longer do, and 12 newly do, all `Dr. nee Smith i MA|V` followed by `Prof.`, `King.` or `St.`, with or without `, MD`. Each is the title-carrying twin of `Dr. nee Smith i MA` / `Dr. nee Smith i V`, which read family 'i' at e0f1a2fa already: the bare-title head leaves no name word, the unguarded stop's class (#548), which the title now reaches because it leaves the clause. ACCEPTED: `Jane Doe (nee Smith Prof.)` reads title 'Prof.', maiden 'Smith' — bracket content ending in a period is suffix-shaped (S1), so the brackets are dropped and the clause is read bare, outside the reach of the delimiter precedence; rules.md#M2 carries it as an Accepted example. MEASURED 2026-09-26, the tree against its parent d9d80492, over three generated grids, and these are dated snapshots. Recipe: grid A is each head in {`Doe, Jane`, `Doe, J.`, `Doe, Prof.`, `Berg, abdul`, `Doe, Jane van`, `Jane Doe`, `J. Doe`, `Dr.`, `abdul`, `Jane van der Berg`, `John`}, then ` nee `, then every sequence of one to three words (repetition allowed) from {Smith, Jones, Prof., Dr., Sir, MA, DO, PhD, V, III, i, do, van, Ma, M.A., King., Rev., ba}, then either nothing or `, PhD`, each text as written, upper-cased and lower-cased, duplicates removed — 327,936 texts; grid B, aimed at the bound-given join, is the same construction over heads {`Dr. abdul`, `abdul rahman`, `Dr. abdul rahman`, `Berg, Dr. abdul`, `Berg, abdul rahman`, `abu`, `Berg, abu`, `Mr. abu`, `Dr.`, `Berg, Dr.`, `abdul`, `Berg, abdul`}, words {Smith, Prof., MA, V, do, van, PhD, Ma, ba, III, Jr., M.A.} and tails {nothing, `, PhD`, `, Jr., MD`}, plus these fourteen texts, as written only: `Jane Doe nee Ph. D. Prof.`, `Jane Doe nee Ph. D. Smith Prof.`, `Jane Doe z domu King. ba`, `Jane Doe z domu Smith Prof.`, `Jane Doe nee King. Prof. ba`, `Jane Doe nee King. Prof.`, `Doe, Jane nee Smith V,`, `Doe, Jane nee Smith V, ` (with the trailing space), `Doe, Jane, nee Smith V`, `Doe, Jane nee Smith V Prof., PhD`, `Doe, Jane nee Smith Prof. V, PhD`, `Doe, Jane nee Smith MA, PhD`, `Doe, Jane nee Smith V (Jr.)`, `Doe, Jane nee Smith V "Bo"` — 173,057 texts. The invariant tested is M2's: in a parse with a non-empty maiden field, every token written after the ` nee ` marker is roled maiden, title or suffix (so the two `z domu` texts are parsed and not checked). Grid C, a four-word probe of the title chain after a family comma, is each head in {`Doe, Jane`, `Jane Doe`, `Doe, J.`, `John Smith`, `Doe, Jane Mary`}, then ` nee `, then every sequence of four words (repetition allowed) from {Smith, Jones, Prof., Dr., King., MA, PhD, V, Jr., ba, do}, then either nothing or `, PhD`, as written only — 146,410 texts. Per grid, the counts are: texts that newly violate the invariant at the tree, texts that violated it at the parent and no longer do, texts that violate it at both. Grid A: 0, 2,012 and 17,854; grid C: 0, 1,368 and 15,534; grid B: 0, 1,222 and 11,149 (all re-measured 2026-09-26 on the tree with the link and particle-title checks below, as narrowed; an intermediate tree whose link check fired on every refused link read grid A 0, 3,772 and 16,094, its extra 1,760 fixes being links the as-written reading refuses too, which are #548's class), one of those last being the check itself counting the quoted nickname in `Doe, Jane nee Smith V "Bo"`. In grids A and B, every other remaining violation has the clause ending immediately before a word of the unambiguous suffix vocabulary written the way the plain suffix-word stop takes it (PhD, III, M.A., Jr., and a lower-case i or v; never a bare capital, which reads as an initial), which is where that stop ends a clause without asking a release question. DEFERRED: the walk's plain suffix-word stop — the one ending the clause at the first suffix word, as distinct from the trailing numeral, credential and title stops — asks no release question at all, and where the words behind it cannot read as post-nominals they land in a name part: on degenerate heads the stop word itself does (`Dr. nee Smith PhD Prof.` reads family 'PhD', `Doe nee Smith Jr. Prof., Jane` family 'Doe Prof.'), and with an ordinary head the words behind it do (`Doe, Jane nee Smith PhD Smith` reads middle 'Smith'). Unchanged here, and what "the clause keeps it" should mean where the kept words would follow a credential the clause itself ended at is its own question; rules.md#M2 carries `Doe nee Smith Jr. Prof., Jane` as an Accepted example meanwhile. The trailing numeral's stop has the same gap before a family comma, where it is made over the peel alone with no release question: `Doe nee Smith V, Jane` reads family 'Doe V', maiden 'Smith' (so did 2.2.0, 2.3.0 and the parent d9d80492; 2.0.0 and 2.1.0 read maiden 'Smith V'), and rules.md#M2 carries it beside the other. Open: #548. COST, measured 2026-09-26 on CPython 3.11.16 as profiler call events in one `Parser.parse` after a warm-up parse, parent → tree: `John Smith` 171 unchanged, as are `Jane Doe Prof.` and `John Smith MA`; `Jane Doe nee Smith` 244 → 246; `Jane Doe nee Smith MA` 377 → 388; `Jane Doe nee Smith Prof.` 276 → 372, the one shape that now runs the chain and a release check it never ran. Recompute: count `sys.setprofile` call events around the second of two `Parser().parse` calls, with the tree and d9d80492 each first on `sys.path`. +- 2026-09-27 (Derek), #544 — S2'S COMPANY CLAUSE DOES NOT REACH ACROSS A CLAUSE, recorded as an Accepted boundary rather than repaired. `Jane Doe Jr. nee Smith Ma` keeps maiden 'Smith Ma' and reports it, the clause's own words standing between 'Jr.' and 'Ma', where the clause-less `Jane Doe Jr. Ma` reads suffix 'Jr. Ma'. tests/v2/test_properties.py's clause-agreement walk pins the pairs that differ for exactly this reason as its `anchored_head` class — 810 of its 20,412 pairs, recorded 2026-09-27, 0 with the anchor off. Inside the clause the company reads as it does anywhere: `Jane Doe nee Smith PhD MEng` ends the clause at 'PhD' and reads suffix 'PhD MEng', maiden 'Smith'. ### N3 — the lone-word nickname rule @@ -616,6 +617,7 @@ Closes #342 (a wordlist question) and #454 (a rules.md question) together, becau - **2026-09-25 (#540) — meng and lac joined SUFFIX_ACRONYMS_AMBIGUOUS (Derek: the marking over removal, and no masks).** #vocabulary-collisions C-i asked at the position the suffix claim acts on, the last word of a name: Meng (孟) is a common Chinese surname and given name, and Lac (Lạc) is a Vietnamese given name — the trailing word in native order, `Nguyễn Văn Lạc` — and a French surname; both are borne in that slot (`wang meng`, `tran lac`, `nguyen van lac`). MEng (Master of Engineering) and LAc (Licensed Acupuncturist) are live credentials, and neither reading is rare enough beside the other to give it the word — the rough balance this entry's criterion marks (where it is uncertain, C-i's default is the marking too). They are the set's first entries longer than two letters, which is the LENGTH clause of the criterion bullet applied, not an exception to it. Neither was checked against a surname when it arrived: `lac` with the af5bdab import (2019-12-11, #93), `meng` with 3e14ea20 (2026-07-01, the German/Dutch title and degree batch, first shipped in 1.3.0). No rule changes: rules.md#S2 reads the two words exactly as it reads `ma` and `ba`. Measured 2026-09-25 on this tree, the parent 120033b5 and the five released wheels (each wheel run as the Gotchas in AGENTS.md prescribe, from a directory outside the checkout under `PYTHONSAFEPATH=1`, asserting `nameparser.__file__`). THE FIX: `wang meng`, `li meng` and `tran lac` read given + family with a `suffix-or-name` report, where every release from 2.0.0 read suffix and no family name — reporting nothing at 2.0.0 through 2.2.0, and `given-or-family` at 2.3.0 and the parent, the one-word-name fork, which was the wrong fork — and 1.4.0 read the family, unflagged. The family-comma spelling returns to 1.4.0 with them, `Wang, Meng` reading given `Meng`, family `Wang` where 2.0.0 through 2.3.0 read suffix `Meng`; and under `FAMILY_FIRST` the bare `Wang Meng` reads given `Meng` where 2.0.0 through 2.3.0 and the parent read family `Wang`, suffix `Meng`, so the comma spelling under the default order and the bare one under `FAMILY_FIRST` still agree with each other. WITH WORDS TO SPARE the credential stays: `john smith meng`, `JOHN SMITH MENG` and `nguyen van lac` keep their suffix, now reported, and so does `john smith m.eng.`, which S2's period gate does not settle, that gate counting one period after each letter. DECIDED (Derek, 2026-09-25): the bare `Wang M.Eng.` reading — family `M.Eng.`, nothing to spare — is accepted and recorded rather than widening the period gate to cover a chunked dotted spelling; S2's period gate counts a member as unambiguous only when it is written one period per letter (`M.A.`), and MEng and LAc are the first members whose conventional dotted spelling is chunked instead. `John Smith M.Eng.` keeps the suffix by the words-to-spare count, unaffected. THE ACCEPTED COST is the one `ma` and `ba` already carry, and it falls on the conventional spellings: in a name written in more than one case S2 reads a member written neither in capitals nor all lower as the name even with words to spare, and MEng and LAc are written exactly that way. `john smith MEng` reads middle `smith`, family `MEng`, where every release from 1.4.0 read suffix `MEng`; `John Smith Meng` and `John Smith LAc` move the same way; and the Title-case `Nguyen Van Lac` reads family `Van Lac` where every release read family `Van`, suffix `Lac` (for this Vietnamese name the gain, the mechanism being the cost's) — only the all-lower `nguyen van lac` keeps its suffix, one case saying nothing. A CREDENTIAL RUN ending in one of them pays the same walk-stopping cost: `Mary Jones PhD MEng` reads middle `Jones PhD`, family `MEng`, and `Jane Doe MS LAc` reads middle `Doe MS`, family `LAc`, where every release read the whole run as a suffix (`PhD, MEng` and `MS, LAc` through 2.2, `PhD MEng` and `MS LAc` in 2.3); `John Smith MEng PhD` reads middle `Smith`, family `MEng`, suffix `PhD`, the unambiguous credential behind the declined pick still peeled; the mechanism is S2's own — the case signal costs a genuine suffix standing behind a name-leaning acronym, the walk stopping at the declined pick rather than continuing past it — pinned by `tests/v2/cases.py`'s `a_declined_ambiguous_pick_stops_the_walk` and, for the family-comma shape, `a_declined_member_ends_the_trailing_run` (`Doe, John MA Ma`). A QUIETER COST sits behind a suffix comma: once the run's last word is a bare ambiguous member, the run is no longer wholly suffix-shaped, so C1 reads the comma as a family comma instead and the run's own words fall into the usual post-comma name slots — `John Smith, PhD MEng` reads given `PhD`, middle `MEng`, family `John Smith`, and `john smith, phd meng` reads given `phd`, family `john smith`, suffix `meng` (the count then peels `meng` with a word to spare); neither case form saves it, `JOHN SMITH, PHD MENG` reading the same way. A title-listed word standing in front of the run reads as a TITLE instead, with no report: `Jane Doe, MS LAc` reads title `MS`, given `LAc`, family `Jane Doe`, unflagged, where every release read suffix `MS LAc`. The path is not new: `john smith, phd ma` and `Jane Doe, MS Ma` already read this way, `ma` having been ambiguous all along. Pinned by `tests/v2/cases.py`'s `comma_credential_run_ending_in_meng_re_reads_the_comma` and `comma_lower_credential_run_ending_in_meng_re_reads_the_comma`. The comma re-read (with its silent title-dual case) and whether a listed member's chunked dotted spelling should pass the period gate are [#544](https://github.com/derek73/python-nameparser/issues/544)'s questions. After a comma, `Smith, MEng` reads given `MEng` (1.4.0's reading; 2.0.0 through 2.3.0 read suffix) — and so does the all-lower `Smith, meng`, a restored reading rather than a cost — and `Smith, John MEng` middle `MEng`, pinned as case rows (`Smith, meng`, `Smith, MEng`, `Smith, John MEng`) and classified by the `fix(#540)` rules, while `John Smith, MEng` keeps the suffix on C1's name-word count; bracketed, `John Smith (MEng)` reads nickname `MEng` with a `suffix-or-nickname` report, S1's escape declining an ambiguous member, where every release read suffix `MEng`. `John Smith MENG` keeps the credential on the capitals lean. The judgment is the one behind rai and cha: a missed credential leaves the letters in a name field where a human can still see them, and a wrong credential reading destroys a surname on every record it appears in. edd and ded are NOT marked: both are borne as GIVEN names (`Edd Smith`, `Ded Gjo Luli`), which is not the position the suffix claim acts on, and no measurement shows either losing a family name — a trailing bearer turning up reopens them. Recomputed with #vocabulary-collisions' recipe on 2026-09-25: suffix_acronyms 608 (unmoved by this change) with 7 ambiguous, 5 before this change (particles 70 with 37 and titles 758, unmoved). - **Declined 2026-09-25 (#540), with measurement: removing the two words, and masking them.** REMOVAL, the rai/cha answer, measured with `Parser(lexicon=Lexicon.default().remove(suffix_acronyms={"meng", "lac"}, suffix_acronyms_ambiguous={"meng", "lac"}))`: `john smith meng` reads middle `smith`, family `meng`, and `JOHN SMITH MENG` family `MENG`, while `john smith m.eng.` stays a suffix by its dotted shape (S3), reported. Removal loses the bare credential outright where the marking keeps it with words to spare, and C-i's default under uncertainty is the marking. MASKS `meng → MEng` and `lac → LAc`, measured with both added to a private lexicon on top of the marking: the exceptions map is role-free (#R4, 2026-09-23), so `wang meng` renders `Wang MEng`, `tran lac` `Tran LAc` and `meng li` `MEng Li` — the mask re-spells the very surnames the marking restores. A suffix-gated mask would reverse the 2026-09-23 role-free decision for two words and is not taken. Without one, the default path renders `wang meng` as `Wang Meng` (the parent gave `Wang MENG`, the credential clause writing a suffix in capitals), `tran lac` as `Tran Lac`, `john smith meng` as `John Smith MENG` and `nguyen van lac` as `Nguyen Van LAC` by the acronym clause, and keeps a writer's `MEng` as written (#R5, 2026-09-24) — `john smith MEng` gives `john smith MEng`, and `John Smith Meng` under `force=True`, the word being the family name there. Pinned by `tests/v2/test_render.py::test_a_listed_acronym_that_is_a_name_word_gets_no_mask`, whose recorded negative control carries the two masks. - **Measurement (2026-09-25, #540).** No corpus name written before this change carries a bare trailing `meng` or `lac` (the `tools/differential/corpus*.jsonl` glob at 120033b5), so the population that could move is this change's own shape-tagged case rows, and every mover is one of them. Classified by `fix(#540)` rules: at each 2.x baseline `wang meng` and `tran lac` move {family, suffix, _ambiguities}, `john smith MEng` and `John Smith MEng PhD` move {middle, family, suffix, _ambiguities}, and `john smith meng` and `nguyen van lac` `_ambiguities` alone; `Smith, John MEng` moves {middle, suffix, _ambiguities} at 2.x and {middle, suffix} at 1.4.0; `Smith, meng` and `Smith, MEng` move {given, suffix, _ambiguities} at 2.x and nothing at 1.4.0 (1.4.0's own reading); `John Smith, PhD MEng` moves {given, middle, family, suffix, _ambiguities} and `john smith, phd meng` {given, family, suffix, _ambiguities} at 2.x (without `_ambiguities` at 1.4.0); `Nguyen Van Lac` moves {family, suffix, _ambiguities} at 2.x and {family, suffix} at 1.4.0; and `Wang M.Eng.` moves {family, suffix, _ambiguities} at 2.x only, 1.4.0 reading the same family for an unrelated reason (its two-piece rule -- a lone word after the given name is the family -- not S2's gate). `nguyen van lac` also diffs at 1.4.0 on `_initials` (`n.` → `n. v.`) for fix(#385/#402)'s reason — its family `van` is a one-particle part — and joins that rule's literal list; the same case-insensitive regex also reaches `Nguyen Van Lac` at every baseline, without explaining anything there (its own family move, at 1.4.0 too, is explained by the Title-case Lac rule instead). Read today's intentional counts off the `corpus:` line of `uv run python tools/differential/compare.py --baseline X`; this change raised them by seven at 1.4.0 and by thirteen at each 2.x baseline. +- **2026-09-27 (#544) — #540's two pending questions are answered, and the #540 bullet stands as it landed.** The chunked dotted spelling passes S2's period gate — `Wang M.Eng.` reads suffix 'M.Eng.' as `Wang M.A.` does, and as 2.0.0 through 2.3.0 read it — and the comma is kept: a credential run after a full name that ends in a member is the credential run by C1's name-word count (`John Smith, PhD MEng` suffix 'PhD MEng'; `Jane Doe, MS LAc` suffix 'MS LAc' where #540 left title 'MS', given 'LAc', at the cost, accepted under S2, of `John Smith, Ms Ma` reading suffix 'Ms Ma'). The walk-stopping cost that bullet lists for a run ENDING in a member goes with them, by S2's company clause: `Mary Jones PhD MEng` and `Jane Doe MS LAc` read the whole run as a suffix again. Where the member LEADS the run nothing speaks for it, so `John Smith MEng PhD` keeps its cost. The `fix(#540)` ledger rules for the comma run and the dotted spelling are retired. `fix(#544)` rules classify the comma-run names; `Wang M.Eng.`, whose roles every release from 2.0.0 read, is classified by `feat(#449)`'s given-or-family report rule at 2.0.0 through 2.2.0 and by the dotted `fix(suffix-routing)` rule beside `Jack M.A.` at 1.4.0, and differs from 2.3.0 not at all. The decision and its measurements are the #544 entry under S2. ### S2 — the case signal at the suffix slot @@ -668,6 +670,16 @@ for n in ('Smith, John','Smith, XYZ'): print(n, calls_for(off.parse, n), calls_f - 2026-09-18 (Derek), #531 — THE COMMA REPORT'S REACH NOW INCLUDES THE GIVEN SEGMENT'S TRAILING SLOT, AND THE OPEN FOLLOW-UP IS CLOSED. The 2026-09-18 bullet above headed THE COMMA REPORT'S REACH IS THE FIRST POST-COMMA PIECE recorded this slot as an open maintainer decision and named the two questions it turned on — whether a middle initial's neighbourhood should start reporting (a noise judgement, rules.md#A1) and whether the silence was a 1.4 parity gap. Both are answered here, and that bullet stands as it landed. The slot now reads and reports: `Doe, John MA` gives suffix `MA` and `Doe, John Ma` keeps middle `Ma`, each saying which way it went. Derek chose to restore the ROLE and report both ways — one rule for both spellings — over a report-only change and over a capitals-only one, because the issue exists in the first place because two spellings of one name disagree. The noise question was settled by MEASURING THE DISAGREEMENT rather than by argument: over a generated sweep of 78 pairs — five listed members and two by-shape tokens in three cased spellings each, plus five controls in one spelling apiece, so 26 words against three name shapes — 48 pairs disagreed about whether the word was a credential or a name, and every one of the 48 disagreed in the same direction, the comma form declining what the comma-less form took. After this change 3 disagree and all three are the lower-case `do` rows P6 owns. The sweep and its allowlist are `tests/v2/test_properties.py`; a slot that answers differently from the same name written without a comma is not a quiet slot, it is an inconsistent one. Recorded under `3-0-reevaluations`' standing rule because v1 parity is LOAD-BEARING for one half of this and explicitly NOT for the other: `Doe, John MA` reads suffix `MA` on the 1.4.0 wheel, so the bare-acronym half RESTORES v1's role and the report is all that is new there — its 1.4.0 ledger rule neither retired nor narrowed to `_ambiguities` (nothing below baseline 2.0 can diff on that pseudo-field) but was RE-POINTED to the four names where the writing declines the credential, which keep the 2.0-era middle name against v1. `Doe, John X.Y.Z.` reads middle `X.Y.Z.` at 1.4.0 too, so the dotted half LEAVES v1 and carries a 1.4.0 ledger rule of its own; it moves to match the comma-less `John Doe X.Y.Z.` and rules.md#S3's shape rule, not to restore anything. BLAST RADIUS, stated the way this log's own rule asks: twenty corpus names move, and only TWO of them were in any corpus before this branch (`Doe, John MA` and `Doe, John X.Y.Z.`, both admitted by #530's own arc) — the other eighteen are this change's own case rows, so what the differential measures on pre-existing data is two names, and the population the rule reaches is a SHAPE (every family-comma listing whose given part ends in a member of this class) that the corpora barely sample. ACCEPTED COSTS, all measured on the differential corpora: `SMITH, JOHN DO` keeps family `DO SMITH` where 1.4.0 read suffix `DO`, paired with the Nascimento record in the bullet above; one-case and caseless names take the credential with no case evidence at all, so `DOE, MARY JO MA`, `doe, john ma`, `田中, 太郎 MA` and `김, 민준 MA` all read a suffix, which is what 1.4.0 read for each of the four — "caseless is inert" holds for the LEAN and not for the outcome, since `is_one_case` answers True for a script with no case and the positional reading then decides; `Doe, John van MA` loses its middle to the family, reading family `van Doe`, suffix `MA` with two reports where it read middle `van MA` in silence, which is `Berg, Jan van Jr.`'s reading arriving through a shape it could not reach before (Derek accepted it 2026-09-18 as P6 working correctly rather than as a cascade to carve out); `Doe, John Prof. MA` gains a TITLE role `Prof.` never had, H5's transparency reaching it once `MA` leaves the walk, landing it on the same answer as `Doe, John MA Prof.`; `Smith, LEED AP` moves under the default-off caps switch alone, to given `LEED`, family `Smith`, suffix `AP` with two reports, so no default reading is at stake; and a genuine middle name that is also a class member is now a credential wherever it ends the given part and is not Title-cased — `DOE, JOHN ED` reads suffix `ED` — which is the same cost the comma-less form has carried since 2.0. NOT REPAIRED HERE and left open: the maiden walk claims `Smith MA` whole in `Doe, Jane nee Smith MA` before this slot exists, so the rule cannot reach it and the name stays silent with maiden `Smith MA` — the same gap #530's close-out saw from the other side with `John Smith nee Jones R.A.I.`, and it is out of scope for #531. - 2026-09-19 (Derek), #533 — THE CLASS'S LAST SILENT TRAILING POSITION IS CLOSED, AND #531'S READING NOW LIVES IN ONE PLACE. The slot list in rules.md#S2 gains the trailing slot of a maiden marker's clause, and #S3's enumeration gains it with the by-shape spellings; the rule that governs what the clause does with those words is M2's, and the entry under `### M2` above is where its reasoning lives. The gap the bullet above left open as out of scope for #531 is the one this closes, from both sides: `Doe, Jane nee Smith MA` now gives maiden 'Smith' with suffix 'MA' and reports, and `John Smith nee Jones R.A.I.` gives suffix 'R.A.I.' again — a RESTORATION rather than a change, since 2.3.0 read it that way (measured on the wheel) and this unreleased cycle moved it into the maiden name when #516 took dotted tokens out of the certain-suffix class, where no corpus file held the name and no gate could see it. Two things belong here rather than under M2. FIRST, `credential_at_the_given_slot` in `_pipeline/_pieces.py` is now where #531's reading of a class member ending the given part is written, with two callers — assign's walk over that part, and the maiden walk's second check over the name a take would leave. Spelling it twice is precisely the "condition written to match it" mechanisms.md#ONE-PREDICATE-PER-QUESTION names, and the drift would have been silent, since each site's own tests would have gone on passing. It is a text-and-tags question, which is what puts it in `_pieces` rather than beside either caller — the destination follows the LAYER, not the topic. It costs one frame PER MEMBER asked at that slot: measured 2026-09-19 against 2f57ff21 per `Parser.parse`, `Doe, John MA` goes 310 → 311 and `Doe, John MA Ma MA`, which asks four times, 439 → 443, while a name with no member there never reaches it and pays nothing. The reference band does not move (412/449 on `uv run python tools/perf/call_count.py`), no test pins 310, and Derek took the trade rather than keep two conditions in step across two stages with no test that asks them both. The maiden path got one frame CHEAPER in the same pass, the numeral reading calling the peel pair it had wrapped rather than the wrapper: `Jane Doe nee Smith` 249 → 248. SECOND, THE ONE-CASE HEAD IS AN ACCEPTED EXCEPTION AND IT IS THE M2 INSTANCE OF #492'S DEFERRED QUESTION about whether a cased suffix token counts as case evidence. `DOE, JANE nee Smith Ma` reads suffix 'Ma' where the clause-less `DOE, JANE Ma` keeps middle 'Ma', because the own-words span stops at the marker — rules.md#P3 puts "a maiden marker's run and every word after it" outside the name's own words from the moment the marker is tagged — so the member's own Title-casing is not in the span `one_case` is computed over and the clause HIDES the contrast. That is THE PREDICATE KEEPS THE JUDGED TOKEN IN THE SPAN, the 2026-09-14 #289/#516 entry at the head of this section, failing structurally rather than by oversight: at this slot the judged token is never in the span, and that entry's own answer — include it — cannot be had here. Widening the span for this one question would change `one_case` for the whole name, and three sites read it, so the exception is accepted instead. Measured 2026-09-19: 114 of 2016 generated pairs — every listed member and both by-shape spellings, in three cased spellings, against sixteen heads and six clause bodies of NAME WORDS ONLY — against 186 allowlisted and 984 disagreeing outside the class before the change, and 0 outside it after. The PR review widened that grid to three markers (`nee`, `née`, `geb.`) and three policies (the default and each 2.4 switch), and the class is INDIFFERENT to both: 1,026 of 18,144, exactly 9x the original in both columns, and still 0 disagreeing outside it. `tests/v2/test_properties.py` carries the sweep with the class defined STRUCTURALLY (`one_case` true of the clause form and false of the clause-less one) rather than as a name list. The recorded control beside it is now the SET and not only its size, as a digest over the members: a count cannot notice a swap, one pair leaving and another arriving, which is the same blindness the count was added to close one level up. - 2026-09-23 #492 — THE DEFERRED QUESTION the 2026-09-14 predicate paragraph and the 2026-09-19 #533 bullet above name — whether a cased suffix token counts as case evidence — IS ANSWERED FOR RENDER ONLY. R5's gate now leaves the suffixes out (decisions.md#R5, 2026-09-23); the parse-time fact this section rests on keeps counting the judged token exactly as written above, so `Jack MA` still leans credential and `DOE, JANE nee Smith Ma` still reads suffix `Ma`. The split is decided and recorded under R5, with the shapes it makes visible there (`jack MA` at this slot, `john e jones III` at P3's). +- 2026-09-27 (Derek), #544 — A CREDENTIAL RUN KEEPS ITS AMBIGUOUS MEMBERS: C1'S NAME-WORD COUNT READS A RUN AS IT READS ONE WORD, AN UNAMBIGUOUS CREDENTIAL IN FRONT OF A MEMBER SPEAKS FOR IT AT EVERY TRAILING SLOT, AND A LISTED MEMBER'S CHUNKED DOTTED SPELLING PASSES THE PERIOD GATE. Three readers never looked past the member itself: C1 flipped a part to the suffix comma by the name-word count only when it was ONE token, and the trailing readers — the no-comma peel, the comma part read wholly as credentials, the given part's trailing slot — met the member first, took its name lean and stopped, so the credential in front of it was never reached. #540 marked `meng` and `lac`, whose conventional spellings lean name, which turned the one-case-row cost this section's 2026-09-15 item (ii) accepted into the ordinary spelling of two degrees. DECIDED, eight forks. (1) THE WHOLE RUN AT C1: two or more words after the comma, every one suffix vocabulary or a class candidate, at least one a candidate, behind two or more name words, is the credential run — all-member runs (`John Smith, Ed Ma`) and a leading member (`John Smith, MEng PhD`) included; offered and declined were an anchored-only rule and the issue's own "last word ambiguous" wording, which contradicted its `Ed Ma` consequence. (2) ALL TRAILING PATHS, ANCHORED IN FRONT: a listed member with an unambiguous credential in front of it, through other members, reads as the credential on the no-comma peel, in the comma part read wholly as credentials and at the given slot; a credential BEHIND it anchors nothing (`Wang Ma PhD` keeps family 'Ma'). (3) THE CHUNKED DOTTED GATE: a listed member written in two or more period-closed letter chunks (`M.Eng.`, `L.Ac.`) passes as `M.A.` does; one trailing period (`Ma.`, `Ed.`) still does not. (4) SINGLE-LETTER ROMAN NUMERALS neither anchor nor count toward a C1 run, S3's single-character retirement being the precedent; they peel as before, and the anchor's half of the exclusion is a matter of their shape, boundary (d) below. (5) NO NEW REPORT WHERE THE WRITING SETTLED IT: a run whose every member is LISTED and leans credential — S2's lean, capitals in a mixed-case name — is not flipped, the family-comma path already reading it as the credential run with the fields and the report it always had (`John Smith, PhD MA`); the alternative, flipping such a run and suppressing only the flip's report, traded 144 report additions for 180 removals over the grid below (measured on the draft, 2026-09-27), `John Smith, MA Jr` losing the report it makes today. Because the lean is the listed set's, a by-shape member in capitals never settles a run: `John Smith, X.Y.Z. MA` flips to suffix 'X.Y.Z. MA' and reports. (6) THE ANCHOR DOES NOT REACH ACROSS A MAIDEN CLAUSE: `Jane Doe Jr. nee Smith Ma` keeps maiden 'Smith Ma' where `Jane Doe Jr. Ma` reads suffix 'Jr. Ma' — rules.md#M2's Accepted boundary, and the M2 bullet of this date. (7) AN ANCHOR OVERRIDES P6 AT THE GIVEN SLOT: `doe, jane v phd do` reads suffix 'v phd do' where it read family 'do doe'; the bullet below amends #531's pairing. (8) THE DUAL EXCLUSION IS SCOPED TO THE GIVEN PART'S HEAD, NOT THE C1 RUN: after a full name the dual counts as the suffix vocabulary it is, as the legacy disjunct already counted it (`John Smith, MS MA`, `John Smith, MD`), and that is what reads the issue's headline `Jane Doe, MS LAc` as suffix 'MS LAc'. + WHAT MAY ANCHOR, four boundaries of the company clause, each a rule. (a) A CONNECTIVE (P3) anchors nothing and ends the run: the generational `i` is also Catalan's conjunction, and `Jane Doe nee Puig i Ma` keeps its link, maiden 'Puig i Ma'. (b) A TITLE/SUFFIX DUAL OPENING THE GIVEN PART after a one-word family comma is a title there and anchors nothing, so the pass starts past the leading title run: `Smith, Ms Ma`, `Nguyen, Sr Ba` and `Smith, MD Do` keep their given names, and `Smith, MD MA Ma` keeps middle 'Ma', no unambiguous credential standing in its run once the title is set aside. Everywhere else a dual anchors like any suffix word (`John Smith MD MEng`, and (8)). (c) THE WALK'S OWN LEADING PIECE NEVER ANCHORS. In a name with no comma before the run, the leading piece is the name word H4's carve-out keeps whatever its vocabulary, and letting it anchor spends that name: `Om Ma` and `PhD Ma` keep family 'Ma' under every name order, where an anchoring `PhD` would leave given 'PhD', suffix 'Ma' and no family at all. One name word in front frees the same word — `John Om Ma` reads suffix 'Om Ma'. A family comma's given part has no such position, its first piece being read by the comma part's own credential reading, so `Smith, PhD MEng` reads family 'Smith', suffix 'PhD MEng'. (d) A SINGLE-LETTER ROMAN NUMERAL anchors nothing because it is INITIAL-SHAPED, written as a middle initial is (`John Smith PhD V Ma` keeps family 'Ma'), not because a generation is no credential: a multi-letter numeral or generational word anchors as any suffix word does (`John Smith PhD III Ma` reads suffix 'PhD III Ma', `abdul Smith Jr Ma` suffix 'Jr Ma'). + ACCEPTED COSTS, each a reading this design chooses. `John Smith, Ms Ma` → suffix 'Ms Ma', reported, as `John Smith, Ms` alone already reads suffix (8). `Om Jr Ma` → given 'Om', suffix 'Jr Ma' and no family name: 'Jr' is not the leading piece, so it anchors 'Ma', and this section's standing Accepted clause already consumes an unambiguous suffix that leaves no family (`Om Jr` → given 'Om', suffix 'Jr'); every release from 2.0.0 made that same role assignment. `doe, jane v phd do` reports `suffix-or-name` where it reported `particle-or-given`, the fork reported being the credential pick rather than P6's attachment (7). And 12 grid parses of the `Smith, MA PhD ba` shape read the whole given part as a credential run, family 'Smith' and no given name, where the parent kept given 'MA'. + ACCEPTED LIMITS, each keeping the member a name word, rules.md#S2's Accepted block and case rows: the merged `Ph. D.` is outside the no-comma walk (`John Smith Ph. D. MEng` family 'MEng'; the comma and given-slot spellings do anchor it, `Smith, Ph. D. MEng` and `Doe, Jane Ph. D. MEng` reading suffix 'Ph. D. MEng'), a title between the credential and the member breaks the run (`John Smith PhD Prof. Ma`), and a particle member P2 has chained before the peel is out of reach (`John Smith PhD Do Do` family 'Do Do'; `Smith, PhD Do Ma` given 'PhD', middle 'Do Ma'). + THE TWO-INPUT CHECK. tests/v2/test_properties.py's M2 clause-agreement walk asks every parse it takes both directions of the company clause: a listed member behind a qualifying credential reads as one, and a member read as a credential because of the word in front has that word in the suffix too. Recorded negative controls, from that test's docstring: 48 failing parses with the anchor off, 30 with the walk's leading piece allowed to anchor, 0 on this tree. + REVERSED BY THIS ENTRY rather than edited: this section's 2026-09-15 ACCEPTED item (ii) (the bullet after this entry says so); the three "wrongly moved" reversals the 2026-09-14 paragraph THE RUN IS THE CAPS CLASS'S ALONE records — `John Smith, Ed Ma`, `John Smith, ma do` and `John Smith, X.Y.Z. A.B.` are credential runs now by C1's own count under every policy that keeps their words in the class (with `unlisted_dotted_suffixes` off, `X.Y.Z.` and `A.B.` are no class members and the third keeps the listing form), while that paragraph's narrowing of the CAPS run test stands; and #540's two pending questions (the `suffix-acronym-collisions` bullet of this date). + MEASURED 2026-09-27, the #544 tree against its parent e10e83b4, py3.11. Recompute: every run of one to three words from {PhD, MD, Jr, MA, Ma, MEng, M.Eng., LAc, Ed, Do, ba, MS, V, Prof.} behind each of `Wang {}`, `John Smith {}`, `John Quincy Smith {}`, `Dr. John Smith {}`, `John Smith, {}`, `John Quincy Smith, {}`, `Smith, {}`, `Doe, Jane {}`, `Jane Doe nee Smith {}`, `Doe, Jane nee Smith {}` and `Jan van der Berg {}`, each written as given, lower-cased and upper-cased and deduped, parsed under the three name orders at otherwise default policy, the seven role fields and the ambiguity kinds compared against the same grid parsed by e10e83b4. 254,496 parses over 84,832 texts; 90,365 move (30,123 texts, 19,303 of them with a field move): 39,801 the anchor freeing a member, 14,676 the C1 run, 35,696 the chunked `M.Eng.` (32,472 of those only losing the report it raised as a pick), 180 a run opened by a lone `v` (`john smith, v phd ma`, suffix 'v phd ma'), and the 12 `Smith, MA PhD ba` parses above. Clean-to-clean — no run word in a name field on either side — 23,454 move, every one `M.Eng.`'s dropped report, and none elsewhere. Two invariants over the same grid, 0 NEW violations of either: a member an unambiguous credential anchors read into a name field, 29,199 → 1,188 (the residue is a particle member P2 has chained and a dual inside a leading title run, `smith, prof. md ma`); an unambiguous credential in a name field, 36,675 → 4,410 (the residue is an interior single-letter `V`, a non-final `Prof.`, and the `PhD` a comma part keeps as its given name when a `V` or a title follows it). The differential corpora at the parent (1,418 names, three orders) move four names, every one intended: `John Smith, PhD MEng`, `john smith, phd meng`, `Wang M.Eng.`, `abdul Smith Jr Ma`. Frames per parse through `Parser().parse`, py3.11: the reference band is unmoved (370/407, `uv run python tools/perf/call_count.py`), and so is every ordinary name measured (`Smith, John Quincy` 260, `Doe, John MA` 283, `Doe, Jane nee Smith PhD` 338); the new questions cost where they are asked — `John Smith Ma` 244 → 250, the anchor pass that finds nothing; `John Smith, MA Jr` 355 → 382, the case fact forced to see the capitals settle the run; `John Smith PhD MEng` 286 → 326 — and `John Smith, PhD MEng` gets cheaper, 358 → 314. + NO MECHANISMS ENTRY IS OWED: the anchor pass is ONE-PREDICATE-PER-QUESTION's `_pieces.credential_anchors`, which `segment_suffix_reading` reads inline for the frame budget under a keep-in-step note, and its linearity is the answer #531's trailing floor and #397's `_run_neighbours` already give — one forward pass per question, never a look-behind per member. tests/v2/test_benchmark.py's `credential_run` shape guards it: the per-member look-behind measured 15.8× for 4× the input at base 800, against 4.13–4.21× on this tree at every base (that module's own record). +- 2026-09-27 (#544) — THE 2026-09-15 ACCEPTED ITEM (ii) IS REVERSED, and the bullet stands as it landed. `abdul Smith Jr Ma` reads given 'abdul', family 'Smith', suffix 'Jr Ma' — 2.3.0's reading — because the unambiguous 'Jr' in front of the Title-case 'Ma' anchors it (the entry above): the peel takes both, and P5's reserve, now seeing the family the join would take, declines the join. Item (i) stands. The case row is now `a_credential_in_front_anchors_a_declined_pick`, and rules.md#S2's Accepted block names the shapes that still keep the company out of reach. +- 2026-09-27 (Derek), #544 — THE #531 PAIRING GAINS A SECOND EXCEPTION, and CAPITALS DECIDE FOR `do` above stands as it landed. That bullet let only a positive credential lean override P6's attachment at the given slot; an unambiguous credential IN FRONT of the member now overrides it as well, the degree being a second and stronger signal: `doe, jane v phd do` reads suffix 'v phd do' and reports `suffix-or-name` where it read family 'do doe' and reported P6's fork. The one-case record the pairing protects has nothing in front of its particle, so `NASCIMENTO, EDSON ARANTES DO` still reads family 'DO NASCIMENTO' and reports `particle-or-given`. ### indic-honorifics — the renunciate class and the Indic honorific vocabulary (2026-09-06, #346/#344/#343) @@ -855,6 +867,7 @@ Excluded (MAIDEN_MARKERS, per nameparser/config/maiden_markers.py): NOT FIXED, and named because this change makes it reachable on more inputs: `revise(n, suffix=n.suffix)` is not the identity on a space-joined run. `Parser.revise` classifies each value by a full sub-parse of that value alone — its own docstring records the limit, the value being "classified ON ITS OWN" — and a bare 'MD PhD' carries no comma for the entry rule to route by, so the two words come back as two entries and the field re-renders 'MD, PhD'. Pre-existing, and the same shape as the `str()` limit above rather than a new one. Measured 2026-09-06 over every corpus name, comparing `revise(parse(n), suffix=parse(n).suffix).suffix` against `parse(n).suffix` at the parent tree and here: 24 of the parent's 1116 names fail and 38 of the 1117 here do, none of the parent's 24 recovering, and the 14 added are exactly the 14 movers — the 13 pre-existing names plus the R1 example. Not this change's to fix: the entry rule reads the commas in a whole name, and what `revise` hands its sub-parse is a field, so closing it means deciding what a field value's separators mean, which is a `revise` question. Answered, and FIXED, by #511 — the next bullet. DECLINED, all four with the evidence: marking the boundary at the core-drop site (#437's MARK-DONT-STRIP shape) — `dropped` already holds the fact with its span, and a second recording of it is the duplication that mechanism exists to prevent, one level up; making the `"joined"` tag role-aware — within a piece it is role-blind and correct for every role, `Smith, Ph. D. Smith` giving `first_list == ['Ph. D.']`, and only the between-piece half was ever a suffix concept; a render-time span scan instead of the recorded tag — `_facade.__setstate__` and `ParsedName.replace()` synthesise span-less tokens, so an unpickled name has nothing to scan and the tag IS the entry structure a pickle carries; and fixing the no-comma path inside group's block as a third branch beside `tail` and `reading`, which is the shape #429 took. - 2026-09-06 #511 — a suffix value handed to `revise()` derives its entries from its own commas, by the rule a whole name uses. `Parser.revise` sub-parsed each value and forced every harvested token to the named role AFTER the sub-parse had run R1's entry pass, which keys on Role.SUFFIX; a bare 'MD PhD' reads there as a title and a family name, so the pass joined nothing and the field rendered 'MD, PhD'. The fix runs the same pass again: `revise` now sub-parses to a ParseState, forces the role on every non-dropped token, and calls `suffix_entries` — the pass, lifted unedited out of post_rules' tail into a function the two callers share (mechanisms.md#ONE-PREDICATE-PER-QUESTION) — over the forced state, then assembles and harvests as before. So a comma in the value parts two credentials and a space joins them. TWO SPELLINGS of the lifted pass, and the reason is the call budget: a first draft had post_rules call the state-in/state-out wrapper, and the second ParseState build cost three more calls per parse against the band tests/v2/test_benchmark.py holds (py3.11, 2026-09-06, tools/perf/call_count.py: 450 calls/name before the move, 451 with the in-place worker `_mark_suffix_entries` that post_rules now calls, 454 with the draft; the facade band tops at 455.9), so the worker writes in place as every other post rule does and the wrapper exists for the one caller with no token list of its own. NO `_run` HELPER for the same reason: a draft routed `parse()` through a private state builder shared with `revise`, and that cost one frame per parse on the hot path (py3.11, 2026-09-06: 415 calls/name against 414 without it, in a 402-418 band), so the four-line construction is spelled twice and `parse()` is untouched. MEASURED 2026-09-06 over the 1117 distinct names in `tools/differential/corpus*.jsonl`, comparing `revise(p, suffix=p.suffix).suffix` against `p.suffix` under the default parser for every name with a non-empty suffix (368 of the 1117): 38 differed before and 1 after; the differential gate is byte-identical at all four baselines, `revise` not being on the compare path. RECOMPUTE: parse each corpus name, skip an empty suffix, revise the parse with its own suffix, count the names whose suffix moved (the script is in the #511 issue body). THE ONE LEFT is '김민준씨, J.씨', and it is not entry structure: the whole-name parse keeps 'J.씨' one glued suffix token, suffix '씨, J.씨', while the sub-parse of the bare value '씨, J.씨' peels the honorific off the initial, so the revised field renders '씨, J. 씨' — right entries, the spurious comma of before ('씨, J., 씨') gone, one word read differently by the value's own parse than by the whole name's. That is the "classified ON ITS OWN" limit `revise`'s docstring has always recorded, and CJK honorific peeling is a W-rule question this change does not move; pinned in `test_revise_reads_a_glued_honorific_on_its_own`. SUPERSEDES the phd-merge acceptance of 'Ph., D.' on this path (its bullet says how). STALE TAGS, tried and backed out: a draft cleared the sub-parse's own "joined" with the forcing and re-derived it, because the pass only ADDS the tag and `revise(n, family="Jones MD PhD")` carries the sub-parse's between-piece suffix mark on 'PhD' onto a FAMILY token. Measured 2026-09-06 against `330ee55`, the clear also destroyed every WITHIN-piece mark on a non-suffix value — 'D.' of `revise(n, family="John Ph. D. Smith")` lost the merge mark the #436 bullet's DECLINED list calls role-blind and correct for every role — while on a ParsedName only the suffix string view reads "joined" (`_text_for`'s suffix_join gate) — the facade's `_list_for` heals it for every role, but no path puts a revised name into a HumanName, the v1 setters going through `replace()`, so wiring those setters onto `revise` is the change that would show the stale mark, as `last_list == ['Jones', 'MD PhD']` — `initials()` and every field string being identical at both trees for every shape measured. So the tags are kept minus FOLDED_TAG as before; the between-piece mark on a forced non-suffix role is a tag-only oddity that predates this change and stays, and for a suffix value nothing depends on the sub-parse's marks, every pair it joined sharing a bucket with no parting token and the pass setting it again; pinned in `test_revise_keeps_the_sub_parses_within_piece_mark`. Dropped tokens keep their role and tags through the forcing, because the pass filters them by index and assemble omits them. ONE MORE LIMIT, pinned in `test_revise_leaves_a_policy_delimiter_unparted_without_a_tail_segment`, and it is the "classified ON ITS OWN" limit again rather than a rule of revise's: a delimiter the policy names through `extra_suffix_delimiters` is dropped, and so parts entries, only on a segment after a comma that the reading of the words makes a tail, and a value with no comma of its own has none — under `Policy(extra_suffix_delimiters=frozenset({" - "}))` the whole name 'Doe, John, MD PhD - FACS' renders 'MD PhD, FACS' while `revise(n, suffix="MD PhD - FACS")` renders 'MD PhD - FACS', the dash surviving as a token and the forced role making it a suffix word; measured 2026-09-06, a value whose own words read with a tail segment does part ('John Doe, MD - FACS' revises to 'John Doe, MD, FACS') and one after a suffix comma does not ('MD, PhD - FACS' stays), which is how the value's words read as a name deciding it. The round-trip is unaffected, the whole-name view having already rendered that boundary as a comma. DECLINED: a list-valued `revise(p, suffix=["MD PhD", "FACS"])`, which widens the API and leaves the string path where it was; documenting the limit with pins alone, the docstring already recording it and #511's measurement showing it reachable on every space-joined run #436 produced; and a text-only read of the delimiter cores inside `revise`, a second rule for a case no round-trip reaches. Out of scope and left as is: `ParsedName.replace(suffix="MD, PhD")` renders 'MD,, PhD', `replace` whitespace-splitting by contract and the facade setters riding on it for v1 parity. +- 2026-09-27 (Derek), #544 — THE NAME-WORD COUNT READS A RUN. The single-token rule this section records for the ambiguous class generalizes to a post-comma part of two or more words, every one suffix vocabulary or a class member (listed or by shape), at least one a member and none a single-letter roman numeral: behind two or more name words the part is the credential run, and the flip reports once over the whole part (`John Smith, Ed Ma`, `Jane Doe, MS LAc`). A title/suffix dual opening the part counts as suffix vocabulary there, and a run whose every member is listed and leans credential is left to the family-comma path, which already reads it whole. The forks and their measurements are the #544 entry under S2. ### T1 — separators, not joiners diff --git a/docs/design/rules.md b/docs/design/rules.md index cbbd41d7..8b35bf50 100644 --- a/docs/design/rules.md +++ b/docs/design/rules.md @@ -1152,7 +1152,7 @@ S2. Rationale: generational suffixes and credentials are recognized and unchanged (decisions.md#v1-xfail-triage: `king` stays a title, for the addressing forms). "Dr Jr" → suffix="Jr" - history: decisions.md#S2 · interacts: H1, H2, H3, H5, C1, S3, P2, P3, P5, P6 · implemented: nameparser/_pipeline/_classify.py, nameparser/_pipeline/_group.py, nameparser/_pipeline/_pieces.py, nameparser/_pipeline/_vocab.py + history: decisions.md#S2 · interacts: H1, H2, H3, H5, C1, S3, P2, P3, P5, P6, M2 · implemented: nameparser/_pipeline/_classify.py, nameparser/_pipeline/_group.py, nameparser/_pipeline/_pieces.py, nameparser/_pipeline/_vocab.py S3. Rationale: credentials are often written run together with periods; the chunks between the periods are what carry the @@ -1645,10 +1645,12 @@ C1. Rationale: a credential run after the comma means the name is in generation rather than a credential. A word of both the title and the suffix vocabulary opening such a part counts as a suffix word there, the name before the comma being complete. Where every word - of this class in the part is written in capitals in a mixed-case - name, the writing has already made each of them the credential - (S2), and the part reads as the credential run on that evidence - rather than on the count. A decision either way at this comma + of this class in the part is a LISTED word written in capitals in + a mixed-case name, the writing has already made each of them the + credential (S2), and the part reads as the credential run on that + evidence rather than on the count; a word of the class by shape + alone carries no such lean, so a part holding one is read by the + count. A decision either way at this comma is reported; for a run of words the decision is the flip to the credential run, and a run the count leaves in the listing form reports only as S2 reads the words in it. It is one of TWO diff --git a/docs/release_log.rst b/docs/release_log.rst index 5d73edc0..d26cfdc8 100644 --- a/docs/release_log.rst +++ b/docs/release_log.rst @@ -12,7 +12,9 @@ Release Log - **Fix a credential acronym that is also a surname being read by position alone.** ``HumanName("Jack MA")`` gives suffix ``MA`` where 2.0 through 2.3 gave last ``MA``, and ``John Smith Ma`` gives last ``Ma`` where they gave suffix ``Ma``. In a name written in more than one case, an ambiguous acronym written in capitals is written the way a credential is written and is read as one even where removing it leaves no surname; one written in any other cased form that is not wholly lower is written the way a surname is written and stays one even where there are words to spare (``John Smith ma`` and ``John Smith ed`` -- all lower, no contrast -- give suffix ``ma``/``ed`` instead). A name written wholly in one case says nothing either way and keeps the reading it had: ``JOHN SMITH MA`` is still a credential, ``ANH DO`` still a surname, ``jack ma`` still a surname. The same reading reaches the comma forms, where the words-to-spare count is now a count of NAME words: ``Smith, MA`` gives last ``Smith``, suffix ``MA``; ``Smith Jr., MA`` keeps last ``Smith``; and ``John Smith, MA``, ``John Smith, Ed``, ``john smith, ma`` and ``JOHN SMITH, MA`` all give a suffix again, which is what 1.4.0 read and 2.0 through 2.3 did not. ``Jack Ma`` and ``Anh Do`` are unchanged. The LEAN is inert on a caseless script, but the comma count above is not -- it asks name-word count, not case -- so ``마틴 킹, MA`` and ``田中 太郎, MA`` also give a suffix again (1.4.0 parity on the suffix, two pre-comma name words each) while the single-token ``毛泽东, MA`` does not move, having no case to write a contrast in either way. See the ``S2`` entry of ``docs/design/decisions.md`` (closes #289) - - **Fix a bare trailing Meng or Lac being read as a credential and losing the family name: meng and lac are now acronyms that are also ordinary names.** ``HumanName("wang meng")`` gives first ``wang``, last ``meng``, and ``parse()`` reports a suffix-or-name ambiguity, where every release from 2.0.0 through 2.3.0 gave suffix ``meng`` and no last name; 1.4.0 read last ``meng``, so this is 1.4.0's answer plus the flag. ``li meng`` and ``tran lac`` move the same way, ``Wang, Meng`` gives first ``Meng``, last ``Wang`` again, and ``Parser(policy=Policy(name_order=FAMILY_FIRST)).parse("Wang Meng")`` gives given ``Meng`` where 2.0.0 through 2.3.0 gave family ``Wang``, suffix ``Meng`` and no given name. With a full name in front the credential reading stays: ``john smith meng`` and ``nguyen van lac`` keep suffix ``meng`` and ``lac``, now flagged. But a Title-case ``Nguyen Van Lac`` gives last ``Van Lac`` where every release gave last ``Van``, suffix ``Lac``. The cost is the marking's own, and it falls on the conventional spellings: a ``MEng`` or ``LAc`` written that way, in a name written in more than one case, is read the way ``John Smith Ma`` is (above), so ``John Smith MEng`` gives middle ``Smith``, last ``MEng``, where every release gave suffix ``MEng``. A credential run ending in one of them goes the same way: ``Mary Jones PhD MEng`` gives middle ``Jones PhD``, last ``MEng``, where every release read the whole run as a suffix (``PhD, MEng`` through 2.2, ``PhD MEng`` in 2.3); ``John Smith MEng PhD`` gives middle ``Smith``, last ``MEng``, suffix ``PhD`` (``rules.md#S2``'s declined pick stops the walk, the cost ``Doe, John MA Ma`` already pays). ``Wang M.Eng.`` gives last ``M.Eng.``, where 2.0 through 2.3 gave suffix ``M.Eng.``: the period gate counts a member as unambiguous only written one period per letter, and ``M.Eng.`` is chunked; ``John Smith M.Eng.`` keeps the suffix. After a comma ``Smith, MEng`` and ``Smith, meng`` give first ``MEng`` and ``meng`` (1.4.0's reading, not the suffix 2.0 through 2.3 gave) and ``Smith, John MEng`` gives middle ``MEng`` (every release gave suffix ``MEng``), and a bracketed ``John Smith (MEng)`` falls through to nickname, where every release gave suffix ``MEng``. A credential run after a comma whose last word is one of them re-reads the comma as a family comma, whatever its letter case: ``John Smith, PhD MEng`` gives first ``PhD``, middle ``MEng``, last ``John Smith`` and ``john smith, phd meng`` gives first ``phd``, last ``john smith``, suffix ``meng``, where every release read the run as a suffix; a title-listed credential in front reads as a title instead (``Jane Doe, MS LAc`` gives title ``MS``, first ``LAc``, unflagged). The path is the one ``john smith, phd ma`` already takes. A lone credential after a comma behind a full name (``John Smith, MEng``) keeps the credential reading, and so does a second comma before a run that ends in one (``John Smith, PhD, MEng`` keeps suffix ``PhD, MEng``); writing it in capitals does not (``JOHN SMITH, PHD MENG`` still re-reads the comma). Meng is a common Chinese surname and given name, Lac a Vietnamese given name (``Nguyen Van Lac``) and a French surname; see the ``suffix-acronym-collisions`` entry of ``docs/design/decisions.md`` (closes #540) + - **Fix a bare trailing Meng or Lac being read as a credential and losing the family name: meng and lac are now acronyms that are also ordinary names.** ``HumanName("wang meng")`` gives first ``wang``, last ``meng``, and ``parse()`` reports a suffix-or-name ambiguity, where every release from 2.0.0 through 2.3.0 gave suffix ``meng`` and no last name; 1.4.0 read last ``meng``, so this is 1.4.0's answer plus the flag. ``li meng`` and ``tran lac`` move the same way, ``Wang, Meng`` gives first ``Meng``, last ``Wang`` again, and ``Parser(policy=Policy(name_order=FAMILY_FIRST)).parse("Wang Meng")`` gives given ``Meng`` where 2.0.0 through 2.3.0 gave family ``Wang``, suffix ``Meng`` and no given name. With a full name in front the credential reading stays: ``john smith meng`` and ``nguyen van lac`` keep suffix ``meng`` and ``lac``, now flagged. But a Title-case ``Nguyen Van Lac`` gives last ``Van Lac`` where every release gave last ``Van``, suffix ``Lac``. The cost is the marking's own, and it falls on the conventional spellings: a ``MEng`` or ``LAc`` written that way, in a name written in more than one case, is read the way ``John Smith Ma`` is (above), so ``John Smith MEng`` gives middle ``Smith``, last ``MEng``, where every release gave suffix ``MEng``. Where one of them LEADS a credential run nothing speaks for it: ``John Smith MEng PhD`` gives middle ``Smith``, last ``MEng``, suffix ``PhD``, where every release read suffix ``MEng PhD`` (``MEng, PhD`` through 2.2); behind a credential, or written with its periods, it is read with the run (the next entry). After a comma ``Smith, MEng`` and ``Smith, meng`` give first ``MEng`` and ``meng`` (1.4.0's reading, not the suffix 2.0 through 2.3 gave) and ``Smith, John MEng`` gives middle ``MEng`` (every release gave suffix ``MEng``), and a bracketed ``John Smith (MEng)`` falls through to nickname, where every release gave suffix ``MEng``. A lone credential after a comma behind a full name (``John Smith, MEng``) keeps the credential reading, and so does a run of them (the next entry). Meng is a common Chinese surname and given name, Lac a Vietnamese given name (``Nguyen Van Lac``) and a French surname; see the ``suffix-acronym-collisions`` entry of ``docs/design/decisions.md`` (closes #540) + + - **Fix a credential run losing the acronyms in it that are also names: a degree in front speaks for the acronym behind it, and a run after a comma is read whole.** ``HumanName("John Smith, Ed Ma")`` gives first ``John``, last ``Smith``, suffix ``Ed Ma``, where 2.0 through 2.3 gave first ``Ed``, middle ``Ma``, last ``John Smith`` -- 1.4.0's reading, restored: with two or more name words before the comma, a part made only of suffix words and acronyms that are also names, and holding no one-letter roman numeral, is a credential run however it is written, as a lone one already was (``John Smith, MA``), and ``parse()`` reports the call. ``john smith, md ma`` and ``John Smith, Ms Ma`` move the same way, where 2.0 through 2.3 gave title ``md``/``Ms`` -- the second is the accepted cost, ``Ms`` read as the suffix word it also is, as ``John Smith, Ms`` alone already reads it -- while one name word before the comma keeps the listing form (``Smith, Ms Ma`` gives title ``Ms``, first ``Ma``). At the end of a name, an acronym standing behind an unambiguous credential is read as that credential's company whatever its case: ``John Smith PhD MEng`` and ``Doe, Jane PhD MEng`` give suffix ``PhD MEng``, the fields every release gave (``PhD, MEng`` through 2.2), now reported, and ``doe, jane v phd do`` gives suffix ``v phd do`` where 2.3.0 gave last ``do doe`` -- a degree in front outranks the particle reading, as capitals already did. Only a credential IN FRONT speaks: ``Wang Ma PhD`` keeps last ``Ma``. A listed acronym written in period-closed chunks is written with its periods, so ``Wang M.Eng.`` gives suffix ``M.Eng.``, as ``Wang M.A.`` does and as 2.0 through 2.3 did. See the #544 entry under ``S2`` in ``docs/design/decisions.md`` (closes #544) - **New Policy field unlisted_dotted_suffixes, on by default: a dotted acronym nobody has listed is read by position.** ``HumanName("John Smith X.Y.Z.")`` gives suffix ``X.Y.Z.`` where every release gave last ``X.Y.Z.``, while ``Jack X.Y.Z.`` keeps its surname, the same words-to-spare rule a listed acronym takes -- and both readings are reported. Case is irrelevant here: the periods are the signal, so ``john smith x.y.z.`` reads the same way. Words the vocabulary does know are untouched (``M.A.``, ``Ph.D.``, ``A.B.C.``), a single trailing period is still not this shape (``John Smith Xyz.`` keeps last ``Xyz.``), and a dotted run at the FRONT of a name is untouched (``J.R.R. Tolkien``). One accident retires with it: a dotted word whose only vocabulary matches were SINGLE ASCII CHARACTERS -- the roman numerals the suffix list holds, and the lone digit ``2`` -- was reading as a generational suffix, so ``Jack X.Y.I.`` gives last ``X.Y.I.`` again, as 1.4.0 read it, while ``Msc.Ed.``, ``JD.CPA`` and ``Lt.Gov.`` are unchanged. The digit is why a dotted VERSION STRING moves with them and moves SILENTLY: ``John Smith 1.4.2`` gives last ``1.4.2`` where 2.3 gave suffix ``1.4.2``, and ``John Smith, 1.4.2`` gives first ``1.4.2``, last ``John Smith``. Such a token reports nothing at any policy -- it is no acronym either, the shape reading wanting every chunk alphabetic -- and a version string read as a credential was the same accident this retirement removes. That retirement is NOT behind this switch and stands either way -- setting it to ``False`` reads an unlisted dotted word as name material by position instead (``John Smith X.Y.Z.`` keeps last ``X.Y.Z.``), the pre-2.4 reading for THAT half alone. See the ``S2`` and ``suffix-acronym-collisions`` entries of ``docs/design/decisions.md`` (closes #516) diff --git a/nameparser/_types.py b/nameparser/_types.py index c1a468cf..4e47feb9 100644 --- a/nameparser/_types.py +++ b/nameparser/_types.py @@ -460,6 +460,14 @@ class AmbiguityKind(StrEnum): #: site and this kind stays out of the way -- "Doe, John do" gives #: family ``do Doe`` and one ``PARTICLE_OR_GIVEN``, never two #: reports of one word. + #: Since #544 C1's name-word count reports its flip once over a + #: whole RUN after the comma ("John Smith, Ed Ma" reads suffix + #: ``Ed Ma``), and a member an unambiguous credential in front of + #: it speaks for (rules.md#S2) is a pick like a counted one, + #: reported by the peel or the given part's trailing slot that took + #: it ("John Smith PhD MEng", "Doe, Jane PhD MEng"). The + #: first-piece emitter still asks its own piece only, so "Smith, + #: PhD MEng" reads suffix ``PhD MEng`` and reports nothing. #: Since 2.4 a maiden marker's clause reports at ITS trailing slot #: too, in both directions: "Doe, Jane nee Smith MA" gives maiden #: ``Smith`` with suffix ``MA`` and says so, "Doe, Jane nee Smith diff --git a/tests/v2/cases.py b/tests/v2/cases.py index 1a7e4bf7..1056ff1a 100644 --- a/tests/v2/cases.py +++ b/tests/v2/cases.py @@ -996,8 +996,10 @@ def _check_cjk_shape_purity(self) -> None: "smith, v ed", {"given": "v", "family": "smith", "suffix": "ed"}, ambiguities=("suffix-or-name",), - notes="a single-letter roman numeral is a generation, not a " - "credential, so it speaks for nothing: 'ed' is read by " + notes="a single-letter roman numeral is initial-shaped, " + "written as a middle initial is, so it speaks for " + "nothing (a multi-letter one such as 'III' would): " + "'ed' is read by " "the given slot's own rule. 1.4.0 read the same; 2.3.0 " "read middle 'ed'"), Case("an_anchored_particle_member_outranks_p6", From 16d835fb7d3dcccc293992ebe7c7332e986c790b Mon Sep 17 00:00:00 2001 From: Derek Gulbranson Date: Mon, 28 Sep 2026 11:15:49 -0700 Subject: [PATCH 06/11] fix(#544): an anchored pick after a family comma reports, and a dual title run keeps its reading A member the company decides reports where it was decided. When a credential in front makes a family comma's given part read wholly as credentials, each member whose own writing declined and which the anchor read as the credential reports suffix-or-name, as the peel's picks do: 'Smith, PhD Ma', 'Smith, PhD MEng', 'Smith, Ph. D. MEng', 'Smith, Dr. PhD LAc', 'john smith, v phd ma' and 'John Smith, PhD Ma Prof.' each report the member. A member its own capitals made the credential stays silent ('Smith, PhD MA'), and the part's first piece is the first-slot report's word, never this one's. `segment_suffix_reading` names those picks and assign emits them. A title/suffix dual in the given part's leading title run is a title there, and no credential behind it speaks for a member of the ambiguous class in that part: the next word is the given name, and the parent's reading stands -- 'Smith, Ms MD Ma' and 'Smith, MD MS Ma' title plus given 'Ma', 'Smith, MD PhD Ma' given 'PhD', middle 'Ma' (reported), 'SMITH, MD MS BA' given 'BA'. A plain title does not do this ('Smith, Dr. PhD LAc' reads suffix 'PhD LAc'), behind a real given name the company reads as ever ('Doe, Jane MD Ma' suffix 'MD Ma'), and past the title run the given slot's company still speaks ('Smith, MD PhD Jr Ma' suffix 'Jr Ma'). A part with no member to speak for, or only members whose capitals decide them, reads as it did ('Smith, MD PhD', 'Smith, Ms MD MA'). rules.md P6 names two exceptions to the attachment, the capitals and S2's company ('doe, jane v phd do'), and limits the one-case sentence to a word with nothing in front of it; S2's own sentence about the particle member and the unreleased #531 release bullet say the same. rules.md: S2's company clause states the dual's leading-title-run scope, the report of a decided member, and the single-letter numeral exclusion as a matter of shape in any case; C1 states the numeral exclusion as shape ('John Smith, III Ma' is a run) and which runs report, naming the silent ones. `is_single_letter_numeral`'s comment gives the shape reason too. Example lines 'Smith, PhD Ma' and 'Smith, MD PhD Ma'. The SUFFIX_OR_NAME docstring and assign's emitter roster say the same. decisions.md: the #544 entry gains forks (9) and (10), restates refinement (b) and boundary (d) (the exclusion is a single-letter roman numeral in any case, which `is_single_letter_numeral` tests; `is_initial_shaped` is False for a lower-case 'v'), drops fork (5)'s unrecomputable counts, states the accepted cost as its shape (a capitalised 'MA' opening a one-word-family part, a degree behind it: 138 grid parses, 46 texts, from given 'MA' to no given name), corrects boundary (c) for the family-first orders, and re-measures the grid against e10e83b4 with the detector's token sets and the attribution order in the recipe: 89,825 of 254,496 parses move, the invariants 29,199 -> 1,530 and 36,675 -> 4,734 with 0 new violations. The #540 addendum names the successor case rows. The release note says which runs report, adds the two readings, and says a one-letter numeral in front no longer stops the part reading whole ('john smith, v phd ma'). Case rows pin 'Smith, PhD Ma', 'John Smith, PhD Ma Prof.' against 'John Smith, PhD Ma', 'Smith, Ms MD Ma' and 'Smith, MD PhD Ma'; 'Smith, PhD MEng' now reports. Unit tests pin the silence for 'Smith, PhD MA' against 'Smith, PhD Ma', and 'Smith, M.D. PhD Ma' (a dotted credential is no title-run dual) reporting 'Ma'. Ledgers: 'Smith, PhD Ma' is a new fix(#544) rule at 2.2.0/2.3.0 and, with 'Smith, PhD MEng', fix(#296/#544)'s at 2.0.0/2.1.0; 'Smith, PhD MEng' joins the anchor rule at 2.2.0/2.3.0; 'John Smith, PhD Ma' joins the run rule; 'Smith, MD PhD Ma' joins fix(#531)'s declining rule at 2.2.0/2.3.0 and is fix(#296/#531)'s at 2.0.0/2.1.0 and fix(#296)'s at 1.4.0; 'Smith, PhD Ma' is fix(#325)'s at 1.4.0, pinned beside 'Smith, PhD MEng'. Gate exit 0 at all five baselines, radar unclassified unchanged. Co-Authored-By: Claude Opus 5.5 --- docs/design/decisions.md | 12 +-- docs/design/rules.md | 85 ++++++++++++------ docs/release_log.rst | 4 +- nameparser/_pipeline/_assign.py | 27 +++++- nameparser/_pipeline/_pieces.py | 80 ++++++++++++----- nameparser/_pipeline/_segment.py | 18 ++-- nameparser/_pipeline/_vocab.py | 14 +-- nameparser/_types.py | 11 ++- tests/v2/cases.py | 79 ++++++++++++++-- tests/v2/pipeline/test_pieces.py | 63 +++++++++++-- tests/v2/pipeline/test_segment.py | 3 +- tests/v2/test_ledger_guards.py | 94 +++++++++++++++++--- tests/v2/test_parser.py | 30 +++++++ tests/v2/test_properties.py | 11 +++ tools/differential/compare.py | 3 + tools/differential/corpus_rules.jsonl | 2 + tools/differential/corpus_shapes.jsonl | 4 + tools/differential/expected_since_1.4.0.toml | 20 +++++ tools/differential/expected_since_2.0.0.toml | 48 +++++++++- tools/differential/expected_since_2.1.0.toml | 48 +++++++++- tools/differential/expected_since_2.2.0.toml | 40 ++++++++- tools/differential/expected_since_2.3.0.toml | 40 ++++++++- 22 files changed, 619 insertions(+), 117 deletions(-) diff --git a/docs/design/decisions.md b/docs/design/decisions.md index de6b7bea..9c68562a 100644 --- a/docs/design/decisions.md +++ b/docs/design/decisions.md @@ -617,7 +617,7 @@ Closes #342 (a wordlist question) and #454 (a rules.md question) together, becau - **2026-09-25 (#540) — meng and lac joined SUFFIX_ACRONYMS_AMBIGUOUS (Derek: the marking over removal, and no masks).** #vocabulary-collisions C-i asked at the position the suffix claim acts on, the last word of a name: Meng (孟) is a common Chinese surname and given name, and Lac (Lạc) is a Vietnamese given name — the trailing word in native order, `Nguyễn Văn Lạc` — and a French surname; both are borne in that slot (`wang meng`, `tran lac`, `nguyen van lac`). MEng (Master of Engineering) and LAc (Licensed Acupuncturist) are live credentials, and neither reading is rare enough beside the other to give it the word — the rough balance this entry's criterion marks (where it is uncertain, C-i's default is the marking too). They are the set's first entries longer than two letters, which is the LENGTH clause of the criterion bullet applied, not an exception to it. Neither was checked against a surname when it arrived: `lac` with the af5bdab import (2019-12-11, #93), `meng` with 3e14ea20 (2026-07-01, the German/Dutch title and degree batch, first shipped in 1.3.0). No rule changes: rules.md#S2 reads the two words exactly as it reads `ma` and `ba`. Measured 2026-09-25 on this tree, the parent 120033b5 and the five released wheels (each wheel run as the Gotchas in AGENTS.md prescribe, from a directory outside the checkout under `PYTHONSAFEPATH=1`, asserting `nameparser.__file__`). THE FIX: `wang meng`, `li meng` and `tran lac` read given + family with a `suffix-or-name` report, where every release from 2.0.0 read suffix and no family name — reporting nothing at 2.0.0 through 2.2.0, and `given-or-family` at 2.3.0 and the parent, the one-word-name fork, which was the wrong fork — and 1.4.0 read the family, unflagged. The family-comma spelling returns to 1.4.0 with them, `Wang, Meng` reading given `Meng`, family `Wang` where 2.0.0 through 2.3.0 read suffix `Meng`; and under `FAMILY_FIRST` the bare `Wang Meng` reads given `Meng` where 2.0.0 through 2.3.0 and the parent read family `Wang`, suffix `Meng`, so the comma spelling under the default order and the bare one under `FAMILY_FIRST` still agree with each other. WITH WORDS TO SPARE the credential stays: `john smith meng`, `JOHN SMITH MENG` and `nguyen van lac` keep their suffix, now reported, and so does `john smith m.eng.`, which S2's period gate does not settle, that gate counting one period after each letter. DECIDED (Derek, 2026-09-25): the bare `Wang M.Eng.` reading — family `M.Eng.`, nothing to spare — is accepted and recorded rather than widening the period gate to cover a chunked dotted spelling; S2's period gate counts a member as unambiguous only when it is written one period per letter (`M.A.`), and MEng and LAc are the first members whose conventional dotted spelling is chunked instead. `John Smith M.Eng.` keeps the suffix by the words-to-spare count, unaffected. THE ACCEPTED COST is the one `ma` and `ba` already carry, and it falls on the conventional spellings: in a name written in more than one case S2 reads a member written neither in capitals nor all lower as the name even with words to spare, and MEng and LAc are written exactly that way. `john smith MEng` reads middle `smith`, family `MEng`, where every release from 1.4.0 read suffix `MEng`; `John Smith Meng` and `John Smith LAc` move the same way; and the Title-case `Nguyen Van Lac` reads family `Van Lac` where every release read family `Van`, suffix `Lac` (for this Vietnamese name the gain, the mechanism being the cost's) — only the all-lower `nguyen van lac` keeps its suffix, one case saying nothing. A CREDENTIAL RUN ending in one of them pays the same walk-stopping cost: `Mary Jones PhD MEng` reads middle `Jones PhD`, family `MEng`, and `Jane Doe MS LAc` reads middle `Doe MS`, family `LAc`, where every release read the whole run as a suffix (`PhD, MEng` and `MS, LAc` through 2.2, `PhD MEng` and `MS LAc` in 2.3); `John Smith MEng PhD` reads middle `Smith`, family `MEng`, suffix `PhD`, the unambiguous credential behind the declined pick still peeled; the mechanism is S2's own — the case signal costs a genuine suffix standing behind a name-leaning acronym, the walk stopping at the declined pick rather than continuing past it — pinned by `tests/v2/cases.py`'s `a_declined_ambiguous_pick_stops_the_walk` and, for the family-comma shape, `a_declined_member_ends_the_trailing_run` (`Doe, John MA Ma`). A QUIETER COST sits behind a suffix comma: once the run's last word is a bare ambiguous member, the run is no longer wholly suffix-shaped, so C1 reads the comma as a family comma instead and the run's own words fall into the usual post-comma name slots — `John Smith, PhD MEng` reads given `PhD`, middle `MEng`, family `John Smith`, and `john smith, phd meng` reads given `phd`, family `john smith`, suffix `meng` (the count then peels `meng` with a word to spare); neither case form saves it, `JOHN SMITH, PHD MENG` reading the same way. A title-listed word standing in front of the run reads as a TITLE instead, with no report: `Jane Doe, MS LAc` reads title `MS`, given `LAc`, family `Jane Doe`, unflagged, where every release read suffix `MS LAc`. The path is not new: `john smith, phd ma` and `Jane Doe, MS Ma` already read this way, `ma` having been ambiguous all along. Pinned by `tests/v2/cases.py`'s `comma_credential_run_ending_in_meng_re_reads_the_comma` and `comma_lower_credential_run_ending_in_meng_re_reads_the_comma`. The comma re-read (with its silent title-dual case) and whether a listed member's chunked dotted spelling should pass the period gate are [#544](https://github.com/derek73/python-nameparser/issues/544)'s questions. After a comma, `Smith, MEng` reads given `MEng` (1.4.0's reading; 2.0.0 through 2.3.0 read suffix) — and so does the all-lower `Smith, meng`, a restored reading rather than a cost — and `Smith, John MEng` middle `MEng`, pinned as case rows (`Smith, meng`, `Smith, MEng`, `Smith, John MEng`) and classified by the `fix(#540)` rules, while `John Smith, MEng` keeps the suffix on C1's name-word count; bracketed, `John Smith (MEng)` reads nickname `MEng` with a `suffix-or-nickname` report, S1's escape declining an ambiguous member, where every release read suffix `MEng`. `John Smith MENG` keeps the credential on the capitals lean. The judgment is the one behind rai and cha: a missed credential leaves the letters in a name field where a human can still see them, and a wrong credential reading destroys a surname on every record it appears in. edd and ded are NOT marked: both are borne as GIVEN names (`Edd Smith`, `Ded Gjo Luli`), which is not the position the suffix claim acts on, and no measurement shows either losing a family name — a trailing bearer turning up reopens them. Recomputed with #vocabulary-collisions' recipe on 2026-09-25: suffix_acronyms 608 (unmoved by this change) with 7 ambiguous, 5 before this change (particles 70 with 37 and titles 758, unmoved). - **Declined 2026-09-25 (#540), with measurement: removing the two words, and masking them.** REMOVAL, the rai/cha answer, measured with `Parser(lexicon=Lexicon.default().remove(suffix_acronyms={"meng", "lac"}, suffix_acronyms_ambiguous={"meng", "lac"}))`: `john smith meng` reads middle `smith`, family `meng`, and `JOHN SMITH MENG` family `MENG`, while `john smith m.eng.` stays a suffix by its dotted shape (S3), reported. Removal loses the bare credential outright where the marking keeps it with words to spare, and C-i's default under uncertainty is the marking. MASKS `meng → MEng` and `lac → LAc`, measured with both added to a private lexicon on top of the marking: the exceptions map is role-free (#R4, 2026-09-23), so `wang meng` renders `Wang MEng`, `tran lac` `Tran LAc` and `meng li` `MEng Li` — the mask re-spells the very surnames the marking restores. A suffix-gated mask would reverse the 2026-09-23 role-free decision for two words and is not taken. Without one, the default path renders `wang meng` as `Wang Meng` (the parent gave `Wang MENG`, the credential clause writing a suffix in capitals), `tran lac` as `Tran Lac`, `john smith meng` as `John Smith MENG` and `nguyen van lac` as `Nguyen Van LAC` by the acronym clause, and keeps a writer's `MEng` as written (#R5, 2026-09-24) — `john smith MEng` gives `john smith MEng`, and `John Smith Meng` under `force=True`, the word being the family name there. Pinned by `tests/v2/test_render.py::test_a_listed_acronym_that_is_a_name_word_gets_no_mask`, whose recorded negative control carries the two masks. - **Measurement (2026-09-25, #540).** No corpus name written before this change carries a bare trailing `meng` or `lac` (the `tools/differential/corpus*.jsonl` glob at 120033b5), so the population that could move is this change's own shape-tagged case rows, and every mover is one of them. Classified by `fix(#540)` rules: at each 2.x baseline `wang meng` and `tran lac` move {family, suffix, _ambiguities}, `john smith MEng` and `John Smith MEng PhD` move {middle, family, suffix, _ambiguities}, and `john smith meng` and `nguyen van lac` `_ambiguities` alone; `Smith, John MEng` moves {middle, suffix, _ambiguities} at 2.x and {middle, suffix} at 1.4.0; `Smith, meng` and `Smith, MEng` move {given, suffix, _ambiguities} at 2.x and nothing at 1.4.0 (1.4.0's own reading); `John Smith, PhD MEng` moves {given, middle, family, suffix, _ambiguities} and `john smith, phd meng` {given, family, suffix, _ambiguities} at 2.x (without `_ambiguities` at 1.4.0); `Nguyen Van Lac` moves {family, suffix, _ambiguities} at 2.x and {family, suffix} at 1.4.0; and `Wang M.Eng.` moves {family, suffix, _ambiguities} at 2.x only, 1.4.0 reading the same family for an unrelated reason (its two-piece rule -- a lone word after the given name is the family -- not S2's gate). `nguyen van lac` also diffs at 1.4.0 on `_initials` (`n.` → `n. v.`) for fix(#385/#402)'s reason — its family `van` is a one-particle part — and joins that rule's literal list; the same case-insensitive regex also reaches `Nguyen Van Lac` at every baseline, without explaining anything there (its own family move, at 1.4.0 too, is explained by the Title-case Lac rule instead). Read today's intentional counts off the `corpus:` line of `uv run python tools/differential/compare.py --baseline X`; this change raised them by seven at 1.4.0 and by thirteen at each 2.x baseline. -- **2026-09-27 (#544) — #540's two pending questions are answered, and the #540 bullet stands as it landed.** The chunked dotted spelling passes S2's period gate — `Wang M.Eng.` reads suffix 'M.Eng.' as `Wang M.A.` does, and as 2.0.0 through 2.3.0 read it — and the comma is kept: a credential run after a full name that ends in a member is the credential run by C1's name-word count (`John Smith, PhD MEng` suffix 'PhD MEng'; `Jane Doe, MS LAc` suffix 'MS LAc' where #540 left title 'MS', given 'LAc', at the cost, accepted under S2, of `John Smith, Ms Ma` reading suffix 'Ms Ma'). The walk-stopping cost that bullet lists for a run ENDING in a member goes with them, by S2's company clause: `Mary Jones PhD MEng` and `Jane Doe MS LAc` read the whole run as a suffix again. Where the member LEADS the run nothing speaks for it, so `John Smith MEng PhD` keeps its cost. The `fix(#540)` ledger rules for the comma run and the dotted spelling are retired. `fix(#544)` rules classify the comma-run names; `Wang M.Eng.`, whose roles every release from 2.0.0 read, is classified by `feat(#449)`'s given-or-family report rule at 2.0.0 through 2.2.0 and by the dotted `fix(suffix-routing)` rule beside `Jack M.A.` at 1.4.0, and differs from 2.3.0 not at all. The decision and its measurements are the #544 entry under S2. +- **2026-09-27 (#544) — #540's two pending questions are answered, and the #540 bullet stands as it landed.** The chunked dotted spelling passes S2's period gate — `Wang M.Eng.` reads suffix 'M.Eng.' as `Wang M.A.` does, and as 2.0.0 through 2.3.0 read it — and the comma is kept: a credential run after a full name that ends in a member is the credential run by C1's name-word count (`John Smith, PhD MEng` suffix 'PhD MEng'; `Jane Doe, MS LAc` suffix 'MS LAc' where #540 left title 'MS', given 'LAc', at the cost, accepted under S2, of `John Smith, Ms Ma` reading suffix 'Ms Ma'). The walk-stopping cost that bullet lists for a run ENDING in a member goes with them, by S2's company clause: `Mary Jones PhD MEng` and `Jane Doe MS LAc` read the whole run as a suffix again. Where the member LEADS the run nothing speaks for it, so `John Smith MEng PhD` keeps its cost. The `fix(#540)` ledger rules for the comma run and the dotted spelling are retired. `fix(#544)` rules classify the comma-run names; `Wang M.Eng.`, whose roles every release from 2.0.0 read, is classified by `feat(#449)`'s given-or-family report rule at 2.0.0 through 2.2.0 and by the dotted `fix(suffix-routing)` rule beside `Jack M.A.` at 1.4.0, and differs from 2.3.0 not at all. The case rows the #540 bullet cites were renamed with the readings they pin: `a_declined_ambiguous_pick_stops_the_walk` is `a_credential_in_front_anchors_a_declined_pick` (`abdul Smith Jr Ma`, suffix 'Jr Ma'), and `comma_credential_run_ending_in_meng_re_reads_the_comma` and `comma_lower_credential_run_ending_in_meng_re_reads_the_comma` are `comma_credential_run_ending_in_meng_keeps_the_suffix_comma` and `comma_lower_credential_run_ending_in_meng_keeps_the_suffix_comma`. The decision and its measurements are the #544 entry under S2. ### S2 — the case signal at the suffix slot @@ -670,13 +670,13 @@ for n in ('Smith, John','Smith, XYZ'): print(n, calls_for(off.parse, n), calls_f - 2026-09-18 (Derek), #531 — THE COMMA REPORT'S REACH NOW INCLUDES THE GIVEN SEGMENT'S TRAILING SLOT, AND THE OPEN FOLLOW-UP IS CLOSED. The 2026-09-18 bullet above headed THE COMMA REPORT'S REACH IS THE FIRST POST-COMMA PIECE recorded this slot as an open maintainer decision and named the two questions it turned on — whether a middle initial's neighbourhood should start reporting (a noise judgement, rules.md#A1) and whether the silence was a 1.4 parity gap. Both are answered here, and that bullet stands as it landed. The slot now reads and reports: `Doe, John MA` gives suffix `MA` and `Doe, John Ma` keeps middle `Ma`, each saying which way it went. Derek chose to restore the ROLE and report both ways — one rule for both spellings — over a report-only change and over a capitals-only one, because the issue exists in the first place because two spellings of one name disagree. The noise question was settled by MEASURING THE DISAGREEMENT rather than by argument: over a generated sweep of 78 pairs — five listed members and two by-shape tokens in three cased spellings each, plus five controls in one spelling apiece, so 26 words against three name shapes — 48 pairs disagreed about whether the word was a credential or a name, and every one of the 48 disagreed in the same direction, the comma form declining what the comma-less form took. After this change 3 disagree and all three are the lower-case `do` rows P6 owns. The sweep and its allowlist are `tests/v2/test_properties.py`; a slot that answers differently from the same name written without a comma is not a quiet slot, it is an inconsistent one. Recorded under `3-0-reevaluations`' standing rule because v1 parity is LOAD-BEARING for one half of this and explicitly NOT for the other: `Doe, John MA` reads suffix `MA` on the 1.4.0 wheel, so the bare-acronym half RESTORES v1's role and the report is all that is new there — its 1.4.0 ledger rule neither retired nor narrowed to `_ambiguities` (nothing below baseline 2.0 can diff on that pseudo-field) but was RE-POINTED to the four names where the writing declines the credential, which keep the 2.0-era middle name against v1. `Doe, John X.Y.Z.` reads middle `X.Y.Z.` at 1.4.0 too, so the dotted half LEAVES v1 and carries a 1.4.0 ledger rule of its own; it moves to match the comma-less `John Doe X.Y.Z.` and rules.md#S3's shape rule, not to restore anything. BLAST RADIUS, stated the way this log's own rule asks: twenty corpus names move, and only TWO of them were in any corpus before this branch (`Doe, John MA` and `Doe, John X.Y.Z.`, both admitted by #530's own arc) — the other eighteen are this change's own case rows, so what the differential measures on pre-existing data is two names, and the population the rule reaches is a SHAPE (every family-comma listing whose given part ends in a member of this class) that the corpora barely sample. ACCEPTED COSTS, all measured on the differential corpora: `SMITH, JOHN DO` keeps family `DO SMITH` where 1.4.0 read suffix `DO`, paired with the Nascimento record in the bullet above; one-case and caseless names take the credential with no case evidence at all, so `DOE, MARY JO MA`, `doe, john ma`, `田中, 太郎 MA` and `김, 민준 MA` all read a suffix, which is what 1.4.0 read for each of the four — "caseless is inert" holds for the LEAN and not for the outcome, since `is_one_case` answers True for a script with no case and the positional reading then decides; `Doe, John van MA` loses its middle to the family, reading family `van Doe`, suffix `MA` with two reports where it read middle `van MA` in silence, which is `Berg, Jan van Jr.`'s reading arriving through a shape it could not reach before (Derek accepted it 2026-09-18 as P6 working correctly rather than as a cascade to carve out); `Doe, John Prof. MA` gains a TITLE role `Prof.` never had, H5's transparency reaching it once `MA` leaves the walk, landing it on the same answer as `Doe, John MA Prof.`; `Smith, LEED AP` moves under the default-off caps switch alone, to given `LEED`, family `Smith`, suffix `AP` with two reports, so no default reading is at stake; and a genuine middle name that is also a class member is now a credential wherever it ends the given part and is not Title-cased — `DOE, JOHN ED` reads suffix `ED` — which is the same cost the comma-less form has carried since 2.0. NOT REPAIRED HERE and left open: the maiden walk claims `Smith MA` whole in `Doe, Jane nee Smith MA` before this slot exists, so the rule cannot reach it and the name stays silent with maiden `Smith MA` — the same gap #530's close-out saw from the other side with `John Smith nee Jones R.A.I.`, and it is out of scope for #531. - 2026-09-19 (Derek), #533 — THE CLASS'S LAST SILENT TRAILING POSITION IS CLOSED, AND #531'S READING NOW LIVES IN ONE PLACE. The slot list in rules.md#S2 gains the trailing slot of a maiden marker's clause, and #S3's enumeration gains it with the by-shape spellings; the rule that governs what the clause does with those words is M2's, and the entry under `### M2` above is where its reasoning lives. The gap the bullet above left open as out of scope for #531 is the one this closes, from both sides: `Doe, Jane nee Smith MA` now gives maiden 'Smith' with suffix 'MA' and reports, and `John Smith nee Jones R.A.I.` gives suffix 'R.A.I.' again — a RESTORATION rather than a change, since 2.3.0 read it that way (measured on the wheel) and this unreleased cycle moved it into the maiden name when #516 took dotted tokens out of the certain-suffix class, where no corpus file held the name and no gate could see it. Two things belong here rather than under M2. FIRST, `credential_at_the_given_slot` in `_pipeline/_pieces.py` is now where #531's reading of a class member ending the given part is written, with two callers — assign's walk over that part, and the maiden walk's second check over the name a take would leave. Spelling it twice is precisely the "condition written to match it" mechanisms.md#ONE-PREDICATE-PER-QUESTION names, and the drift would have been silent, since each site's own tests would have gone on passing. It is a text-and-tags question, which is what puts it in `_pieces` rather than beside either caller — the destination follows the LAYER, not the topic. It costs one frame PER MEMBER asked at that slot: measured 2026-09-19 against 2f57ff21 per `Parser.parse`, `Doe, John MA` goes 310 → 311 and `Doe, John MA Ma MA`, which asks four times, 439 → 443, while a name with no member there never reaches it and pays nothing. The reference band does not move (412/449 on `uv run python tools/perf/call_count.py`), no test pins 310, and Derek took the trade rather than keep two conditions in step across two stages with no test that asks them both. The maiden path got one frame CHEAPER in the same pass, the numeral reading calling the peel pair it had wrapped rather than the wrapper: `Jane Doe nee Smith` 249 → 248. SECOND, THE ONE-CASE HEAD IS AN ACCEPTED EXCEPTION AND IT IS THE M2 INSTANCE OF #492'S DEFERRED QUESTION about whether a cased suffix token counts as case evidence. `DOE, JANE nee Smith Ma` reads suffix 'Ma' where the clause-less `DOE, JANE Ma` keeps middle 'Ma', because the own-words span stops at the marker — rules.md#P3 puts "a maiden marker's run and every word after it" outside the name's own words from the moment the marker is tagged — so the member's own Title-casing is not in the span `one_case` is computed over and the clause HIDES the contrast. That is THE PREDICATE KEEPS THE JUDGED TOKEN IN THE SPAN, the 2026-09-14 #289/#516 entry at the head of this section, failing structurally rather than by oversight: at this slot the judged token is never in the span, and that entry's own answer — include it — cannot be had here. Widening the span for this one question would change `one_case` for the whole name, and three sites read it, so the exception is accepted instead. Measured 2026-09-19: 114 of 2016 generated pairs — every listed member and both by-shape spellings, in three cased spellings, against sixteen heads and six clause bodies of NAME WORDS ONLY — against 186 allowlisted and 984 disagreeing outside the class before the change, and 0 outside it after. The PR review widened that grid to three markers (`nee`, `née`, `geb.`) and three policies (the default and each 2.4 switch), and the class is INDIFFERENT to both: 1,026 of 18,144, exactly 9x the original in both columns, and still 0 disagreeing outside it. `tests/v2/test_properties.py` carries the sweep with the class defined STRUCTURALLY (`one_case` true of the clause form and false of the clause-less one) rather than as a name list. The recorded control beside it is now the SET and not only its size, as a digest over the members: a count cannot notice a swap, one pair leaving and another arriving, which is the same blindness the count was added to close one level up. - 2026-09-23 #492 — THE DEFERRED QUESTION the 2026-09-14 predicate paragraph and the 2026-09-19 #533 bullet above name — whether a cased suffix token counts as case evidence — IS ANSWERED FOR RENDER ONLY. R5's gate now leaves the suffixes out (decisions.md#R5, 2026-09-23); the parse-time fact this section rests on keeps counting the judged token exactly as written above, so `Jack MA` still leans credential and `DOE, JANE nee Smith Ma` still reads suffix `Ma`. The split is decided and recorded under R5, with the shapes it makes visible there (`jack MA` at this slot, `john e jones III` at P3's). -- 2026-09-27 (Derek), #544 — A CREDENTIAL RUN KEEPS ITS AMBIGUOUS MEMBERS: C1'S NAME-WORD COUNT READS A RUN AS IT READS ONE WORD, AN UNAMBIGUOUS CREDENTIAL IN FRONT OF A MEMBER SPEAKS FOR IT AT EVERY TRAILING SLOT, AND A LISTED MEMBER'S CHUNKED DOTTED SPELLING PASSES THE PERIOD GATE. Three readers never looked past the member itself: C1 flipped a part to the suffix comma by the name-word count only when it was ONE token, and the trailing readers — the no-comma peel, the comma part read wholly as credentials, the given part's trailing slot — met the member first, took its name lean and stopped, so the credential in front of it was never reached. #540 marked `meng` and `lac`, whose conventional spellings lean name, which turned the one-case-row cost this section's 2026-09-15 item (ii) accepted into the ordinary spelling of two degrees. DECIDED, eight forks. (1) THE WHOLE RUN AT C1: two or more words after the comma, every one suffix vocabulary or a class candidate, at least one a candidate, behind two or more name words, is the credential run — all-member runs (`John Smith, Ed Ma`) and a leading member (`John Smith, MEng PhD`) included; offered and declined were an anchored-only rule and the issue's own "last word ambiguous" wording, which contradicted its `Ed Ma` consequence. (2) ALL TRAILING PATHS, ANCHORED IN FRONT: a listed member with an unambiguous credential in front of it, through other members, reads as the credential on the no-comma peel, in the comma part read wholly as credentials and at the given slot; a credential BEHIND it anchors nothing (`Wang Ma PhD` keeps family 'Ma'). (3) THE CHUNKED DOTTED GATE: a listed member written in two or more period-closed letter chunks (`M.Eng.`, `L.Ac.`) passes as `M.A.` does; one trailing period (`Ma.`, `Ed.`) still does not. (4) SINGLE-LETTER ROMAN NUMERALS neither anchor nor count toward a C1 run, S3's single-character retirement being the precedent; they peel as before, and the anchor's half of the exclusion is a matter of their shape, boundary (d) below. (5) NO NEW REPORT WHERE THE WRITING SETTLED IT: a run whose every member is LISTED and leans credential — S2's lean, capitals in a mixed-case name — is not flipped, the family-comma path already reading it as the credential run with the fields and the report it always had (`John Smith, PhD MA`); the alternative, flipping such a run and suppressing only the flip's report, traded 144 report additions for 180 removals over the grid below (measured on the draft, 2026-09-27), `John Smith, MA Jr` losing the report it makes today. Because the lean is the listed set's, a by-shape member in capitals never settles a run: `John Smith, X.Y.Z. MA` flips to suffix 'X.Y.Z. MA' and reports. (6) THE ANCHOR DOES NOT REACH ACROSS A MAIDEN CLAUSE: `Jane Doe Jr. nee Smith Ma` keeps maiden 'Smith Ma' where `Jane Doe Jr. Ma` reads suffix 'Jr. Ma' — rules.md#M2's Accepted boundary, and the M2 bullet of this date. (7) AN ANCHOR OVERRIDES P6 AT THE GIVEN SLOT: `doe, jane v phd do` reads suffix 'v phd do' where it read family 'do doe'; the bullet below amends #531's pairing. (8) THE DUAL EXCLUSION IS SCOPED TO THE GIVEN PART'S HEAD, NOT THE C1 RUN: after a full name the dual counts as the suffix vocabulary it is, as the legacy disjunct already counted it (`John Smith, MS MA`, `John Smith, MD`), and that is what reads the issue's headline `Jane Doe, MS LAc` as suffix 'MS LAc'. - WHAT MAY ANCHOR, four boundaries of the company clause, each a rule. (a) A CONNECTIVE (P3) anchors nothing and ends the run: the generational `i` is also Catalan's conjunction, and `Jane Doe nee Puig i Ma` keeps its link, maiden 'Puig i Ma'. (b) A TITLE/SUFFIX DUAL OPENING THE GIVEN PART after a one-word family comma is a title there and anchors nothing, so the pass starts past the leading title run: `Smith, Ms Ma`, `Nguyen, Sr Ba` and `Smith, MD Do` keep their given names, and `Smith, MD MA Ma` keeps middle 'Ma', no unambiguous credential standing in its run once the title is set aside. Everywhere else a dual anchors like any suffix word (`John Smith MD MEng`, and (8)). (c) THE WALK'S OWN LEADING PIECE NEVER ANCHORS. In a name with no comma before the run, the leading piece is the name word H4's carve-out keeps whatever its vocabulary, and letting it anchor spends that name: `Om Ma` and `PhD Ma` keep family 'Ma' under every name order, where an anchoring `PhD` would leave given 'PhD', suffix 'Ma' and no family at all. One name word in front frees the same word — `John Om Ma` reads suffix 'Om Ma'. A family comma's given part has no such position, its first piece being read by the comma part's own credential reading, so `Smith, PhD MEng` reads family 'Smith', suffix 'PhD MEng'. (d) A SINGLE-LETTER ROMAN NUMERAL anchors nothing because it is INITIAL-SHAPED, written as a middle initial is (`John Smith PhD V Ma` keeps family 'Ma'), not because a generation is no credential: a multi-letter numeral or generational word anchors as any suffix word does (`John Smith PhD III Ma` reads suffix 'PhD III Ma', `abdul Smith Jr Ma` suffix 'Jr Ma'). - ACCEPTED COSTS, each a reading this design chooses. `John Smith, Ms Ma` → suffix 'Ms Ma', reported, as `John Smith, Ms` alone already reads suffix (8). `Om Jr Ma` → given 'Om', suffix 'Jr Ma' and no family name: 'Jr' is not the leading piece, so it anchors 'Ma', and this section's standing Accepted clause already consumes an unambiguous suffix that leaves no family (`Om Jr` → given 'Om', suffix 'Jr'); every release from 2.0.0 made that same role assignment. `doe, jane v phd do` reports `suffix-or-name` where it reported `particle-or-given`, the fork reported being the credential pick rather than P6's attachment (7). And 12 grid parses of the `Smith, MA PhD ba` shape read the whole given part as a credential run, family 'Smith' and no given name, where the parent kept given 'MA'. +- 2026-09-27 (Derek), #544 — A CREDENTIAL RUN KEEPS ITS AMBIGUOUS MEMBERS: C1'S NAME-WORD COUNT READS A RUN AS IT READS ONE WORD, AN UNAMBIGUOUS CREDENTIAL IN FRONT OF A MEMBER SPEAKS FOR IT AT EVERY TRAILING SLOT, AND A LISTED MEMBER'S CHUNKED DOTTED SPELLING PASSES THE PERIOD GATE. Three readers never looked past the member itself: C1 flipped a part to the suffix comma by the name-word count only when it was ONE token, and the trailing readers — the no-comma peel, the comma part read wholly as credentials, the given part's trailing slot — met the member first, took its name lean and stopped, so the credential in front of it was never reached. #540 marked `meng` and `lac`, whose conventional spellings lean name, which turned the one-case-row cost this section's 2026-09-15 item (ii) accepted into the ordinary spelling of two degrees. DECIDED, ten forks, (9) and (10) on 2026-09-28. (1) THE WHOLE RUN AT C1: two or more words after the comma, every one suffix vocabulary or a class candidate, at least one a candidate, behind two or more name words, is the credential run — all-member runs (`John Smith, Ed Ma`) and a leading member (`John Smith, MEng PhD`) included; offered and declined were an anchored-only rule and the issue's own "last word ambiguous" wording, which contradicted its `Ed Ma` consequence. (2) ALL TRAILING PATHS, ANCHORED IN FRONT: a listed member with an unambiguous credential in front of it, through other members, reads as the credential on the no-comma peel, in the comma part read wholly as credentials and at the given slot; a credential BEHIND it anchors nothing (`Wang Ma PhD` keeps family 'Ma'). (3) THE CHUNKED DOTTED GATE: a listed member written in two or more period-closed letter chunks (`M.Eng.`, `L.Ac.`) passes as `M.A.` does; one trailing period (`Ma.`, `Ed.`) still does not. (4) SINGLE-LETTER ROMAN NUMERALS neither anchor nor count toward a C1 run, S3's single-character retirement being the precedent; they peel as before, and the anchor's half of the exclusion is a matter of their shape, boundary (d) below. (5) NO NEW REPORT WHERE THE WRITING SETTLED IT: a run whose every member is LISTED and leans credential — S2's lean, capitals in a mixed-case name — is not flipped, the family-comma path already reading it as the credential run with the fields and the report it always had (`John Smith, PhD MA`); the alternative, flipping such a run and suppressing only the flip's report, removes reports the family-comma path makes today, `John Smith, MA Jr` losing the one it makes for 'MA'. Because the lean is the listed set's, a by-shape member in capitals never settles a run: `John Smith, X.Y.Z. MA` flips to suffix 'X.Y.Z. MA' and reports. (6) THE ANCHOR DOES NOT REACH ACROSS A MAIDEN CLAUSE: `Jane Doe Jr. nee Smith Ma` keeps maiden 'Smith Ma' where `Jane Doe Jr. Ma` reads suffix 'Jr. Ma' — rules.md#M2's Accepted boundary, and the M2 bullet of this date. (7) AN ANCHOR OVERRIDES P6 AT THE GIVEN SLOT: `doe, jane v phd do` reads suffix 'v phd do' where it read family 'do doe'; the bullet below amends #531's pairing. (8) THE DUAL EXCLUSION IS SCOPED TO THE GIVEN PART'S LEADING TITLE RUN, NOT THE C1 RUN: after a full name the dual counts as the suffix vocabulary it is, as the legacy disjunct already counted it (`John Smith, MS MA`, `John Smith, MD`), and that is what reads the issue's headline `Jane Doe, MS LAc` as suffix 'MS LAc'. (9) A MEMBER THE COMPANY DECIDES REPORTS WHERE IT WAS DECIDED, in a family comma's part read wholly as credentials as at the peel: each member whose own writing declined and which the credential in front made the credential reports `suffix-or-name` at that reading. `Smith, PhD Ma`, `Smith, PhD MEng` and `Smith, Ph. D. MEng` report the member; `Smith, Dr. PhD LAc` and `John Quincy Smith, Prof. PhD LAc` report 'LAc'; `john smith, v phd ma` reports 'ma'; `John Smith, PhD Ma Prof.`, which C1 does not flip for its trailing title, reports 'Ma' where its twin `John Smith, PhD Ma` reports C1's flip once over the part. A member its own capitals made the credential stays silent (`Smith, PhD MA`), and the part's first piece, which nothing stands in front of, is the first-slot report's word and never this one's. (10) A DUAL IN THE GIVEN PART'S LEADING TITLE RUN SILENCES THE WHOLE PART: refinement (b) below reads such a dual as a title that speaks for nothing, and in a part whose leading title run holds one no credential behind the dual speaks for a member of the ambiguous class either, so the company never reads such a part wholly as credentials: the titles give it its reading, the dual a title and the next word the given name, and behind that given name the given slot reads as it reads any (`Smith, MD PhD Jr Ma` reads suffix 'Jr Ma', as `Smith, MD Jane Jr Ma` does). A part with no member to speak for, or only members whose capitals decide them, reads as it did: `Smith, MD PhD` suffix 'MD PhD', `Smith, Ms MD MA` suffix 'Ms MD MA'. `Smith, Ms MD Ma`, `Smith, MD MS Ma` and `SMITH, MD MS BA` read title 'Ms MD' / 'MD MS', given 'Ma' / 'BA'; `Smith, MD PhD Ma` reads title 'MD', given 'PhD', middle 'Ma', the given slot reporting 'Ma'; e10e83b4 read each of them so. The run is the walk's leading title run, so a dual behind a plain title stands in it too (`Smith, Prof. MD Ma` and `smith, prof. md ma` read title 'Prof. MD', given 'Ma'; `Smith, Dr. MD PhD Ma` keeps given 'PhD'), while a plain title alone silences nothing (`Smith, Dr. PhD LAc` reads title 'Dr.', suffix 'PhD LAc'). Behind a real given name the company reads as ever (`Doe, Jane MD Ma` and `Doe, Jane Ms Ma` read suffix 'MD Ma' and 'Ms Ma'), and past the title run a dual speaks like any suffix word (`Smith, PhD Ms Ma` reads suffix 'PhD Ms Ma'); `Smith, Ms Ma`, `Nguyen, Sr Ba` and `Smith, MD Do` are unchanged. + WHAT MAY ANCHOR, four boundaries of the company clause, each a rule. (a) A CONNECTIVE (P3) anchors nothing and ends the run: the generational `i` is also Catalan's conjunction, and `Jane Doe nee Puig i Ma` keeps its link, maiden 'Puig i Ma'. (b) A TITLE/SUFFIX DUAL IN THE GIVEN PART'S LEADING TITLE RUN after a one-word family comma is a title there and anchors nothing — and, by (10), no credential behind it speaks for a member in that part — so the given slot's pass starts past the leading title run: `Smith, Ms Ma`, `Nguyen, Sr Ba` and `Smith, MD Do` keep their given names, and `Smith, MD MA Ma` keeps middle 'Ma', no unambiguous credential standing in its run once the title is set aside. Everywhere else a dual anchors like any suffix word (`John Smith MD MEng`, and (8)). (c) THE WALK'S OWN LEADING PIECE NEVER ANCHORS. In a name with no comma before the run, the leading piece is the name word H4's carve-out keeps whatever its vocabulary, and letting it anchor spends that name: in `Om Ma` and `PhD Ma` 'Ma' stays a name word under every name order (family 'Ma' in the default order, given 'Ma' under both family-first orders), where an anchoring `PhD` would leave given 'PhD', suffix 'Ma' and no family at all. One name word in front frees the same word — `John Om Ma` reads suffix 'Om Ma'. A family comma's given part has no such position, its first piece being read by the comma part's own credential reading, so `Smith, PhD MEng` reads family 'Smith', suffix 'PhD MEng'. (d) A SINGLE-LETTER ROMAN NUMERAL, IN ANY CASE (`V`, `v`, `I.`), anchors nothing for its SHAPE — one letter is how a middle initial is written, and S3 retired single-character vocabulary matches for the same reason (`John Smith PhD V Ma` keeps family 'Ma') — not because a generation is no credential. The rule is stated by that shape and the code tests exactly it (`_vocab.is_single_letter_numeral`); "initial-shaped" was the wrong word for it, `is_initial_shaped` answering False for a lower-case `v` that the exclusion covers too. And the exclusion is not about generations: a multi-letter numeral or generational word anchors as any suffix word does (`John Smith PhD III Ma` reads suffix 'PhD III Ma', `abdul Smith Jr Ma` suffix 'Jr Ma'). + ACCEPTED COSTS, each a reading this design chooses. `John Smith, Ms Ma` → suffix 'Ms Ma', reported, as `John Smith, Ms` alone already reads suffix (8). `Om Jr Ma` → given 'Om', suffix 'Jr Ma' and no family name: 'Jr' is not the leading piece, so it anchors 'Ma', and this section's standing Accepted clause already consumes an unambiguous suffix that leaves no family (`Om Jr` → given 'Om', suffix 'Jr'); every release from 2.0.0 made that same role assignment. `doe, jane v phd do` reports `suffix-or-name` where it reported `particle-or-given`, the fork reported being the credential pick rather than P6's attachment (7). And a one-word family comma whose given part OPENS with a capitalised `MA` in a mixed-case name, with an unambiguous credential behind it, reads the whole part as a credential run, family 'Smith' and no given name, where the parent kept given 'MA': `Smith, MA PhD Ma` read given 'MA', middle 'Ma', suffix 'PhD' and reads suffix 'MA PhD Ma'. 'MA' leans credential on its capitals, so it is read as the credential it is written as, and what follows it is then the company's (or, for `M.Eng.`, the chunked gate's). Measured 2026-09-28 on the grid below: 138 parses (46 texts, three orders each) go from given 'MA' to no given name, 72 of them with no `M.Eng.` in the run; 12 of the 138 are the parses the attribution below leaves unattributed, the rest falling in the anchor and `M.Eng.` classes. Each reports 'MA' at the first slot, and a member behind the credential reports as the company's pick (9). ACCEPTED LIMITS, each keeping the member a name word, rules.md#S2's Accepted block and case rows: the merged `Ph. D.` is outside the no-comma walk (`John Smith Ph. D. MEng` family 'MEng'; the comma and given-slot spellings do anchor it, `Smith, Ph. D. MEng` and `Doe, Jane Ph. D. MEng` reading suffix 'Ph. D. MEng'), a title between the credential and the member breaks the run (`John Smith PhD Prof. Ma`), and a particle member P2 has chained before the peel is out of reach (`John Smith PhD Do Do` family 'Do Do'; `Smith, PhD Do Ma` given 'PhD', middle 'Do Ma'). - THE TWO-INPUT CHECK. tests/v2/test_properties.py's M2 clause-agreement walk asks every parse it takes both directions of the company clause: a listed member behind a qualifying credential reads as one, and a member read as a credential because of the word in front has that word in the suffix too. Recorded negative controls, from that test's docstring: 48 failing parses with the anchor off, 30 with the walk's leading piece allowed to anchor, 0 on this tree. + THE TWO-INPUT CHECK. tests/v2/test_properties.py's M2 clause-agreement walk asks every parse it takes both directions of the company clause: a listed member behind a qualifying credential reads as one, and a member read as a credential because of the word in front has that word in the suffix too. Recorded negative controls, from that test's docstring: 48 failing parses with the anchor off, 30 with the walk's leading piece allowed to anchor, 0 on this tree; re-run 2026-09-28 after (9) and (10), unmoved. The walk holds no comma head that opens its part with a dual, so (10) is pinned by case rows and a unit test instead (`a_dual_opening_the_given_part_turns_the_anchor_off`, `a_dual_opening_the_given_part_silences_a_later_degree`, test_pieces' `test_a_dual_in_the_leading_title_run_turns_the_anchor_off`). REVERSED BY THIS ENTRY rather than edited: this section's 2026-09-15 ACCEPTED item (ii) (the bullet after this entry says so); the three "wrongly moved" reversals the 2026-09-14 paragraph THE RUN IS THE CAPS CLASS'S ALONE records — `John Smith, Ed Ma`, `John Smith, ma do` and `John Smith, X.Y.Z. A.B.` are credential runs now by C1's own count under every policy that keeps their words in the class (with `unlisted_dotted_suffixes` off, `X.Y.Z.` and `A.B.` are no class members and the third keeps the listing form), while that paragraph's narrowing of the CAPS run test stands; and #540's two pending questions (the `suffix-acronym-collisions` bullet of this date). - MEASURED 2026-09-27, the #544 tree against its parent e10e83b4, py3.11. Recompute: every run of one to three words from {PhD, MD, Jr, MA, Ma, MEng, M.Eng., LAc, Ed, Do, ba, MS, V, Prof.} behind each of `Wang {}`, `John Smith {}`, `John Quincy Smith {}`, `Dr. John Smith {}`, `John Smith, {}`, `John Quincy Smith, {}`, `Smith, {}`, `Doe, Jane {}`, `Jane Doe nee Smith {}`, `Doe, Jane nee Smith {}` and `Jan van der Berg {}`, each written as given, lower-cased and upper-cased and deduped, parsed under the three name orders at otherwise default policy, the seven role fields and the ambiguity kinds compared against the same grid parsed by e10e83b4. 254,496 parses over 84,832 texts; 90,365 move (30,123 texts, 19,303 of them with a field move): 39,801 the anchor freeing a member, 14,676 the C1 run, 35,696 the chunked `M.Eng.` (32,472 of those only losing the report it raised as a pick), 180 a run opened by a lone `v` (`john smith, v phd ma`, suffix 'v phd ma'), and the 12 `Smith, MA PhD ba` parses above. Clean-to-clean — no run word in a name field on either side — 23,454 move, every one `M.Eng.`'s dropped report, and none elsewhere. Two invariants over the same grid, 0 NEW violations of either: a member an unambiguous credential anchors read into a name field, 29,199 → 1,188 (the residue is a particle member P2 has chained and a dual inside a leading title run, `smith, prof. md ma`); an unambiguous credential in a name field, 36,675 → 4,410 (the residue is an interior single-letter `V`, a non-final `Prof.`, and the `PhD` a comma part keeps as its given name when a `V` or a title follows it). The differential corpora at the parent (1,418 names, three orders) move four names, every one intended: `John Smith, PhD MEng`, `john smith, phd meng`, `Wang M.Eng.`, `abdul Smith Jr Ma`. Frames per parse through `Parser().parse`, py3.11: the reference band is unmoved (370/407, `uv run python tools/perf/call_count.py`), and so is every ordinary name measured (`Smith, John Quincy` 260, `Doe, John MA` 283, `Doe, Jane nee Smith PhD` 338); the new questions cost where they are asked — `John Smith Ma` 244 → 250, the anchor pass that finds nothing; `John Smith, MA Jr` 355 → 382, the case fact forced to see the capitals settle the run; `John Smith PhD MEng` 286 → 326 — and `John Smith, PhD MEng` gets cheaper, 358 → 314. + MEASURED 2026-09-28, the #544 tree against its parent e10e83b4, py3.11, after (9) and (10); the 2026-09-27 figures this paragraph first carried were taken before them and are superseded. Recompute: every run of one to three words from {PhD, MD, Jr, MA, Ma, MEng, M.Eng., LAc, Ed, Do, ba, MS, V, Prof.} behind each of `Wang {}`, `John Smith {}`, `John Quincy Smith {}`, `Dr. John Smith {}`, `John Smith, {}`, `John Quincy Smith, {}`, `Smith, {}`, `Doe, Jane {}`, `Jane Doe nee Smith {}`, `Doe, Jane nee Smith {}` and `Jan van der Berg {}`, each written as given, lower-cased and upper-cased and deduped, parsed under the three name orders at otherwise default policy, the seven role fields and the ambiguity kinds compared against the same grid parsed by e10e83b4. THE DETECTOR, which both the invariants and the attribution read, looks only at the run's own words, compared case-free: {phd, md, jr, ms, m.eng.} are the unambiguous credentials, {ma, meng, lac, ed, do, ba} the members; a member is ANCHORED where an unambiguous credential stands in front of it in the run with nothing but members between, except that `ms` or `md` opening a comma part anchors nothing; and a run word is IN A NAME FIELD where it appears among the space-split words of given, middle, family or maiden more often than the run's unconstrained occurrences of it account for. The two invariants count parses with an anchored member in a name field, and with an unambiguous credential in one. THE ATTRIBUTION puts each moved parse in ONE class, the first that holds, in this order: THE ANCHOR, where the tree's two violation counts are each no higher than the parent's and one is lower; THE CHUNKED `M.Eng.`, where the run holds `M.Eng.`; A LONE `v`, where re-running the first test with `v` counted as an unambiguous credential makes it hold; THE C1 RUN, a comma shape with two or more name words before the comma whose run is only credentials and members, at least one a member; and anything left unattributed. 254,496 parses over 84,832 texts; 89,825 move (29,943 texts, 19,123 of them with a field move): 39,261 the anchor, 14,676 the C1 run, 35,696 the chunked `M.Eng.` (32,472 of those only losing the report it raised as a pick), 180 a lone `v` (`john smith, v phd ma`, suffix 'v phd ma'), and 12 unattributed, the `Smith, MA PhD ba` parses above. Clean-to-clean — no run word in a name field on either side — 23,454 move, every one `M.Eng.`'s dropped report, and none elsewhere. The two invariants, 0 NEW violations of either: an anchored member in a name field, 29,199 → 1,530 (the residue is a particle member P2 has chained, a dual inside a leading title run, `smith, prof. md ma`, and the member behind a credential in a part a dual opens, `Smith, MD PhD Ma`, which (10) keeps a name); an unambiguous credential in a name field, 36,675 → 4,734 (the residue is an interior single-letter `V`, a non-final `Prof.`, the `PhD` a comma part keeps as its given name when a `V` or a title follows it, and the same `PhD` behind a dual that opens the part, (10) again). Against the tree before (9) and (10), the same grid moves 5,076 parses: 4,536 (1,512 texts, every one a comma shape whose part reads wholly as credentials) gain a `suffix-or-name` report and move nothing else, and 540 (180 texts, every one `Smith, ` then `MD` or `MS` opening the run, in each case form) return to e10e83b4's reading, fields and reports alike, from a silent suffix reading of the whole part. The differential corpora at the parent (1,418 names, three orders) move four names, every one intended: `John Smith, PhD MEng`, `john smith, phd meng`, `Wang M.Eng.`, `abdul Smith Jr Ma`. Frames per parse through `Parser().parse`, py3.11: the reference band is unmoved (370/407, `uv run python tools/perf/call_count.py`), and so is every ordinary name measured (`Smith, John Quincy` 260, `Doe, John MA` 283, `Doe, Jane nee Smith PhD` 338); the new questions cost where they are asked — `John Smith Ma` 244 → 250, the anchor pass that finds nothing; `John Smith, MA Jr` 355 → 382, the case fact forced to see the capitals settle the run; `John Smith PhD MEng` 286 → 326; `Smith, MD PhD Ma` 349 → 362, the given slot's pass that (10) leaves with nothing to find — and `John Smith, PhD MEng` gets cheaper, 358 → 314, as does `Smith, PhD Ma`, 286 → 271. NO MECHANISMS ENTRY IS OWED: the anchor pass is ONE-PREDICATE-PER-QUESTION's `_pieces.credential_anchors`, which `segment_suffix_reading` reads inline for the frame budget under a keep-in-step note, and its linearity is the answer #531's trailing floor and #397's `_run_neighbours` already give — one forward pass per question, never a look-behind per member. tests/v2/test_benchmark.py's `credential_run` shape guards it: the per-member look-behind measured 15.8× for 4× the input at base 800, against 4.13–4.21× on this tree at every base (that module's own record). - 2026-09-27 (#544) — THE 2026-09-15 ACCEPTED ITEM (ii) IS REVERSED, and the bullet stands as it landed. `abdul Smith Jr Ma` reads given 'abdul', family 'Smith', suffix 'Jr Ma' — 2.3.0's reading — because the unambiguous 'Jr' in front of the Title-case 'Ma' anchors it (the entry above): the peel takes both, and P5's reserve, now seeing the family the join would take, declines the join. Item (i) stands. The case row is now `a_credential_in_front_anchors_a_declined_pick`, and rules.md#S2's Accepted block names the shapes that still keep the company out of reach. - 2026-09-27 (Derek), #544 — THE #531 PAIRING GAINS A SECOND EXCEPTION, and CAPITALS DECIDE FOR `do` above stands as it landed. That bullet let only a positive credential lean override P6's attachment at the given slot; an unambiguous credential IN FRONT of the member now overrides it as well, the degree being a second and stronger signal: `doe, jane v phd do` reads suffix 'v phd do' and reports `suffix-or-name` where it read family 'do doe' and reported P6's fork. The one-case record the pairing protects has nothing in front of its particle, so `NASCIMENTO, EDSON ARANTES DO` still reads family 'DO NASCIMENTO' and reports `particle-or-given`. diff --git a/docs/design/rules.md b/docs/design/rules.md index 8b35bf50..18673d43 100644 --- a/docs/design/rules.md +++ b/docs/design/rules.md @@ -785,18 +785,22 @@ P6. Rationale: a particle ending the name has nothing to link the word is BOTH a particle and suffix vocabulary, this attachment outranks the suffix reading (S2): a trailing abbreviation after a family comma is the tussenvoegsel far more - often than the decoration it collides with. One exception, and it - is where the capitals speak: a word of the AMBIGUOUS credential - class, written in capitals in a name written in more than one - case, reads as the credential and this attachment stands down — - unless a particle stands immediately in front of it, the two - being one particle run by then, which this rule takes whole. - Every other spelling of such a word attaches as it did before, - and the kind rule below gives it this rule's particle fork rather - than S2's credential one. In a name written wholly in one case - the two readings cannot be told apart and the particle keeps it, - which is right about a Portuguese record and wrong about a - credential; the report is how a caller finds the second. + often than the decoration it collides with. Two exceptions. The + first is where the capitals speak: a word of the AMBIGUOUS + credential class, written in capitals in a name written in more + than one case, reads as the credential and this attachment stands + down — unless a particle stands immediately in front of it, the + two being one particle run by then, which this rule takes whole. + The second is S2's company: such a word standing behind an + unambiguous credential in one run of suffix words reads as the + credential in any spelling and any case, reported as S2's + credential fork. Every other spelling of such a word attaches as + it did before, and the kind rule below gives it this rule's + particle fork rather than S2's credential one. In a name written + wholly in one case, with nothing in front of the word to speak + for it, the two readings cannot be told apart and the particle + keeps it, which is right about a Portuguese record and wrong + about a credential; the report is how a caller finds the second. "Jong, Anke de" → family="de Jong" "Beethoven, Ludwig van" → family="van Beethoven" "Berg, Jan vd" → family="vd Berg" @@ -1006,7 +1010,8 @@ S2. Rationale: generational suffixes and credentials are recognized name text, asked nothing and reporting nothing. One member of this class is particle vocabulary as well, and where it stands alone at this slot P6 decides it: the capitals take it as the - credential and every other spelling attaches to the family, + credential, as does an unambiguous credential in front of it (the + company below), and every other spelling attaches to the family, reported there as P6's fork rather than as this one. Behind another particle it does not stand alone — the two are one particle run by then — and the run attaches whatever the capitals @@ -1031,15 +1036,25 @@ S2. Rationale: generational suffixes and credentials are recognized the count, at every trailing slot this rule names: the degree in front says what the run is. Only a credential in FRONT speaks; one behind the member says nothing about it. A connective (P3) and a - single-letter roman numeral — initial-shaped, as a bare middle - initial is, unlike a multi-letter one such as 'III' — speak for - nothing and end the run; so does any name word. A word of both - the title and the suffix vocabulary opening - the given part is a title there, and speaks for nothing either, - though it speaks wherever else it stands. At the trailing slot of - the given part this company outranks P6's attachment, as the - capitals do. It does not reach across a maiden marker's clause - (M2), whose name words stand between. + single-letter roman numeral, in any case — one letter, the shape + a bare middle initial is written in, where a multi-letter one + such as 'III' speaks — speak for nothing and end the run; so does + any name word. A word of both the title and the suffix vocabulary + standing in the given part's leading title run is a title there + and speaks for nothing, and no credential behind it speaks for a + word of this class in that part: the title reading makes the next + word the given name, and from there the part reads as any given + part does, the given part's own company included. A part holding + no such word, or only words whose capitals decide them, reads as + it did. Anywhere else such a word speaks like any suffix word. + At the trailing slot of the given part this company outranks P6's + attachment, as the capitals do. It does not reach across a maiden + marker's clause (M2), whose name words stand between. A member + the company decides reports the fork as a counted pick does, + wherever it stands: at the trailing slots above, and in a part + after a family comma that the company leaves holding no name + word, which reads wholly as the credential run. A member its own + capitals already made the credential reports nothing new. An unlisted word joins this same ambiguous class by SHAPE where the caller asks for it. Two or more period-separated chunks is one such shape, admitted by default (S3); an unlisted all-caps @@ -1074,6 +1089,9 @@ S2. Rationale: generational suffixes and credentials are recognized "Wang Ma PhD" → family="Ma" · boundary "Smith, PhD MEng" → suffix="PhD MEng" "Smith, Ms Ma" → given="Ma" · boundary + "Smith, PhD Ma" → suffix="PhD Ma" + "Smith, PhD Ma" → ambiguities=("suffix-or-name",) + "Smith, MD PhD Ma" → given="PhD" · boundary "Doe, Jane PhD MEng" → suffix="PhD MEng" "doe, jane v phd do" → suffix="v phd do" "Smith, MA" → suffix="MA" @@ -1641,10 +1659,13 @@ C1. Rationale: a credential run after the comma means the name is in credential there too (S2). The same count reads a part of two or more words as the credential run when every word of it is a suffix word or a word of this class, at least one of them of this - class, and none of them a single-letter roman numeral, which is a - generation rather than a credential. A word of both the title and - the suffix vocabulary opening such a part counts as a suffix word - there, the name before the comma being complete. Where every word + class, and none of them a single-letter roman numeral, in any + case: one letter is the shape a middle initial is written in, + and S3 retires single-character matches for the same reason, + while a multi-letter numeral or generational word stands in a + run like any suffix word ('John Smith, III Ma'). A word of both + the title and the suffix vocabulary opening such a part counts as + a suffix word there, the name before the comma being complete. Where every word of this class in the part is a LISTED word written in capitals in a mixed-case name, the writing has already made each of them the credential (S2), and the part reads as the credential run on that @@ -1652,9 +1673,15 @@ C1. Rationale: a credential run after the comma means the name is in alone carries no such lean, so a part holding one is read by the count. A decision either way at this comma is reported; for a run of words the decision is the flip to the - credential run, and a run the count leaves in the listing form - reports only as S2 reads the words in it. It is one of TWO - places the comma's own decision is reported, the other being the word trailing the given part after + credential run, reported once over the whole part. A run the + count leaves in the listing form reports as S2 reads the words + in it: a word of this class read as the credential because a + credential in front speaks for it reports, and one its own + capitals made the credential does not, so a run whose every such + word is written in capitals reads whole in silence + ('John Smith, PhD MA', 'Smith, PhD MA'), as does a part read as + titles before a lone given name ('Smith, Ms MD Ma'). It is one + of TWO places the comma's own decision is reported, the other being the word trailing the given part after it (S2), which is a second decision about a second word and never the same fork twice; an attachment decided after a family comma (P6) reports on its own. C2's comma-structure flag reports what the diff --git a/docs/release_log.rst b/docs/release_log.rst index d26cfdc8..b46cf9ef 100644 --- a/docs/release_log.rst +++ b/docs/release_log.rst @@ -14,7 +14,7 @@ Release Log - **Fix a bare trailing Meng or Lac being read as a credential and losing the family name: meng and lac are now acronyms that are also ordinary names.** ``HumanName("wang meng")`` gives first ``wang``, last ``meng``, and ``parse()`` reports a suffix-or-name ambiguity, where every release from 2.0.0 through 2.3.0 gave suffix ``meng`` and no last name; 1.4.0 read last ``meng``, so this is 1.4.0's answer plus the flag. ``li meng`` and ``tran lac`` move the same way, ``Wang, Meng`` gives first ``Meng``, last ``Wang`` again, and ``Parser(policy=Policy(name_order=FAMILY_FIRST)).parse("Wang Meng")`` gives given ``Meng`` where 2.0.0 through 2.3.0 gave family ``Wang``, suffix ``Meng`` and no given name. With a full name in front the credential reading stays: ``john smith meng`` and ``nguyen van lac`` keep suffix ``meng`` and ``lac``, now flagged. But a Title-case ``Nguyen Van Lac`` gives last ``Van Lac`` where every release gave last ``Van``, suffix ``Lac``. The cost is the marking's own, and it falls on the conventional spellings: a ``MEng`` or ``LAc`` written that way, in a name written in more than one case, is read the way ``John Smith Ma`` is (above), so ``John Smith MEng`` gives middle ``Smith``, last ``MEng``, where every release gave suffix ``MEng``. Where one of them LEADS a credential run nothing speaks for it: ``John Smith MEng PhD`` gives middle ``Smith``, last ``MEng``, suffix ``PhD``, where every release read suffix ``MEng PhD`` (``MEng, PhD`` through 2.2); behind a credential, or written with its periods, it is read with the run (the next entry). After a comma ``Smith, MEng`` and ``Smith, meng`` give first ``MEng`` and ``meng`` (1.4.0's reading, not the suffix 2.0 through 2.3 gave) and ``Smith, John MEng`` gives middle ``MEng`` (every release gave suffix ``MEng``), and a bracketed ``John Smith (MEng)`` falls through to nickname, where every release gave suffix ``MEng``. A lone credential after a comma behind a full name (``John Smith, MEng``) keeps the credential reading, and so does a run of them (the next entry). Meng is a common Chinese surname and given name, Lac a Vietnamese given name (``Nguyen Van Lac``) and a French surname; see the ``suffix-acronym-collisions`` entry of ``docs/design/decisions.md`` (closes #540) - - **Fix a credential run losing the acronyms in it that are also names: a degree in front speaks for the acronym behind it, and a run after a comma is read whole.** ``HumanName("John Smith, Ed Ma")`` gives first ``John``, last ``Smith``, suffix ``Ed Ma``, where 2.0 through 2.3 gave first ``Ed``, middle ``Ma``, last ``John Smith`` -- 1.4.0's reading, restored: with two or more name words before the comma, a part made only of suffix words and acronyms that are also names, and holding no one-letter roman numeral, is a credential run however it is written, as a lone one already was (``John Smith, MA``), and ``parse()`` reports the call. ``john smith, md ma`` and ``John Smith, Ms Ma`` move the same way, where 2.0 through 2.3 gave title ``md``/``Ms`` -- the second is the accepted cost, ``Ms`` read as the suffix word it also is, as ``John Smith, Ms`` alone already reads it -- while one name word before the comma keeps the listing form (``Smith, Ms Ma`` gives title ``Ms``, first ``Ma``). At the end of a name, an acronym standing behind an unambiguous credential is read as that credential's company whatever its case: ``John Smith PhD MEng`` and ``Doe, Jane PhD MEng`` give suffix ``PhD MEng``, the fields every release gave (``PhD, MEng`` through 2.2), now reported, and ``doe, jane v phd do`` gives suffix ``v phd do`` where 2.3.0 gave last ``do doe`` -- a degree in front outranks the particle reading, as capitals already did. Only a credential IN FRONT speaks: ``Wang Ma PhD`` keeps last ``Ma``. A listed acronym written in period-closed chunks is written with its periods, so ``Wang M.Eng.`` gives suffix ``M.Eng.``, as ``Wang M.A.`` does and as 2.0 through 2.3 did. See the #544 entry under ``S2`` in ``docs/design/decisions.md`` (closes #544) + - **Fix a credential run losing the acronyms in it that are also names: a degree in front speaks for the acronym behind it, and a run after a comma is read whole.** ``HumanName("John Smith, Ed Ma")`` gives first ``John``, last ``Smith``, suffix ``Ed Ma``, where 2.0 through 2.3 gave first ``Ed``, middle ``Ma``, last ``John Smith`` -- 1.4.0's reading, restored: with two or more name words before the comma, a part made only of suffix words and acronyms that are also names, and holding no one-letter roman numeral, is a credential run however it is written, as a lone one already was (``John Smith, MA``), and ``parse()`` reports the call wherever the writing left it open: a run whose every such acronym is written in capitals in a mixed-case name is the credential run without a report, the capitals having decided it (``John Smith, PhD MA``, ``John Smith, MS MA``). A one-letter numeral keeps a part out of that rule, but not out of the next one: a degree behind the letter still speaks for the acronyms after it and the part is read whole, so ``john smith, v phd ma`` gives first ``john``, last ``smith``, suffix ``v phd ma``, where 2.3 gave first ``v``, middle ``ma``, last ``john smith``. ``john smith, md ma`` and ``John Smith, Ms Ma`` move the same way, where 2.0 through 2.3 gave title ``md``/``Ms`` -- the second is the accepted cost, ``Ms`` read as the suffix word it also is, as ``John Smith, Ms`` alone already reads it -- while one name word before the comma keeps the listing form (``Smith, Ms Ma`` gives title ``Ms``, first ``Ma``). At the end of a name, an acronym standing behind an unambiguous credential is read as that credential's company whatever its case: ``John Smith PhD MEng`` and ``Doe, Jane PhD MEng`` give suffix ``PhD MEng``, the fields every release gave (``PhD, MEng`` through 2.2), now reported, and ``doe, jane v phd do`` gives suffix ``v phd do`` where 2.3.0 gave last ``do doe`` -- a degree in front outranks the particle reading, as capitals already did. Only a credential IN FRONT speaks: ``Wang Ma PhD`` keeps last ``Ma``. After a one-word family comma the part it speaks for reads wholly as credentials and the acronym it decided is reported: ``Smith, PhD Ma`` gives last ``Smith``, suffix ``PhD Ma``, where 2.3 gave first ``PhD``, middle ``Ma``. A title that is also a credential (``MD``, ``Ms``) opening that part stays a title and nothing in the part speaks, so ``Smith, MD PhD Ma`` keeps title ``MD``, first ``PhD``, middle ``Ma`` and ``Smith, Ms MD Ma`` title ``Ms MD``, first ``Ma``, as 2.3 read them. A listed acronym written in period-closed chunks is written with its periods, so ``Wang M.Eng.`` gives suffix ``M.Eng.``, as ``Wang M.A.`` does and as 2.0 through 2.3 did. See the #544 entry under ``S2`` in ``docs/design/decisions.md`` (closes #544) - **New Policy field unlisted_dotted_suffixes, on by default: a dotted acronym nobody has listed is read by position.** ``HumanName("John Smith X.Y.Z.")`` gives suffix ``X.Y.Z.`` where every release gave last ``X.Y.Z.``, while ``Jack X.Y.Z.`` keeps its surname, the same words-to-spare rule a listed acronym takes -- and both readings are reported. Case is irrelevant here: the periods are the signal, so ``john smith x.y.z.`` reads the same way. Words the vocabulary does know are untouched (``M.A.``, ``Ph.D.``, ``A.B.C.``), a single trailing period is still not this shape (``John Smith Xyz.`` keeps last ``Xyz.``), and a dotted run at the FRONT of a name is untouched (``J.R.R. Tolkien``). One accident retires with it: a dotted word whose only vocabulary matches were SINGLE ASCII CHARACTERS -- the roman numerals the suffix list holds, and the lone digit ``2`` -- was reading as a generational suffix, so ``Jack X.Y.I.`` gives last ``X.Y.I.`` again, as 1.4.0 read it, while ``Msc.Ed.``, ``JD.CPA`` and ``Lt.Gov.`` are unchanged. The digit is why a dotted VERSION STRING moves with them and moves SILENTLY: ``John Smith 1.4.2`` gives last ``1.4.2`` where 2.3 gave suffix ``1.4.2``, and ``John Smith, 1.4.2`` gives first ``1.4.2``, last ``John Smith``. Such a token reports nothing at any policy -- it is no acronym either, the shape reading wanting every chunk alphabetic -- and a version string read as a credential was the same accident this retirement removes. That retirement is NOT behind this switch and stands either way -- setting it to ``False`` reads an unlisted dotted word as name material by position instead (``John Smith X.Y.Z.`` keeps last ``X.Y.Z.``), the pre-2.4 reading for THAT half alone. See the ``S2`` and ``suffix-acronym-collisions`` entries of ``docs/design/decisions.md`` (closes #516) @@ -22,7 +22,7 @@ Release Log - **The comma's own decision about an ambiguous credential is now reported.** ``parse("Smith, MA").ambiguities`` names ``suffix-or-name``, and so does every other decision at the ambiguous credential class -- before or after a comma, in either direction, with no new ``AmbiguityKind`` (the family-comma attachment fork already reported this way, e.g. ``parse("Berg, Jan vd")``). One report per decision: ``Smith, Ma`` reports that the word was kept as the given name just as ``Smith, MA`` reports that it was taken as a credential. The reading a SURNAME PARTICLE swallows is reported too, which no release before this one did: ``John van der Berg Ma`` gives last ``van der Berg Ma`` and names ``suffix-or-name``, where the chain took a word the credential reading had considered. ONE report goes away, because a comma segment the parser reads as a credential run is no longer called unrecognized: ``Steven Hardman, MD, DO, DDS`` no longer reports ``comma-structure``, on its written case. That is the whole of the losses over the differential corpora -- ``John Smith, MD, R.A.I.`` is quieted on its shape by the same change, but it never reported at 2.3.0 either, having only carried the flag inside this release's own development. The other movement an upgrader sees is a SWAP rather than a loss: ``Jack X.Y.I.`` reported ``given-or-family`` at 2.3.0 and reports ``suffix-or-name`` here, the dotted retirement above having handed it to the ambiguous class. Everything else at this class is a GAIN, which is what the rest of this bullet describes. Two slots this bullet left silent no longer are, and the two bullets below close them: a credential trailing the GIVEN part of a family-comma listing now reads as a credential and reports either way, and so does one ending a maiden marker's clause. See the ``S2`` and ``C1`` entries of ``docs/design/decisions.md`` - - **Fix a credential ending the given part of a family-comma listing being read as a middle name in silence.** ``HumanName("Doe, John MA")`` gives first ``John``, last ``Doe``, suffix ``MA``, where 2.0 through 2.3 gave middle ``MA`` -- and 1.4.0 gave the suffix, so this restores v1's reading for that half. The comma has already named the family and the first word after it is the given name, so the words-to-spare count that governs the comma-less form is satisfied by construction and the writing decides alone: ``Doe, John Ma`` keeps middle ``Ma``, written the way a name is written, and ``Doe, John Ed`` keeps middle ``Ed``. Either reading is now reported, and the report belongs to the SPELLING rather than to the fields -- a declined name re-rendered without its comma, ``John Ma Doe``, re-parses to those same three fields and reports nothing, the word no longer standing where the question is asked. A name word behind the credential still ends its reach and stays silent -- ``Doe, John MA Smith`` gives middle ``MA Smith`` and reports nothing -- while a credential run or a trailing title is transparent to it: ``Doe, John MA PhD`` gives suffix ``MA PhD`` and ``Doe, John MA Prof.`` gives title ``Prof.`` with suffix ``MA``. Two second-order movements an upgrader may see, both consequences of the word leaving the given part rather than of this rule reaching further: ``Doe, John Prof. MA`` now gives title ``Prof.`` where it gave middle ``Prof. MA``, the trailing-title chain reaching a word the credential used to hide; and ``Doe, John van MA`` gives last ``van Doe`` with suffix ``MA`` where it gave middle ``van MA``, the surname-particle rule reaching a particle the same way. A name written wholly in one case says nothing either way and takes the credential, which is what 1.4.0 read: ``DOE, MARY JO MA``, ``doe, john ma``, ``田中, 太郎 MA`` and ``김, 민준 MA`` all give a suffix. The unlisted dotted spelling moves with them without the parity claim -- ``Doe, John X.Y.Z.`` gives suffix ``X.Y.Z.`` where 1.4.0 and 2.3.0 both gave a middle name -- to match the comma-less ``John Doe X.Y.Z.``. One word is carved out: ``do`` is the only member of this class that is also a surname particle, so capitals decide it and the particle reading keeps every other spelling. ``Doe, John DO`` gives suffix ``DO``, while ``Doe, John do``, ``Doe, John Do``, ``DOE, JOHN DO`` and ``doe, john do`` are unchanged and keep the particle-or-given report they already had. In a name written wholly in one case the two cannot be told apart, so ``SMITH, JOHN DO`` keeps last ``DO SMITH`` as ``NASCIMENTO, EDSON ARANTES DO`` does -- right about the Portuguese record, wrong about the osteopath, and the report is how a caller finds the second. See the ``S2`` and ``P6`` entries of ``docs/design/decisions.md`` (closes #531) + - **Fix a credential ending the given part of a family-comma listing being read as a middle name in silence.** ``HumanName("Doe, John MA")`` gives first ``John``, last ``Doe``, suffix ``MA``, where 2.0 through 2.3 gave middle ``MA`` -- and 1.4.0 gave the suffix, so this restores v1's reading for that half. The comma has already named the family and the first word after it is the given name, so the words-to-spare count that governs the comma-less form is satisfied by construction and the writing decides alone: ``Doe, John Ma`` keeps middle ``Ma``, written the way a name is written, and ``Doe, John Ed`` keeps middle ``Ed``. Either reading is now reported, and the report belongs to the SPELLING rather than to the fields -- a declined name re-rendered without its comma, ``John Ma Doe``, re-parses to those same three fields and reports nothing, the word no longer standing where the question is asked. A name word behind the credential still ends its reach and stays silent -- ``Doe, John MA Smith`` gives middle ``MA Smith`` and reports nothing -- while a credential run or a trailing title is transparent to it: ``Doe, John MA PhD`` gives suffix ``MA PhD`` and ``Doe, John MA Prof.`` gives title ``Prof.`` with suffix ``MA``. Two second-order movements an upgrader may see, both consequences of the word leaving the given part rather than of this rule reaching further: ``Doe, John Prof. MA`` now gives title ``Prof.`` where it gave middle ``Prof. MA``, the trailing-title chain reaching a word the credential used to hide; and ``Doe, John van MA`` gives last ``van Doe`` with suffix ``MA`` where it gave middle ``van MA``, the surname-particle rule reaching a particle the same way. A name written wholly in one case says nothing either way and takes the credential, which is what 1.4.0 read: ``DOE, MARY JO MA``, ``doe, john ma``, ``田中, 太郎 MA`` and ``김, 민준 MA`` all give a suffix. The unlisted dotted spelling moves with them without the parity claim -- ``Doe, John X.Y.Z.`` gives suffix ``X.Y.Z.`` where 1.4.0 and 2.3.0 both gave a middle name -- to match the comma-less ``John Doe X.Y.Z.``. One word is carved out: ``do`` is the only member of this class that is also a surname particle, so capitals decide it and, with nothing in front of it, the particle reading keeps every other spelling (a degree in front is the other exception, the next-but-one entry). ``Doe, John DO`` gives suffix ``DO``, while ``Doe, John do``, ``Doe, John Do``, ``DOE, JOHN DO`` and ``doe, john do`` are unchanged and keep the particle-or-given report they already had. In a name written wholly in one case the two cannot be told apart, so ``SMITH, JOHN DO`` keeps last ``DO SMITH`` as ``NASCIMENTO, EDSON ARANTES DO`` does -- right about the Portuguese record, wrong about the osteopath, and the report is how a caller finds the second. See the ``S2`` and ``P6`` entries of ``docs/design/decisions.md`` (closes #531) - **Fix a maiden marker's clause swallowing a trailing credential in silence.** ``HumanName("Jane Doe nee Smith MA")`` gives maiden ``Smith`` with suffix ``MA``, where 2.0 through 2.3 gave maiden ``Smith MA`` and said nothing; 1.4.0 read the ``MA`` as a suffix too. ``Doe, Jane nee Smith MA`` moves with it, and so do the one-case spellings ``JANE DOE NEE SMITH MA`` and ``jane doe nee smith ma``. The words a marker takes now end where a trailing credential begins, which is what the marker's other two stops -- a suffix word, a trailing roman numeral -- have always done. Until this release it was the last trailing position in the library where a word of the ambiguous credential class was read without a report, and it was order-sensitive besides: ``Jane Doe nee Smith MA PhD`` gave maiden ``Smith MA`` while ``Jane Doe nee Smith PhD MA`` gave maiden ``Smith``, so whether the word was read at all depended on which side of the unambiguous credential the writer put it. Both now give maiden ``Smith``, with suffix ``MA PhD`` and ``PhD MA``. The writing still decides, exactly as it does for the same word ending a name with no clause: ``Jane Doe nee Smith Ma`` keeps maiden ``Smith Ma``, and ``Jane Doe nee Yo-Yo Ma`` keeps a two-word birth surname whole. The one member of this class that is also a surname particle keeps the carve-out it has outside a clause -- ``Doe, Jane nee Smith DO`` gives suffix ``DO`` while ``Doe, Jane nee Smith do`` and ``Doe, Jane nee Smith Do`` keep maiden ``Smith do`` and ``Smith Do``, and the comma-less ``Jane Doe nee Smith do`` gives suffix ``do`` as ``John Doe do`` does. Either reading is now reported, and there is no third: a word the clause gives up reads as a post-nominal, or the clause keeps it and says so. A name word behind the credential ends its reach and stays silent -- ``Jane Doe nee MA Smith`` gives maiden ``MA Smith`` and reports nothing -- and this stop never takes the first word after the marker, whatever its writing says: ``Jane Doe nee MA`` keeps maiden ``MA`` and reports, the marker having announced a name where there would otherwise be none, and ``Jane Doe nee MA PhD`` keeps it too. That differs on purpose from what a certain post-nominal gets there, ``Jane Smith nee PhD`` and ``Jane Smith nee V`` leaving the marker standing as an ordinary word as before. Where no trailing rule reads the clause's tail nothing is decided and the clause keeps every word: ``Smith nee Jones MA, Jane`` and ``Smith, John, Jr nee Jones MA`` both keep maiden ``Jones MA``, unchanged and with no ``suffix-or-name`` report. A title written behind or in front of the credential usually does not change how the credential is read; where something in or ahead of the clause could take the title it can -- see the trailing-title bullet below. The clause also keeps a word it cannot promise a credential reading for, which is where three shapes that look like they should move do not. Where the part the word would land in holds no name of its own there is nothing to read it as a credential, so ``Doe, Dr. nee Smith MA`` and ``Jane Doe, Jr nee Smith MA`` both keep maiden ``Smith MA`` and report. Where a join would swallow it first the same applies, and it is the birth name that would lose the word: ``Berg, abdul nee Jones MA`` keeps maiden ``Jones MA`` rather than reading first ``abdul MA``, and ``Berg, Jane van der nee Smith DO`` keeps maiden ``Smith DO`` rather than letting the particle chain carry the ``DO`` into last ``van der DO Berg``. Each of those reads as 2.3.0 read it. Delimiters settle the question outright and always did: ``HumanName("Jane Doe (nee Smith MA)")`` keeps the whole span as the maiden name and reports nothing, the writer having drawn the boundary, while ``Jane Doe (nee Smith) MA`` gives suffix ``MA`` for the word left outside it. One name is a restoration rather than a change: ``John Smith nee Jones R.A.I.`` gives suffix ``R.A.I.`` again, as 2.3.0 read it, this unreleased cycle having moved it into the maiden name when the unlisted-dotted reading above took the word out of the certain-suffix class. See the ``M2`` and ``S2`` entries of ``docs/design/decisions.md`` (closes #533) diff --git a/nameparser/_pipeline/_assign.py b/nameparser/_pipeline/_assign.py index d6f99538..a2d172c6 100644 --- a/nameparser/_pipeline/_assign.py +++ b/nameparser/_pipeline/_assign.py @@ -45,13 +45,15 @@ particles_ambiguous token with more pieces following ("Van Johnson", and since #367 "Dr. Van Johnson" too, a title no longer displacing the particle out of that position) -- whatever role name_order assigns. -Emits SUFFIX_OR_NAME at FIVE sites: the trailing roman numeral, each +Emits SUFFIX_OR_NAME at SIX sites: the trailing roman numeral, each ambiguous acronym the trailing peel had to resolve, the bare-suffix carve-out where an input that is nothing but post-nominal vocabulary gets its first word made into the name (H4's suffix half, #491), -- since #289 -- the FAMILY-COMMA path's own read of the first -post-comma piece, and -- since #531 -- the class member ENDING that -path's given part, which the first-piece emitter could never reach. +post-comma piece, -- since #531 -- the class member ENDING that +path's given part, which the first-piece emitter could never reach, +and -- since #544 -- each member a credential in front of it made the +credential in a post-comma part read wholly as credentials. Further emitters of the same kind live in `_segment.py`, `_group.py` and `_post_rules.py`; they are not assign's and are not counted here. And @@ -539,9 +541,11 @@ def assign(state: ParseState) -> ParseState: # positional read peels a trailing suffix first: 'Smith Jr., # Mr.' has two pieces and one name, and read positionally lost # its family (the code review). + anchored_picks: list[int] = [] reading = segment_suffix_reading( state.pieces[1], state.piece_tags[1], tokens, - state.policy.lenient_comma_suffixes, state.one_case) + state.policy.lenient_comma_suffixes, state.one_case, + anchored_picks) # rules.md#C1's exception, scoped to the ambiguous credential # class: this is the first report of the comma's OWN decision # (listing or credential run), where the writing left the @@ -901,6 +905,21 @@ def reads_as_a_suffix(m: int, titled: tuple[int, ...]) -> bool: _set_roles(tokens, piece, Role.SUFFIX if reading[k] else Role.TITLE) n = len(pieces) + # rules.md#S2's company, reported where it decided: a + # member the anchor read as a credential after its own + # writing declined is a pick, as the peel's are (#544). + # Never piece 0, which nothing stands in front of, so + # never the first-piece report's word above; a member + # whose capitals lean credential is not among them + # ('Smith, PhD MA' stays silent). + for k in anchored_picks: + i2 = pieces[k][0] + ambiguities.append(PendingAmbiguity( + AmbiguityKind.SUFFIX_OR_NAME, + f"{tokens[i2].text!r} behind a credential after " + f"the comma is also an ordinary name word; read " + f"as a credential", + (i2,))) else: n = _peel_leading_titles(pieces, ptags, tokens) # rules.md#H5: "the title is TRANSPARENT to the suffix diff --git a/nameparser/_pipeline/_pieces.py b/nameparser/_pipeline/_pieces.py index a12b4d1e..347c14e8 100644 --- a/nameparser/_pipeline/_pieces.py +++ b/nameparser/_pipeline/_pieces.py @@ -292,16 +292,18 @@ def _anchors(piece: Sequence[int], tokens: Sequence[WorkToken]) -> bool: A connective anchors nothing because between two name words it is a link -- the generational 'i' is also Catalan's 'i' (rules.md#P3) -- and 'Jane Doe nee Puig i Ma' must keep its clause. A - single-letter numeral anchors nothing because it is - INITIAL-SHAPED, the same shape a middle initial writes in ('V' can - be one) -- not because a numeral itself is no credential, since a - MULTI-letter one ('Jr', 'III') anchors like any other suffix piece - (`is_single_letter_numeral`). A title/suffix - DUAL ('ms', 'md', 'sr') does anchor, except at the head of the - given part, where it stands in title position ('Smith, Ms Ma' is - Ms. Ma Smith); that exclusion is the callers', which start their - walks past the leading title run or test the head themselves - (`segment_suffix_reading`). + single-letter roman numeral, in any case ('V', 'v', 'I.'), anchors + nothing for its SHAPE: one letter is the shape a middle initial + is written in ('V' can be one), and S3 retired single-character + vocabulary matches for the same reason. Not because a numeral is + no credential -- a MULTI-letter one ('Jr', 'III') anchors like + any other suffix piece (`is_single_letter_numeral`). A title/suffix + DUAL ('ms', 'md', 'sr') does anchor, except in the given part's + leading title run, where it stands in title position ('Smith, Ms + Ma' is Ms. Ma Smith); that exclusion is the callers', which start + their walks past the leading title run or, in + `segment_suffix_reading`, read no anchor at all in a part whose + leading title run holds one. Asked only of a piece `is_suffix_piece` accepted, and only once a member's own writing has declined, so an ordinary name never pays @@ -363,6 +365,7 @@ def segment_suffix_reading(pieces: Sequence[Sequence[int]], tokens: Sequence[WorkToken], lenient: bool, one_case: bool | None, + anchored: list[int] | None = None, ) -> tuple[bool, ...] | None: """How each piece of a no-name segment reads: True a suffix, False a title. None when the segment holds a name word and so is not a @@ -379,8 +382,27 @@ class by SHAPE takes the count instead, which is decided at the A listed member ANCHORED by an unambiguous credential in front of it in the same run reads as a credential too, whatever its writing (#544, `credential_anchors`): 'Smith, PhD MEng' is family 'Smith' - with two degrees. A dual opening the part is a title there and - anchors nothing ('Smith, Ms Ma' keeps its given name). + with two degrees. A title/suffix dual standing in the part's + LEADING TITLE RUN -- every piece ahead of it a title or another + such dual -- is a title there, and in a part that holds one this + reading anchors nothing at all, not by the dual and not by a + later credential: the dual makes the next name-position word the + given name, and reading the part wholly as credentials would + stack the anchor's guess on the dual's ('Smith, Ms Ma', 'Smith, + Ms MD Ma' and 'Smith, MD PhD Ma' keep a given name, 'Smith, + Prof. MD Ma' too). The walk then reads the part, and its given + slot's company starts past that given name ('Smith, MD PhD Jr + Ma' reads suffix 'Jr Ma'). A plain title ahead of the credential + does not stop this reading ('Smith, Dr. PhD LAc' reads title + 'Dr.', suffix 'PhD LAc'). + + `anchored`, when the caller passes a list, receives the index of + every piece the anchor read as a credential after its own writing + declined -- the picks the caller reports, since this decides them + (mechanisms.md#AMBIGUITY-AT-THE-DECISION-SITE). A member whose own + capitals lean credential is not among them. Meaningful only when + the answer is not None: a segment the walk abandons part-way may + have appended to it first. ONE answer for two readers, both in _assign.py -- the no-name gate and the router -- because they must agree piece for piece. #429 @@ -429,6 +451,11 @@ class by SHAPE takes the count instead, which is decided at the # members and cleared by anything else; asked `_anchors` only when # a member's writing has declined. Keep the two in step. anchor: Sequence[int] | None = None + # Whether every piece so far stands in the part's leading title + # run (titles, and title/suffix duals), and whether a dual has + # stood there: once one has, nothing in the part anchors. + leading = True + dual_led = False for piece, tags in zip(pieces, ptags): # the verdict just recorded IS "stands behind a suffix" -- keeping # a separate flag meant maintaining that equality by hand at three @@ -436,27 +463,32 @@ class by SHAPE takes the count instead, which is decided at the # have diverged silently after_suffix = bool(out) and out[-1] if is_suffix_piece(piece, tags, tokens): - # a title/suffix dual with nothing read as a suffix ahead of - # it stands in the part's title position and anchors nothing - # ('Smith, Ms Ma'); `any` is a builtin, not a frame - anchor = (None if (len(piece) == 1 - and "vocab:title" in tokens[piece[0]].tags - and not any(out)) - else piece) + if (leading and len(piece) == 1 + and "vocab:title" in tokens[piece[0]].tags): + dual_led = True + else: + leading = False + anchor = piece out.append(True) continue member = (len(piece) == 1 and AMBIGUOUS_ACRONYM_TAG in tokens[piece[0]].tags) if not member: anchor = None - if member and (listed_lean(tokens[piece[0]], one_case) - == "credential" - or (anchor is not None - and SHAPE_ACRONYM_TAG not in tokens[piece[0]].tags - and _anchors(anchor, tokens))): + if member and listed_lean(tokens[piece[0]], one_case) \ + == "credential": + leading = False + out.append(True) + elif (member and anchor is not None and not dual_led + and SHAPE_ACRONYM_TAG not in tokens[piece[0]].tags + and _anchors(anchor, tokens)): + leading = False + if anchored is not None: + anchored.append(len(out)) out.append(True) elif (lenient and after_suffix and _numeral_behind_the_initial_veto(piece, tokens)): + leading = False out.append(True) elif is_leading_title(piece, tags, tokens): out.append(False) diff --git a/nameparser/_pipeline/_segment.py b/nameparser/_pipeline/_segment.py index 4debccfc..7e2012c5 100644 --- a/nameparser/_pipeline/_segment.py +++ b/nameparser/_pipeline/_segment.py @@ -240,13 +240,17 @@ def class_run(seg: tuple[int, ...]) -> bool: # word of this class, at least one of them of this class" -- the # single-token rule above generalized to RUNS (#544): 'John Smith, # PhD MEng' is the credential run the single-token 'John Smith, - # MEng' already is. A single-letter roman numeral voids the run (a - # generation is not a credential, `is_single_letter_numeral`), - # while a title/suffix DUAL opening the part counts as the suffix - # vocabulary it is: with a full name before the comma the name is - # complete, and the legacy disjunct below already counts a dual - # that way ('John Smith, MS MA'). The given part's head is where a - # dual reads as a title ('Smith, Ms Ma'), and that exclusion is + # MEng' already is. A single-letter roman numeral, in any case, + # voids the run for its SHAPE -- one letter is how a middle initial + # is written, and S3 retired single-character vocabulary matches + # for the same reason (`is_single_letter_numeral`) -- not for + # being a generation: 'John Smith, III Ma' and 'John Smith, Jr Ma' + # are runs. A title/suffix DUAL opening the part counts as the + # suffix vocabulary it is: with a full name before the comma the + # name is complete, and the legacy disjunct below already counts a + # dual that way ('John Smith, MS MA'). The given part's leading + # title run is where a dual reads as a title ('Smith, Ms Ma', + # 'Smith, MD PhD Ma'), and that exclusion is # `_pieces.segment_suffix_reading`'s, one name word before the # comma never reaching the flip. # diff --git a/nameparser/_pipeline/_vocab.py b/nameparser/_pipeline/_vocab.py index 692b5dfb..f9b23176 100644 --- a/nameparser/_pipeline/_vocab.py +++ b/nameparser/_pipeline/_vocab.py @@ -207,12 +207,14 @@ def is_trailing_numeral_suffix(text: str, preceding: str) -> bool: and not is_initial_shaped(preceding)) -# #544: a single-letter roman numeral ('V', 'v', 'I.') is about a -# GENERATION, not a credential -- rules.md#S3 retires the same class -# from the chunk rule for the same reason -- so it neither anchors an -# ambiguous member behind it nor counts toward C1's multi-word -# credential run. It still peels exactly as before; this answers only -# those two questions. +# #544: a single-letter roman numeral ('V', 'v', 'I.'), in any case, +# is excluded for its SHAPE -- one letter is how a middle initial is +# written, and rules.md#S3 retires single-character matches for the +# same reason -- not for being a generation: a multi-letter numeral +# ('III', 'Jr') anchors and runs like any suffix word. So it neither +# anchors an ambiguous member behind it nor counts toward C1's +# multi-word credential run. It still peels exactly as before; this +# answers only those two questions. def is_single_letter_numeral(text: str) -> bool: """One letter, optionally followed by periods, that is a roman numeral ('V', 'v', 'I.', 'X').""" diff --git a/nameparser/_types.py b/nameparser/_types.py index 4e47feb9..66c92c5e 100644 --- a/nameparser/_types.py +++ b/nameparser/_types.py @@ -464,10 +464,13 @@ class AmbiguityKind(StrEnum): #: whole RUN after the comma ("John Smith, Ed Ma" reads suffix #: ``Ed Ma``), and a member an unambiguous credential in front of #: it speaks for (rules.md#S2) is a pick like a counted one, - #: reported by the peel or the given part's trailing slot that took - #: it ("John Smith PhD MEng", "Doe, Jane PhD MEng"). The - #: first-piece emitter still asks its own piece only, so "Smith, - #: PhD MEng" reads suffix ``PhD MEng`` and reports nothing. + #: reported where it was decided: by the peel, by the given part's + #: trailing slot, or by the reading of a part after a family comma + #: that the company leaves holding no name word ("John Smith PhD + #: MEng", "Doe, Jane PhD MEng" and "Smith, PhD MEng" each report + #: ``MEng``). A member its own capitals already made the + #: credential reports nothing new, so "Smith, PhD MA" and "John + #: Smith, PhD MA" read suffix ``PhD MA`` in silence. #: Since 2.4 a maiden marker's clause reports at ITS trailing slot #: too, in both directions: "Doe, Jane nee Smith MA" gives maiden #: ``Smith`` with suffix ``MA`` and says so, "Doe, Jane nee Smith diff --git a/tests/v2/cases.py b/tests/v2/cases.py index 1056ff1a..4cd29a3d 100644 --- a/tests/v2/cases.py +++ b/tests/v2/cases.py @@ -950,13 +950,54 @@ def _check_cjk_shape_purity(self) -> None: "Smith, PhD MEng", {"family": "Smith", "suffix": "PhD MEng"}, classification="fix(#544)", + ambiguities=("suffix-or-name",), notes="one name word before the comma, so C1 reads the " "listing form, and the part holds no name word once " - "'PhD' anchors 'MEng': the credential run, whole, with " - "no report, 'PhD' not being a member. 2.3.0 read the " - "same; #540 had read given 'PhD', middle 'MEng', and " - "1.4.0 title 'PhD', first 'MEng'", + "'PhD' anchors 'MEng': the credential run, whole, and " + "'MEng' reports as the pick the anchor made, its own " + "writing having declined. 2.3.0 read the same fields " + "without the report; #540 had read given 'PhD', middle " + "'MEng', reporting 'MEng', and 1.4.0 title 'PhD', first " + "'MEng'", + shape=2), + Case("an_anchored_pick_after_a_one_word_family_reports", + "Smith, PhD Ma", + {"family": "Smith", "suffix": "PhD Ma"}, + classification="fix(#544)", + ambiguities=("suffix-or-name",), + notes="the part is read wholly as credentials because 'PhD' " + "speaks for the Title-case 'Ma' (S2's company), and the " + "pick is reported where the anchor made it. 2.2.0 " + "and 2.3.0 read given 'PhD', middle 'Ma'; 1.4.0 through " + "2.1.0 title 'PhD', first 'Ma'. 'Smith, PhD MA' reads the " + "same fields in silence, the capitals having decided " + "the member", shape=2), + Case("an_anchored_pick_before_a_trailing_title_reports", + "John Smith, PhD Ma Prof.", + {"title": "Prof.", "given": "John", "family": "Smith", + "suffix": "PhD Ma"}, + classification="fix(#544)", + ambiguities=("suffix-or-name",), + notes="the trailing title keeps C1's run test from flipping " + "the comma, so the family-comma path reads the part " + "wholly as credentials plus the title (H5), 'John " + "Smith' keeping its positional read, and the member " + "the anchor decided reports on that path: one report " + "for 'Ma', where the twin 'John Smith, PhD Ma' reports " + "C1's flip once over the whole part. 2.3.0 read title " + "'Prof.', given 'PhD', middle 'Ma', family 'John " + "Smith', and 2.2.0 middle 'Ma Prof.'; 1.4.0 through " + "2.1.0 title 'PhD', first 'Ma', middle 'Prof.'"), + Case("an_anchored_pick_twin_without_the_title_reports_the_flip", + "John Smith, PhD Ma", + {"given": "John", "family": "Smith", "suffix": "PhD Ma"}, + ambiguities=("suffix-or-name",), + notes="C1's run rule, reporting once over the whole part; the " + "pair to the row above. 1.4.0 read the same; 2.2.0 " + "and 2.3.0 read given 'PhD', middle 'Ma', family 'John " + "Smith', and 2.0.0 and 2.1.0 title 'PhD', given 'Ma'", + shape=3), Case("a_degree_in_the_given_part_anchors_the_member", "Doe, Jane PhD MEng", {"given": "Jane", "family": "Doe", "suffix": "PhD MEng"}, @@ -992,13 +1033,37 @@ def _check_cjk_shape_purity(self) -> None: notes="the same exclusion for the other chunked member. " "1.4.0 read the same; 2.3.0 read suffix 'MD MEng', " "the reading #540's marking moved"), + Case("a_dual_opening_the_given_part_turns_the_anchor_off", + "Smith, Ms MD Ma", + {"title": "Ms MD", "given": "Ma", "family": "Smith"}, + notes="a dual opening the given part is a title there, and a " + "part whose leading title run holds one is anchored by " + "nothing, the second dual standing in that same run. " + "1.4.0 and 2.3.0 read the same", + shape=2), + Case("a_dual_opening_the_given_part_silences_a_later_degree", + "Smith, MD PhD Ma", + {"title": "MD", "given": "PhD", "middle": "Ma", + "family": "Smith"}, + classification="fix(#296)", + ambiguities=("suffix-or-name",), + notes="the title reading of the opening dual makes 'PhD' the " + "given name, and reading the whole part as credentials " + "instead would stack the anchor's guess on the dual's, " + "so 'PhD' speaks for nothing here and 'Ma' ends the " + "given part as a middle name, reported by the given " + "slot. 2.2.0 and 2.3.0 read the same fields; 1.4.0 " + "through 2.1.0 title 'MD PhD', first 'Ma', 'phd' being " + "title vocabulary until #296", + shape=2), Case("a_numeral_in_front_anchors_nothing", "smith, v ed", {"given": "v", "family": "smith", "suffix": "ed"}, ambiguities=("suffix-or-name",), - notes="a single-letter roman numeral is initial-shaped, " - "written as a middle initial is, so it speaks for " - "nothing (a multi-letter one such as 'III' would): " + notes="a single-letter roman numeral, in any case, speaks " + "for nothing for its shape, one letter being how a " + "middle initial is written (a multi-letter one such as " + "'III' would speak): " "'ed' is read by " "the given slot's own rule. 1.4.0 read the same; 2.3.0 " "read middle 'ed'"), diff --git a/tests/v2/pipeline/test_pieces.py b/tests/v2/pipeline/test_pieces.py index a6876646..2b013e19 100644 --- a/tests/v2/pipeline/test_pieces.py +++ b/tests/v2/pipeline/test_pieces.py @@ -130,6 +130,54 @@ def test_strict_ends_the_run_at_the_initial_shaped_numeral() -> None: assert segment_suffix_reading(*args, False, state.one_case) is None +def _comma_part_reading(text: str) -> tuple[tuple[bool, ...] | None, + list[int]]: + state = _through_group(text) + picks: list[int] = [] + reading = segment_suffix_reading( + state.pieces[1], state.piece_tags[1], list(state.tokens), True, + state.one_case, picks) + return reading, picks + + +def test_the_reading_names_the_picks_its_anchor_made() -> None: + """#544: the members the anchor read as credentials after their + own writing declined -- the picks assign reports. A member whose + capitals lean credential decided itself and is not among them, + and nothing stands in front of piece 0 to anchor it.""" + assert _comma_part_reading("Smith, PhD Ma") == ((True, True), [1]) + assert _comma_part_reading("Smith, PhD MA") == ((True, True), []) + assert _comma_part_reading("Smith, Dr. PhD Ed Ma") == ( + (False, True, True, True), [2, 3]) + assert _comma_part_reading("Smith, MA PhD ba") == ( + (True, True, True), [2]) + + +def test_a_dual_in_the_leading_title_run_turns_the_anchor_off() -> None: + """#544: a title/suffix dual standing in the given part's leading + title run is a title there, and this reading then anchors nothing + in the part -- neither by that dual, nor by a second one in the + same run, nor by a credential behind them -- so the walk reads it + as the parent did, its given slot's company starting past the + given name. A plain title does not do this, and a dual behind a + credential anchors like any suffix word.""" + for text in ("Smith, Ms Ma", "Smith, Ms MD Ma", "Smith, MD PhD Ma", + "Smith, MD MS Ma", "SMITH, MD MS BA", + "Smith, Prof. MD Ma", "smith, prof. md ma", + "Smith, Dr. MD PhD Ma"): + assert _comma_part_reading(text)[0] is None, text + assert _comma_part_reading("Smith, Dr. PhD LAc") == ( + (False, True, True), [2]) + assert _comma_part_reading("Smith, PhD Ms Ma") == ( + (True, True, True), [2]) + # with no member to anchor, the dual reads as the suffix it is + assert _comma_part_reading("Smith, MD PhD") == ((True, True), []) + # and the walk's given slot, past the given name, reads its own + # company as behind any given name + assert parse("Smith, MD PhD Jr Ma").suffix == "Jr Ma" + assert parse("Smith, MD PhD Jr Ma").given == "PhD" + + def _leading(text: str) -> int: state = _through_group(text) return leading_titles(state.pieces[0], state.piece_tags[0], @@ -403,9 +451,10 @@ def test_credential_anchors_reads_in_front_and_through_members() -> None: def test_credential_anchors_stops_at_a_numeral_or_a_name_word() -> None: # the piece straight behind the credential is marked whatever it # is; what matters is the member past it, which a single-letter - # numeral (initial-shaped, like a bare middle initial) or a name - # word leaves unanchored. A multi-letter numeral is not this - # shape and anchors like any other suffix piece (pinned below). + # roman numeral (one letter, in any case, the shape a bare middle + # initial is written in) or a name word leaves unanchored. A + # multi-letter numeral is not this shape and anchors like any + # other suffix piece (pinned below). assert _anchored_words("John Smith PhD v Ma") == ["v"] assert _anchored_words("John Smith PhD Jones Ma") == ["Jones"] assert _anchored_words("John Smith PhD III Ma") == ["III", "Ma"] @@ -414,8 +463,8 @@ def test_credential_anchors_stops_at_a_numeral_or_a_name_word() -> None: def test_a_connective_or_a_numeral_suffix_word_anchors_nothing() -> None: # the generational 'i' is also Catalan's conjunction, and between # two name words it is a link (rules.md#P3); a single-letter roman - # numeral is INITIAL-SHAPED, the same shape a middle initial - # writes in -- not excluded for being "no credential", since a + # numeral, in any case, is ONE LETTER, the shape a middle initial + # is written in -- not excluded for being "no credential", since a # multi-letter numeral ('III') anchors like any other suffix word # (pinned in `test_credential_anchors_stops_at_a_numeral...` and # `test_a_multi_letter_numeral_anchors_like_any_suffix_word` @@ -437,6 +486,10 @@ def test_a_connective_or_a_numeral_suffix_word_anchors_nothing() -> None: # peel's Accepted limit, the merged piece standing outside its walk) assert parse("Doe, Jane Ph. D. MEng").suffix == "Ph. D. MEng" assert parse("Smith, Ph. D. MEng").suffix == "Ph. D. MEng" + # and the member it speaks for reports as the pick it is + assert [(a.kind.value, [t.text for t in a.tokens]) + for a in parse("Smith, Ph. D. MEng").ambiguities] == [ + ("suffix-or-name", ["MEng"])] def test_a_multi_letter_numeral_anchors_like_any_suffix_word() -> None: diff --git a/tests/v2/pipeline/test_segment.py b/tests/v2/pipeline/test_segment.py index 4e0c12cc..b4a212ed 100644 --- a/tests/v2/pipeline/test_segment.py +++ b/tests/v2/pipeline/test_segment.py @@ -238,7 +238,8 @@ def test_the_name_word_count_reads_a_run_as_it_reads_one_word() -> None: def test_the_run_test_declines_a_name_word_and_a_numeral() -> None: # a name word anywhere in the part, or a word that is neither # vocabulary nor a member, keeps the listing form -- and so does a - # single-letter roman numeral, a generation rather than a credential + # single-letter roman numeral, in any case, for its one-letter + # shape (a multi-letter one runs: 'John Smith, III Ma') for text in ("John Smith, Jones Ma", "John Smith, PhD Jones Ma", "John Smith, V Ma", "John Smith, PhD v Ma", "John Smith, PhD Ma.", "John Smith, J. Ma"): diff --git a/tests/v2/test_ledger_guards.py b/tests/v2/test_ledger_guards.py index ad82b5cd..c42fedbc 100644 --- a/tests/v2/test_ledger_guards.py +++ b/tests/v2/test_ledger_guards.py @@ -1271,6 +1271,16 @@ def test_case_shape_ids_exist_in_the_inventory() -> None: "fix(#544) a degree in front outranks P6's attachment for a particle member": ("doe, jane do", "NASCIMENTO, EDSON ARANTES DO", "Dr. doe, jane v phd do"), + # 2026-09-28: the one-word-family-comma rule is fix(#544) at 2.2.0 + # and 2.3.0 and fix(#296/#544) at 2.0.0 and 2.1.0, so its key drops + # the tag as the anchor rule's does; the dual-title rules are + # fix(#296) at 1.4.0 and fix(#296/#531) at 2.0.0 and 2.1.0, keyed by + # the words they share. Walls: the capitals-settled member, the + # dual in the leading title run, and a superstring each. + "#544) a credential in front anchors a member after a one-word family comma, and the pick reports": + ("Smith, PhD MA", "Smith, MD PhD Ma", "Dr. Smith, PhD Ma"), + "phd is not a prenominal, so behind a dual title it is the given name": + ("Smith, MD PhD MA", "Smith, Ms MD Ma", "Dr. Smith, MD PhD Ma"), # The esq boundary is every spelling SUFFIX_WORDS still carries, # in each of the three positions the corpora write it in. "change(suffix-acronym-collisions) esq leaves the acronym set": @@ -2481,6 +2491,12 @@ class _LatinCopy(NamedTuple): "Smith nee Jones, Jane MA", "doe, john ma"}), frozenset({"Doe, John Ed", "Doe, John MA Ma", "Doe, John Ma", "Doe, Mary Jo Ma"}), + # 2026-09-28: the 2.2.0 and 2.3.0 copies of the declining rule + # gain 'Smith, MD PhD Ma', the member ending a given part that a + # dual title opens; 1.4.0's fix(given-part-trailing-slot) keeps + # the four above. + frozenset({"Doe, John Ed", "Doe, John MA Ma", "Doe, John Ma", + "Doe, Mary Jo Ma", "Smith, MD PhD Ma"}), # The third set. What selects these five is a WORD that is both # particle and credential vocabulary standing last after a family # comma -- `mc` and `do` -- and a member copying PARTICLES would @@ -3014,9 +3030,17 @@ class _LatinCopy(NamedTuple): "Jane Doe nee Smith PhD MEng"}), frozenset({r"jack\s+m\.a\.", r"wang\s+m\.eng\."}), frozenset({"John Smith, Ed Ma", "John Smith, Ms Ma", - r"John Smith, X\.Y\.Z\. MA", "john smith, md ma"}), + "John Smith, PhD Ma", r"John Smith, X\.Y\.Z\. MA", + "john smith, md ma"}), frozenset({"Doe, Jane PhD MEng", "Doe, Jane nee Smith PhD MEng", "Jane Doe nee Smith PhD MEng", "John Smith PhD MEng"}), + # 2026-09-28: the anchor rule at 2.2.0 and 2.3.0 gains 'Smith, PhD + # MEng', whose roles those baselines read; at 2.0.0 and 2.1.0 the + # name moves roles and is the one-word-family-comma pair's. + frozenset({"Doe, Jane PhD MEng", "Doe, Jane nee Smith PhD MEng", + "Jane Doe nee Smith PhD MEng", "John Smith PhD MEng", + "Smith, PhD MEng"}), + frozenset({"Smith, PhD MEng", "Smith, PhD Ma"}), frozenset({"Jane Doe, MS LAc", "John Smith, MD MEng", "John Smith, MEng PhD", "John Smith, PhD MEng", "john smith, phd meng"}), @@ -3687,8 +3711,11 @@ def _claim(rule: dict) -> _Claim: # 2026-09-27, #544: 380 -> 392; gains 'Doe, Jane PhD MEng', # 'Doe, Jane nee Smith PhD MEng', 'Jane Doe, MS LAc', 'John # Smith, Ed Ma' and 8 more. + # 2026-09-28, #544: 392 -> 396; gains 'John Smith, PhD Ma', + # 'Smith, MD PhD Ma', 'Smith, Ms MD Ma', 'Smith, PhD Ma'. + # Reach, verified name by name. "fix(comma-family) lone post-comma piece routes to suffix/title, not first": - _Claim(392, ('given', 'suffix', 'title'), "86492f9f87ba", None), + _Claim(396, ('given', 'suffix', 'title'), "1c2cfbb3f881", None), "fix(comma-family) a comma followed only by titles keeps the given/family split": _Claim(2, ('family', 'given'), "5bd9c6d96c38", None), "fix(comma-family) a comma followed only by titles keeps the given/family split, the C1 example": @@ -3726,8 +3753,9 @@ def _claim(rule: dict) -> _Claim: "fix(#296) a lone post-comma credential is a suffix": _Claim(23, ('family', 'given', 'suffix', 'title'), "54c1ae9911e1", None), # 2026-09-27, #544: 6 -> 7; gains 'Smith, PhD MEng'. + # 2026-09-28, #544: 7 -> 8; gains 'Smith, PhD Ma'. "fix(#325) a split credential followed by another suffix after a one-word family comma reads as suffixes": - _Claim(7, ('given', 'suffix', 'title'), "6ab773e59aeb", None), + _Claim(8, ('given', 'suffix', 'title'), "81e29a5745e7", None), "fix(#325) a credential run across a second comma reads as suffixes": _Claim(1, ('suffix', 'title'), "f025c5f70a4e", None), "fix(#367) an inferred title no longer displaces a leading particle either": @@ -3775,8 +3803,9 @@ def _claim(rule: dict) -> _Claim: # 2026-09-27, #544: 380 -> 392; gains 'Doe, Jane PhD MEng', # 'Doe, Jane nee Smith PhD MEng', 'Jane Doe, MS LAc', 'John # Smith, Ed Ma' and 8 more. + # 2026-09-28, #544: 392 -> 396; the same four comma names. "fix(comma-precomma-family) pre-comma run reads as family, not given": - _Claim(392, ('family', 'given'), "86492f9f87ba", None), + _Claim(396, ('family', 'given'), "1c2cfbb3f881", None), # 2026-09-20, #397: retitled in place, reach and digest # unchanged -- the rule keeps 'Carod i', which the landing # leaves byte-identical. @@ -4389,6 +4418,9 @@ def _claim(rule: dict) -> _Claim: # 2026-09-27, #544: new, 1; gains 'John Smith, X.Y.Z. MA'. "fix(#544) a run of ambiguous members after a suffix comma reads by the name-word count": _Claim(1, ('family', 'given', 'suffix'), "23c792cefe18", ('DEFAULT',)), + # 2026-09-28, #544: new, 1; 'Smith, MD PhD Ma'. + "fix(#296) phd is not a prenominal, so behind a dual title it is the given name": + _Claim(1, ('given', 'middle', 'title'), "8e9913f5df8b", None), }, "expected_since_2.0.0.toml": { # The ph removal (#459/#521): one literal name, the cases.py @@ -4665,8 +4697,9 @@ def _claim(rule: dict) -> _Claim: "fix(#296) a lone post-comma credential is a suffix": _Claim(23, ('suffix', 'title'), "54c1ae9911e1", None), # 2026-09-27, #544: 6 -> 7; gains 'Smith, PhD MEng'. + # 2026-09-28, #544: 7 -> 8; gains 'Smith, PhD Ma'. "fix(#325) a split credential followed by another suffix after a one-word family comma reads as suffixes": - _Claim(7, ('given', 'suffix', 'title'), "6ab773e59aeb", None), + _Claim(8, ('given', 'suffix', 'title'), "81e29a5745e7", None), "fix(#325) a credential run across a second comma reads as suffixes": _Claim(1, ('suffix', 'title'), "f025c5f70a4e", None), "fix(#296) a glued honorific before a lone credential: the credential is the postnominal": @@ -5071,8 +5104,9 @@ def _claim(rule: dict) -> _Claim: _Claim(5, ('_ambiguities',), "bac52af569e4", ('DEFAULT',)), # 2026-09-27, #544: new, 4; gains 'John Smith, Ed Ma', 'John # Smith, Ms Ma', 'John Smith, X.Y.Z. MA', 'john smith, md ma'. + # 2026-09-28, #544: 4 -> 5; gains 'John Smith, PhD Ma'. "fix(#544) a run of ambiguous members after a suffix comma reads by the name-word count": - _Claim(4, ('_ambiguities', 'family', 'given', 'middle', 'suffix', 'title'), "65222cc6fe3d", ('DEFAULT',)), + _Claim(5, ('_ambiguities', 'family', 'given', 'middle', 'suffix', 'title'), "4123358beccd", ('DEFAULT',)), # 2026-09-27, #544: new, 4; gains 'Doe, Jane PhD MEng', 'Doe, # Jane nee Smith PhD MEng', 'Jane Doe nee Smith PhD MEng', # 'John Smith PhD MEng'. Relabelled the same day @@ -5083,6 +5117,12 @@ def _claim(rule: dict) -> _Claim: # 2026-09-27, #544: new, 1; gains 'doe, jane v phd do'. "fix(#544) a degree in front outranks P6's attachment for a particle member": _Claim(1, ('_ambiguities', 'middle', 'suffix'), "a2b8cea490a7", ('DEFAULT',)), + # 2026-09-28, #544: new, 2; 'Smith, PhD MEng', 'Smith, PhD Ma'. + "fix(#296/#544) a credential in front anchors a member after a one-word family comma, and the pick reports": + _Claim(2, ('_ambiguities', 'given', 'suffix', 'title'), "9fe346e1f150", ('DEFAULT',)), + # 2026-09-28, #544: new, 1; 'Smith, MD PhD Ma'. + "fix(#296/#531) phd is not a prenominal, so behind a dual title it is the given name and the member ending the part reports": + _Claim(1, ('_ambiguities', 'given', 'middle', 'title'), "8e9913f5df8b", ('DEFAULT',)), }, # The 2.3 cycle's first rule, and a facade-only render fix: every # role is identical, so `_initials` alone. Reach and digest as in @@ -5341,8 +5381,9 @@ def _claim(rule: dict) -> _Claim: # {middle, suffix} instead. "fix(#531) a credential ending the given part of a family-comma listing reads as a credential": _Claim(15, ('_ambiguities', 'middle', 'suffix'), "f17532bb4ff1", ('DEFAULT',)), + # 2026-09-28, #544: 4 -> 5; gains 'Smith, MD PhD Ma'. "fix(#531) a member the writing declines keeps its name reading and reports the fork": - _Claim(4, ('_ambiguities',), "c6d26d145ee1", ('DEFAULT',)), + _Claim(5, ('_ambiguities',), "f73bc2fc5408", ('DEFAULT',)), "fix(#531) capitals take the do collision from the family-comma particle attachment": _Claim(1, ('_ambiguities', 'family', 'suffix'), "8ad64f404621", ('DEFAULT',)), "fix(#531) the trailing slot's positional reading reaches a caseless script": @@ -5499,18 +5540,23 @@ def _claim(rule: dict) -> _Claim: _Claim(5, ('_ambiguities',), "bac52af569e4", ('DEFAULT',)), # 2026-09-27, #544: new, 4; gains 'John Smith, Ed Ma', 'John # Smith, Ms Ma', 'John Smith, X.Y.Z. MA', 'john smith, md ma'. + # 2026-09-28, #544: 4 -> 5; gains 'John Smith, PhD Ma'. "fix(#544) a run of ambiguous members after a suffix comma reads by the name-word count": - _Claim(4, ('_ambiguities', 'family', 'given', 'middle', 'suffix', 'title'), "65222cc6fe3d", ('DEFAULT',)), + _Claim(5, ('_ambiguities', 'family', 'given', 'middle', 'suffix', 'title'), "4123358beccd", ('DEFAULT',)), # 2026-09-27, #544: new, 4; gains 'Doe, Jane PhD MEng', 'Doe, # Jane nee Smith PhD MEng', 'Jane Doe nee Smith PhD MEng', # 'John Smith PhD MEng'. Relabelled the same day # fix(#436/#437/#544), the `suffix` it claims being R1's # spacing alone; reach, roles and digest unchanged. + # 2026-09-28, #544: 4 -> 5; gains 'Smith, PhD MEng'. "fix(#436/#437/#544) an unambiguous credential in front anchors the member behind it": - _Claim(4, ('_ambiguities', 'suffix'), "c9d1d53dee20", ('DEFAULT',)), + _Claim(5, ('_ambiguities', 'suffix'), "5807b060ae87", ('DEFAULT',)), # 2026-09-27, #544: new, 1; gains 'doe, jane v phd do'. "fix(#544) a degree in front outranks P6's attachment for a particle member": _Claim(1, ('_ambiguities', 'family', 'suffix'), "a2b8cea490a7", ('DEFAULT',)), + # 2026-09-28, #544: new, 1; 'Smith, PhD Ma'. + "fix(#544) a credential in front anchors a member after a one-word family comma, and the pick reports": + _Claim(1, ('_ambiguities', 'given', 'middle', 'suffix'), "b2b939a7b814", ('DEFAULT',)), }, "expected_since_2.1.0.toml": { # The ph removal (#459/#521): one literal name, the cases.py @@ -5763,8 +5809,9 @@ def _claim(rule: dict) -> _Claim: "fix(#296) a lone post-comma credential is a suffix": _Claim(23, ('suffix', 'title'), "54c1ae9911e1", None), # 2026-09-27, #544: 6 -> 7; gains 'Smith, PhD MEng'. + # 2026-09-28, #544: 7 -> 8; gains 'Smith, PhD Ma'. "fix(#325) a split credential followed by another suffix after a one-word family comma reads as suffixes": - _Claim(7, ('given', 'suffix', 'title'), "6ab773e59aeb", None), + _Claim(8, ('given', 'suffix', 'title'), "81e29a5745e7", None), "fix(#325) a credential run across a second comma reads as suffixes": _Claim(1, ('suffix', 'title'), "f025c5f70a4e", None), "fix(#296) a glued honorific before a lone credential: the credential is the postnominal": @@ -6160,8 +6207,9 @@ def _claim(rule: dict) -> _Claim: _Claim(5, ('_ambiguities',), "bac52af569e4", ('DEFAULT',)), # 2026-09-27, #544: new, 4; gains 'John Smith, Ed Ma', 'John # Smith, Ms Ma', 'John Smith, X.Y.Z. MA', 'john smith, md ma'. + # 2026-09-28, #544: 4 -> 5; gains 'John Smith, PhD Ma'. "fix(#544) a run of ambiguous members after a suffix comma reads by the name-word count": - _Claim(4, ('_ambiguities', 'family', 'given', 'middle', 'suffix', 'title'), "65222cc6fe3d", ('DEFAULT',)), + _Claim(5, ('_ambiguities', 'family', 'given', 'middle', 'suffix', 'title'), "4123358beccd", ('DEFAULT',)), # 2026-09-27, #544: new, 4; gains 'Doe, Jane PhD MEng', 'Doe, # Jane nee Smith PhD MEng', 'Jane Doe nee Smith PhD MEng', # 'John Smith PhD MEng'. Relabelled the same day @@ -6172,6 +6220,12 @@ def _claim(rule: dict) -> _Claim: # 2026-09-27, #544: new, 1; gains 'doe, jane v phd do'. "fix(#544) a degree in front outranks P6's attachment for a particle member": _Claim(1, ('_ambiguities', 'middle', 'suffix'), "a2b8cea490a7", ('DEFAULT',)), + # 2026-09-28, #544: new, 2; 'Smith, PhD MEng', 'Smith, PhD Ma'. + "fix(#296/#544) a credential in front anchors a member after a one-word family comma, and the pick reports": + _Claim(2, ('_ambiguities', 'given', 'suffix', 'title'), "9fe346e1f150", ('DEFAULT',)), + # 2026-09-28, #544: new, 1; 'Smith, MD PhD Ma'. + "fix(#296/#531) phd is not a prenominal, so behind a dual title it is the given name and the member ending the part reports": + _Claim(1, ('_ambiguities', 'given', 'middle', 'title'), "8e9913f5df8b", ('DEFAULT',)), }, "expected_since_2.3.0.toml": { # The ph removal (#459/#521): one literal name, the cases.py @@ -6290,8 +6344,9 @@ def _claim(rule: dict) -> _Claim: # {middle, suffix} instead. "fix(#531) a credential ending the given part of a family-comma listing reads as a credential": _Claim(15, ('_ambiguities', 'middle', 'suffix'), "f17532bb4ff1", ('DEFAULT',)), + # 2026-09-28, #544: 4 -> 5; gains 'Smith, MD PhD Ma'. "fix(#531) a member the writing declines keeps its name reading and reports the fork": - _Claim(4, ('_ambiguities',), "c6d26d145ee1", ('DEFAULT',)), + _Claim(5, ('_ambiguities',), "f73bc2fc5408", ('DEFAULT',)), "fix(#531) capitals take the do collision from the family-comma particle attachment": _Claim(1, ('_ambiguities', 'family', 'suffix'), "8ad64f404621", ('DEFAULT',)), "fix(#531) the trailing slot's positional reading reaches a caseless script": @@ -6437,16 +6492,21 @@ def _claim(rule: dict) -> _Claim: _Claim(5, ('_ambiguities',), "bac52af569e4", ('DEFAULT',)), # 2026-09-27, #544: new, 4; gains 'John Smith, Ed Ma', 'John # Smith, Ms Ma', 'John Smith, X.Y.Z. MA', 'john smith, md ma'. + # 2026-09-28, #544: 4 -> 5; gains 'John Smith, PhD Ma'. "fix(#544) a run of ambiguous members after a suffix comma reads by the name-word count": - _Claim(4, ('_ambiguities', 'family', 'given', 'middle', 'suffix', 'title'), "65222cc6fe3d", ('DEFAULT',)), + _Claim(5, ('_ambiguities', 'family', 'given', 'middle', 'suffix', 'title'), "4123358beccd", ('DEFAULT',)), # 2026-09-27, #544: new, 4; gains 'Doe, Jane PhD MEng', 'Doe, # Jane nee Smith PhD MEng', 'Jane Doe nee Smith PhD MEng', # 'John Smith PhD MEng'. + # 2026-09-28, #544: 4 -> 5; gains 'Smith, PhD MEng'. "fix(#544) an unambiguous credential in front anchors the member behind it": - _Claim(4, ('_ambiguities',), "c9d1d53dee20", ('DEFAULT',)), + _Claim(5, ('_ambiguities',), "5807b060ae87", ('DEFAULT',)), # 2026-09-27, #544: new, 1; gains 'doe, jane v phd do'. "fix(#544) a degree in front outranks P6's attachment for a particle member": _Claim(1, ('_ambiguities', 'family', 'suffix'), "a2b8cea490a7", ('DEFAULT',)), + # 2026-09-28, #544: new, 1; 'Smith, PhD Ma'. + "fix(#544) a credential in front anchors a member after a one-word family comma, and the pick reports": + _Claim(1, ('_ambiguities', 'given', 'middle', 'suffix'), "b2b939a7b814", ('DEFAULT',)), }, } @@ -6908,6 +6968,12 @@ def test_every_rule_claims_the_recorded_share_of_the_corpus() -> None: "Smith, PhD MEng": "fix(#325) a split credential followed by another suffix " "after a one-word family comma reads as suffixes", + # 2026-09-28: 'Smith, PhD Ma' likewise -- 1.4.0 title 'PhD', + # first 'Ma', collapsing whole into `suffix` -- the same + # contest, won by the same file order on the same argument. + "Smith, PhD Ma": + "fix(#325) a split credential followed by another suffix " + "after a one-word family comma reads as suffixes", # PAIR B, three names, OVERLAPPING `fields`. This is the pair # #498 was filed on: `fix(#271/#272/#298)` declares # {family, given, middle} and `fix(cjk-delimited-nickname)` diff --git a/tests/v2/test_parser.py b/tests/v2/test_parser.py index b0006e4a..1294f228 100644 --- a/tests/v2/test_parser.py +++ b/tests/v2/test_parser.py @@ -251,6 +251,36 @@ def test_every_ambiguous_acronym_in_a_name_is_reported() -> None: ["JD", "MA"] +def _reports(text: str) -> list[tuple[AmbiguityKind, list[str]]]: + return [(a.kind, [t.text for t in a.tokens]) + for a in parse(text).ambiguities] + + +def test_a_member_the_company_decides_after_a_family_comma_reports() -> None: + # #544: a part after a one-word family comma that a credential in + # front reads wholly as credentials reports each member it decided + # -- and only those: a member whose capitals already lean + # credential decided itself, so the same fields are silent + n = parse("Smith, PhD Ma") + assert (n.family, n.given, n.suffix) == ("Smith", "", "PhD Ma") + assert _reports("Smith, PhD Ma") == [ + (AmbiguityKind.SUFFIX_OR_NAME, ["Ma"])] + n = parse("Smith, PhD MA") + assert (n.family, n.given, n.suffix) == ("Smith", "", "PhD MA") + assert n.ambiguities == () + + +def test_a_dotted_credential_is_no_title_run_dual() -> None: + # 'M.D.' carries the suffix reading and not the title one, so it + # does not stand in the given part's leading title run: 'PhD' + # behind it still speaks for 'Ma', and the pick reports (the parent + # read given 'M.D.', middle 'Ma', suffix 'PhD') + n = parse("Smith, M.D. PhD Ma") + assert (n.family, n.given, n.suffix) == ("Smith", "", "M.D. PhD Ma") + assert _reports("Smith, M.D. PhD Ma") == [ + (AmbiguityKind.SUFFIX_OR_NAME, ["Ma"])] + + def test_ambiguous_acronym_detail_names_the_role_it_got() -> None: # the unpeeled piece is the last NAME piece, which is the family # name only under GIVEN_FIRST -- FAMILY_FIRST puts it in given, so diff --git a/tests/v2/test_properties.py b/tests/v2/test_properties.py index 20dd528c..9ab83b24 100644 --- a/tests/v2/test_properties.py +++ b/tests/v2/test_properties.py @@ -383,6 +383,17 @@ def _outside_its_company(name: ParsedName) -> list[str]: reads family 'Ma', because 'PhD' is the name the reserve kept, not a credential run's own member). + A dual in the given part's leading title run also keeps a + credential BEHIND it from speaking for a member in that part + ('Smith, MD PhD Ma' keeps given 'PhD', middle 'Ma'), and that + exemption is NOT modelled here: no head of the + walk opens a comma part with a title/suffix dual, so it would be + code no parse reaches. The case rows + `a_dual_opening_the_given_part_turns_the_anchor_off` and + `a_dual_opening_the_given_part_silences_a_later_degree`, and + test_pieces' `test_a_dual_in_the_leading_title_run_turns_the_anchor_off`, + pin it instead. + Deliberately does NOT require the front to already be in the SUFFIX role: that was tried and it blinded the check to exactly the no-comma half of the defect this exists for -- with the diff --git a/tools/differential/compare.py b/tools/differential/compare.py index 2d71d222..814fa18b 100644 --- a/tools/differential/compare.py +++ b/tools/differential/compare.py @@ -1811,6 +1811,9 @@ class _ShapeMismatch(NamedTuple): # 'Smith, PhD Jr.' above -- the unambiguous PhD anchors the # MEng behind it, so the run collapses whole into `suffix`. "Smith, PhD MEng": ("given", "suffix", "title"), + # 2026-09-28: 'Smith, PhD Ma', the same run with the Title-case + # member of the older marking, on the same terms. + "Smith, PhD Ma": ("given", "suffix", "title"), # #528's two, adjudicated 2026-09-13, and kept BELOW #498's # block so the three cohorts read down the dict in the order # the PROVENANCE note above tells them. Both are contested by diff --git a/tools/differential/corpus_rules.jsonl b/tools/differential/corpus_rules.jsonl index c63db184..d77be1aa 100644 --- a/tools/differential/corpus_rules.jsonl +++ b/tools/differential/corpus_rules.jsonl @@ -328,6 +328,7 @@ "Smith, Jr." "Smith, MA" "Smith, MD PhD" +"Smith, MD PhD Ma" "Smith, Ma" "Smith, Major. John" "Smith, Ms Ma" @@ -338,6 +339,7 @@ "Smith, Ph. D. Jr." "Smith, PhD" "Smith, PhD MEng" +"Smith, PhD Ma" "Smith, Sr." "Smith, de Mesnil Jean" "Smith. John" diff --git a/tools/differential/corpus_shapes.jsonl b/tools/differential/corpus_shapes.jsonl index 348a9578..c21d879f 100644 --- a/tools/differential/corpus_shapes.jsonl +++ b/tools/differential/corpus_shapes.jsonl @@ -214,10 +214,13 @@ {"name": "Smith, John V, Jr.", "shape": 2} {"name": "Smith, John, Extra, Jr.", "shape": 2} {"name": "Smith, MA", "shape": 2} +{"name": "Smith, MD PhD Ma", "shape": 2} {"name": "Smith, MEng", "shape": 2} {"name": "Smith, Ma", "shape": 2} +{"name": "Smith, Ms MD Ma", "shape": 2} {"name": "Smith, Ms Ma", "shape": 2} {"name": "Smith, PhD MEng", "shape": 2} +{"name": "Smith, PhD Ma", "shape": 2} {"name": "Smith, meng", "shape": 2} {"name": "de la Vega, Juan", "shape": 2} {"name": "doe, jane v phd do", "shape": 2} @@ -238,6 +241,7 @@ {"name": "John Smith, Ms Ma", "shape": 3} {"name": "John Smith, PhD", "shape": 3} {"name": "John Smith, PhD MEng", "shape": 3} +{"name": "John Smith, PhD Ma", "shape": 3} {"name": "John Smith, X.Y.Z. MA", "shape": 3} {"name": "Steven Hardman, MD, DO, DDS", "shape": 3} {"name": "john smith, ma", "shape": 3} diff --git a/tools/differential/expected_since_1.4.0.toml b/tools/differential/expected_since_1.4.0.toml index 9eada5d4..294c6f19 100644 --- a/tools/differential/expected_since_1.4.0.toml +++ b/tools/differential/expected_since_1.4.0.toml @@ -753,6 +753,9 @@ issue = "fix(#325) a split credential followed by another suffix after a one-wor # company clause), so the run after the one-word family comma collapses # whole into `suffix` where the tree before #544 read given 'PhD', # middle 'MEng'. +# 2026-09-28: 'Smith, PhD Ma' likewise, the same run with the Title-case +# member of the older marking; both are pinned in _RECORDED_DIFFS +# beside 'Smith, PhD Jr.'. name_regex = "(?i)^smith,\\s*ph\\.?\\s?d\\.?\\s" fields = ["title", "given", "suffix"] @@ -4751,3 +4754,20 @@ issue = "fix(#540) a Title-case Lac behind a particle is the family name" name_regex = "^Nguyen Van Lac$" fields = ["family", "suffix"] orders = ["DEFAULT"] + +[[change]] +issue = "fix(#296) phd is not a prenominal, so behind a dual title it is the given name" +# 'Smith, MD PhD Ma': title 'MD', given 'PhD', middle 'Ma', where v1 +# read title 'MD PhD', first 'Ma'. 'phd' left TITLES (#296), so the +# title run ends at the dual 'MD' and 'PhD' is the given name, with +# 'Ma' ending the given part as a middle name (#531's slot, which keeps +# a Title-case member a name). #544 leaves the reading where it was: +# nothing in a part whose leading title run holds a dual speaks for the +# member (rules.md#S2). +# +# Literal. Probes: 'Smith, MD PhD MA' (the capitals make the member a +# credential and the part reads whole), 'Smith, Ms MD Ma' (v1's +# reading, unchanged) and the superstring 'Dr. Smith, MD PhD Ma' are +# _MUST_NOT_MATCH. +name_regex = "^Smith, MD PhD Ma$" +fields = ["title", "given", "middle"] diff --git a/tools/differential/expected_since_2.0.0.toml b/tools/differential/expected_since_2.0.0.toml index beeacd85..2b5b1f08 100644 --- a/tools/differential/expected_since_2.0.0.toml +++ b/tools/differential/expected_since_2.0.0.toml @@ -1080,6 +1080,10 @@ issue = "fix(#325) a split credential followed by another suffix after a one-wor # company clause), so the run after the one-word family comma collapses # whole into `suffix` where the tree before #544 read given 'PhD', # middle 'MEng'. +# 2026-09-28: the regex still reaches 'Smith, PhD MEng', and now 'Smith, +# PhD Ma', but explains neither: each reports the pick its member is, +# and this rule declares no `_ambiguities`. Both are the +# fix(#296/#544) rule's at the foot of this file. name_regex = "(?i)^smith,\\s*ph\\.?\\s?d\\.?\\s" fields = ["title", "given", "suffix"] @@ -3710,11 +3714,15 @@ issue = "fix(#544) a run of ambiguous members after a suffix comma reads by the # LISTED member only -- so its run flips too. 1.4.0 read the listed- # member names as suffixes as well. # +# 2026-09-28: 'John Smith, PhD Ma' joins, the same run with an +# unambiguous word in front of the member, which the count reads whole +# like the rest. +# # Literal; `fields` is the union the names move ('title' is the duals'). # Probes: 'Smith, Ed Ma' (one name word before the comma keeps the # listing form), 'Smith, Ms Ma' (a dual opening the given part is a # title) and the superstring 'Dr. John Smith, Ed Ma' are _MUST_NOT_MATCH. -name_regex = "^(?:John Smith, Ed Ma|John Smith, Ms Ma|John Smith, X\\.Y\\.Z\\. MA|john smith, md ma)$" +name_regex = "^(?:John Smith, Ed Ma|John Smith, Ms Ma|John Smith, PhD Ma|John Smith, X\\.Y\\.Z\\. MA|john smith, md ma)$" fields = ["given", "middle", "family", "suffix", "title", "_ambiguities"] orders = ["DEFAULT"] @@ -3773,3 +3781,41 @@ issue = "fix(#540) a Title-case Lac behind a particle is the family name" name_regex = "^Nguyen Van Lac$" fields = ["family", "suffix", "_ambiguities"] orders = ["DEFAULT"] + +[[change]] +issue = "fix(#296/#544) a credential in front anchors a member after a one-word family comma, and the pick reports" +# 'Smith, PhD Ma' and 'Smith, PhD MEng': family 'Smith', suffix 'PhD +# Ma' / 'PhD MEng', each reporting its member, where this baseline read +# title 'PhD' and given 'Ma' (suffix 'MEng'). Two changes make the one +# diff, so the label is joint: 'phd' left TITLES (#296, the 2.2 audit), +# which is what lets the part after the comma read as a credential run +# at all, and rules.md#S2's company clause (#544), 'PhD' speaking for +# the member behind it and the pick reporting where it was made. The +# fix(#325) rule above reaches both names by its regex and explains +# neither any longer, declaring no `_ambiguities`. +# +# Literal. Probes: 'Smith, PhD MA' (the capitals decide the member), +# 'Smith, MD PhD Ma' (a dual in the leading title run: nothing in the +# part speaks) and the superstring 'Dr. Smith, PhD Ma' are +# _MUST_NOT_MATCH. +name_regex = "^(?:Smith, PhD MEng|Smith, PhD Ma)$" +fields = ["title", "given", "suffix", "_ambiguities"] +orders = ["DEFAULT"] + +[[change]] +issue = "fix(#296/#531) phd is not a prenominal, so behind a dual title it is the given name and the member ending the part reports" +# 'Smith, MD PhD Ma': title 'MD', given 'PhD', middle 'Ma', reporting +# 'Ma', where this baseline read title 'MD PhD', given 'Ma'. 'phd' left +# TITLES (#296), so the title run ends at the dual 'MD' and 'PhD' is +# the given name; the Title-case 'Ma' ending the given part is #531's +# slot, which keeps it a name and reports the fork. #544 leaves the +# reading where it was: nothing in a part whose leading title run holds +# a dual speaks for the member (rules.md#S2). +# +# Literal. Probes: 'Smith, MD PhD MA' (the capitals make the member a +# credential and the part reads whole), 'Smith, Ms MD Ma' (the title +# run leaves 'Ma' the given name) and the superstring 'Dr. Smith, MD +# PhD Ma' are _MUST_NOT_MATCH. +name_regex = "^Smith, MD PhD Ma$" +fields = ["title", "given", "middle", "_ambiguities"] +orders = ["DEFAULT"] diff --git a/tools/differential/expected_since_2.1.0.toml b/tools/differential/expected_since_2.1.0.toml index 44417b48..9e8efb93 100644 --- a/tools/differential/expected_since_2.1.0.toml +++ b/tools/differential/expected_since_2.1.0.toml @@ -746,6 +746,10 @@ issue = "fix(#325) a split credential followed by another suffix after a one-wor # company clause), so the run after the one-word family comma collapses # whole into `suffix` where the tree before #544 read given 'PhD', # middle 'MEng'. +# 2026-09-28: the regex still reaches 'Smith, PhD MEng', and now 'Smith, +# PhD Ma', but explains neither: each reports the pick its member is, +# and this rule declares no `_ambiguities`. Both are the +# fix(#296/#544) rule's at the foot of this file. name_regex = "(?i)^smith,\\s*ph\\.?\\s?d\\.?\\s" fields = ["title", "given", "suffix"] @@ -3621,11 +3625,15 @@ issue = "fix(#544) a run of ambiguous members after a suffix comma reads by the # LISTED member only -- so its run flips too. 1.4.0 read the listed- # member names as suffixes as well. # +# 2026-09-28: 'John Smith, PhD Ma' joins, the same run with an +# unambiguous word in front of the member, which the count reads whole +# like the rest. +# # Literal; `fields` is the union the names move ('title' is the duals'). # Probes: 'Smith, Ed Ma' (one name word before the comma keeps the # listing form), 'Smith, Ms Ma' (a dual opening the given part is a # title) and the superstring 'Dr. John Smith, Ed Ma' are _MUST_NOT_MATCH. -name_regex = "^(?:John Smith, Ed Ma|John Smith, Ms Ma|John Smith, X\\.Y\\.Z\\. MA|john smith, md ma)$" +name_regex = "^(?:John Smith, Ed Ma|John Smith, Ms Ma|John Smith, PhD Ma|John Smith, X\\.Y\\.Z\\. MA|john smith, md ma)$" fields = ["given", "middle", "family", "suffix", "title", "_ambiguities"] orders = ["DEFAULT"] @@ -3684,3 +3692,41 @@ issue = "fix(#540) a Title-case Lac behind a particle is the family name" name_regex = "^Nguyen Van Lac$" fields = ["family", "suffix", "_ambiguities"] orders = ["DEFAULT"] + +[[change]] +issue = "fix(#296/#544) a credential in front anchors a member after a one-word family comma, and the pick reports" +# 'Smith, PhD Ma' and 'Smith, PhD MEng': family 'Smith', suffix 'PhD +# Ma' / 'PhD MEng', each reporting its member, where this baseline read +# title 'PhD' and given 'Ma' (suffix 'MEng'). Two changes make the one +# diff, so the label is joint: 'phd' left TITLES (#296, the 2.2 audit), +# which is what lets the part after the comma read as a credential run +# at all, and rules.md#S2's company clause (#544), 'PhD' speaking for +# the member behind it and the pick reporting where it was made. The +# fix(#325) rule above reaches both names by its regex and explains +# neither any longer, declaring no `_ambiguities`. +# +# Literal. Probes: 'Smith, PhD MA' (the capitals decide the member), +# 'Smith, MD PhD Ma' (a dual in the leading title run: nothing in the +# part speaks) and the superstring 'Dr. Smith, PhD Ma' are +# _MUST_NOT_MATCH. +name_regex = "^(?:Smith, PhD MEng|Smith, PhD Ma)$" +fields = ["title", "given", "suffix", "_ambiguities"] +orders = ["DEFAULT"] + +[[change]] +issue = "fix(#296/#531) phd is not a prenominal, so behind a dual title it is the given name and the member ending the part reports" +# 'Smith, MD PhD Ma': title 'MD', given 'PhD', middle 'Ma', reporting +# 'Ma', where this baseline read title 'MD PhD', given 'Ma'. 'phd' left +# TITLES (#296), so the title run ends at the dual 'MD' and 'PhD' is +# the given name; the Title-case 'Ma' ending the given part is #531's +# slot, which keeps it a name and reports the fork. #544 leaves the +# reading where it was: nothing in a part whose leading title run holds +# a dual speaks for the member (rules.md#S2). +# +# Literal. Probes: 'Smith, MD PhD MA' (the capitals make the member a +# credential and the part reads whole), 'Smith, Ms MD Ma' (the title +# run leaves 'Ma' the given name) and the superstring 'Dr. Smith, MD +# PhD Ma' are _MUST_NOT_MATCH. +name_regex = "^Smith, MD PhD Ma$" +fields = ["title", "given", "middle", "_ambiguities"] +orders = ["DEFAULT"] diff --git a/tools/differential/expected_since_2.2.0.toml b/tools/differential/expected_since_2.2.0.toml index 6cc62125..c5c5d271 100644 --- a/tools/differential/expected_since_2.2.0.toml +++ b/tools/differential/expected_since_2.2.0.toml @@ -1201,9 +1201,16 @@ issue = "fix(#531) a member the writing declines keeps its name reading and repo # fix(given-part-trailing-slot), whose comment records that #531 # changed which names it holds rather than retiring it. # +# 2026-09-28: 'Smith, MD PhD Ma' joins -- a title/suffix dual opening the +# given part is a title there, 'PhD' is the given name, and the +# Title-case 'Ma' ending the part is this slot's declining half, +# reported. No role moves from this baseline; #544 keeps it so +# (rules.md#S2: nothing in a part whose leading title run holds a dual +# speaks for the member). +# # Literal-anchored: the class is the slot's declining half, and a # regex for it would claim the thirteen movers above. -name_regex = "^(?:Doe, John Ed|Doe, John MA Ma|Doe, John Ma|Doe, Mary Jo Ma)$" +name_regex = "^(?:Doe, John Ed|Doe, John MA Ma|Doe, John Ma|Doe, Mary Jo Ma|Smith, MD PhD Ma)$" fields = ["_ambiguities"] orders = ["DEFAULT"] @@ -2016,11 +2023,15 @@ issue = "fix(#544) a run of ambiguous members after a suffix comma reads by the # LISTED member only -- so its run flips too. 1.4.0 read the listed- # member names as suffixes as well. # +# 2026-09-28: 'John Smith, PhD Ma' joins, the same run with an +# unambiguous word in front of the member, which the count reads whole +# like the rest. +# # Literal; `fields` is the union the names move ('title' is the duals'). # Probes: 'Smith, Ed Ma' (one name word before the comma keeps the # listing form), 'Smith, Ms Ma' (a dual opening the given part is a # title) and the superstring 'Dr. John Smith, Ed Ma' are _MUST_NOT_MATCH. -name_regex = "^(?:John Smith, Ed Ma|John Smith, Ms Ma|John Smith, X\\.Y\\.Z\\. MA|john smith, md ma)$" +name_regex = "^(?:John Smith, Ed Ma|John Smith, Ms Ma|John Smith, PhD Ma|John Smith, X\\.Y\\.Z\\. MA|john smith, md ma)$" fields = ["given", "middle", "family", "suffix", "title", "_ambiguities"] orders = ["DEFAULT"] @@ -2043,10 +2054,15 @@ issue = "fix(#436/#437/#544) an unambiguous credential in front anchors the memb # report #544 adds. The 2.3.0 ledger, whose baseline already spaces the # run, carries the same rule as #544's alone. # +# 2026-09-28: 'Smith, PhD MEng' joins. The one-word family comma leaves +# its part holding no name word once 'PhD' speaks for 'MEng', so the +# part reads wholly as the credential run -- the fields this baseline +# read -- and the pick the company made reports where it was made. +# # Literal. Probes: 'Wang Ma PhD' (the credential BEHIND the member speaks # for nothing) and the superstring 'Dr. John Smith PhD MEng' are # _MUST_NOT_MATCH. -name_regex = "^(?:Doe, Jane PhD MEng|Doe, Jane nee Smith PhD MEng|Jane Doe nee Smith PhD MEng|John Smith PhD MEng)$" +name_regex = "^(?:Doe, Jane PhD MEng|Doe, Jane nee Smith PhD MEng|Jane Doe nee Smith PhD MEng|John Smith PhD MEng|Smith, PhD MEng)$" fields = ["suffix", "_ambiguities"] orders = ["DEFAULT"] @@ -2079,3 +2095,21 @@ issue = "fix(#540) a Title-case Lac behind a particle is the family name" name_regex = "^Nguyen Van Lac$" fields = ["family", "suffix", "_ambiguities"] orders = ["DEFAULT"] + +[[change]] +issue = "fix(#544) a credential in front anchors a member after a one-word family comma, and the pick reports" +# 'Smith, PhD Ma': family 'Smith', suffix 'PhD Ma', reporting the +# Title-case 'Ma', where this baseline read given 'PhD', middle 'Ma'. +# rules.md#S2's company clause: 'PhD' in front speaks for 'Ma', so the +# part after the one-word family comma holds no name word and reads +# wholly as the credential run, and "a member the company decides +# reports the fork as a counted pick does". 'Smith, PhD MEng', whose +# roles this baseline already read, is the anchor rule's above. +# +# Literal. Probes: 'Smith, PhD MA' (the capitals decide the member, +# nothing new to report), 'Smith, MD PhD Ma' (a dual in the leading +# title run: nothing in the part speaks) and the superstring 'Dr. +# Smith, PhD Ma' are _MUST_NOT_MATCH. +name_regex = "^Smith, PhD Ma$" +fields = ["given", "middle", "suffix", "_ambiguities"] +orders = ["DEFAULT"] diff --git a/tools/differential/expected_since_2.3.0.toml b/tools/differential/expected_since_2.3.0.toml index 7c3c43b5..b7dc4cea 100644 --- a/tools/differential/expected_since_2.3.0.toml +++ b/tools/differential/expected_since_2.3.0.toml @@ -489,9 +489,16 @@ issue = "fix(#531) a member the writing declines keeps its name reading and repo # fix(given-part-trailing-slot), whose comment records that #531 # changed which names it holds rather than retiring it. # +# 2026-09-28: 'Smith, MD PhD Ma' joins -- a title/suffix dual opening the +# given part is a title there, 'PhD' is the given name, and the +# Title-case 'Ma' ending the part is this slot's declining half, +# reported. No role moves from this baseline; #544 keeps it so +# (rules.md#S2: nothing in a part whose leading title run holds a dual +# speaks for the member). +# # Literal-anchored: the class is the slot's declining half, and a # regex for it would claim the thirteen movers above. -name_regex = "^(?:Doe, John Ed|Doe, John MA Ma|Doe, John Ma|Doe, Mary Jo Ma)$" +name_regex = "^(?:Doe, John Ed|Doe, John MA Ma|Doe, John Ma|Doe, Mary Jo Ma|Smith, MD PhD Ma)$" fields = ["_ambiguities"] orders = ["DEFAULT"] @@ -1283,11 +1290,15 @@ issue = "fix(#544) a run of ambiguous members after a suffix comma reads by the # LISTED member only -- so its run flips too. 1.4.0 read the listed- # member names as suffixes as well. # +# 2026-09-28: 'John Smith, PhD Ma' joins, the same run with an +# unambiguous word in front of the member, which the count reads whole +# like the rest. +# # Literal; `fields` is the union the names move ('title' is the duals'). # Probes: 'Smith, Ed Ma' (one name word before the comma keeps the # listing form), 'Smith, Ms Ma' (a dual opening the given part is a # title) and the superstring 'Dr. John Smith, Ed Ma' are _MUST_NOT_MATCH. -name_regex = "^(?:John Smith, Ed Ma|John Smith, Ms Ma|John Smith, X\\.Y\\.Z\\. MA|john smith, md ma)$" +name_regex = "^(?:John Smith, Ed Ma|John Smith, Ms Ma|John Smith, PhD Ma|John Smith, X\\.Y\\.Z\\. MA|john smith, md ma)$" fields = ["given", "middle", "family", "suffix", "title", "_ambiguities"] orders = ["DEFAULT"] @@ -1304,10 +1315,15 @@ issue = "fix(#544) an unambiguous credential in front anchors the member behind # writes as the writer spaced it (#436/#437), so one rule explains the # whole of each diff. # +# 2026-09-28: 'Smith, PhD MEng' joins. The one-word family comma leaves +# its part holding no name word once 'PhD' speaks for 'MEng', so the +# part reads wholly as the credential run -- the fields this baseline +# read -- and the pick the company made reports where it was made. +# # Literal. Probes: 'Wang Ma PhD' (the credential BEHIND the member speaks # for nothing) and the superstring 'Dr. John Smith PhD MEng' are # _MUST_NOT_MATCH. -name_regex = "^(?:Doe, Jane PhD MEng|Doe, Jane nee Smith PhD MEng|Jane Doe nee Smith PhD MEng|John Smith PhD MEng)$" +name_regex = "^(?:Doe, Jane PhD MEng|Doe, Jane nee Smith PhD MEng|Jane Doe nee Smith PhD MEng|John Smith PhD MEng|Smith, PhD MEng)$" fields = ["_ambiguities"] orders = ["DEFAULT"] @@ -1340,3 +1356,21 @@ issue = "fix(#540) a Title-case Lac behind a particle is the family name" name_regex = "^Nguyen Van Lac$" fields = ["family", "suffix", "_ambiguities"] orders = ["DEFAULT"] + +[[change]] +issue = "fix(#544) a credential in front anchors a member after a one-word family comma, and the pick reports" +# 'Smith, PhD Ma': family 'Smith', suffix 'PhD Ma', reporting the +# Title-case 'Ma', where this baseline read given 'PhD', middle 'Ma'. +# rules.md#S2's company clause: 'PhD' in front speaks for 'Ma', so the +# part after the one-word family comma holds no name word and reads +# wholly as the credential run, and "a member the company decides +# reports the fork as a counted pick does". 'Smith, PhD MEng', whose +# roles this baseline already read, is the anchor rule's above. +# +# Literal. Probes: 'Smith, PhD MA' (the capitals decide the member, +# nothing new to report), 'Smith, MD PhD Ma' (a dual in the leading +# title run: nothing in the part speaks) and the superstring 'Dr. +# Smith, PhD Ma' are _MUST_NOT_MATCH. +name_regex = "^Smith, PhD Ma$" +fields = ["given", "middle", "suffix", "_ambiguities"] +orders = ["DEFAULT"] From a83cb812f671062c0256d22dbbae3510a6b70db4 Mon Sep 17 00:00:00 2001 From: Derek Gulbranson Date: Mon, 28 Sep 2026 11:50:28 -0700 Subject: [PATCH 07/11] fix(#544): a part read wholly as suffixes reports no name reading of its words group's particle chain runs over every comma segment, and both of its emitters reported inside a tail segment, which assign reads wholly as suffixes: particle-or-given for a particle chained behind a word of both the title and particle vocabularies (every release from 2.0.0 through 2.3.0: 'John Smith, Jr., Freiherr von Richthofen' reports it on 'von', a suffix token), and suffix-or-name when the chain takes an ambiguous acronym into the name (unreleased; the C1 run rule made it reachable behind one comma, so 'John Smith, PhD Do Ma' carried the flip and a second report on 'Ma'). Each named a reading the parse never made. A family comma's tails were already silent. group now hands the chain no report list in a tail segment after either comma. The maiden channel is untouched: a tail's reader is NONE, so the maiden walk reports nothing there. Measured against the same tree with the old condition restored: no differential-corpus name moves (the gate is unchanged at all five baselines); over a comma grid of 11,049 texts x 3 orders, 360 parses lose one suffix-or-name report each, every one on a suffix-role token, with 0 field moves and 0 reports added. That grid holds no word of both the title and particle vocabularies, so the particle-or-given half is pinned by a case row ('John Smith, Jr., Freiherr von Richthofen'); a wider sweep over such words removes particle-or-given reports too, again only on suffix-role tokens and with no field move. Frame counts are unchanged for the reference list and the call_count band (370/407). rules.md#C2 states the boundary; two case rows pin it beside the rows the chain still reports on; decisions.md carries a dated C1 bullet with the recompute recipe; the release log gains a Fix bullet; AGENTS.md's decision-site note names the tail case. Co-Authored-By: Claude Opus 5.5 --- AGENTS.md | 2 +- docs/design/decisions.md | 1 + docs/design/rules.md | 21 +++++++++++++++++++-- docs/release_log.rst | 2 ++ nameparser/_pipeline/_group.py | 25 ++++++++++++++++--------- tests/v2/cases.py | 26 ++++++++++++++++++++++++++ 6 files changed, 65 insertions(+), 12 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index ea8c8abf..f21b0647 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -314,7 +314,7 @@ The 2.0 rewrite lands as underscore-private modules alongside the v1 code. These - **Method organization**, fixed section order in every class: fields + `__post_init__` validation → alternative constructors → dunders (construction/equality → protocol → operators) → properties → public methods by concern (access → editing → comparison → rendering delegates) → private helpers last, except a helper serving exactly one section may sit at that section's head. Sanctioned deviation, facade layer only: `HumanName` and the shim `Constants` organize by v1 concern groups (`# -- render defaults --`, `# -- config / parsing --`, `# -- fields --`, ..., dunders and pickle last) — the classes mirror v1's own surface and die in 3.0; the canonical order still binds every core type. - **Validation is eager and fail-loud**: every `raise` states the offending value, the expected form, and the fix. Exception taxonomy: wrong type — including wrong element type inside a collection, bare `str` where an iterable of strings is expected, or a `Mapping` where a plain iterable is expected — raises `TypeError`; well-typed but unacceptable values raise `ValueError`; failed enum lookups stay `ValueError` for any input (stdlib `EnumType` precedent). **When the message hands the reader code to paste, that code has to survive a type checker** — nameparser ships `py.typed`. #337's segmenterless warning offered `Policy(segment_scripts=())`, an `arg-type` error, because these fields are annotated with what they STORE rather than everything the constructor accepts. Prefer the `frozenset()` / `()` spellings in messages and docstrings, and pin the offered spelling in a test — the warning tests matched on `ja_segmenter` and never checked the actionable half of the message. **A warning emitted in `Parser.__post_init__` needs `parser_for` to re-emit it from its own frame** (the `catch_warnings(record=True)` block at its return): `__post_init__`'s `stacklevel` is sized for direct `Parser(...)` construction, and through `parser_for`'s extra frame the default one-line rendering attributes the warning to the library's own `return Parser(...)` — the exact call the message tells the user to change becomes invisible. No single stacklevel serves both entry points; a new construction warning gets the re-emission for free, but a new CONSTRUCTION SITE for `Parser` inside this package needs its own re-emission or its callers get library-attributed warnings (#337 review). - **Guard, hint, and emit for the WHOLE family, and parametrize the test over it**: a check added to one member of a set belongs on all of it, and the test must sweep the family, not one example. This session shipped `_reject_str_and_mapping` on `Policy` but not `PolicyPatch`, the bytes decode hint on three of five config entry points, and a regex-sync roster missing four of its copies — each a separate follow-up bug that a `{class} × {field} × {bad-value}` parametrization would have caught and a per-example test hid. When you find you're guarding member N, grep for the other members first. -- **Ambiguities are emitted at the DECISION site**: an `Ambiguity` records a fork the parse had to call, not a token that sits in an ambiguous vocabulary. Emit where the branch is taken — the trailing-suffix peel in `_assign`, the delimiter escape's follow-up in `classify` — never by scanning for a `vocab:*-ambiguous` tag. The same tagged token is a genuine fork in one position and unremarkable in another (`do` mid-name in "Joao da Silva do Amaral de Souza" chooses nothing). **A branch that runs but changes nothing is not a decision either** -- the prefix chain's `merge(k, j)` executes even when `j == k + 1`, folding a piece into itself, and keying on "the code got here" reported a fork for all ambiguous particles on "Do Van Jr." (`Dr.` when that was written, before #367 made a plain title transparent and put the shape out of the loop's reach entirely), where the particle stayed a lone leading name piece — the GIVEN name under the default order, the family name under `FAMILY_FIRST` — and `_assign` reported the same token again. Check that the branch actually claimed something (`j > k + 1`) before recording. Structure often settles the question before it arises, which is why `PARTICLE_OR_GIVEN` is not emitted on the `FAMILY_COMMA` path's WHOLLY-FAMILY read -- the comma fixed which piece is the family -- and `SUFFIX_OR_NAME` is not emitted for "Ma, Jack". Read that scope narrowly: the comma settles nothing about a particle trailing the given name, so P6's attachment in `post_rules` decides that fork on the same path and reports it (#405), in the kind naming the reading it OVERRODE, which is the reading assign made and not the word's vocabulary: `SUFFIX_OR_NAME` where assign had read the run as a post-nominal (`vd`, `mc`), else `PARTICLE_OR_GIVEN` where the run holds an ambiguous particle (`van`, and `do`, which is in the suffix vocabulary too but in its AMBIGUOUS half, so no credential reading was overridden), else silence. The decision site also has the token index and the detail text in hand, which the tag scan would have to reconstruct. **If a fork's two branches are taken in DIFFERENT stages, every one of them needs the emitter** -- `PARTICLE_OR_GIVEN` is decided in `_assign` when the ambiguous particle stays a lone leading piece, in `_group` when something shifts it off the name's leading piece and the prefix chain claims it, and in `post_rules` when P6's attachment takes a trailing particle into the family after a comma, so all three report; for two years only the first did. What can still do the shifting is narrow, and #367 is why: a plain title no longer can (`Dr. Van Johnson` reads as `Van Johnson` does and reports from `_assign`), so the `_group` emitter needs a word that is BOTH a title and a particle — measured, `TITLES ∩ particles_ambiguous` is `{freiherr, st}` in the default vocabulary (`do` left TITLES in #296's audit; decisions.md's Excluded block records the before and after), plus any overlap a caller's config creates — standing ahead of the chained particle as the LEADING NAME word. Titles may precede it, so `Dr. St van Johnson` reaches the emitter and `St van Johnson` does too; a given name may not, so `Jan Freiherr von Richthofen` does not reach it while `Freiherr von Richthofen` and `Dr. Freiherr von Richthofen` do. Two shapes that look like they should reach it and do NOT, both measured by stepping `STAGES` and watching where `ambiguities` grows: `Dr. Do van Johnson` and `Do St Johnson` report from `assign`, not `group`, because `do` is no longer a title and so stays the leading name piece assign reports on — a both-vocabulary word CHAINED (`Jan St Johnson`) reports nothing at all. When checking whether that emitter is dead, a both-vocabulary word in the leading name position is the thing to look for, and the answer is that it is not dead. The stage-ownership map in `tests/v2/pipeline/test_state.py` must list `ambiguities` for each such stage, and it passes vacuously until a case row exercises the path, so add the row too. Report BOTH directions of a two-way fork — "John Smith MA" (read as a suffix) and "Jack MA" (read as the family name) are equally guesses. Every kind needs a trigger in `tests/v2/test_contracts.py::_AMBIGUITY_TRIGGERS` (an explicit `None`, strict-xfail, while reserved), and case-table rows pin expected kinds exactly, so a new emitter shows up in both immediately. **Pin the decision, not the vocabulary**: the only titled-particle test used an UNAMBIGUOUS particle, so it walked the right code path and proved nothing about the branch under test -- two criticals passed 1539 tests. A row contrasting the two readings ("John Smith V" against "John Smith B") is what makes an emitter's absence meaningful. +- **Ambiguities are emitted at the DECISION site**: an `Ambiguity` records a fork the parse had to call, not a token that sits in an ambiguous vocabulary. Emit where the branch is taken — the trailing-suffix peel in `_assign`, the delimiter escape's follow-up in `classify` — never by scanning for a `vocab:*-ambiguous` tag. The same tagged token is a genuine fork in one position and unremarkable in another (`do` mid-name in "Joao da Silva do Amaral de Souza" chooses nothing). **A branch that runs but changes nothing is not a decision either** -- the prefix chain's `merge(k, j)` executes even when `j == k + 1`, folding a piece into itself, and keying on "the code got here" reported a fork for all ambiguous particles on "Do Van Jr." (`Dr.` when that was written, before #367 made a plain title transparent and put the shape out of the loop's reach entirely), where the particle stayed a lone leading name piece — the GIVEN name under the default order, the family name under `FAMILY_FIRST` — and `_assign` reported the same token again. Check that the branch actually claimed something (`j > k + 1`) before recording. Structure often settles the question before it arises, which is why `PARTICLE_OR_GIVEN` is not emitted on the `FAMILY_COMMA` path's WHOLLY-FAMILY read -- the comma fixed which piece is the family -- and `SUFFIX_OR_NAME` is not emitted for "Ma, Jack". A tail segment is the same case from the other side: assign reads it wholly as suffixes, so group's two chain emitters are handed no report list there, after either comma — `John Smith, Jr., Freiherr von Richthofen` still chains `von` and reports only `comma-structure` (rules.md#C2, 2026-09-28). Read that scope narrowly: the comma settles nothing about a particle trailing the given name, so P6's attachment in `post_rules` decides that fork on the same path and reports it (#405), in the kind naming the reading it OVERRODE, which is the reading assign made and not the word's vocabulary: `SUFFIX_OR_NAME` where assign had read the run as a post-nominal (`vd`, `mc`), else `PARTICLE_OR_GIVEN` where the run holds an ambiguous particle (`van`, and `do`, which is in the suffix vocabulary too but in its AMBIGUOUS half, so no credential reading was overridden), else silence. The decision site also has the token index and the detail text in hand, which the tag scan would have to reconstruct. **If a fork's two branches are taken in DIFFERENT stages, every one of them needs the emitter** -- `PARTICLE_OR_GIVEN` is decided in `_assign` when the ambiguous particle stays a lone leading piece, in `_group` when something shifts it off the name's leading piece and the prefix chain claims it, and in `post_rules` when P6's attachment takes a trailing particle into the family after a comma, so all three report; for two years only the first did. What can still do the shifting is narrow, and #367 is why: a plain title no longer can (`Dr. Van Johnson` reads as `Van Johnson` does and reports from `_assign`), so the `_group` emitter needs a word that is BOTH a title and a particle — measured, `TITLES ∩ particles_ambiguous` is `{freiherr, st}` in the default vocabulary (`do` left TITLES in #296's audit; decisions.md's Excluded block records the before and after), plus any overlap a caller's config creates — standing ahead of the chained particle as the LEADING NAME word. Titles may precede it, so `Dr. St van Johnson` reaches the emitter and `St van Johnson` does too; a given name may not, so `Jan Freiherr von Richthofen` does not reach it while `Freiherr von Richthofen` and `Dr. Freiherr von Richthofen` do. Two shapes that look like they should reach it and do NOT, both measured by stepping `STAGES` and watching where `ambiguities` grows: `Dr. Do van Johnson` and `Do St Johnson` report from `assign`, not `group`, because `do` is no longer a title and so stays the leading name piece assign reports on — a both-vocabulary word CHAINED (`Jan St Johnson`) reports nothing at all. When checking whether that emitter is dead, a both-vocabulary word in the leading name position is the thing to look for, and the answer is that it is not dead. The stage-ownership map in `tests/v2/pipeline/test_state.py` must list `ambiguities` for each such stage, and it passes vacuously until a case row exercises the path, so add the row too. Report BOTH directions of a two-way fork — "John Smith MA" (read as a suffix) and "Jack MA" (read as the family name) are equally guesses. Every kind needs a trigger in `tests/v2/test_contracts.py::_AMBIGUITY_TRIGGERS` (an explicit `None`, strict-xfail, while reserved), and case-table rows pin expected kinds exactly, so a new emitter shows up in both immediately. **Pin the decision, not the vocabulary**: the only titled-particle test used an UNAMBIGUOUS particle, so it walked the right code path and proved nothing about the branch under test -- two criticals passed 1539 tests. A row contrasting the two readings ("John Smith V" against "John Smith B") is what makes an emitter's absence meaningful. - **A kind is worth adding only if a reader would hesitate too**: the test is not "does the code take a branch" but whether a person reading that input would genuinely be unsure. "Smith, John V" reads as a middle initial to anyone -- the comma settles it -- so reporting it would be noise that teaches callers to ignore the field, which costs more than the missing report. Reachability of the second branch is necessary, not sufficient. Prefer leaving a fork silent and documenting the omission over emitting on input nobody finds ambiguous. - **Parser owns config-dependent conveniences**: `Parser.matches`/`Parser.capitalized`/`Parser.revise` exist because the `ParsedName` equivalents fall back to DEFAULT config for str/omitted arguments (documented loudly in both docstrings). `revise` harvests tokens from a full sub-parse of each replacement value (tags kept minus `FOLDED_TAG`, roles forced, the R1 entry pass `suffix_entries` re-run over the forced state so a suffix value's entries follow its own commas, ambiguities discarded); the merge tail is shared with `replace()` via `ParsedName._with_field_tokens`. `Parser.capitalized` delegates through `name.capitalized(self.lexicon)` specifically so `_parser` never imports `_render` — keep it that way. - **Per-word vocabulary fields warn on multi-word entries** (`_normset`/`_normpairs` via `_warn_dead_entry`, UserWarning, never a raise — see the given_name_titles Gotcha for why raising is wrong). `given_name_titles` is the one multi-word-matched field and is exempt; `_edit` passes `warn=False` (add() warns once via the new instance's `__post_init__`; remove() stores nothing). The default vocabulary and every locale pack must stay warning-free (`test_default_lexicon_builds_warning_free`, `test_pack_vocabulary_entries_are_single_words`). diff --git a/docs/design/decisions.md b/docs/design/decisions.md index 9c68562a..9d006ffc 100644 --- a/docs/design/decisions.md +++ b/docs/design/decisions.md @@ -868,6 +868,7 @@ Excluded (MAIDEN_MARKERS, per nameparser/config/maiden_markers.py): DECLINED, all four with the evidence: marking the boundary at the core-drop site (#437's MARK-DONT-STRIP shape) — `dropped` already holds the fact with its span, and a second recording of it is the duplication that mechanism exists to prevent, one level up; making the `"joined"` tag role-aware — within a piece it is role-blind and correct for every role, `Smith, Ph. D. Smith` giving `first_list == ['Ph. D.']`, and only the between-piece half was ever a suffix concept; a render-time span scan instead of the recorded tag — `_facade.__setstate__` and `ParsedName.replace()` synthesise span-less tokens, so an unpickled name has nothing to scan and the tag IS the entry structure a pickle carries; and fixing the no-comma path inside group's block as a third branch beside `tail` and `reading`, which is the shape #429 took. - 2026-09-06 #511 — a suffix value handed to `revise()` derives its entries from its own commas, by the rule a whole name uses. `Parser.revise` sub-parsed each value and forced every harvested token to the named role AFTER the sub-parse had run R1's entry pass, which keys on Role.SUFFIX; a bare 'MD PhD' reads there as a title and a family name, so the pass joined nothing and the field rendered 'MD, PhD'. The fix runs the same pass again: `revise` now sub-parses to a ParseState, forces the role on every non-dropped token, and calls `suffix_entries` — the pass, lifted unedited out of post_rules' tail into a function the two callers share (mechanisms.md#ONE-PREDICATE-PER-QUESTION) — over the forced state, then assembles and harvests as before. So a comma in the value parts two credentials and a space joins them. TWO SPELLINGS of the lifted pass, and the reason is the call budget: a first draft had post_rules call the state-in/state-out wrapper, and the second ParseState build cost three more calls per parse against the band tests/v2/test_benchmark.py holds (py3.11, 2026-09-06, tools/perf/call_count.py: 450 calls/name before the move, 451 with the in-place worker `_mark_suffix_entries` that post_rules now calls, 454 with the draft; the facade band tops at 455.9), so the worker writes in place as every other post rule does and the wrapper exists for the one caller with no token list of its own. NO `_run` HELPER for the same reason: a draft routed `parse()` through a private state builder shared with `revise`, and that cost one frame per parse on the hot path (py3.11, 2026-09-06: 415 calls/name against 414 without it, in a 402-418 band), so the four-line construction is spelled twice and `parse()` is untouched. MEASURED 2026-09-06 over the 1117 distinct names in `tools/differential/corpus*.jsonl`, comparing `revise(p, suffix=p.suffix).suffix` against `p.suffix` under the default parser for every name with a non-empty suffix (368 of the 1117): 38 differed before and 1 after; the differential gate is byte-identical at all four baselines, `revise` not being on the compare path. RECOMPUTE: parse each corpus name, skip an empty suffix, revise the parse with its own suffix, count the names whose suffix moved (the script is in the #511 issue body). THE ONE LEFT is '김민준씨, J.씨', and it is not entry structure: the whole-name parse keeps 'J.씨' one glued suffix token, suffix '씨, J.씨', while the sub-parse of the bare value '씨, J.씨' peels the honorific off the initial, so the revised field renders '씨, J. 씨' — right entries, the spurious comma of before ('씨, J., 씨') gone, one word read differently by the value's own parse than by the whole name's. That is the "classified ON ITS OWN" limit `revise`'s docstring has always recorded, and CJK honorific peeling is a W-rule question this change does not move; pinned in `test_revise_reads_a_glued_honorific_on_its_own`. SUPERSEDES the phd-merge acceptance of 'Ph., D.' on this path (its bullet says how). STALE TAGS, tried and backed out: a draft cleared the sub-parse's own "joined" with the forcing and re-derived it, because the pass only ADDS the tag and `revise(n, family="Jones MD PhD")` carries the sub-parse's between-piece suffix mark on 'PhD' onto a FAMILY token. Measured 2026-09-06 against `330ee55`, the clear also destroyed every WITHIN-piece mark on a non-suffix value — 'D.' of `revise(n, family="John Ph. D. Smith")` lost the merge mark the #436 bullet's DECLINED list calls role-blind and correct for every role — while on a ParsedName only the suffix string view reads "joined" (`_text_for`'s suffix_join gate) — the facade's `_list_for` heals it for every role, but no path puts a revised name into a HumanName, the v1 setters going through `replace()`, so wiring those setters onto `revise` is the change that would show the stale mark, as `last_list == ['Jones', 'MD PhD']` — `initials()` and every field string being identical at both trees for every shape measured. So the tags are kept minus FOLDED_TAG as before; the between-piece mark on a forced non-suffix role is a tag-only oddity that predates this change and stays, and for a suffix value nothing depends on the sub-parse's marks, every pair it joined sharing a bucket with no parting token and the pass setting it again; pinned in `test_revise_keeps_the_sub_parses_within_piece_mark`. Dropped tokens keep their role and tags through the forcing, because the pass filters them by index and assemble omits them. ONE MORE LIMIT, pinned in `test_revise_leaves_a_policy_delimiter_unparted_without_a_tail_segment`, and it is the "classified ON ITS OWN" limit again rather than a rule of revise's: a delimiter the policy names through `extra_suffix_delimiters` is dropped, and so parts entries, only on a segment after a comma that the reading of the words makes a tail, and a value with no comma of its own has none — under `Policy(extra_suffix_delimiters=frozenset({" - "}))` the whole name 'Doe, John, MD PhD - FACS' renders 'MD PhD, FACS' while `revise(n, suffix="MD PhD - FACS")` renders 'MD PhD - FACS', the dash surviving as a token and the forced role making it a suffix word; measured 2026-09-06, a value whose own words read with a tail segment does part ('John Doe, MD - FACS' revises to 'John Doe, MD, FACS') and one after a suffix comma does not ('MD, PhD - FACS' stays), which is how the value's words read as a name deciding it. The round-trip is unaffected, the whole-name view having already rendered that boundary as a comma. DECLINED: a list-valued `revise(p, suffix=["MD PhD", "FACS"])`, which widens the API and leaves the string path where it was; documenting the limit with pins alone, the docstring already recording it and #511's measurement showing it reachable on every space-joined run #436 produced; and a text-only read of the delimiter cores inside `revise`, a second rule for a case no round-trip reaches. Out of scope and left as is: `ParsedName.replace(suffix="MD, PhD")` renders 'MD,, PhD', `replace` whitespace-splitting by contract and the facade setters riding on it for v1 parity. - 2026-09-27 (Derek), #544 — THE NAME-WORD COUNT READS A RUN. The single-token rule this section records for the ambiguous class generalizes to a post-comma part of two or more words, every one suffix vocabulary or a class member (listed or by shape), at least one a member and none a single-letter roman numeral: behind two or more name words the part is the credential run, and the flip reports once over the whole part (`John Smith, Ed Ma`, `Jane Doe, MS LAc`). A title/suffix dual opening the part counts as suffix vocabulary there, and a run whose every member is listed and leans credential is left to the family-comma path, which already reads it whole. The forks and their measurements are the #544 entry under S2. +- 2026-09-28 (Derek), #544 — A PART READ WHOLLY AS SUFFIXES REPORTS NO NAME READING OF ITS WORDS. group's particle chain runs over every comma segment, and its two emitters — `particle-or-given` when a particle behind a word of both the title and the particle vocabulary chains (since 2.0.0, de264af1) and `suffix-or-name` when the chain takes an ambiguous acronym into the name (#289/#516, 59d8f38a, in no release) — reported inside a TAIL segment, which assign reads wholly as suffixes. Each such report named a reading the parse never made, against rules.md#A1's "A report names the reading the parse took". A family comma's tail was already silent, since group hands the chain no report list anywhere after a family comma; the suffix comma's tails were not. Measured on the released wheels: `John Smith, Jr., Freiherr von Richthofen` reports `particle-or-given` on 'von', a token in the suffix role, at 2.0.0, 2.1.0, 2.2.0 and 2.3.0; `John Smith, Jr., PhD van Ma` and `John Smith, Jr., PhD Do Ma` report it on 'van' and 'Do' at 2.0.0 and 2.1.0 only. The `suffix-or-name` half reached `John Smith, Jr., PhD Do Ma`, `John Smith, MA, PhD Do Ma` and `John Smith, Jr., PhD van Ma` on master (e10e83b4), and the run rule of the bullet above made it reachable behind ONE comma: `John Smith, PhD Do Ma` carried C1's flip and a second report on 'Ma'. FIXED by scope, not by a new test at the emitter: group passes the chain no report list in a tail segment either, so both emitters go quiet there together, and rules.md#C2 states the boundary for any part consumed wholly as suffixes. The maiden channel is a separate parameter and is untouched: a tail segment's reader is NONE, so the maiden walk reports nothing there to begin with. No mechanisms.md entry: this is AMBIGUITY-AT-THE-DECISION-SITE's own contract (a report fires only where the parse chose between live readings) applied to a stage whose reading a later stage overrides for the whole segment. MEASURED 2026-09-28, the tree against the same tree with `None if family_comma else ambiguities` restored in `group()` (the comparator), each parse recorded as its seven fields plus `(kind, [(token text, token role)])` per report, under all three name orders: 0 of the 1441 differential-corpus names move, so the gate has nothing to classify; over a comma grid — the prefixes `John Smith, `, `John Smith, Jr., ` and `Smith, John, ` times every run of one to three words drawn with repetition from {PhD, MA, Ma, Do, van, de, Jr, MEng, Ed, y, i}, each text as written, lowercased and uppercased, deduplicated to 11,049 texts — 360 parses (120 texts, every one of them mixed case) lose one `suffix-or-name` report apiece, every removed report on a token in the suffix role, 0 reports added, 0 field moves. 21 of the 120 texts carry one comma and are the run rule's reach; the other 99 carry two and moved the same way (297 parses) when the same one-line change was applied to master e10e83b4; the `Smith, John, ` prefix moves nothing, being a family comma. The grid holds no word of both the title and the particle vocabulary, so the `particle-or-given` half is witnessed by the case row alone. Pinned by the case rows `a_credential_run_after_the_comma_reports_no_chain_fork` and `a_part_past_the_second_reports_no_particle_fork`; the rows the chain still reports on outside a tail are `the_chain_reports_the_acronym_it_takes` and `titled_particle_chain_survives_a_title_that_is_also_a_particle`. ### T1 — separators, not joiners diff --git a/docs/design/rules.md b/docs/design/rules.md index 18673d43..4510c4d6 100644 --- a/docs/design/rules.md +++ b/docs/design/rules.md @@ -496,7 +496,7 @@ P2. Rationale: a particle is written as part of the surname it (#132's ask) has it as the surnames view rather than the family field. "Vincent van Gogh van Beethoven" → surnames="van Gogh van Beethoven" - history: decisions.md#P2 · interacts: P1, P4, H5, M2, S2 · implemented: nameparser/_pipeline/_group.py, nameparser/_pipeline/_post_rules.py + history: decisions.md#P2 · interacts: P1, P4, H5, M2, S2, C2 · implemented: nameparser/_pipeline/_group.py, nameparser/_pipeline/_post_rules.py P3. Rationale: connective words ("y", "of the") bind name words into one name part; but a single letter in a short name is more @@ -1812,6 +1812,13 @@ C2. Rationale: text beyond the recognized comma parts should be S2's other by-shape half, the unlisted all-caps word, does not reach here under its switch either: the shape a tail segment is recognized by is the dotted one alone. + A part the parse consumes wholly as suffixes raises no report + about reading a word of it as a name, whether it is the part + after a suffix comma (C1) or a part beyond the second: nothing + in it is read as one, so a particle chain run over it (P2) + reports neither a particle chained onto a name word nor an + acronym taken into the name. What such a part reports is its + own — C1's flip and this rule's flag. "John Smith, MD, Bart" → suffix="MD, Bart" "John Smith, MD,, Jr." → suffix="MD, Jr." · boundary "John Smith, MD, R.A.I." → suffix="MD, R.A.I." @@ -1821,7 +1828,17 @@ C2. Rationale: text beyond the recognized comma parts should be "Steven Hardman, MD, DO, DDS" → ambiguities=() "STEVEN HARDMAN, MD, DO, DDS" → ambiguities=("comma-structure",) · boundary "John Smith, MD, XYZ" unlisted_caps_suffixes-on → ambiguities=("comma-structure",) - history: decisions.md#C1, decisions.md#S2 · interacts: C1, S2, S3 · implemented: nameparser/_pipeline/_segment.py + Accepted: the no-name-reading clause carries no example line of + its own. What it moves is a report with no field beside it, and + an example line would enter the rules corpus for that report + alone; its executable witnesses are the case rows + a_credential_run_after_the_comma_reports_no_chain_fork and + a_part_past_the_second_reports_no_particle_fork in + tests/v2/cases.py, read beside the two rows the chain still + reports on outside a tail: + the_chain_reports_the_acronym_it_takes and + titled_particle_chain_survives_a_title_that_is_also_a_particle. + history: decisions.md#C1, decisions.md#S2 · interacts: C1, P2, S2, S3 · implemented: nameparser/_pipeline/_segment.py, nameparser/_pipeline/_group.py ## Name order (O) diff --git a/docs/release_log.rst b/docs/release_log.rst index b46cf9ef..04430c78 100644 --- a/docs/release_log.rst +++ b/docs/release_log.rst @@ -26,6 +26,8 @@ Release Log - **Fix a maiden marker's clause swallowing a trailing credential in silence.** ``HumanName("Jane Doe nee Smith MA")`` gives maiden ``Smith`` with suffix ``MA``, where 2.0 through 2.3 gave maiden ``Smith MA`` and said nothing; 1.4.0 read the ``MA`` as a suffix too. ``Doe, Jane nee Smith MA`` moves with it, and so do the one-case spellings ``JANE DOE NEE SMITH MA`` and ``jane doe nee smith ma``. The words a marker takes now end where a trailing credential begins, which is what the marker's other two stops -- a suffix word, a trailing roman numeral -- have always done. Until this release it was the last trailing position in the library where a word of the ambiguous credential class was read without a report, and it was order-sensitive besides: ``Jane Doe nee Smith MA PhD`` gave maiden ``Smith MA`` while ``Jane Doe nee Smith PhD MA`` gave maiden ``Smith``, so whether the word was read at all depended on which side of the unambiguous credential the writer put it. Both now give maiden ``Smith``, with suffix ``MA PhD`` and ``PhD MA``. The writing still decides, exactly as it does for the same word ending a name with no clause: ``Jane Doe nee Smith Ma`` keeps maiden ``Smith Ma``, and ``Jane Doe nee Yo-Yo Ma`` keeps a two-word birth surname whole. The one member of this class that is also a surname particle keeps the carve-out it has outside a clause -- ``Doe, Jane nee Smith DO`` gives suffix ``DO`` while ``Doe, Jane nee Smith do`` and ``Doe, Jane nee Smith Do`` keep maiden ``Smith do`` and ``Smith Do``, and the comma-less ``Jane Doe nee Smith do`` gives suffix ``do`` as ``John Doe do`` does. Either reading is now reported, and there is no third: a word the clause gives up reads as a post-nominal, or the clause keeps it and says so. A name word behind the credential ends its reach and stays silent -- ``Jane Doe nee MA Smith`` gives maiden ``MA Smith`` and reports nothing -- and this stop never takes the first word after the marker, whatever its writing says: ``Jane Doe nee MA`` keeps maiden ``MA`` and reports, the marker having announced a name where there would otherwise be none, and ``Jane Doe nee MA PhD`` keeps it too. That differs on purpose from what a certain post-nominal gets there, ``Jane Smith nee PhD`` and ``Jane Smith nee V`` leaving the marker standing as an ordinary word as before. Where no trailing rule reads the clause's tail nothing is decided and the clause keeps every word: ``Smith nee Jones MA, Jane`` and ``Smith, John, Jr nee Jones MA`` both keep maiden ``Jones MA``, unchanged and with no ``suffix-or-name`` report. A title written behind or in front of the credential usually does not change how the credential is read; where something in or ahead of the clause could take the title it can -- see the trailing-title bullet below. The clause also keeps a word it cannot promise a credential reading for, which is where three shapes that look like they should move do not. Where the part the word would land in holds no name of its own there is nothing to read it as a credential, so ``Doe, Dr. nee Smith MA`` and ``Jane Doe, Jr nee Smith MA`` both keep maiden ``Smith MA`` and report. Where a join would swallow it first the same applies, and it is the birth name that would lose the word: ``Berg, abdul nee Jones MA`` keeps maiden ``Jones MA`` rather than reading first ``abdul MA``, and ``Berg, Jane van der nee Smith DO`` keeps maiden ``Smith DO`` rather than letting the particle chain carry the ``DO`` into last ``van der DO Berg``. Each of those reads as 2.3.0 read it. Delimiters settle the question outright and always did: ``HumanName("Jane Doe (nee Smith MA)")`` keeps the whole span as the maiden name and reports nothing, the writer having drawn the boundary, while ``Jane Doe (nee Smith) MA`` gives suffix ``MA`` for the word left outside it. One name is a restoration rather than a change: ``John Smith nee Jones R.A.I.`` gives suffix ``R.A.I.`` again, as 2.3.0 read it, this unreleased cycle having moved it into the maiden name when the unlisted-dotted reading above took the word out of the certain-suffix class. See the ``M2`` and ``S2`` entries of ``docs/design/decisions.md`` (closes #533) + - **Fix a comma part read wholly as suffixes reporting a particle in it as chained onto a name.** ``parse("John Smith, Jr., Freiherr von Richthofen").ambiguities`` names ``comma-structure`` alone, where 2.0 through 2.3 also named ``particle-or-given`` for ``von`` -- a word the same parse had put in the suffix, so the report described a reading it never made. No field moves. A part the parser consumes as suffixes, after a suffix comma or past the second comma, reports what the part is and nothing about its words as names, and the particle chain's credential-acronym report this release adds (above) keeps the same bound: ``John Smith, PhD Do Ma`` reports the comma's decision once and not again for ``Ma``. See the 2026-09-28 bullet of the ``C1`` entry in ``docs/design/decisions.md`` + - **Fix a trailing title after a maiden marker being read as part of the maiden name.** ``HumanName("Jane Doe nee Smith Prof.")`` gives title ``Prof.`` with maiden ``Smith``, where 2.0 through 2.3 gave maiden ``Smith Prof.``, and ``Mary Smith née Jones Prof.`` moves the same way. A credential or roman numeral in front of the title is read as it is with the title absent, so the two spellings ``Jane Doe nee Smith MA Prof.`` and ``Jane Doe nee Smith Prof. MA`` now agree -- title ``Prof.``, maiden ``Smith``, suffix ``MA`` -- where 2.3.0 kept every word in the maiden name (``Smith MA Prof.`` and ``Smith Prof. MA``), and ``Jane Doe nee Smith V Prof.`` gives suffix ``V``. Where the clause keeps the word in front of the title they still differ, since a title cannot leave without the words behind it: ``Doe nee Smith ba Prof.`` gives title ``Prof.`` with maiden ``Smith ba``, while ``Doe nee Smith Prof. ba`` keeps maiden ``Smith Prof. ba`` as 2.3.0 kept all three words in both. They can also differ where something in or ahead of the clause could take the title -- a particle, a bound given name, or a name left with no name word: ``Jane Doe nee Smith do MA Prof.`` keeps maiden ``Smith do MA`` while ``Jane Doe nee Smith do Prof. MA`` gives maiden ``Smith do``, suffix ``MA``, both with title ``Prof.``. A period behind an ordinary surname that is also a title makes it one here as it does at the end of any name: ``Jane Doe nee Smith King.`` gives title ``King.`` with maiden ``Smith``, where 2.3.0 gave maiden ``Smith King.``. The given part after a family comma reads it the same way: ``Doe, Jane nee Smith MA Prof.`` gives title ``Prof.`` too. A title straight after the marker stays the maiden name, the marker having announced one: ``Jane Doe nee King.`` keeps maiden ``King.``, and ``Jane Doe nee Prof. Dr.`` keeps maiden ``Prof.`` and gives title ``Dr.``, where 2.3.0 gave maiden ``Prof. Dr.``. Where the title itself would end the clause, the clause keeps it wherever giving it up would put it in a name part, and each of these reads as 2.3.0 read it: before a family comma (``Doe nee Smith Prof., Jane`` keeps maiden ``Smith Prof.``), where no name word would be left in front of it (``Dr. nee Jones Smith Prof.``), and behind a particle whose chain would take it (``Jane van der Berg nee Smith Prof.``). Brackets around a clause ending in a period are dropped before any of this is read, as in 2.3.0, so ``Jane Doe (nee Smith Prof.)`` gives title ``Prof.`` too. One reading moves because of the name the title is now read against: ``abdul nee Smith Dr.`` gives title ``Dr.`` with last ``abdul``, as ``abdul Dr.`` does, where 2.3.0 gave first ``abdul`` with maiden ``Smith Dr.``. See the ``M2`` entry of ``docs/design/decisions.md`` (closes #535) - **Fix a roman numeral ending a maiden clause landing in the first or middle name.** ``HumanName("Berg, abdul nee Smith V")`` gives first ``abdul`` with maiden ``Smith V``, where 2.2 and 2.3 gave first ``abdul V`` -- a word of the birth name joined into the current given name -- and 2.0 and 2.1 gave first ``abdul nee``. ``Doe, Jane nee Smith V, PhD`` gives maiden ``Smith V``, as 2.0 and 2.1 did, where 2.2 and 2.3 gave middle ``V``: after a family comma a lone numeral ending the given part is a suffix only when no further comma part follows it, and the clause now asks that as the given part itself does. Without the credential tail the numeral still goes to the suffix (``Doe, Jane nee Smith V`` gives maiden ``Smith``, suffix ``V``). A numeral straight after the marker can leave the marker an ordinary word, as ``Jane Smith née V`` does, and where it does a title behind the numeral no longer changes that: ``Jane Doe nee V Prof.`` gives title ``Prof.``, suffix ``V``, middle ``Doe``, last ``nee``, where 2.3.0 gave maiden ``V Prof.``. Elsewhere the clause keeps it -- ``Dr. nee V`` keeps maiden ``V`` as before -- and ``Doe, Jane nee V, PhD`` now gives maiden ``V``, suffix ``PhD``, as 2.0 and 2.1 did, where 2.2 and 2.3 gave middle ``nee V`` -- the same further-comma-part condition as ``Doe, Jane nee Smith V, PhD`` above -- with ``Doe, Jane nee V, Jr.`` moving the same way. See the ``M2`` entry of ``docs/design/decisions.md`` (#535) diff --git a/nameparser/_pipeline/_group.py b/nameparser/_pipeline/_group.py index 0d6a8b41..a0bed881 100644 --- a/nameparser/_pipeline/_group.py +++ b/nameparser/_pipeline/_group.py @@ -1321,15 +1321,16 @@ def _group_segment(seg: tuple[int, ...], additional: int, # segment's structure, and a default would be this module guessing # what that caller already knows. group() passes `None` on the # first for the chain emitter after a family comma -- the comma - # fixed the family, so that fork is settled -- and #533's is not - # that fork: a credential ending a maiden clause is a question the + # fixed the family, so that fork is settled -- and in a tail + # segment, which assign reads wholly as suffixes. #533's fork is + # neither: a credential ending a maiden clause is a question the # comma settles nothing about, which is why the two channels are # two parameters. They are given the SAME list wherever nothing is - # suppressed, which is every segment that is NOT after a family - # comma; what the split buys is the other case, where `None` on - # the first must not reach the second -- a maiden channel - # defaulting to whatever the first was would let a caller passing - # `ambiguities=None` silence both (the review's finding). + # suppressed, which is every segment that is neither after a + # family comma nor a tail; what the split buys is the other case, + # where `None` on the first must not reach the second -- a maiden + # channel defaulting to whatever the first was would let a caller + # passing `ambiguities=None` silence both (the review's finding). def title(k: int) -> bool: return is_title_piece(pieces[k], ptags[k], tokens) @@ -2038,7 +2039,13 @@ def group(state: ParseState) -> ParseState: bound_join = BoundJoin.STRICT # Suppressed after a family comma for the same reason _assign # suppresses it there: the family name is already fixed, so - # there is no fork left to report. + # there is no fork left to report. Suppressed in a tail segment + # as well, after either comma: assign reads that segment + # wholly as suffixes, so a chain report there -- a particle + # chained onto a name piece, or an acronym taken into the name + # -- names a reading the parse never takes. rules.md#C2: "a part + # the parse consumes wholly as suffixes raises no report about + # reading a word of it as a name" tail = tail_start is not None and seg_idx >= tail_start seg_cores = cores if tail else frozenset() # #533: which rule reads what the maiden walk would leave, off @@ -2054,7 +2061,7 @@ def group(state: ParseState) -> ParseState: reader = TailReader.TRAILING pieces, ptags, taken = _group_segment( seg, additional, tokens, bound_join, - None if family_comma else ambiguities, + None if (family_comma or tail) else ambiguities, seg_cores, state.lexicon.given_name_titles, opens_the_name=(seg_idx == 0 and not family_comma), diff --git a/tests/v2/cases.py b/tests/v2/cases.py index 4cd29a3d..c7c2b5e9 100644 --- a/tests/v2/cases.py +++ b/tests/v2/cases.py @@ -1523,6 +1523,18 @@ def _check_cjk_shape_purity(self) -> None: "'first piece that is not a title' test broke: it " "skipped 'St'/'Do'/'Freiherr' and collapsed the " "untitled 'St John Smith' into one given name"), + Case("a_part_past_the_second_reports_no_particle_fork", + "John Smith, Jr., Freiherr von Richthofen", + {"given": "John", "family": "Smith", + "suffix": "Jr., Freiherr von Richthofen"}, + ambiguities=("comma-structure",), + notes="the row above as a part past the second comma, which " + "is consumed wholly as suffixes (C2): 'von' still " + "chains in group, but it ends a suffix, so the " + "particle-or-given report that 'von' was chained onto " + "a name piece named a reading the parse never makes. " + "Every 2.x release through 2.3.0 carried it; the " + "comma-structure flag is the part's own report"), Case("titled_ambiguous_particle_no_op_chain", "St Van Jr.", {"title": "St", "family": "Van", "suffix": "Jr."}, notes="the piece after the particle is a suffix, so the chain " @@ -2330,6 +2342,20 @@ def _check_cjk_shape_purity(self) -> None: "and the emitter's `j > k + 1` floor is what both have " "to clear", shape=1), + Case("a_credential_run_after_the_comma_reports_no_chain_fork", + "John Smith, PhD Do Ma", + {"given": "John", "family": "Smith", "suffix": "PhD Do Ma"}, + ambiguities=("suffix-or-name",), + notes="the row above behind a suffix comma: the part after it " + "is the credential run C1 flips to, and assign reads a " + "tail segment wholly as suffixes, so 'Ma' ends a suffix. " + "The chain still runs there ('Do' takes 'Ma') but may " + "not report taking it into the name, a reading the " + "parse never makes (rules.md#C2); the one report is " + "C1's flip over the whole part. Carried a second, " + "contradicting suffix-or-name on 'Ma' until 2026-09-28. " + "1.4.0 read the same fields; 2.3.0 read given 'PhD', " + "middle 'Do Ma', family 'John Smith'"), Case("the_chain_reports_the_by_shape_half_too", "John van der Berg X.Y.Z.", {"given": "John", "family": "van der Berg X.Y.Z."}, From 364cc337ad34dc4c6cf561b328c3ce8b4cfa8ca4 Mon Sep 17 00:00:00 2001 From: Derek Gulbranson Date: Mon, 28 Sep 2026 12:08:05 -0700 Subject: [PATCH 08/11] perf(#544): the C1 run test ends on a name word before the numeral test A comma name with two or more words on each side of the comma enters the run test, and its first name word ends the run; the numeral test stood ahead of the name-word test and cost that ordinary name a frame for nothing. The name-word test goes first -- a one-letter roman numeral is suffix vocabulary, so it never folds as a name word and the order cannot change a reading. 'Doe Smith, Jane Q.' and 'Garcia Lopez, Maria Jose' now cost 318 and 331 frames, one over e10e83b4 (run_word_fold's own call), where they cost 320 and 333; the run_word_fold docstring and decisions.md's #544 frame line say so. A comment in segment claimed three runs are credential runs "at every policy"; with unlisted_dotted_suffixes off 'X.Y.Z.' is a name word, so it now carries the qualifier the decisions entry already states. Co-Authored-By: Claude Opus 5.5 --- docs/design/decisions.md | 2 +- nameparser/_pipeline/_segment.py | 11 ++++++++--- nameparser/_pipeline/_vocab.py | 26 ++++++++++++-------------- 3 files changed, 21 insertions(+), 18 deletions(-) diff --git a/docs/design/decisions.md b/docs/design/decisions.md index 9d006ffc..d468acae 100644 --- a/docs/design/decisions.md +++ b/docs/design/decisions.md @@ -676,7 +676,7 @@ for n in ('Smith, John','Smith, XYZ'): print(n, calls_for(off.parse, n), calls_f ACCEPTED LIMITS, each keeping the member a name word, rules.md#S2's Accepted block and case rows: the merged `Ph. D.` is outside the no-comma walk (`John Smith Ph. D. MEng` family 'MEng'; the comma and given-slot spellings do anchor it, `Smith, Ph. D. MEng` and `Doe, Jane Ph. D. MEng` reading suffix 'Ph. D. MEng'), a title between the credential and the member breaks the run (`John Smith PhD Prof. Ma`), and a particle member P2 has chained before the peel is out of reach (`John Smith PhD Do Do` family 'Do Do'; `Smith, PhD Do Ma` given 'PhD', middle 'Do Ma'). THE TWO-INPUT CHECK. tests/v2/test_properties.py's M2 clause-agreement walk asks every parse it takes both directions of the company clause: a listed member behind a qualifying credential reads as one, and a member read as a credential because of the word in front has that word in the suffix too. Recorded negative controls, from that test's docstring: 48 failing parses with the anchor off, 30 with the walk's leading piece allowed to anchor, 0 on this tree; re-run 2026-09-28 after (9) and (10), unmoved. The walk holds no comma head that opens its part with a dual, so (10) is pinned by case rows and a unit test instead (`a_dual_opening_the_given_part_turns_the_anchor_off`, `a_dual_opening_the_given_part_silences_a_later_degree`, test_pieces' `test_a_dual_in_the_leading_title_run_turns_the_anchor_off`). REVERSED BY THIS ENTRY rather than edited: this section's 2026-09-15 ACCEPTED item (ii) (the bullet after this entry says so); the three "wrongly moved" reversals the 2026-09-14 paragraph THE RUN IS THE CAPS CLASS'S ALONE records — `John Smith, Ed Ma`, `John Smith, ma do` and `John Smith, X.Y.Z. A.B.` are credential runs now by C1's own count under every policy that keeps their words in the class (with `unlisted_dotted_suffixes` off, `X.Y.Z.` and `A.B.` are no class members and the third keeps the listing form), while that paragraph's narrowing of the CAPS run test stands; and #540's two pending questions (the `suffix-acronym-collisions` bullet of this date). - MEASURED 2026-09-28, the #544 tree against its parent e10e83b4, py3.11, after (9) and (10); the 2026-09-27 figures this paragraph first carried were taken before them and are superseded. Recompute: every run of one to three words from {PhD, MD, Jr, MA, Ma, MEng, M.Eng., LAc, Ed, Do, ba, MS, V, Prof.} behind each of `Wang {}`, `John Smith {}`, `John Quincy Smith {}`, `Dr. John Smith {}`, `John Smith, {}`, `John Quincy Smith, {}`, `Smith, {}`, `Doe, Jane {}`, `Jane Doe nee Smith {}`, `Doe, Jane nee Smith {}` and `Jan van der Berg {}`, each written as given, lower-cased and upper-cased and deduped, parsed under the three name orders at otherwise default policy, the seven role fields and the ambiguity kinds compared against the same grid parsed by e10e83b4. THE DETECTOR, which both the invariants and the attribution read, looks only at the run's own words, compared case-free: {phd, md, jr, ms, m.eng.} are the unambiguous credentials, {ma, meng, lac, ed, do, ba} the members; a member is ANCHORED where an unambiguous credential stands in front of it in the run with nothing but members between, except that `ms` or `md` opening a comma part anchors nothing; and a run word is IN A NAME FIELD where it appears among the space-split words of given, middle, family or maiden more often than the run's unconstrained occurrences of it account for. The two invariants count parses with an anchored member in a name field, and with an unambiguous credential in one. THE ATTRIBUTION puts each moved parse in ONE class, the first that holds, in this order: THE ANCHOR, where the tree's two violation counts are each no higher than the parent's and one is lower; THE CHUNKED `M.Eng.`, where the run holds `M.Eng.`; A LONE `v`, where re-running the first test with `v` counted as an unambiguous credential makes it hold; THE C1 RUN, a comma shape with two or more name words before the comma whose run is only credentials and members, at least one a member; and anything left unattributed. 254,496 parses over 84,832 texts; 89,825 move (29,943 texts, 19,123 of them with a field move): 39,261 the anchor, 14,676 the C1 run, 35,696 the chunked `M.Eng.` (32,472 of those only losing the report it raised as a pick), 180 a lone `v` (`john smith, v phd ma`, suffix 'v phd ma'), and 12 unattributed, the `Smith, MA PhD ba` parses above. Clean-to-clean — no run word in a name field on either side — 23,454 move, every one `M.Eng.`'s dropped report, and none elsewhere. The two invariants, 0 NEW violations of either: an anchored member in a name field, 29,199 → 1,530 (the residue is a particle member P2 has chained, a dual inside a leading title run, `smith, prof. md ma`, and the member behind a credential in a part a dual opens, `Smith, MD PhD Ma`, which (10) keeps a name); an unambiguous credential in a name field, 36,675 → 4,734 (the residue is an interior single-letter `V`, a non-final `Prof.`, the `PhD` a comma part keeps as its given name when a `V` or a title follows it, and the same `PhD` behind a dual that opens the part, (10) again). Against the tree before (9) and (10), the same grid moves 5,076 parses: 4,536 (1,512 texts, every one a comma shape whose part reads wholly as credentials) gain a `suffix-or-name` report and move nothing else, and 540 (180 texts, every one `Smith, ` then `MD` or `MS` opening the run, in each case form) return to e10e83b4's reading, fields and reports alike, from a silent suffix reading of the whole part. The differential corpora at the parent (1,418 names, three orders) move four names, every one intended: `John Smith, PhD MEng`, `john smith, phd meng`, `Wang M.Eng.`, `abdul Smith Jr Ma`. Frames per parse through `Parser().parse`, py3.11: the reference band is unmoved (370/407, `uv run python tools/perf/call_count.py`), and so is every ordinary name measured (`Smith, John Quincy` 260, `Doe, John MA` 283, `Doe, Jane nee Smith PhD` 338); the new questions cost where they are asked — `John Smith Ma` 244 → 250, the anchor pass that finds nothing; `John Smith, MA Jr` 355 → 382, the case fact forced to see the capitals settle the run; `John Smith PhD MEng` 286 → 326; `Smith, MD PhD Ma` 349 → 362, the given slot's pass that (10) leaves with nothing to find — and `John Smith, PhD MEng` gets cheaper, 358 → 314, as does `Smith, PhD Ma`, 286 → 271. + MEASURED 2026-09-28, the #544 tree against its parent e10e83b4, py3.11, after (9) and (10); the 2026-09-27 figures this paragraph first carried were taken before them and are superseded. Recompute: every run of one to three words from {PhD, MD, Jr, MA, Ma, MEng, M.Eng., LAc, Ed, Do, ba, MS, V, Prof.} behind each of `Wang {}`, `John Smith {}`, `John Quincy Smith {}`, `Dr. John Smith {}`, `John Smith, {}`, `John Quincy Smith, {}`, `Smith, {}`, `Doe, Jane {}`, `Jane Doe nee Smith {}`, `Doe, Jane nee Smith {}` and `Jan van der Berg {}`, each written as given, lower-cased and upper-cased and deduped, parsed under the three name orders at otherwise default policy, the seven role fields and the ambiguity kinds compared against the same grid parsed by e10e83b4. THE DETECTOR, which both the invariants and the attribution read, looks only at the run's own words, compared case-free: {phd, md, jr, ms, m.eng.} are the unambiguous credentials, {ma, meng, lac, ed, do, ba} the members; a member is ANCHORED where an unambiguous credential stands in front of it in the run with nothing but members between, except that `ms` or `md` opening a comma part anchors nothing; and a run word is IN A NAME FIELD where it appears among the space-split words of given, middle, family or maiden more often than the run's unconstrained occurrences of it account for. The two invariants count parses with an anchored member in a name field, and with an unambiguous credential in one. THE ATTRIBUTION puts each moved parse in ONE class, the first that holds, in this order: THE ANCHOR, where the tree's two violation counts are each no higher than the parent's and one is lower; THE CHUNKED `M.Eng.`, where the run holds `M.Eng.`; A LONE `v`, where re-running the first test with `v` counted as an unambiguous credential makes it hold; THE C1 RUN, a comma shape with two or more name words before the comma whose run is only credentials and members, at least one a member; and anything left unattributed. 254,496 parses over 84,832 texts; 89,825 move (29,943 texts, 19,123 of them with a field move): 39,261 the anchor, 14,676 the C1 run, 35,696 the chunked `M.Eng.` (32,472 of those only losing the report it raised as a pick), 180 a lone `v` (`john smith, v phd ma`, suffix 'v phd ma'), and 12 unattributed, the `Smith, MA PhD ba` parses above. Clean-to-clean — no run word in a name field on either side — 23,454 move, every one `M.Eng.`'s dropped report, and none elsewhere. The two invariants, 0 NEW violations of either: an anchored member in a name field, 29,199 → 1,530 (the residue is a particle member P2 has chained, a dual inside a leading title run, `smith, prof. md ma`, and the member behind a credential in a part a dual opens, `Smith, MD PhD Ma`, which (10) keeps a name); an unambiguous credential in a name field, 36,675 → 4,734 (the residue is an interior single-letter `V`, a non-final `Prof.`, the `PhD` a comma part keeps as its given name when a `V` or a title follows it, and the same `PhD` behind a dual that opens the part, (10) again). Against the tree before (9) and (10), the same grid moves 5,076 parses: 4,536 (1,512 texts, every one a comma shape whose part reads wholly as credentials) gain a `suffix-or-name` report and move nothing else, and 540 (180 texts, every one `Smith, ` then `MD` or `MS` opening the run, in each case form) return to e10e83b4's reading, fields and reports alike, from a silent suffix reading of the whole part. The differential corpora at the parent (1,418 names, three orders) move four names, every one intended: `John Smith, PhD MEng`, `john smith, phd meng`, `Wang M.Eng.`, `abdul Smith Jr Ma`. Frames per parse through `Parser().parse`, py3.11: the reference band is unmoved (370/407, `uv run python tools/perf/call_count.py`), and so is every ordinary name measured (`Smith, John Quincy` 260, `Doe, John MA` 283, `Doe, Jane nee Smith PhD` 338) except a comma name with two or more words on each side of the comma, which enters the C1 run test and pays `run_word_fold`'s one call before its first name word ends it (`Doe Smith, Jane Q.` 317 → 318, `Garcia Lopez, Maria Jose` 330 → 331, measured 2026-09-28); the new questions cost where they are asked — `John Smith Ma` 244 → 250, the anchor pass that finds nothing; `John Smith, MA Jr` 355 → 382, the case fact forced to see the capitals settle the run; `John Smith PhD MEng` 286 → 326; `Smith, MD PhD Ma` 349 → 362, the given slot's pass that (10) leaves with nothing to find — and `John Smith, PhD MEng` gets cheaper, 358 → 314, as does `Smith, PhD Ma`, 286 → 271. NO MECHANISMS ENTRY IS OWED: the anchor pass is ONE-PREDICATE-PER-QUESTION's `_pieces.credential_anchors`, which `segment_suffix_reading` reads inline for the frame budget under a keep-in-step note, and its linearity is the answer #531's trailing floor and #397's `_run_neighbours` already give — one forward pass per question, never a look-behind per member. tests/v2/test_benchmark.py's `credential_run` shape guards it: the per-member look-behind measured 15.8× for 4× the input at base 800, against 4.13–4.21× on this tree at every base (that module's own record). - 2026-09-27 (#544) — THE 2026-09-15 ACCEPTED ITEM (ii) IS REVERSED, and the bullet stands as it landed. `abdul Smith Jr Ma` reads given 'abdul', family 'Smith', suffix 'Jr Ma' — 2.3.0's reading — because the unambiguous 'Jr' in front of the Title-case 'Ma' anchors it (the entry above): the peel takes both, and P5's reserve, now seeing the family the join would take, declines the join. Item (i) stands. The case row is now `a_credential_in_front_anchors_a_declined_pick`, and rules.md#S2's Accepted block names the shapes that still keep the company out of reach. - 2026-09-27 (Derek), #544 — THE #531 PAIRING GAINS A SECOND EXCEPTION, and CAPITALS DECIDE FOR `do` above stands as it landed. That bullet let only a positive credential lean override P6's attachment at the given slot; an unambiguous credential IN FRONT of the member now overrides it as well, the degree being a second and stronger signal: `doe, jane v phd do` reads suffix 'v phd do' and reports `suffix-or-name` where it read family 'do doe' and reported P6's fork. The one-case record the pairing protects has nothing in front of its particle, so `NASCIMENTO, EDSON ARANTES DO` still reads family 'DO NASCIMENTO' and reports `particle-or-given`. diff --git a/nameparser/_pipeline/_segment.py b/nameparser/_pipeline/_segment.py index 7e2012c5..f311e3a8 100644 --- a/nameparser/_pipeline/_segment.py +++ b/nameparser/_pipeline/_segment.py @@ -216,7 +216,9 @@ def class_run(seg: tuple[int, ...]) -> bool: # un-narrowed call per token moved `'John Smith, Ed Ma'`, `'John # Smith, ma do'` and `'John Smith, X.Y.Z. A.B.'` to a credential # run through the CAPS switch. Since #544 all three are credential - # runs at every policy, by the listed-and-dotted run test below and + # runs under every policy that keeps their words in the class + # (`unlisted_dotted_suffixes=False` leaves 'X.Y.Z.' a name word), + # by the listed-and-dotted run test below and # its name-word count -- a different question, asked with the # switch off too, so this narrowing still stands. # @@ -286,10 +288,13 @@ def class_run(seg: tuple[int, ...]) -> bool: settled = (settled and text.isupper() and (fold == "member" or ambiguous_class_member(text, lexicon))) - elif is_single_letter_numeral(text): - break + # "reject" first: a name word ends the run without the + # numeral test's frame, and the numeral cannot be a + # "reject" (it is suffix vocabulary, so it folds "defer") elif fold == "reject": break + elif is_single_letter_numeral(text): + break else: rest.append(text) else: diff --git a/nameparser/_pipeline/_vocab.py b/nameparser/_pipeline/_vocab.py index f9b23176..8a1fc85c 100644 --- a/nameparser/_pipeline/_vocab.py +++ b/nameparser/_pipeline/_vocab.py @@ -606,20 +606,18 @@ def run_word_fold( (its docstring carries a negative control: with one acceptance path dropped, the sweep fails). - Measured (2026-09-27, #544, by a - profiler-frame count over one parse), for an ORDINARY comma name - that enters the run loop and breaks on its very first token -- - 'Doe Smith, Jane Q.', 'Garcia Lopez, Maria Jose': asking - `ambiguous_class_candidate` of that token directly, with this - whole gate skipped, costs 325 and 338 frames; this named function - costs 319 and 332; the ORIGINAL hand-inlined gate (before it was - a named function at all) cost 317 and 330. So the gate itself - saves 8 frames against asking the predicate directly; making it a - named, testable call gives back 2 of those 8 (the call's own - frame); net 6 -- still the frame a comma name with no credential - in it pays less than it would with no gate at all, and the price - of a gate a test can reach on its own rather than one hand-copied - at the call site. + Measured (2026-09-28, #544, by a profiler-frame count over one + parse, py3.11), for an ORDINARY comma name that enters the run + loop and breaks on its very first token -- 'Doe Smith, Jane Q.', + 'Garcia Lopez, Maria Jose': the tree costs 318 and 331 frames + against e10e83b4's 317 and 330 (no run rule at all), the one + frame being this call's own; the loop tests "reject" before the + numeral so a name word pays nothing more. Asking + `ambiguous_class_candidate` of every token directly instead, with + this gate skipped, cost 325 and 338 when measured on 2026-09-27 + (the numeral test then stood ahead of "reject") -- the price of a + gate a test can reach on its own rather than one hand-copied at + the call site is that one frame. """ core = text.rstrip(".") if not (text.isascii() and "." not in core): From 080810cbef4d89a629d28403c802123a51dd753b Mon Sep 17 00:00:00 2001 From: Derek Gulbranson Date: Mon, 28 Sep 2026 13:41:34 -0700 Subject: [PATCH 09/11] fix(#544): a particle anchors nothing, and the anchor pass runs only where a suffix piece is in reach A word of both the particle and the unambiguous suffix vocabulary ('vd', 'mc') heads the family name behind it (rules.md#P2), so it speaks for no member of the ambiguous class behind it (rules.md#S2's company clause). Anchoring there split the family around a suffix: 'Smith vd Ma, John' read family 'Smith Ma', suffix 'vd', and 'Jan vd Ma' suffix 'vd Ma' with no family. Both, and 'Smith Mc Ma, John' and 'D. Mc Ba Ed, Smith', read as e10e83b4 reads them again. A particle member behind a credential is still spoken for ('doe, jane v phd do' keeps suffix 'v phd do'). segment_suffix_reading asks credential_anchors (first_kept=False) instead of carrying a hand-kept copy of it; a comma part holding a name word returns before any member is asked, so a plain comma name pays no frame for it. _pieces.anchor_in_reach answers, from tags alone, where the pass must answer False -- past the lone members in front, the first piece is no suffix piece -- and is asked only before the pass exists. An ordinary name ending in a declined member stops paying for the pass ('John Smith Ma' 250 -> 246 frames, 'Doe, John Q. Ma' 328 -> 320). On the no-comma peel its reach starts past the kept leading piece, which is what made the by-shape guard there unreachable; it is removed. A tail segment reads wholly as suffixes outside a maiden clause standing in it ('Jane Doe, PhD, Jr nee van Ma' keeps maiden 'van Ma'); rules.md#C2, AGENTS.md and _group's comments say so. The comma run's report names a run as holding a word that is also a name word. Case rows pin the particle exclusion and the two maiden-walk release checks' anchor thunks; each row fails with its thunk replaced by None. A frame-count test holds the run test's reject-before-numeral order. decisions.md's #544 entry gains a dated addendum with the measurements; the property test's second recorded control reads 42. Co-Authored-By: Claude Opus 5.5 --- AGENTS.md | 2 +- docs/design/decisions.md | 1 + docs/design/rules.md | 22 +++-- nameparser/_pipeline/_assign.py | 15 +++- nameparser/_pipeline/_group.py | 22 +++-- nameparser/_pipeline/_pieces.py | 146 ++++++++++++++++++++++--------- nameparser/_pipeline/_segment.py | 39 +++++---- nameparser/_pipeline/_vocab.py | 16 ++-- tests/v2/cases.py | 45 ++++++++++ tests/v2/pipeline/test_pieces.py | 54 +++++++++++- tests/v2/pipeline/test_vocab.py | 15 ++-- tests/v2/test_benchmark.py | 33 +++++-- tests/v2/test_ledger_guards.py | 8 +- tests/v2/test_properties.py | 44 ++++++---- 14 files changed, 334 insertions(+), 128 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index f21b0647..c32ec491 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -314,7 +314,7 @@ The 2.0 rewrite lands as underscore-private modules alongside the v1 code. These - **Method organization**, fixed section order in every class: fields + `__post_init__` validation → alternative constructors → dunders (construction/equality → protocol → operators) → properties → public methods by concern (access → editing → comparison → rendering delegates) → private helpers last, except a helper serving exactly one section may sit at that section's head. Sanctioned deviation, facade layer only: `HumanName` and the shim `Constants` organize by v1 concern groups (`# -- render defaults --`, `# -- config / parsing --`, `# -- fields --`, ..., dunders and pickle last) — the classes mirror v1's own surface and die in 3.0; the canonical order still binds every core type. - **Validation is eager and fail-loud**: every `raise` states the offending value, the expected form, and the fix. Exception taxonomy: wrong type — including wrong element type inside a collection, bare `str` where an iterable of strings is expected, or a `Mapping` where a plain iterable is expected — raises `TypeError`; well-typed but unacceptable values raise `ValueError`; failed enum lookups stay `ValueError` for any input (stdlib `EnumType` precedent). **When the message hands the reader code to paste, that code has to survive a type checker** — nameparser ships `py.typed`. #337's segmenterless warning offered `Policy(segment_scripts=())`, an `arg-type` error, because these fields are annotated with what they STORE rather than everything the constructor accepts. Prefer the `frozenset()` / `()` spellings in messages and docstrings, and pin the offered spelling in a test — the warning tests matched on `ja_segmenter` and never checked the actionable half of the message. **A warning emitted in `Parser.__post_init__` needs `parser_for` to re-emit it from its own frame** (the `catch_warnings(record=True)` block at its return): `__post_init__`'s `stacklevel` is sized for direct `Parser(...)` construction, and through `parser_for`'s extra frame the default one-line rendering attributes the warning to the library's own `return Parser(...)` — the exact call the message tells the user to change becomes invisible. No single stacklevel serves both entry points; a new construction warning gets the re-emission for free, but a new CONSTRUCTION SITE for `Parser` inside this package needs its own re-emission or its callers get library-attributed warnings (#337 review). - **Guard, hint, and emit for the WHOLE family, and parametrize the test over it**: a check added to one member of a set belongs on all of it, and the test must sweep the family, not one example. This session shipped `_reject_str_and_mapping` on `Policy` but not `PolicyPatch`, the bytes decode hint on three of five config entry points, and a regex-sync roster missing four of its copies — each a separate follow-up bug that a `{class} × {field} × {bad-value}` parametrization would have caught and a per-example test hid. When you find you're guarding member N, grep for the other members first. -- **Ambiguities are emitted at the DECISION site**: an `Ambiguity` records a fork the parse had to call, not a token that sits in an ambiguous vocabulary. Emit where the branch is taken — the trailing-suffix peel in `_assign`, the delimiter escape's follow-up in `classify` — never by scanning for a `vocab:*-ambiguous` tag. The same tagged token is a genuine fork in one position and unremarkable in another (`do` mid-name in "Joao da Silva do Amaral de Souza" chooses nothing). **A branch that runs but changes nothing is not a decision either** -- the prefix chain's `merge(k, j)` executes even when `j == k + 1`, folding a piece into itself, and keying on "the code got here" reported a fork for all ambiguous particles on "Do Van Jr." (`Dr.` when that was written, before #367 made a plain title transparent and put the shape out of the loop's reach entirely), where the particle stayed a lone leading name piece — the GIVEN name under the default order, the family name under `FAMILY_FIRST` — and `_assign` reported the same token again. Check that the branch actually claimed something (`j > k + 1`) before recording. Structure often settles the question before it arises, which is why `PARTICLE_OR_GIVEN` is not emitted on the `FAMILY_COMMA` path's WHOLLY-FAMILY read -- the comma fixed which piece is the family -- and `SUFFIX_OR_NAME` is not emitted for "Ma, Jack". A tail segment is the same case from the other side: assign reads it wholly as suffixes, so group's two chain emitters are handed no report list there, after either comma — `John Smith, Jr., Freiherr von Richthofen` still chains `von` and reports only `comma-structure` (rules.md#C2, 2026-09-28). Read that scope narrowly: the comma settles nothing about a particle trailing the given name, so P6's attachment in `post_rules` decides that fork on the same path and reports it (#405), in the kind naming the reading it OVERRODE, which is the reading assign made and not the word's vocabulary: `SUFFIX_OR_NAME` where assign had read the run as a post-nominal (`vd`, `mc`), else `PARTICLE_OR_GIVEN` where the run holds an ambiguous particle (`van`, and `do`, which is in the suffix vocabulary too but in its AMBIGUOUS half, so no credential reading was overridden), else silence. The decision site also has the token index and the detail text in hand, which the tag scan would have to reconstruct. **If a fork's two branches are taken in DIFFERENT stages, every one of them needs the emitter** -- `PARTICLE_OR_GIVEN` is decided in `_assign` when the ambiguous particle stays a lone leading piece, in `_group` when something shifts it off the name's leading piece and the prefix chain claims it, and in `post_rules` when P6's attachment takes a trailing particle into the family after a comma, so all three report; for two years only the first did. What can still do the shifting is narrow, and #367 is why: a plain title no longer can (`Dr. Van Johnson` reads as `Van Johnson` does and reports from `_assign`), so the `_group` emitter needs a word that is BOTH a title and a particle — measured, `TITLES ∩ particles_ambiguous` is `{freiherr, st}` in the default vocabulary (`do` left TITLES in #296's audit; decisions.md's Excluded block records the before and after), plus any overlap a caller's config creates — standing ahead of the chained particle as the LEADING NAME word. Titles may precede it, so `Dr. St van Johnson` reaches the emitter and `St van Johnson` does too; a given name may not, so `Jan Freiherr von Richthofen` does not reach it while `Freiherr von Richthofen` and `Dr. Freiherr von Richthofen` do. Two shapes that look like they should reach it and do NOT, both measured by stepping `STAGES` and watching where `ambiguities` grows: `Dr. Do van Johnson` and `Do St Johnson` report from `assign`, not `group`, because `do` is no longer a title and so stays the leading name piece assign reports on — a both-vocabulary word CHAINED (`Jan St Johnson`) reports nothing at all. When checking whether that emitter is dead, a both-vocabulary word in the leading name position is the thing to look for, and the answer is that it is not dead. The stage-ownership map in `tests/v2/pipeline/test_state.py` must list `ambiguities` for each such stage, and it passes vacuously until a case row exercises the path, so add the row too. Report BOTH directions of a two-way fork — "John Smith MA" (read as a suffix) and "Jack MA" (read as the family name) are equally guesses. Every kind needs a trigger in `tests/v2/test_contracts.py::_AMBIGUITY_TRIGGERS` (an explicit `None`, strict-xfail, while reserved), and case-table rows pin expected kinds exactly, so a new emitter shows up in both immediately. **Pin the decision, not the vocabulary**: the only titled-particle test used an UNAMBIGUOUS particle, so it walked the right code path and proved nothing about the branch under test -- two criticals passed 1539 tests. A row contrasting the two readings ("John Smith V" against "John Smith B") is what makes an emitter's absence meaningful. +- **Ambiguities are emitted at the DECISION site**: an `Ambiguity` records a fork the parse had to call, not a token that sits in an ambiguous vocabulary. Emit where the branch is taken — the trailing-suffix peel in `_assign`, the delimiter escape's follow-up in `classify` — never by scanning for a `vocab:*-ambiguous` tag. The same tagged token is a genuine fork in one position and unremarkable in another (`do` mid-name in "Joao da Silva do Amaral de Souza" chooses nothing). **A branch that runs but changes nothing is not a decision either** -- the prefix chain's `merge(k, j)` executes even when `j == k + 1`, folding a piece into itself, and keying on "the code got here" reported a fork for all ambiguous particles on "Do Van Jr." (`Dr.` when that was written, before #367 made a plain title transparent and put the shape out of the loop's reach entirely), where the particle stayed a lone leading name piece — the GIVEN name under the default order, the family name under `FAMILY_FIRST` — and `_assign` reported the same token again. Check that the branch actually claimed something (`j > k + 1`) before recording. Structure often settles the question before it arises, which is why `PARTICLE_OR_GIVEN` is not emitted on the `FAMILY_COMMA` path's WHOLLY-FAMILY read -- the comma fixed which piece is the family -- and `SUFFIX_OR_NAME` is not emitted for "Ma, Jack". A tail segment is the same case from the other side: assign reads it wholly as suffixes outside any maiden clause standing in it (`Jane Doe, PhD, Jr nee van Ma` keeps maiden `van Ma`), so group's two chain emitters are handed no report list there, after either comma — `John Smith, Jr., Freiherr von Richthofen` still chains `von` and reports only `comma-structure` (rules.md#C2, 2026-09-28). Read that scope narrowly: the comma settles nothing about a particle trailing the given name, so P6's attachment in `post_rules` decides that fork on the same path and reports it (#405), in the kind naming the reading it OVERRODE, which is the reading assign made and not the word's vocabulary: `SUFFIX_OR_NAME` where assign had read the run as a post-nominal (`vd`, `mc`), else `PARTICLE_OR_GIVEN` where the run holds an ambiguous particle (`van`, and `do`, which is in the suffix vocabulary too but in its AMBIGUOUS half, so no credential reading was overridden), else silence. The decision site also has the token index and the detail text in hand, which the tag scan would have to reconstruct. **If a fork's two branches are taken in DIFFERENT stages, every one of them needs the emitter** -- `PARTICLE_OR_GIVEN` is decided in `_assign` when the ambiguous particle stays a lone leading piece, in `_group` when something shifts it off the name's leading piece and the prefix chain claims it, and in `post_rules` when P6's attachment takes a trailing particle into the family after a comma, so all three report; for two years only the first did. What can still do the shifting is narrow, and #367 is why: a plain title no longer can (`Dr. Van Johnson` reads as `Van Johnson` does and reports from `_assign`), so the `_group` emitter needs a word that is BOTH a title and a particle — measured, `TITLES ∩ particles_ambiguous` is `{freiherr, st}` in the default vocabulary (`do` left TITLES in #296's audit; decisions.md's Excluded block records the before and after), plus any overlap a caller's config creates — standing ahead of the chained particle as the LEADING NAME word. Titles may precede it, so `Dr. St van Johnson` reaches the emitter and `St van Johnson` does too; a given name may not, so `Jan Freiherr von Richthofen` does not reach it while `Freiherr von Richthofen` and `Dr. Freiherr von Richthofen` do. Two shapes that look like they should reach it and do NOT, both measured by stepping `STAGES` and watching where `ambiguities` grows: `Dr. Do van Johnson` and `Do St Johnson` report from `assign`, not `group`, because `do` is no longer a title and so stays the leading name piece assign reports on — a both-vocabulary word CHAINED (`Jan St Johnson`) reports nothing at all. When checking whether that emitter is dead, a both-vocabulary word in the leading name position is the thing to look for, and the answer is that it is not dead. The stage-ownership map in `tests/v2/pipeline/test_state.py` must list `ambiguities` for each such stage, and it passes vacuously until a case row exercises the path, so add the row too. Report BOTH directions of a two-way fork — "John Smith MA" (read as a suffix) and "Jack MA" (read as the family name) are equally guesses. Every kind needs a trigger in `tests/v2/test_contracts.py::_AMBIGUITY_TRIGGERS` (an explicit `None`, strict-xfail, while reserved), and case-table rows pin expected kinds exactly, so a new emitter shows up in both immediately. **Pin the decision, not the vocabulary**: the only titled-particle test used an UNAMBIGUOUS particle, so it walked the right code path and proved nothing about the branch under test -- two criticals passed 1539 tests. A row contrasting the two readings ("John Smith V" against "John Smith B") is what makes an emitter's absence meaningful. - **A kind is worth adding only if a reader would hesitate too**: the test is not "does the code take a branch" but whether a person reading that input would genuinely be unsure. "Smith, John V" reads as a middle initial to anyone -- the comma settles it -- so reporting it would be noise that teaches callers to ignore the field, which costs more than the missing report. Reachability of the second branch is necessary, not sufficient. Prefer leaving a fork silent and documenting the omission over emitting on input nobody finds ambiguous. - **Parser owns config-dependent conveniences**: `Parser.matches`/`Parser.capitalized`/`Parser.revise` exist because the `ParsedName` equivalents fall back to DEFAULT config for str/omitted arguments (documented loudly in both docstrings). `revise` harvests tokens from a full sub-parse of each replacement value (tags kept minus `FOLDED_TAG`, roles forced, the R1 entry pass `suffix_entries` re-run over the forced state so a suffix value's entries follow its own commas, ambiguities discarded); the merge tail is shared with `replace()` via `ParsedName._with_field_tokens`. `Parser.capitalized` delegates through `name.capitalized(self.lexicon)` specifically so `_parser` never imports `_render` — keep it that way. - **Per-word vocabulary fields warn on multi-word entries** (`_normset`/`_normpairs` via `_warn_dead_entry`, UserWarning, never a raise — see the given_name_titles Gotcha for why raising is wrong). `given_name_titles` is the one multi-word-matched field and is exempt; `_edit` passes `warn=False` (add() warns once via the new instance's `__post_init__`; remove() stores nothing). The default vocabulary and every locale pack must stay warning-free (`test_default_lexicon_builds_warning_free`, `test_pack_vocabulary_entries_are_single_words`). diff --git a/docs/design/decisions.md b/docs/design/decisions.md index d468acae..e790ae85 100644 --- a/docs/design/decisions.md +++ b/docs/design/decisions.md @@ -678,6 +678,7 @@ for n in ('Smith, John','Smith, XYZ'): print(n, calls_for(off.parse, n), calls_f REVERSED BY THIS ENTRY rather than edited: this section's 2026-09-15 ACCEPTED item (ii) (the bullet after this entry says so); the three "wrongly moved" reversals the 2026-09-14 paragraph THE RUN IS THE CAPS CLASS'S ALONE records — `John Smith, Ed Ma`, `John Smith, ma do` and `John Smith, X.Y.Z. A.B.` are credential runs now by C1's own count under every policy that keeps their words in the class (with `unlisted_dotted_suffixes` off, `X.Y.Z.` and `A.B.` are no class members and the third keeps the listing form), while that paragraph's narrowing of the CAPS run test stands; and #540's two pending questions (the `suffix-acronym-collisions` bullet of this date). MEASURED 2026-09-28, the #544 tree against its parent e10e83b4, py3.11, after (9) and (10); the 2026-09-27 figures this paragraph first carried were taken before them and are superseded. Recompute: every run of one to three words from {PhD, MD, Jr, MA, Ma, MEng, M.Eng., LAc, Ed, Do, ba, MS, V, Prof.} behind each of `Wang {}`, `John Smith {}`, `John Quincy Smith {}`, `Dr. John Smith {}`, `John Smith, {}`, `John Quincy Smith, {}`, `Smith, {}`, `Doe, Jane {}`, `Jane Doe nee Smith {}`, `Doe, Jane nee Smith {}` and `Jan van der Berg {}`, each written as given, lower-cased and upper-cased and deduped, parsed under the three name orders at otherwise default policy, the seven role fields and the ambiguity kinds compared against the same grid parsed by e10e83b4. THE DETECTOR, which both the invariants and the attribution read, looks only at the run's own words, compared case-free: {phd, md, jr, ms, m.eng.} are the unambiguous credentials, {ma, meng, lac, ed, do, ba} the members; a member is ANCHORED where an unambiguous credential stands in front of it in the run with nothing but members between, except that `ms` or `md` opening a comma part anchors nothing; and a run word is IN A NAME FIELD where it appears among the space-split words of given, middle, family or maiden more often than the run's unconstrained occurrences of it account for. The two invariants count parses with an anchored member in a name field, and with an unambiguous credential in one. THE ATTRIBUTION puts each moved parse in ONE class, the first that holds, in this order: THE ANCHOR, where the tree's two violation counts are each no higher than the parent's and one is lower; THE CHUNKED `M.Eng.`, where the run holds `M.Eng.`; A LONE `v`, where re-running the first test with `v` counted as an unambiguous credential makes it hold; THE C1 RUN, a comma shape with two or more name words before the comma whose run is only credentials and members, at least one a member; and anything left unattributed. 254,496 parses over 84,832 texts; 89,825 move (29,943 texts, 19,123 of them with a field move): 39,261 the anchor, 14,676 the C1 run, 35,696 the chunked `M.Eng.` (32,472 of those only losing the report it raised as a pick), 180 a lone `v` (`john smith, v phd ma`, suffix 'v phd ma'), and 12 unattributed, the `Smith, MA PhD ba` parses above. Clean-to-clean — no run word in a name field on either side — 23,454 move, every one `M.Eng.`'s dropped report, and none elsewhere. The two invariants, 0 NEW violations of either: an anchored member in a name field, 29,199 → 1,530 (the residue is a particle member P2 has chained, a dual inside a leading title run, `smith, prof. md ma`, and the member behind a credential in a part a dual opens, `Smith, MD PhD Ma`, which (10) keeps a name); an unambiguous credential in a name field, 36,675 → 4,734 (the residue is an interior single-letter `V`, a non-final `Prof.`, the `PhD` a comma part keeps as its given name when a `V` or a title follows it, and the same `PhD` behind a dual that opens the part, (10) again). Against the tree before (9) and (10), the same grid moves 5,076 parses: 4,536 (1,512 texts, every one a comma shape whose part reads wholly as credentials) gain a `suffix-or-name` report and move nothing else, and 540 (180 texts, every one `Smith, ` then `MD` or `MS` opening the run, in each case form) return to e10e83b4's reading, fields and reports alike, from a silent suffix reading of the whole part. The differential corpora at the parent (1,418 names, three orders) move four names, every one intended: `John Smith, PhD MEng`, `john smith, phd meng`, `Wang M.Eng.`, `abdul Smith Jr Ma`. Frames per parse through `Parser().parse`, py3.11: the reference band is unmoved (370/407, `uv run python tools/perf/call_count.py`), and so is every ordinary name measured (`Smith, John Quincy` 260, `Doe, John MA` 283, `Doe, Jane nee Smith PhD` 338) except a comma name with two or more words on each side of the comma, which enters the C1 run test and pays `run_word_fold`'s one call before its first name word ends it (`Doe Smith, Jane Q.` 317 → 318, `Garcia Lopez, Maria Jose` 330 → 331, measured 2026-09-28); the new questions cost where they are asked — `John Smith Ma` 244 → 250, the anchor pass that finds nothing; `John Smith, MA Jr` 355 → 382, the case fact forced to see the capitals settle the run; `John Smith PhD MEng` 286 → 326; `Smith, MD PhD Ma` 349 → 362, the given slot's pass that (10) leaves with nothing to find — and `John Smith, PhD MEng` gets cheaper, 358 → 314, as does `Smith, PhD Ma`, 286 → 271. NO MECHANISMS ENTRY IS OWED: the anchor pass is ONE-PREDICATE-PER-QUESTION's `_pieces.credential_anchors`, which `segment_suffix_reading` reads inline for the frame budget under a keep-in-step note, and its linearity is the answer #531's trailing floor and #397's `_run_neighbours` already give — one forward pass per question, never a look-behind per member. tests/v2/test_benchmark.py's `credential_run` shape guards it: the per-member look-behind measured 15.8× for 4× the input at base 800, against 4.13–4.21× on this tree at every base (that module's own record). + ADDENDUM 2026-09-28 (Derek). (e) A PARTICLE ANCHORS NOTHING, a fifth boundary beside the four WHAT MAY ANCHOR lists: a word of both the particle and the unambiguous suffix vocabulary (`vd` and `mc` in the shipped lexicon; `do` sits in the ambiguous half, so it is a member and never an anchor) heads the family name behind it (P2), and anchoring there split the family around a suffix — `Smith vd Ma, John` read family 'Smith Ma', suffix 'vd'; `Jan vd Ma` suffix 'vd Ma' and no family; `Smith Mc Ma, John` and `D. Mc Ba Ed, Smith` the same way. Each reads as e10e83b4 reads it again (family 'Smith vd Ma', 'vd Ma', 'Smith Mc Ma', 'D. Mc Ba Ed'). The exclusion is of the word in front only: a particle MEMBER behind a credential is still spoken for (`doe, jane v phd do` keeps suffix 'v phd do'), and a C1 run holding the particle is C1's reading rather than the anchor's (`Jan Berg, vd Ma` reads suffix 'vd Ma', as `Jan Berg, vd` reads suffix 'vd'). Measured over the grid of MEASURED above with `vd` and `Mc` added to its words (127,578 texts, 382,734 parses), against the tree before this addendum: 9,426 parses (3,142 texts) move, every one holding `vd` or `Mc` directly in front of a member; 8,940 return to e10e83b4's reading exactly, and the other 486 all hold `M.Eng.`, whose own change is (3) — 406 of them take e10e83b4's fields and differ only in the pick `M.Eng.` no longer reports, and 80 take the fields e10e83b4 gives the same text with `PhD` in `M.Eng.`'s place (`Jane Doe nee Smith M.Eng. vd Ma` reads middle 'Doe M.Eng.', family 'vd Ma', as `Jane Doe nee Smith PhD vd Ma` reads middle 'Doe PhD'). The differential corpora move 0 of 1441 names, and the grid of MEASURED without the two words moves 0 parses, so every figure there stands. THE PASS IS ASKED ONLY WHERE A SUFFIX PIECE IS IN REACH: `_pieces.anchor_in_reach` walks back through the lone members in front of a declined member on tags alone and answers False wherever the pass must — the first piece past the members is no suffix piece, or there is none — and it is asked only before the pass exists, one lookup answering after that, so a run of members stays linear. An ordinary name ending in a declined member stops paying for the pass (frames, py3.11, before → after, e10e83b4 in brackets): `John Smith Ma` 250 → 246 (244), `Smith, John Ma` 285 → 279 (275), `Doe, Jane MA do` 374 → 369 (363), `Doe, John Q. Ma` 328 → 320 (316), `Jan vd Ma` 308 → 243 (238); a name the pass does read pays the test on top, `John Smith PhD MEng` 326 → 329 and `Doe, Jane PhD MEng` 359 → 362. On the no-comma peel the reach starts past the walk's leading piece, which never anchors — so a by-shape member, which reaches the pass only at the peel's second position (a lean of None with a word to spare is consumed first), finds nothing in reach, and the by-shape guard that stood there was deleted as unreachable. `segment_suffix_reading` calls `credential_anchors` (`first_kept=False`, a comma part's first piece being a run member like any other) in place of the inline copy NO MECHANISMS describes, and the keep-in-step note goes with it: a part holding a name word returns before any member is asked, so a plain comma name pays nothing (`Smith, John` 182, `Doe, John MA` 283, `Smith, J. Q.` 246, all unmoved), and a part the company decides pays two frames over the copy (`Smith, PhD MEng` 273 → 275, `Smith, PhD Ma` 271 → 273). THE TWO-INPUT CHECK's second control reads 42, not 30, with no change to the property: the walk's leading piece is now held out of the pass twice on the no-comma peel, by `credential_anchors` and by the reach test, either alone suffices, and with both removed the by-shape heads the deleted guard held back (`PhD X.Y.Z.`) fail beside the listed ones; the first control is unmoved at 48. And the 2026-09-28 C2 bullet's "a TAIL segment, which assign reads wholly as suffixes" holds outside a maiden clause standing in it — `Jane Doe, PhD, Jr nee van Ma` keeps maiden 'van Ma' — which rules.md#C2's statement now says. - 2026-09-27 (#544) — THE 2026-09-15 ACCEPTED ITEM (ii) IS REVERSED, and the bullet stands as it landed. `abdul Smith Jr Ma` reads given 'abdul', family 'Smith', suffix 'Jr Ma' — 2.3.0's reading — because the unambiguous 'Jr' in front of the Title-case 'Ma' anchors it (the entry above): the peel takes both, and P5's reserve, now seeing the family the join would take, declines the join. Item (i) stands. The case row is now `a_credential_in_front_anchors_a_declined_pick`, and rules.md#S2's Accepted block names the shapes that still keep the company out of reach. - 2026-09-27 (Derek), #544 — THE #531 PAIRING GAINS A SECOND EXCEPTION, and CAPITALS DECIDE FOR `do` above stands as it landed. That bullet let only a positive credential lean override P6's attachment at the given slot; an unambiguous credential IN FRONT of the member now overrides it as well, the degree being a second and stronger signal: `doe, jane v phd do` reads suffix 'v phd do' and reports `suffix-or-name` where it read family 'do doe' and reported P6's fork. The one-case record the pairing protects has nothing in front of its particle, so `NASCIMENTO, EDSON ARANTES DO` still reads family 'DO NASCIMENTO' and reports `particle-or-given`. diff --git a/docs/design/rules.md b/docs/design/rules.md index 4510c4d6..a8818cad 100644 --- a/docs/design/rules.md +++ b/docs/design/rules.md @@ -1035,11 +1035,14 @@ S2. Rationale: generational suffixes and credentials are recognized on — reads as the credential whatever its writing and whatever the count, at every trailing slot this rule names: the degree in front says what the run is. Only a credential in FRONT speaks; one - behind the member says nothing about it. A connective (P3) and a - single-letter roman numeral, in any case — one letter, the shape - a bare middle initial is written in, where a multi-letter one - such as 'III' speaks — speak for nothing and end the run; so does - any name word. A word of both the title and the suffix vocabulary + behind the member says nothing about it. A connective (P3), a + particle (P2) — a word of both the particle and the suffix + vocabulary heads the family name behind it — and a single-letter + roman numeral, in any case — one letter, the shape a bare middle + initial is written in, where a multi-letter one such as 'III' + speaks — speak for nothing and end the run; so does any name + word. A particle of this class standing behind a credential is + still spoken for: the exclusion is of the word in front. A word of both the title and the suffix vocabulary standing in the given part's leading title run is a title there and speaks for nothing, and no credential behind it speaks for a word of this class in that part: the title reading makes the next @@ -1815,10 +1818,11 @@ C2. Rationale: text beyond the recognized comma parts should be A part the parse consumes wholly as suffixes raises no report about reading a word of it as a name, whether it is the part after a suffix comma (C1) or a part beyond the second: nothing - in it is read as one, so a particle chain run over it (P2) - reports neither a particle chained onto a name word nor an - acronym taken into the name. What such a part reports is its - own — C1's flip and this rule's flag. + in it is read as one outside a maiden clause standing in it + (M2), so a particle chain run over it (P2) reports neither a + particle chained onto a name word nor an acronym taken into the + name. What such a part reports is its own — C1's flip and this + rule's flag. "John Smith, MD, Bart" → suffix="MD, Bart" "John Smith, MD,, Jr." → suffix="MD, Jr." · boundary "John Smith, MD, R.A.I." → suffix="MD, R.A.I." diff --git a/nameparser/_pipeline/_assign.py b/nameparser/_pipeline/_assign.py index a2d172c6..5e346603 100644 --- a/nameparser/_pipeline/_assign.py +++ b/nameparser/_pipeline/_assign.py @@ -52,8 +52,9 @@ -- since #289 -- the FAMILY-COMMA path's own read of the first post-comma piece, -- since #531 -- the class member ENDING that path's given part, which the first-piece emitter could never reach, -and -- since #544 -- each member a credential in front of it made the -credential in a post-comma part read wholly as credentials. +and -- since #544 -- each member of a post-comma part read wholly as +credentials that was read as one only because a credential in front +of it anchors it. Further emitters of the same kind live in `_segment.py`, `_group.py` and `_post_rules.py`; they are not assign's and are not counted here. And @@ -73,7 +74,7 @@ effective_script, is_suffix_lenient, resolve_script_set, ) from nameparser._pipeline._pieces import ( - credential_anchors, credential_at_the_given_slot, + anchor_in_reach, credential_anchors, credential_at_the_given_slot, is_suffix_piece, leading_titles, peel_walk, segment_suffix_reading, tail_reading, trailing_titles, ) @@ -688,6 +689,14 @@ def previous_kept(m: int, titled: tuple[int, ...]) -> int: def anchored(m: int, titled: tuple[int, ...]) -> bool: memo = anchor_memo.get(titled) if memo is None: + # nothing in front the pass could read as an + # anchor: the ordinary 'Smith, John Ma' stops here. + # Asked only before the pass exists -- once it + # does it answers in one lookup, where the reach + # test walks back through the run + if not anchor_in_reach(range(m - 1, -1, -1), pieces, + ptags, tokens, titled): + return False # from past the leading title run: a title/suffix # dual opening the part is a TITLE there and # anchors nothing ('Smith, MD MA Ma') diff --git a/nameparser/_pipeline/_group.py b/nameparser/_pipeline/_group.py index a0bed881..5373f089 100644 --- a/nameparser/_pipeline/_group.py +++ b/nameparser/_pipeline/_group.py @@ -44,7 +44,7 @@ from nameparser._lexicon import _run_addresses_by_given from nameparser._pipeline._pieces import ( - credential_anchors, credential_at_the_given_slot, + anchor_in_reach, credential_anchors, credential_at_the_given_slot, is_leading_title, is_suffix_piece, is_title_piece, is_trailing_title_word, Peel, leading_titles, peel_trailing, peel_walk, tail_reading, @@ -421,6 +421,11 @@ def _release_reads_off(view: Sequence[Sequence[int]], def anchored_at(q: int) -> bool: if not anchor_cell: + # the reach test first, and only before the pass + # exists (`_pieces.anchor_in_reach`) + if not anchor_in_reach(range(q - 1, -1, -1), view, + view_tags, tokens): + return False lead = leading_titles(view, view_tags, tokens) anchor_cell.append([False] * lead + credential_anchors( range(lead, len(view)), view, view_tags, tokens)) @@ -866,7 +871,9 @@ def _maiden_take(pieces: Sequence[Sequence[int]], # past the leading title run. takes = credential_at_the_given_slot( tokens[head[0]], one_case, - lambda: credential_anchors( + lambda: anchor_in_reach( + range(at - 1, -1, -1), view, view_tags, tokens) + and credential_anchors( range(min(leading_titles(view, view_tags, tokens), at), at + 1), view, view_tags, tokens)[-1]) @@ -1322,7 +1329,8 @@ def _group_segment(seg: tuple[int, ...], additional: int, # what that caller already knows. group() passes `None` on the # first for the chain emitter after a family comma -- the comma # fixed the family, so that fork is settled -- and in a tail - # segment, which assign reads wholly as suffixes. #533's fork is + # segment, which assign reads wholly as suffixes outside a maiden + # clause standing in it. #533's fork is # neither: a credential ending a maiden clause is a question the # comma settles nothing about, which is why the two channels are # two parameters. They are given the SAME list wherever nothing is @@ -1330,7 +1338,7 @@ def _group_segment(seg: tuple[int, ...], additional: int, # family comma nor a tail; what the split buys is the other case, # where `None` on the first must not reach the second -- a maiden # channel defaulting to whatever the first was would let a caller - # passing `ambiguities=None` silence both (the review's finding). + # passing `ambiguities=None` silence both. def title(k: int) -> bool: return is_title_piece(pieces[k], ptags[k], tokens) @@ -2040,8 +2048,10 @@ def group(state: ParseState) -> ParseState: # Suppressed after a family comma for the same reason _assign # suppresses it there: the family name is already fixed, so # there is no fork left to report. Suppressed in a tail segment - # as well, after either comma: assign reads that segment - # wholly as suffixes, so a chain report there -- a particle + # as well, after either comma: assign reads that segment, + # outside a maiden clause standing in it ('Jane Doe, PhD, Jr + # nee van Ma' keeps maiden 'van Ma'), wholly as suffixes, so a + # chain report there -- a particle # chained onto a name piece, or an acronym taken into the name # -- names a reading the parse never takes. rules.md#C2: "a part # the parse consumes wholly as suffixes raises no report about diff --git a/nameparser/_pipeline/_pieces.py b/nameparser/_pipeline/_pieces.py index 347c14e8..88f807f2 100644 --- a/nameparser/_pipeline/_pieces.py +++ b/nameparser/_pipeline/_pieces.py @@ -52,7 +52,8 @@ """ from __future__ import annotations -from collections.abc import Callable, Mapping, Sequence, Set +from collections.abc import ( + Callable, Container, Iterable, Mapping, Sequence, Set) from typing import NamedTuple from nameparser._pipeline._state import ( @@ -287,18 +288,27 @@ def _numeral_behind_the_initial_veto(piece: Sequence[int], def _anchors(piece: Sequence[int], tokens: Sequence[WorkToken]) -> bool: """Whether a suffix piece may ANCHOR an ambiguous member behind it - (#544): not a connective, and not a single-letter roman numeral. + (#544): not a connective, not a particle, and not a single-letter + roman numeral. A connective anchors nothing because between two name words it is a link -- the generational 'i' is also Catalan's 'i' (rules.md#P3) - -- and 'Jane Doe nee Puig i Ma' must keep its clause. A - single-letter roman numeral, in any case ('V', 'v', 'I.'), anchors - nothing for its SHAPE: one letter is the shape a middle initial - is written in ('V' can be one), and S3 retired single-character - vocabulary matches for the same reason. Not because a numeral is - no credential -- a MULTI-letter one ('Jr', 'III') anchors like - any other suffix piece (`is_single_letter_numeral`). A title/suffix - DUAL ('ms', 'md', 'sr') does anchor, except in the given part's + -- and 'Jane Doe nee Puig i Ma' must keep its clause. A particle + anchors nothing for the same reason from the other side: it joins + the name word behind it (rules.md#P2), so a word that is both + particle and suffix vocabulary ('vd', 'mc') standing in front of + a member is the head of a family name, not a credential list -- + 'Smith vd Ma, John' keeps family 'Smith vd Ma' and 'Jan vd Ma' + family 'vd Ma'. A particle MEMBER is still anchored by a + credential in front of it ('doe, jane v phd do'); the exclusion is + of the anchor only. A single-letter roman numeral, in any case + ('V', 'v', 'I.'), anchors nothing for its SHAPE: one letter is + the shape a middle initial is written in ('V' can be one), and S3 + retired single-character vocabulary matches for the same reason. + Not because a numeral is no credential -- a multi-letter suffix + word ('III', 'Jr') anchors like any other suffix piece + (`is_single_letter_numeral`). A title/suffix DUAL ('ms', 'md', + 'sr') does anchor, except in the given part's leading title run, where it stands in title position ('Smith, Ms Ma' is Ms. Ma Smith); that exclusion is the callers', which start their walks past the leading title run or, in @@ -312,6 +322,7 @@ def _anchors(piece: Sequence[int], tokens: Sequence[WorkToken]) -> bool: return True tok = tokens[piece[0]] return ("conjunction" not in tok.tags + and "particle" not in tok.tags and not is_single_letter_numeral(tok.text)) @@ -324,7 +335,8 @@ def _anchors(piece: Sequence[int], tokens: Sequence[WorkToken]) -> bool: def credential_anchors(order: Sequence[int], pieces: Sequence[Sequence[int]], ptags: Sequence[Set[str]], - tokens: Sequence[WorkToken]) -> list[bool]: + tokens: Sequence[WorkToken], + first_kept: bool = True) -> list[bool]: """For each position of `order` (piece indices in text order), whether a lone listed member standing there is ANCHORED: the contiguous run of pieces in FRONT of it that are suffix pieces or @@ -335,18 +347,21 @@ def credential_anchors(order: Sequence[int], The caller decides where the walk starts, and starts it past the leading title run, where a title/suffix dual reads as a title - ('Smith, MD MA Ma' keeps its middle name). `order`'s own leading - position is always the name the reserve keeps -- `rest[0]` at the - no-comma peel, the given part's own first piece at the other call - sites -- so ITS writing never anchors what stands behind it, - whatever vocabulary it carries: 'PhD Ma' keeps its parent reading, - family 'Ma', rather than reading as a credential list headed by - the given name itself.""" + ('Smith, MD MA Ma' keeps its middle name). Where `order` holds the + name part, its own leading position is the name the reserve keeps + -- `rest[0]` at the no-comma peel, the given part's own first + piece at the given-slot call sites -- so ITS writing never anchors + what stands behind it, whatever vocabulary it carries: 'PhD Ma' + keeps the count's reading, family 'Ma', rather than reading as a + credential list headed by the given name itself. `first_kept` + False is the one caller with no reserve, `segment_suffix_reading`, + asking whether a comma part is a credential run at all: there the + first piece is a run member like any other ('Smith, PhD MEng').""" out: list[bool] = [] anchor = False for pos, idx in enumerate(order): out.append(anchor) - if pos == 0: + if pos == 0 and first_kept: continue piece = pieces[idx] if is_suffix_piece(piece, ptags[idx], tokens): @@ -360,6 +375,44 @@ def credential_anchors(order: Sequence[int], return out +def anchor_in_reach(back: Iterable[int], + pieces: Sequence[Sequence[int]], + ptags: Sequence[Set[str]], + tokens: Sequence[WorkToken], + skip: Container[int] = ()) -> bool: + """False where `credential_anchors` must answer False for the + member standing just behind `back` -- the piece indices in front + of it, nearest first, `skip` spliced out: past the lone members + in front, the first other piece is no suffix piece, or there is + none. True means only "ask the pass". + + Exact in that direction because the test is weaker than the + pass's: a suffix piece carries "suffix" in its ptags or is one + token carrying "vocab:suffix" (`is_suffix_piece`), so a piece + failing both can neither anchor nor carry a run on. Read from tags + alone, with no frame per piece, so the ordinary name ending in a + Title-case member ('John Smith Ma', 'Doe, John Q. Ma') spends one + frame here rather than the pass's one per piece. Callers pass the + positions the pass itself would read or a superset reaching + further back: the walk meets any anchor the pass would find before + it reaches past the pass's start, so reaching further can cost a + pass that finds nothing and never a reading.""" + for q in back: + if q in skip: + continue + piece = pieces[q] + if "suffix" in ptags[q]: + return True + if len(piece) != 1: + return False + tags = tokens[piece[0]].tags + if "vocab:suffix" in tags: + return True + if AMBIGUOUS_ACRONYM_TAG not in tags: + return False + return False + + def segment_suffix_reading(pieces: Sequence[Sequence[int]], ptags: Sequence[Set[str]], tokens: Sequence[WorkToken], @@ -444,13 +497,12 @@ class by SHAPE takes the count instead, which is decided at the if not pieces: return None out: list[bool] = [] - # #544: `credential_anchors`' own reading, carried inline because - # this walk is already the forward pass it would make, and a call - # per family-comma segment 1 is a frame every comma name pays. The - # most recent suffix piece of the current run, kept through lone - # members and cleared by anything else; asked `_anchors` only when - # a member's writing has declined. Keep the two in step. - anchor: Sequence[int] | None = None + # #544: `credential_anchors` over the whole part, computed the + # first time a member's writing declines. Every piece in front of + # such a member has already been read as a suffix, a title, a + # member or a numeral -- a name word returns None first -- so a + # comma part holding a name word never pays for the pass. + anchors: list[bool] | None = None # Whether every piece so far stands in the part's leading title # run (titles, and title/suffix duals), and whether a dual has # stood there: once one has, nothing in the part anchors. @@ -468,20 +520,20 @@ class by SHAPE takes the count instead, which is decided at the dual_led = True else: leading = False - anchor = piece out.append(True) continue member = (len(piece) == 1 and AMBIGUOUS_ACRONYM_TAG in tokens[piece[0]].tags) - if not member: - anchor = None if member and listed_lean(tokens[piece[0]], one_case) \ == "credential": leading = False out.append(True) - elif (member and anchor is not None and not dual_led + elif (member and not dual_led and SHAPE_ACRONYM_TAG not in tokens[piece[0]].tags - and _anchors(anchor, tokens)): + and (anchors := anchors if anchors is not None + else credential_anchors( + range(len(pieces)), pieces, ptags, tokens, + first_kept=False))[len(out)]): leading = False if anchored is not None: anchored.append(len(out)) @@ -741,17 +793,27 @@ def peel_trailing(rest: Sequence[int], pieces: Sequence[Sequence[int]], # #544: or an unambiguous credential stands IN FRONT of it # in the same run -- 'John Smith PhD MEng' is a list of # degrees, not a family name 'MEng'. Computed once per - # walk, and only when a member has declined, so an ordinary - # name never pays for it; a LISTED member only, the - # by-shape half taking the count as above. Still a pick: - # the fork is reported as a counted member's is. - if SHAPE_ACRONYM_TAG not in tokens[piece[0]].tags: - if anchors is None: - anchors = credential_anchors(rest, pieces, ptags, - tokens) - if anchors[k - 1]: - k -= 1 - continue + # walk, and only when a member has declined and a suffix + # piece is in reach, so an ordinary name never pays for + # it. Still a pick: the fork is reported as a counted + # member's is. + # + # `rest[k - 2:0:-1]` is what stands in front of the member + # down to, not including, `rest[0]`, the name the reserve + # keeps, which never anchors (`credential_anchors`). So a + # member at k == 2 finds nothing in reach -- and that is + # every by-shape member that gets here, a lean of None + # with k >= 3 having been consumed above, so the by-shape + # half takes the count with no test of its own. + # The pass, once built, answers first: asked of every + # member, the reach test would walk back through the run + # each time, one look-behind per member. + if anchors is None and anchor_in_reach( + rest[k - 2:0:-1], pieces, ptags, tokens): + anchors = credential_anchors(rest, pieces, ptags, tokens) + if anchors is not None and anchors[k - 1]: + k -= 1 + continue break return Peel(k, numeral, tuple(picks)) diff --git a/nameparser/_pipeline/_segment.py b/nameparser/_pipeline/_segment.py index f311e3a8..4a46627a 100644 --- a/nameparser/_pipeline/_segment.py +++ b/nameparser/_pipeline/_segment.py @@ -34,14 +34,13 @@ stage's own lazy gate supplies. The #544 run test reads _vocab.run_word_fold, _vocab.ambiguous_class_candidate (its "ask" fallback), _vocab.ambiguous_class_member (the settled-lean check for -a fold-deferred member), _vocab.is_single_letter_numeral, -_vocab.ambiguous_lean and _vocab.is_wholly_suffix (once, over the -leftovers) -- six calls, not the two `ambiguous_class_candidate`/ -`is_wholly_suffix` a reader of the earlier, un-named fold might -expect. `run_word_fold` ANSWERS membership for a simple token -outright (its own docstring proves it agrees with -`ambiguous_class_candidate` there); it is a shortcut around a real -predicate, not a condition gating a second call to one. +a member admitted through that "ask" fallback), +_vocab.is_single_letter_numeral, _vocab.ambiguous_lean and +_vocab.is_wholly_suffix (once, over the leftovers). `run_word_fold` +ANSWERS membership for a simple token outright (its own docstring +proves it agrees with `ambiguous_class_candidate` there); it is a +shortcut around a real predicate, not a condition gating a second +call to one. Implements rules C1 and C2 of docs/design/rules.md, cited at the decision site below; history in decisions.md#C1. @@ -267,9 +266,9 @@ def class_run(seg: tuple[int, ...]) -> bool: # without calling them at all, and only a non-simple token, or one # `run_word_fold` defers, still takes the real predicate -- # `ambiguous_class_candidate` for membership, `is_wholly_suffix` - # once over the leftovers. A part holding a name word ('Doe, Jane - # Q. Public') breaks on that word's "reject" without entering a - # Python frame for either predicate. + # once over the leftovers. A part holding a name word ('Doe Smith, + # Jane Q. Public') breaks on that word's "reject" without entering + # a Python frame for either predicate. if not candidate and len(groups[1]) >= 2 and len(groups[0]) >= 2: members: list[str] = [] rest: list[str] = [] @@ -350,7 +349,7 @@ def class_run(seg: tuple[int, ...]) -> bool: # the attachment. rules.md#C1's comma-quiet policy gains its # exception for this class and no other. # - # `disp`/the index tuple cover the WHOLE post-comma part -- + # The quoted text and the index tuple cover the WHOLE post-comma part -- # the same text as before for the single-token listed and # dotted halves (a join of one element is that element), while # the caps half's run ('LEED AP') is the first time this class @@ -362,12 +361,20 @@ def class_run(seg: tuple[int, ...]) -> bool: # comma name (`John Smith, MA`, `John Smith, A.B.`, `Davis # Royce, Ed`) +2 frames at the DEFAULT policy, a path this # switch must not touch at all (#516 review round, F4). - disp = (state.tokens[groups[1][0]].text if len(groups[1]) == 1 - else " ".join(state.tokens[i].text for i in groups[1])) + # + # A RUN (#544) is named as holding such a word rather than + # being one: in 'John Smith, PhD MEng' only 'MEng' is also a + # name word. + if len(groups[1]) == 1: + what = (f"{state.tokens[groups[1][0]].text!r} after the " + f"comma is also an ordinary name word") + else: + what = (f"{' '.join(state.tokens[i].text for i in groups[1])!r}" + f" after the comma holds a word that is also an " + f"ordinary name word") ambiguities.append(PendingAmbiguity( AmbiguityKind.SUFFIX_OR_NAME, - f"{disp!r} after the comma is also an " - f"ordinary name word; the part before the comma holds " + f"{what}; the part before the comma holds " f"{pre_comma_names} name words, so it is read as a " f"credential run", groups[1])) diff --git a/nameparser/_pipeline/_vocab.py b/nameparser/_pipeline/_vocab.py index 8a1fc85c..3b46e034 100644 --- a/nameparser/_pipeline/_vocab.py +++ b/nameparser/_pipeline/_vocab.py @@ -210,8 +210,8 @@ def is_trailing_numeral_suffix(text: str, preceding: str) -> bool: # #544: a single-letter roman numeral ('V', 'v', 'I.'), in any case, # is excluded for its SHAPE -- one letter is how a middle initial is # written, and rules.md#S3 retires single-character matches for the -# same reason -- not for being a generation: a multi-letter numeral -# ('III', 'Jr') anchors and runs like any suffix word. So it neither +# same reason -- not for being a generation: a multi-letter suffix +# word ('III', 'Jr') anchors and runs like any other. So it neither # anchors an ambiguous member behind it nor counts toward C1's # multi-word credential run. It still peels exactly as before; this # answers only those two questions. @@ -550,8 +550,7 @@ class at all). # #544's inline gate for the comma-run test's common token, named and -# tested here rather than hand-copied at the call site (quality-review -# finding on the first cut): a SIMPLE token -- ASCII, no INTERIOR +# tested here rather than hand-copied at the call site: a SIMPLE token -- ASCII, no INTERIOR # period -- is fully resolved from the vocabulary sets directly, at # less cost than either real predicate it stands in for; a non-simple # token is left to them ("ask"). @@ -612,12 +611,9 @@ def run_word_fold( 'Garcia Lopez, Maria Jose': the tree costs 318 and 331 frames against e10e83b4's 317 and 330 (no run rule at all), the one frame being this call's own; the loop tests "reject" before the - numeral so a name word pays nothing more. Asking - `ambiguous_class_candidate` of every token directly instead, with - this gate skipped, cost 325 and 338 when measured on 2026-09-27 - (the numeral test then stood ahead of "reject") -- the price of a - gate a test can reach on its own rather than one hand-copied at - the call site is that one frame. + numeral so a name word pays nothing more. That one frame is the + price of a gate a test can reach on its own rather than one + hand-copied at the call site. """ core = text.rstrip(".") if not (text.isascii() and "." not in core): diff --git a/tests/v2/cases.py b/tests/v2/cases.py index c7c2b5e9..687554d5 100644 --- a/tests/v2/cases.py +++ b/tests/v2/cases.py @@ -933,6 +933,26 @@ def _check_cjk_shape_purity(self) -> None: "piece before the peel, so no lone member stands behind " "the degree. Unchanged by #544; 1.4.0 and 2.3.0 read " "the same"), + Case("a_particle_in_front_of_a_member_anchors_nothing", + "Jan vd Ma", + {"given": "Jan", "family": "vd Ma"}, + classification="fix(#289)", + ambiguities=("suffix-or-name",), + notes="S2's company clause: 'vd' is particle AND suffix " + "vocabulary, and a particle is the head of the family " + "name behind it rather than a credential that could " + "speak for 'Ma', so the Title-case member's own writing " + "decides and the chain keeps it. 2.3.0 read suffix 'vd " + "Ma', and 1.4.0 last 'vd', suffix 'Ma'"), + Case("a_particle_in_front_of_a_member_anchors_nothing_before_a_comma", + "Smith vd Ma, John", + {"given": "John", "family": "Smith vd Ma"}, + classification="fix(#289)", + notes="the same exclusion in the part before a family comma, " + "where a particle standing as the anchor split the " + "family around a suffix 'vd'. 2.3.0 read family 'Smith " + "Ma', suffix 'vd', and 1.4.0 last 'Smith', suffix 'vd, " + "Ma'"), Case("the_anchor_does_not_reach_across_a_maiden_clause", "Jane Doe Jr. nee Smith Ma", {"given": "Jane", "family": "Doe", "suffix": "Jr.", @@ -1019,6 +1039,31 @@ def _check_cjk_shape_purity(self) -> None: "read the same; #540 had read middle 'MEng', and 1.4.0 " "had no maiden routing", shape=2), + Case("the_clause_gives_up_a_member_a_later_suffix_anchors", + "Doe, Jane nee Smith MA Jr Ed", + {"given": "Jane", "family": "Doe", "suffix": "MA Jr Ed", + "maiden": "Smith"}, + classification="fix(#544)", + ambiguities=("suffix-or-name", "suffix-or-name"), + notes="the release check the maiden walk asks at the member " + "it stops on reads the span behind it the way assign's " + "given slot does, anchors included: 'Jr' speaks for the " + "Title-case 'Ed', so every word behind 'MA' reads as a " + "suffix and the clause gives 'MA' up. 2.3.0 read " + "middle 'Ed', suffix 'Jr', maiden 'Smith MA', and 1.4.0 " + "had no maiden routing"), + Case("the_clause_gives_up_a_title_a_later_suffix_anchors_past", + "Doe, Jane nee Smith Dr. Jr Ma", + {"title": "Dr.", "given": "Jane", "family": "Doe", + "suffix": "Jr Ma", "maiden": "Smith"}, + classification="fix(#544)", + ambiguities=("suffix-or-name",), + notes="the same release check at a title the walk stops on: " + "the given part's title chain runs from the end through " + "'Ma' only because 'Jr' anchors it, and so reaches " + "'Dr.', which the clause gives up. 2.3.0 read middle " + "'Ma', suffix 'Jr', maiden 'Smith Dr.', and 1.4.0 had " + "no maiden routing"), Case("a_dual_opening_the_given_part_is_a_title_there", "Smith, Ms Ma", {"title": "Ms", "given": "Ma", "family": "Smith"}, diff --git a/tests/v2/pipeline/test_pieces.py b/tests/v2/pipeline/test_pieces.py index 2b013e19..2e610699 100644 --- a/tests/v2/pipeline/test_pieces.py +++ b/tests/v2/pipeline/test_pieces.py @@ -17,7 +17,8 @@ from nameparser._pipeline._classify import classify from nameparser._pipeline._group import group from nameparser._pipeline._pieces import ( - _anchors, _numeral_behind_the_initial_veto, credential_anchors, + _anchors, _numeral_behind_the_initial_veto, anchor_in_reach, + credential_anchors, credential_at_the_given_slot, is_leading_title, leading_titles, own_words, peel_trailing, peel_walk, segment_suffix_reading, trailing_titles, @@ -389,9 +390,8 @@ def test_the_walks_own_leading_piece_never_anchors_what_follows_it( # anchor a member behind it, because that leading position is # always the name H4's carve-out keeps. Anchoring it let the run # collapse entirely: 'PhD Ma' read given 'PhD', suffix 'Ma', - # losing the family outright, rather than 'PhD Ma' keeping its - # parent reading (given 'PhD', family 'Ma', decided by the count - # alone, as it was before this commit existed). + # losing the family outright, rather than 'PhD Ma' keeping the + # reading the count alone gives it (given 'PhD', family 'Ma'). for text, family in (("Om Ma", "Ma"), ("PhD Ma", "Ma"), ("Jr Ma", "Ma")): rest, pieces, ptags, tokens = _peel_inputs(text) @@ -480,6 +480,52 @@ def test_a_connective_or_a_numeral_suffix_word_anchors_nothing() -> None: state = _through_group("John Smith v Ma") v = next(i for i, t in enumerate(state.tokens) if t.text == "v") assert not _anchors((v,), state.tokens) + # a particle that is also suffix vocabulary is the head of the + # family name behind it, not a credential: the same 'Jr' tagged a + # particle anchors nothing either + tokens[jr] = dataclasses.replace( + tokens[jr], tags=(tokens[jr].tags - {"conjunction"}) | {"particle"}) + assert not _anchors((jr,), tokens) + + +@pytest.mark.parametrize("text, fields", [ + ("Jan vd Ma", {"given": "Jan", "family": "vd Ma"}), + ("Smith vd Ma, John", {"given": "John", "family": "Smith vd Ma"}), + ("Smith Mc Ma, John", {"given": "John", "family": "Smith Mc Ma"}), + ("D. Mc Ba Ed, Smith", {"given": "Smith", "family": "D. Mc Ba Ed"}), +]) +def test_a_particle_in_suffix_vocabulary_anchors_nothing( + text: str, fields: dict[str, str]) -> None: + # #544: 'vd' and 'mc' are particle AND unambiguous suffix + # vocabulary; standing in front of a member they head the family + # name, and anchoring there split it around a suffix ('Smith vd + # Ma, John' read family 'Smith Ma', suffix 'vd'). A particle + # MEMBER is still anchored by a credential in front of it. + n = parse(text) + assert {k: v for k, v in n.as_dict().items() if v} == fields + assert parse("doe, jane v phd do").suffix == "v phd do" + + +def test_anchor_in_reach_is_false_only_where_the_pass_is() -> None: + # the reach test is a necessary condition for the pass: False + # where the first piece in front past the lone members is no + # suffix piece, True (ask the pass) otherwise -- including where + # the pass then answers False, a non-anchoring suffix piece + # ('v') or the kept leading piece being in reach + for text, expect in (("John Smith Ma", False), + ("John Smith Ed Ma", False), + ("John Smith PhD Ma", True), + ("John Smith PhD Ed Ma", True), + ("John Smith PhD v Ma", True)): + rest, pieces, ptags, tokens = _peel_inputs(text) + back = rest[len(rest) - 2::-1] + assert anchor_in_reach(back, pieces, ptags, tokens) is expect, text + if not expect: + assert not credential_anchors(rest, pieces, ptags, tokens)[-1] + # `skip` splices pieces out of the walk + rest, pieces, ptags, tokens = _peel_inputs("John Smith PhD Ma") + assert not anchor_in_reach(rest[2::-1], pieces, ptags, tokens, + skip={rest[2]}) # a merged split credential is ONE suffix piece of two tokens, and # it anchors as the unsplit spelling does wherever it is in the # walk -- after a comma ('John Smith Ph. D. MEng' is the no-comma diff --git a/tests/v2/pipeline/test_vocab.py b/tests/v2/pipeline/test_vocab.py index cfe220f1..248a1378 100644 --- a/tests/v2/pipeline/test_vocab.py +++ b/tests/v2/pipeline/test_vocab.py @@ -397,7 +397,8 @@ def test_run_word_fold_agrees_with_the_real_predicates() -> None: vocabulary plus a few name-word controls, in lower/Title/UPPER case, bare and with one trailing period, under both `Policy.lenient_comma_suffixes` settings and one delimiter-core - policy (11,880 built cases) -- is asked, with NO skip of its own: + policy (11,880 built cases, measured 2026-09-28) -- is asked, + with NO skip of its own: whichever verdict `run_word_fold` returns is checked against the real predicates, or, for "ask", left unchecked here and pinned instead by the fixed non-simple list below. "member" must agree @@ -411,11 +412,11 @@ def test_run_word_fold_agrees_with_the_real_predicates() -> None: 576 of the 11,880 built cases, every Hebrew/Devanagari/Bengali/ CJK honorific in the shipped vocabulary among them, times three policies) settles to "ask" here like any other non-simple token, - checked by nothing but its own verdict -- no longer skipped - before `run_word_fold` is even called, which is what let a drift - in the SIMPLE-token gate go unseen (11,304 of the 11,880 cases - are actually asserted against the real predicates; the rest are - "ask" and pinned only by the fixed list below). + checked by nothing but its own verdict: 11,304 of the 11,880 + cases are asserted against the real predicates, and the rest are + "ask" and pinned only by the fixed list below. Nothing is skipped + before `run_word_fold` is called, so a drift in the SIMPLE-token + gate cannot hide behind a filter. A fixed list of non-simple tokens pins "ask" directly, since the built sweep above never asserts it: an interior-period acronym @@ -882,7 +883,7 @@ def test_a_listed_member_written_in_period_closed_chunks_is_dotted( def test_is_single_letter_numeral() -> None: - """#544: the generation class C1's run and the anchor both leave + """#544: the one-letter shape C1's run and the anchor both leave out -- one letter, periods allowed, that is a roman numeral.""" for text in ("V", "v", "I", "I.", "x", "X."): assert is_single_letter_numeral(text), text diff --git a/tests/v2/test_benchmark.py b/tests/v2/test_benchmark.py index f6105823..8649e595 100644 --- a/tests/v2/test_benchmark.py +++ b/tests/v2/test_benchmark.py @@ -22,6 +22,7 @@ import sys import time from collections.abc import Callable +from types import FrameType import pytest @@ -238,8 +239,10 @@ def test_a_thousand_names_still_parse_in_reasonable_time( # its writing, so the peel asks # `credential_anchors` (#544), and every # 'Ma' stands behind a 'PhD' that anchors -# it: 39 of the 40 words of 'PhD Ma ' x20 -# read as the suffix. 'PhD MA ' reads the +# it: 38 of the 40 words of 'PhD Ma ' x20 +# read as the suffix, the reserve keeping +# the first 'PhD' and 'Ma' as given and +# family. 'PhD MA ' reads the # same and never asks the pass at all, the # capitals deciding each member first _SHAPES = { @@ -454,8 +457,9 @@ def test_policy_gated_cost_grows_no_worse_than_linearly( _RUN_HUGE_MAX_RATIO = 5.0 -def _frames_for(text: str) -> int: - """Python frame entries for ONE parse of `text`. +def _frames_for(text: str, only: str | None = None) -> int: + """Python frame entries for ONE parse of `text` -- of every + function, or of the one named `only`. One parse, not a mean: this measures growth between two inputs, and the count is deterministic for a given (tree, interpreter) -- see @@ -465,9 +469,10 @@ def _frames_for(text: str) -> int: parse("warm up the caches") calls = 0 - def counter(frame: object, event: str, arg: object) -> None: + def counter(frame: FrameType, event: str, arg: object) -> None: nonlocal calls - if event == "call": + if event == "call" and (only is None + or frame.f_code.co_name == only): calls += 1 sys.setprofile(counter) @@ -665,3 +670,19 @@ def test_a_link_costs_what_it_is_pinned_at() -> None: f"see it: check whether `_group._between_name_words` still " f"answers both sides of a link in one call, then move the " f"baseline deliberately (#397)") + + +def test_a_name_word_ends_the_comma_run_before_the_numeral_test() -> None: + """rules.md#C1's run test (#544) asks `run_word_fold`'s "reject" + before `is_single_letter_numeral`, so an ordinary comma name whose + part holds a name word pays no frame for the numeral test. The + fold is asked (the reachability probe: the name enters the run + loop at all, two words standing before the comma) and the numeral + test never is. RECORDED NEGATIVE CONTROL: with the two tests in + the other order, `is_single_letter_numeral` is entered once for + 'Doe Smith, Jane Q.' (measured 2026-09-28).""" + if sys.getprofile() is not None: + pytest.skip("a profile hook is already installed; this test owns it") + text = "Doe Smith, Jane Q." + assert _frames_for(text, only="run_word_fold") >= 1 + assert _frames_for(text, only="is_single_letter_numeral") == 0 diff --git a/tests/v2/test_ledger_guards.py b/tests/v2/test_ledger_guards.py index c42fedbc..1350256c 100644 --- a/tests/v2/test_ledger_guards.py +++ b/tests/v2/test_ledger_guards.py @@ -1218,8 +1218,7 @@ def test_case_shape_ids_exist_in_the_inventory() -> None: "John Smith Ph.D."), # #540's rules, keyed on the full issue since all of them carry # `fix(#540)` -- eight when written, six since #544 (2026-09-27) - # read the comma run and the chunked dotted spelling again, which - # retired the comma-cost rule the sentences below still name. + # read the comma run and the chunked dotted spelling again. # Each wall is the other rules' names plus the # spellings a case-blind widening would reach: 'meng li' leads # with the word and nothing moved; the one-case 'john smith meng' @@ -1227,9 +1226,8 @@ def test_case_shape_ids_exist_in_the_inventory() -> None: # no business with either. Its third probe, 'John Smith MENG', is # not one-case either -- MENG's own capitals keep the credential # on the capitals lean, in a name otherwise written Title-case. - # The report rule must not reach the names whose ROLES move, and - # the comma-cost rule must not reach the bare-word, no-comma or - # all-lower spellings. The lone-word comma rule must not reach + # The report rule must not reach the names whose ROLES move. The + # lone-word comma rule must not reach # its own capitals-lean probe or the two-word comma rule's name. # Every #540 rule carries one superstring probe too, after a # mutant dropping the ^...$ anchors passed every other guard here. diff --git a/tests/v2/test_properties.py b/tests/v2/test_properties.py index 9ab83b24..53c81f1d 100644 --- a/tests/v2/test_properties.py +++ b/tests/v2/test_properties.py @@ -285,7 +285,8 @@ def test_the_comma_agreement_exceptions_are_all_still_exceptions( #: (a caps member leans credential on both sides and a lower one is #: counted on both, so neither disagrees); 0 with the anchor off, which #: is the recorded negative control. The one-case class's 1,026 is -#: unmoved by the two heads #544 added, both being mixed case. +#: unmoved by the three heads #544 added ('Jane Doe PhD', 'Doe, Jane +#: PhD', 'PhD'), all three being mixed case. _MAIDEN_ANCHORED_HEAD_EXCEPTIONS = 810 _MAIDEN_ANCHORED_HEAD_DIGEST = ( "c5a8f5277e49fe583e0edc834346f623dd02459845ccd6ab9325ecfe7a167c7c") @@ -308,11 +309,12 @@ def _qualifying_front(toks: list[tuple[Token, tuple[int, int]]], i: int, member of the ambiguous class) through lone members to the piece that would speak for their company, and return it -- with its OWN index, which a caller's role question needs -- if it structurally - qualifies: unambiguous suffix vocabulary, not a connective, not a - single-letter roman numeral, and nothing but a comma or a maiden - marker stands between it and `toks[i]` (the company does not - reach across a clause, rules.md#M2). None if no such piece exists - or it fails one of those tests. + qualifies: unambiguous suffix vocabulary, not tagged an initial, + not a connective, not a particle, not a single-letter roman + numeral, and neither a comma nor a maiden marker stands between it + and `toks[i]` (the company is read within one comma part, and does + not reach across a clause, rules.md#M2). None if no such piece exists or it fails one + of those tests. Says NOTHING about either token's ROLE -- not `toks[i]`'s, and not the front's. That is deliberate: a caller checking whether the @@ -335,6 +337,7 @@ def _qualifying_front(toks: list[tuple[Token, tuple[int, int]]], i: int, or "vocab:suffix" not in front.tags or "initial" in front.tags or "conjunction" in front.tags + or "particle" in front.tags or (len(letters) == 1 and letters.lower() in "ivx")): return None return front, j @@ -349,8 +352,9 @@ def _reserve_kept(toks: list[tuple[Token, tuple[int, int]]], j: int, is this shape (the reserve keeps 'PhD'); 'Smith, PhD Ma' is NOT -- a family already exists from before the comma, so nothing here is H4's carve-out, whatever `segment_suffix_reading` does with - the given part on its own account (not modelled by this walk; see - the docstring below on where that shape is pinned instead). A + the given part on its own account (not modelled by this walk; + `_outside_its_company`'s docstring names where that shape is + pinned instead). A comma anywhere before `toks[j]` therefore answers False outright. A maiden clause crossed on the way answers nothing on its own: a @@ -439,10 +443,10 @@ def _credential_without_suffix_role( word stands in front of it must have that FRONT word in the SUFFIX role too -- the front cannot itself be reserved as the given or family name while lending its credential-ness to the - member behind it. 'PhD Ma' read given 'PhD', suffix 'Ma', family - '' before this fix: 'Ma' inherited PhD's company while PhD itself - was handed back to the given slot the walk had to reserve, losing - the family entirely. Same `_qualifying_front` walk-back as + member behind it. 'PhD Ma' reading given 'PhD', suffix 'Ma', + family '' is the shape this catches: 'Ma' inheriting PhD's company + while PhD itself is handed back to the given slot the walk has to + reserve, losing the family entirely (the recorded control below). Same `_qualifying_front` walk-back as `_outside_its_company`, checked the other way; the two properties then ask DIFFERENT role questions of the front on purpose (see `_qualifying_front`'s docstring) -- this one does NOT exempt a @@ -535,13 +539,15 @@ def test_a_maiden_clause_does_not_change_how_a_trailing_word_reads( another: `assert not company` is the company check's recorded negative control, `assert not orphaned` the converse's. RECORDED NEGATIVE CONTROL for `assert not company`: with - `credential_anchors` answering False and `segment_suffix_reading`'s - inline anchor off, it fails on 48 of the walk's parses ('Jane Doe - Jr. Ma' reading family 'Ma'); 0 here. RECORDED NEGATIVE CONTROL - for `assert not orphaned`: with only `credential_anchors`' - leading-position exclusion removed (the position-0 defect this - commit fixes), it fails on 30 of the walk's plain-form parses - ('PhD Ma' reading given 'PhD', suffix 'Ma'); 0 here. + `credential_anchors` answering False, it fails on 48 of the walk's + parses ('Jane Doe Jr. Ma' reading family 'Ma'); 0 here. RECORDED + NEGATIVE CONTROL for `assert not orphaned`: the walk's own leading + piece never anchors, and the no-comma peel holds that twice -- + `credential_anchors` skips the position, and the peel's reach test + (`anchor_in_reach`) never looks at it. With both removed it fails + on 42 of the walk's plain-form parses ('PhD Ma' reading given + 'PhD', suffix 'Ma', and the by-shape 'PhD X.Y.Z.' alike); with + either one alone removed, 0; 0 here (measured 2026-09-28). """ members = ("ba", "do", "ed", "jd", "ma", "x.y.z.", "r.a.i.") heads = ("Jane Doe", "Doe, Jane", "John", "J.", "Dr.", "Jane", From 5c8e8acb0763fdd720b23aa555cb398c809cd814 Mon Sep 17 00:00:00 2001 From: Derek Gulbranson Date: Mon, 28 Sep 2026 13:57:12 -0700 Subject: [PATCH 10/11] perf(#544): a member opening a comma part asks no anchor pass rules.md#S2's company clause speaks only for a member with a credential in front of it, so the first piece of a comma part is never anchored. segment_suffix_reading now builds its credential_anchors pass only where anchor_in_reach finds a suffix piece in front of the declined member; a member opening the part asks neither. 'Smith, Ed' and 'Smith, Ma' go back to 213 frames, 'Smith, Ed John' to 270, and credential_anchors is entered 0 times for each, which a frame-count test pins. A sweep test holds anchor_in_reach's one-way exactness -- False implies credential_anchors False at every lone member -- over the case table's texts and every run of one to three words behind two heads (4,497 texts, 0.23s); with the reach test reading a "vocab:suffix" token as no suffix piece, 878 positions fail. rules.md#S2's reworded sentence is reflowed to the rule body's width, and decisions.md's #544 addendum carries the new figures. Co-Authored-By: Claude Opus 5.5 --- docs/design/decisions.md | 2 +- docs/design/rules.md | 17 ++++---- nameparser/_pipeline/_pieces.py | 15 +++++-- tests/v2/pipeline/test_pieces.py | 68 +++++++++++++++++++++++++++++++- tests/v2/test_benchmark.py | 16 ++++++++ 5 files changed, 104 insertions(+), 14 deletions(-) diff --git a/docs/design/decisions.md b/docs/design/decisions.md index e790ae85..3e4dcdad 100644 --- a/docs/design/decisions.md +++ b/docs/design/decisions.md @@ -678,7 +678,7 @@ for n in ('Smith, John','Smith, XYZ'): print(n, calls_for(off.parse, n), calls_f REVERSED BY THIS ENTRY rather than edited: this section's 2026-09-15 ACCEPTED item (ii) (the bullet after this entry says so); the three "wrongly moved" reversals the 2026-09-14 paragraph THE RUN IS THE CAPS CLASS'S ALONE records — `John Smith, Ed Ma`, `John Smith, ma do` and `John Smith, X.Y.Z. A.B.` are credential runs now by C1's own count under every policy that keeps their words in the class (with `unlisted_dotted_suffixes` off, `X.Y.Z.` and `A.B.` are no class members and the third keeps the listing form), while that paragraph's narrowing of the CAPS run test stands; and #540's two pending questions (the `suffix-acronym-collisions` bullet of this date). MEASURED 2026-09-28, the #544 tree against its parent e10e83b4, py3.11, after (9) and (10); the 2026-09-27 figures this paragraph first carried were taken before them and are superseded. Recompute: every run of one to three words from {PhD, MD, Jr, MA, Ma, MEng, M.Eng., LAc, Ed, Do, ba, MS, V, Prof.} behind each of `Wang {}`, `John Smith {}`, `John Quincy Smith {}`, `Dr. John Smith {}`, `John Smith, {}`, `John Quincy Smith, {}`, `Smith, {}`, `Doe, Jane {}`, `Jane Doe nee Smith {}`, `Doe, Jane nee Smith {}` and `Jan van der Berg {}`, each written as given, lower-cased and upper-cased and deduped, parsed under the three name orders at otherwise default policy, the seven role fields and the ambiguity kinds compared against the same grid parsed by e10e83b4. THE DETECTOR, which both the invariants and the attribution read, looks only at the run's own words, compared case-free: {phd, md, jr, ms, m.eng.} are the unambiguous credentials, {ma, meng, lac, ed, do, ba} the members; a member is ANCHORED where an unambiguous credential stands in front of it in the run with nothing but members between, except that `ms` or `md` opening a comma part anchors nothing; and a run word is IN A NAME FIELD where it appears among the space-split words of given, middle, family or maiden more often than the run's unconstrained occurrences of it account for. The two invariants count parses with an anchored member in a name field, and with an unambiguous credential in one. THE ATTRIBUTION puts each moved parse in ONE class, the first that holds, in this order: THE ANCHOR, where the tree's two violation counts are each no higher than the parent's and one is lower; THE CHUNKED `M.Eng.`, where the run holds `M.Eng.`; A LONE `v`, where re-running the first test with `v` counted as an unambiguous credential makes it hold; THE C1 RUN, a comma shape with two or more name words before the comma whose run is only credentials and members, at least one a member; and anything left unattributed. 254,496 parses over 84,832 texts; 89,825 move (29,943 texts, 19,123 of them with a field move): 39,261 the anchor, 14,676 the C1 run, 35,696 the chunked `M.Eng.` (32,472 of those only losing the report it raised as a pick), 180 a lone `v` (`john smith, v phd ma`, suffix 'v phd ma'), and 12 unattributed, the `Smith, MA PhD ba` parses above. Clean-to-clean — no run word in a name field on either side — 23,454 move, every one `M.Eng.`'s dropped report, and none elsewhere. The two invariants, 0 NEW violations of either: an anchored member in a name field, 29,199 → 1,530 (the residue is a particle member P2 has chained, a dual inside a leading title run, `smith, prof. md ma`, and the member behind a credential in a part a dual opens, `Smith, MD PhD Ma`, which (10) keeps a name); an unambiguous credential in a name field, 36,675 → 4,734 (the residue is an interior single-letter `V`, a non-final `Prof.`, the `PhD` a comma part keeps as its given name when a `V` or a title follows it, and the same `PhD` behind a dual that opens the part, (10) again). Against the tree before (9) and (10), the same grid moves 5,076 parses: 4,536 (1,512 texts, every one a comma shape whose part reads wholly as credentials) gain a `suffix-or-name` report and move nothing else, and 540 (180 texts, every one `Smith, ` then `MD` or `MS` opening the run, in each case form) return to e10e83b4's reading, fields and reports alike, from a silent suffix reading of the whole part. The differential corpora at the parent (1,418 names, three orders) move four names, every one intended: `John Smith, PhD MEng`, `john smith, phd meng`, `Wang M.Eng.`, `abdul Smith Jr Ma`. Frames per parse through `Parser().parse`, py3.11: the reference band is unmoved (370/407, `uv run python tools/perf/call_count.py`), and so is every ordinary name measured (`Smith, John Quincy` 260, `Doe, John MA` 283, `Doe, Jane nee Smith PhD` 338) except a comma name with two or more words on each side of the comma, which enters the C1 run test and pays `run_word_fold`'s one call before its first name word ends it (`Doe Smith, Jane Q.` 317 → 318, `Garcia Lopez, Maria Jose` 330 → 331, measured 2026-09-28); the new questions cost where they are asked — `John Smith Ma` 244 → 250, the anchor pass that finds nothing; `John Smith, MA Jr` 355 → 382, the case fact forced to see the capitals settle the run; `John Smith PhD MEng` 286 → 326; `Smith, MD PhD Ma` 349 → 362, the given slot's pass that (10) leaves with nothing to find — and `John Smith, PhD MEng` gets cheaper, 358 → 314, as does `Smith, PhD Ma`, 286 → 271. NO MECHANISMS ENTRY IS OWED: the anchor pass is ONE-PREDICATE-PER-QUESTION's `_pieces.credential_anchors`, which `segment_suffix_reading` reads inline for the frame budget under a keep-in-step note, and its linearity is the answer #531's trailing floor and #397's `_run_neighbours` already give — one forward pass per question, never a look-behind per member. tests/v2/test_benchmark.py's `credential_run` shape guards it: the per-member look-behind measured 15.8× for 4× the input at base 800, against 4.13–4.21× on this tree at every base (that module's own record). - ADDENDUM 2026-09-28 (Derek). (e) A PARTICLE ANCHORS NOTHING, a fifth boundary beside the four WHAT MAY ANCHOR lists: a word of both the particle and the unambiguous suffix vocabulary (`vd` and `mc` in the shipped lexicon; `do` sits in the ambiguous half, so it is a member and never an anchor) heads the family name behind it (P2), and anchoring there split the family around a suffix — `Smith vd Ma, John` read family 'Smith Ma', suffix 'vd'; `Jan vd Ma` suffix 'vd Ma' and no family; `Smith Mc Ma, John` and `D. Mc Ba Ed, Smith` the same way. Each reads as e10e83b4 reads it again (family 'Smith vd Ma', 'vd Ma', 'Smith Mc Ma', 'D. Mc Ba Ed'). The exclusion is of the word in front only: a particle MEMBER behind a credential is still spoken for (`doe, jane v phd do` keeps suffix 'v phd do'), and a C1 run holding the particle is C1's reading rather than the anchor's (`Jan Berg, vd Ma` reads suffix 'vd Ma', as `Jan Berg, vd` reads suffix 'vd'). Measured over the grid of MEASURED above with `vd` and `Mc` added to its words (127,578 texts, 382,734 parses), against the tree before this addendum: 9,426 parses (3,142 texts) move, every one holding `vd` or `Mc` directly in front of a member; 8,940 return to e10e83b4's reading exactly, and the other 486 all hold `M.Eng.`, whose own change is (3) — 406 of them take e10e83b4's fields and differ only in the pick `M.Eng.` no longer reports, and 80 take the fields e10e83b4 gives the same text with `PhD` in `M.Eng.`'s place (`Jane Doe nee Smith M.Eng. vd Ma` reads middle 'Doe M.Eng.', family 'vd Ma', as `Jane Doe nee Smith PhD vd Ma` reads middle 'Doe PhD'). The differential corpora move 0 of 1441 names, and the grid of MEASURED without the two words moves 0 parses, so every figure there stands. THE PASS IS ASKED ONLY WHERE A SUFFIX PIECE IS IN REACH: `_pieces.anchor_in_reach` walks back through the lone members in front of a declined member on tags alone and answers False wherever the pass must — the first piece past the members is no suffix piece, or there is none — and it is asked only before the pass exists, one lookup answering after that, so a run of members stays linear. An ordinary name ending in a declined member stops paying for the pass (frames, py3.11, before → after, e10e83b4 in brackets): `John Smith Ma` 250 → 246 (244), `Smith, John Ma` 285 → 279 (275), `Doe, Jane MA do` 374 → 369 (363), `Doe, John Q. Ma` 328 → 320 (316), `Jan vd Ma` 308 → 243 (238); a name the pass does read pays the test on top, `John Smith PhD MEng` 326 → 329 and `Doe, Jane PhD MEng` 359 → 362. On the no-comma peel the reach starts past the walk's leading piece, which never anchors — so a by-shape member, which reaches the pass only at the peel's second position (a lean of None with a word to spare is consumed first), finds nothing in reach, and the by-shape guard that stood there was deleted as unreachable. `segment_suffix_reading` calls `credential_anchors` (`first_kept=False`, a comma part's first piece being a run member like any other) in place of the inline copy NO MECHANISMS describes, and the keep-in-step note goes with it: a part holding a name word returns before any member is asked, so a plain comma name pays nothing (`Smith, John` 182, `Doe, John MA` 283, `Smith, J. Q.` 246, all unmoved), and a part the company decides pays two frames over the copy (`Smith, PhD MEng` 273 → 275, `Smith, PhD Ma` 271 → 273). THE TWO-INPUT CHECK's second control reads 42, not 30, with no change to the property: the walk's leading piece is now held out of the pass twice on the no-comma peel, by `credential_anchors` and by the reach test, either alone suffices, and with both removed the by-shape heads the deleted guard held back (`PhD X.Y.Z.`) fail beside the listed ones; the first control is unmoved at 48. And the 2026-09-28 C2 bullet's "a TAIL segment, which assign reads wholly as suffixes" holds outside a maiden clause standing in it — `Jane Doe, PhD, Jr nee van Ma` keeps maiden 'van Ma' — which rules.md#C2's statement now says. + ADDENDUM 2026-09-28 (Derek). (e) A PARTICLE ANCHORS NOTHING, a fifth boundary beside the four WHAT MAY ANCHOR lists: a word of both the particle and the unambiguous suffix vocabulary (`vd` and `mc` in the shipped lexicon; `do` sits in the ambiguous half, so it is a member and never an anchor) heads the family name behind it (P2), and anchoring there split the family around a suffix — `Smith vd Ma, John` read family 'Smith Ma', suffix 'vd'; `Jan vd Ma` suffix 'vd Ma' and no family; `Smith Mc Ma, John` and `D. Mc Ba Ed, Smith` the same way. Each reads as e10e83b4 reads it again (family 'Smith vd Ma', 'vd Ma', 'Smith Mc Ma', 'D. Mc Ba Ed'). The exclusion is of the word in front only: a particle MEMBER behind a credential is still spoken for (`doe, jane v phd do` keeps suffix 'v phd do'), and a C1 run holding the particle is C1's reading rather than the anchor's (`Jan Berg, vd Ma` reads suffix 'vd Ma', as `Jan Berg, vd` reads suffix 'vd'). Measured over the grid of MEASURED above with `vd` and `Mc` added to its words (127,578 texts, 382,734 parses), against the tree before this addendum: 9,426 parses (3,142 texts) move, every one holding `vd` or `Mc` directly in front of a member; 8,940 return to e10e83b4's reading exactly, and the other 486 all hold `M.Eng.`, whose own change is (3) — 406 of them take e10e83b4's fields and differ only in the pick `M.Eng.` no longer reports, and 80 take the fields e10e83b4 gives the same text with `PhD` in `M.Eng.`'s place (`Jane Doe nee Smith M.Eng. vd Ma` reads middle 'Doe M.Eng.', family 'vd Ma', as `Jane Doe nee Smith PhD vd Ma` reads middle 'Doe PhD'). The differential corpora move 0 of 1441 names, and the grid of MEASURED without the two words moves 0 parses, so every figure there stands. THE PASS IS ASKED ONLY WHERE A SUFFIX PIECE IS IN REACH: `_pieces.anchor_in_reach` walks back through the lone members in front of a declined member on tags alone and answers False wherever the pass must — the first piece past the members is no suffix piece, or there is none — and it is asked only before the pass exists, one lookup answering after that, so a run of members stays linear. An ordinary name ending in a declined member stops paying for the pass (frames, py3.11, before → after, e10e83b4 in brackets): `John Smith Ma` 250 → 246 (244), `Smith, John Ma` 285 → 279 (275), `Doe, Jane MA do` 374 → 369 (363), `Doe, John Q. Ma` 328 → 320 (316), `Jan vd Ma` 308 → 243 (238); a name the pass does read pays the test on top, `John Smith PhD MEng` 326 → 329 and `Doe, Jane PhD MEng` 359 → 362. On the no-comma peel the reach starts past the walk's leading piece, which never anchors — so a by-shape member, which reaches the pass only at the peel's second position (a lean of None with a word to spare is consumed first), finds nothing in reach, and the by-shape guard that stood there was deleted as unreachable. `segment_suffix_reading` calls `credential_anchors` (`first_kept=False`, a comma part's first piece being a run member like any other) in place of the inline copy NO MECHANISMS describes, and the keep-in-step note goes with it. The pass is built only where the reach test finds a suffix piece in front of a declined member, which no member opening the part has, and a part whose name word stands ahead of any member returns before one is asked: a plain comma name pays nothing (`Smith, John` 182, `Doe, John MA` 283, `Smith, J. Q.` 246, `Smith, Ed` and `Smith, Ma` 213, `Smith, Ed John` 270, all as before the call replaced the copy; built on every declined member instead, the last three paid 215, 215 and 273), a title in front pays the reach test alone (`Smith, Dr. Ma` 253 → 254), and a part the company decides pays three frames over the copy (`Smith, PhD MEng` 273 → 276, `Smith, PhD Ma` 271 → 274; `Smith, PhD Ma John` 330 → 335, its pass built for 'Ma' before 'John' ends the part). THE TWO-INPUT CHECK's second control reads 42, not 30, with no change to the property: the walk's leading piece is now held out of the pass twice on the no-comma peel, by `credential_anchors` and by the reach test, either alone suffices, and with both removed the by-shape heads the deleted guard held back (`PhD X.Y.Z.`) fail beside the listed ones; the first control is unmoved at 48. And the 2026-09-28 C2 bullet's "a TAIL segment, which assign reads wholly as suffixes" holds outside a maiden clause standing in it — `Jane Doe, PhD, Jr nee van Ma` keeps maiden 'van Ma' — which rules.md#C2's statement now says. The reach test's one-way exactness is held by tests/v2/pipeline/test_pieces.py's `test_anchor_in_reach_never_hides_an_anchor`, over the case table's texts and runs of one to three words behind two heads; its recorded negative control, the reach test reading a `vocab:suffix` token as no suffix piece, fails at 878 member positions. - 2026-09-27 (#544) — THE 2026-09-15 ACCEPTED ITEM (ii) IS REVERSED, and the bullet stands as it landed. `abdul Smith Jr Ma` reads given 'abdul', family 'Smith', suffix 'Jr Ma' — 2.3.0's reading — because the unambiguous 'Jr' in front of the Title-case 'Ma' anchors it (the entry above): the peel takes both, and P5's reserve, now seeing the family the join would take, declines the join. Item (i) stands. The case row is now `a_credential_in_front_anchors_a_declined_pick`, and rules.md#S2's Accepted block names the shapes that still keep the company out of reach. - 2026-09-27 (Derek), #544 — THE #531 PAIRING GAINS A SECOND EXCEPTION, and CAPITALS DECIDE FOR `do` above stands as it landed. That bullet let only a positive credential lean override P6's attachment at the given slot; an unambiguous credential IN FRONT of the member now overrides it as well, the degree being a second and stronger signal: `doe, jane v phd do` reads suffix 'v phd do' and reports `suffix-or-name` where it read family 'do doe' and reported P6's fork. The one-case record the pairing protects has nothing in front of its particle, so `NASCIMENTO, EDSON ARANTES DO` still reads family 'DO NASCIMENTO' and reports `particle-or-given`. diff --git a/docs/design/rules.md b/docs/design/rules.md index a8818cad..c1356afa 100644 --- a/docs/design/rules.md +++ b/docs/design/rules.md @@ -1042,14 +1042,15 @@ S2. Rationale: generational suffixes and credentials are recognized initial is written in, where a multi-letter one such as 'III' speaks — speak for nothing and end the run; so does any name word. A particle of this class standing behind a credential is - still spoken for: the exclusion is of the word in front. A word of both the title and the suffix vocabulary - standing in the given part's leading title run is a title there - and speaks for nothing, and no credential behind it speaks for a - word of this class in that part: the title reading makes the next - word the given name, and from there the part reads as any given - part does, the given part's own company included. A part holding - no such word, or only words whose capitals decide them, reads as - it did. Anywhere else such a word speaks like any suffix word. + still spoken for: the exclusion is of the word in front. A word + of both the title and the suffix vocabulary standing in the given + part's leading title run is a title there and speaks for + nothing, and no credential behind it speaks for a word of this + class in that part: the title reading makes the next word the + given name, and from there the part reads as any given part does, + the given part's own company included. A part holding no such + word, or only words whose capitals decide them, reads as it did. + Anywhere else such a word speaks like any suffix word. At the trailing slot of the given part this company outranks P6's attachment, as the capitals do. It does not reach across a maiden marker's clause (M2), whose name words stand between. A member diff --git a/nameparser/_pipeline/_pieces.py b/nameparser/_pipeline/_pieces.py index 88f807f2..02d9e6a0 100644 --- a/nameparser/_pipeline/_pieces.py +++ b/nameparser/_pipeline/_pieces.py @@ -498,10 +498,13 @@ class by SHAPE takes the count instead, which is decided at the return None out: list[bool] = [] # #544: `credential_anchors` over the whole part, computed the - # first time a member's writing declines. Every piece in front of - # such a member has already been read as a suffix, a title, a - # member or a numeral -- a name word returns None first -- so a - # comma part holding a name word never pays for the pass. + # first time a member's writing declines with a suffix piece in + # reach in front of it (`anchor_in_reach`, asked only while the + # pass does not exist yet). A member opening the part has nothing + # in front and asks neither ('Smith, Ed', 'Smith, Ma John'); a + # name word in front returns None before the member is reached; + # and a title in front ends the reach test, so 'Smith, Dr. Ma' + # pays that one test and no pass. anchors: list[bool] | None = None # Whether every piece so far stands in the part's leading title # run (titles, and title/suffix duals), and whether a dual has @@ -530,6 +533,10 @@ class by SHAPE takes the count instead, which is decided at the out.append(True) elif (member and not dual_led and SHAPE_ACRONYM_TAG not in tokens[piece[0]].tags + and (anchors is not None + or (bool(out) and anchor_in_reach( + range(len(out) - 1, -1, -1), pieces, ptags, + tokens))) and (anchors := anchors if anchors is not None else credential_anchors( range(len(pieces)), pieces, ptags, tokens, diff --git a/tests/v2/pipeline/test_pieces.py b/tests/v2/pipeline/test_pieces.py index 2e610699..6c1f2211 100644 --- a/tests/v2/pipeline/test_pieces.py +++ b/tests/v2/pipeline/test_pieces.py @@ -6,6 +6,7 @@ no parse can produce, and the stability its readers rest on. """ import dataclasses +import itertools from collections.abc import Sequence, Set import pytest @@ -24,11 +25,15 @@ trailing_titles, ) from nameparser._pipeline._segment import segment -from nameparser._pipeline._state import ParseState, WorkToken +from nameparser._pipeline._state import ( + AMBIGUOUS_ACRONYM_TAG, ParseState, WorkToken, +) from nameparser._pipeline._tokenize import tokenize from nameparser._pipeline._vocab import is_one_case, is_title_shaped, tag_marker_runs from nameparser._policy import Policy +from ..cases import CASES + def _through_group(text: str, policy: Policy = Policy()) -> ParseState: state = ParseState(original=text, lexicon=Lexicon.default(), @@ -756,3 +761,64 @@ def test_tag_marker_runs_answers_in_ascending_index_order() -> None: state.lexicon.maiden_markers, folded)) assert keys == sorted(keys), text + + +#: The run words of the reach sweep below: unambiguous credentials, +#: members (one a particle too), the two particles of the suffix +#: vocabulary, a one-letter numeral, titles and a name word. +_REACH_WORDS = ("PhD", "Jr", "MA", "Ma", "Ed", "Do", "vd", "Mc", "V", + "Prof.", "Dr.", "Jones") + + +def _reach_failures(texts: Sequence[str]) -> list[str]: + """Every (text, segment, position, first_kept, start) at which + `anchor_in_reach` answers False while `credential_anchors` anchors + the lone member standing there -- the direction the reach test + must never get wrong, since a False skips the pass. Asked of every + segment of every text, over `order` both from the segment's first + piece and from past its leading title run, and with the leading + position both kept and read, the reach walking everything in + front each time (the superset every caller passes).""" + out = [] + for text in texts: + state = _through_group(text) + tokens = state.tokens + for seg, (pieces, ptags) in enumerate( + zip(state.pieces, state.piece_tags)): + for start in {0, leading_titles(pieces, ptags, tokens)}: + order = range(start, len(pieces)) + for first_kept in (True, False): + anchors = credential_anchors(order, pieces, ptags, + tokens, first_kept) + for pos, m in enumerate(order): + piece = pieces[m] + if not (len(piece) == 1 and AMBIGUOUS_ACRONYM_TAG + in tokens[piece[0]].tags): + continue + if anchors[pos] and not anchor_in_reach( + range(m - 1, -1, -1), pieces, ptags, + tokens): + out.append(f"{text!r} seg {seg} at {m} " + f"first_kept={first_kept} " + f"start={start}") + return out + + +def test_anchor_in_reach_never_hides_an_anchor() -> None: + """`anchor_in_reach` False implies `credential_anchors` False, at + every lone member of every segment, over the case table's texts + and every run of one to three `_REACH_WORDS` behind 'John Smith ' + and 'Doe, Jane ' (4,497 distinct texts, 742 of them the table's; + 0.23s on py3.11, measured 2026-09-28). RECORDED NEGATIVE CONTROL: + with the reach test reading a "vocab:suffix" token as no suffix + piece (returning False there), 878 positions fail (measured + 2026-09-28).""" + texts = {case.text for case in CASES if case.policy is None + and case.locale is None} + for n in (1, 2, 3): + for run in itertools.product(_REACH_WORDS, repeat=n): + for head in ("John Smith ", "Doe, Jane "): + texts.add(head + " ".join(run)) + failures = _reach_failures(sorted(texts)) + assert not failures, (f"{len(failures)} anchored member(s) the " + f"reach test hides:\n" + "\n".join(failures[:15])) diff --git a/tests/v2/test_benchmark.py b/tests/v2/test_benchmark.py index 8649e595..350675ce 100644 --- a/tests/v2/test_benchmark.py +++ b/tests/v2/test_benchmark.py @@ -686,3 +686,19 @@ def test_a_name_word_ends_the_comma_run_before_the_numeral_test() -> None: text = "Doe Smith, Jane Q." assert _frames_for(text, only="run_word_fold") >= 1 assert _frames_for(text, only="is_single_letter_numeral") == 0 + + +def test_a_member_opening_a_comma_part_asks_no_anchor_pass() -> None: + """rules.md#S2's company clause speaks only for a member with a + credential in FRONT of it, so a member opening the part after a + one-word family comma ('Smith, Ed') is never anchored, and + `segment_suffix_reading` builds no `credential_anchors` pass for + it: the pass is built only where `anchor_in_reach` finds a suffix + piece in front. RECORDED NEGATIVE CONTROL: with the pass built + for every member whose writing declines, `credential_anchors` is + entered once for each of these (measured 2026-09-28).""" + if sys.getprofile() is not None: + pytest.skip("a profile hook is already installed; this test owns it") + for text in ("Smith, Ed", "Smith, Ma", "Smith, Ed John"): + assert parse(text).family == "Smith", text + assert _frames_for(text, only="credential_anchors") == 0, text From a075420a96ebaee3fadce98fade24ebbb523a14d Mon Sep 17 00:00:00 2001 From: Derek Gulbranson Date: Mon, 28 Sep 2026 22:00:31 -0700 Subject: [PATCH 11/11] refactor(#544): one given-slot anchor query, and a flatter anchor branch `_pieces.given_slot_anchors` is now the one home of the given slot's anchor pass, indexed by piece: assign's given slot, group's GIVEN_SLOT reader and the maiden clause's take each built it by hand, with three spellings of the leading-title start and of the pieces before it. Each caller keeps its own reach test and its own cache; assign's starts at the `n` `_peel_leading_titles` already counted, rather than counting the run again, and group's cache is a `nonlocal` rather than a one-slot list. `segment_suffix_reading`'s anchor branch reads as `peel_trailing`'s does: build the pass if a suffix piece is in reach, then ask it, each early branch ending in `continue`. Its leading-title-run flags stay walk-local, with a comment saying why: `leading_titles` counts a period-shaped suffix ('Esq.') into the run, which would move 'Smith, Esq. MD Ma'. `segment`'s run loop ends on a name word or a single-letter numeral in one branch, and the #516 narrowing comment keeps only why it stands. In test_properties, `_credential_without_suffix_role` forces `_one_case(name.original)` itself instead of taking a thunk both callers built from that same text, the walk's paired `company` / `orphaned` updates share a local `audit`, and the literal "shape:acronym" is the imported SHAPE_ACRONYM_TAG. Readings are byte-identical (all seven fields, ambiguity kinds and token indices) over a 27,662-name grid under three policies and over tests/v2/cases.py plus a 76,128-name maiden/title grid. Frames: the plain names are unchanged, the anchored given-slot names drop 2-5 ('Doe, Jane PhD Ma' 360 -> 357, 'Doe, Dr. Jane PhD Ma' 419 -> 414); call_count stays 370/407. Both recorded controls in the maiden walk still reproduce (48 company, 42 orphaned). Co-Authored-By: Claude Opus 5.5 --- nameparser/_pipeline/_assign.py | 22 +++++------- nameparser/_pipeline/_group.py | 23 +++++++------ nameparser/_pipeline/_pieces.py | 58 +++++++++++++++++++++++--------- nameparser/_pipeline/_segment.py | 17 +++------- tests/v2/test_properties.py | 57 ++++++++++++------------------- 5 files changed, 90 insertions(+), 87 deletions(-) diff --git a/nameparser/_pipeline/_assign.py b/nameparser/_pipeline/_assign.py index 5e346603..d6898f0e 100644 --- a/nameparser/_pipeline/_assign.py +++ b/nameparser/_pipeline/_assign.py @@ -74,7 +74,7 @@ effective_script, is_suffix_lenient, resolve_script_set, ) from nameparser._pipeline._pieces import ( - anchor_in_reach, credential_anchors, credential_at_the_given_slot, + anchor_in_reach, credential_at_the_given_slot, given_slot_anchors, is_suffix_piece, leading_titles, peel_walk, segment_suffix_reading, tail_reading, trailing_titles, ) @@ -680,11 +680,11 @@ def previous_kept(m: int, titled: tuple[int, ...]) -> int: #: text and tags identical. floors: dict[tuple[int, ...], tuple[int, bool]] = {} #: #544's anchors, per `titled` value like `floors` and for - #: the same reason: `credential_anchors` over the pieces the + #: the same reason: `given_slot_anchors` over the pieces the #: chain kept, computed once, the first time a member's own #: writing declines, so a run of members is read in one #: forward pass rather than one look-behind per member. - anchor_memo: dict[tuple[int, ...], dict[int, bool]] = {} + anchor_memo: dict[tuple[int, ...], list[bool]] = {} def anchored(m: int, titled: tuple[int, ...]) -> bool: memo = anchor_memo.get(titled) @@ -697,17 +697,13 @@ def anchored(m: int, titled: tuple[int, ...]) -> bool: if not anchor_in_reach(range(m - 1, -1, -1), pieces, ptags, tokens, titled): return False - # from past the leading title run: a title/suffix - # dual opening the part is a TITLE there and - # anchors nothing ('Smith, MD MA Ma') - order = [q for q in range( - leading_titles(pieces, ptags, tokens), - len(pieces)) - if q not in titled] - memo = dict(zip(order, credential_anchors( - order, pieces, ptags, tokens))) + # from past the leading title run, which `n` already + # counts: this closure is reached only from the walk + # below `_peel_leading_titles` sets it on + memo = given_slot_anchors(pieces, ptags, tokens, n, + skip=titled) anchor_memo[titled] = memo - return memo.get(m, False) + return memo[m] def trailing_floor(m: int, titled: tuple[int, ...]) -> int: """Where the trailing suffix run starts, walked as far diff --git a/nameparser/_pipeline/_group.py b/nameparser/_pipeline/_group.py index 5373f089..b515bc02 100644 --- a/nameparser/_pipeline/_group.py +++ b/nameparser/_pipeline/_group.py @@ -44,7 +44,7 @@ from nameparser._lexicon import _run_addresses_by_given from nameparser._pipeline._pieces import ( - anchor_in_reach, credential_anchors, credential_at_the_given_slot, + anchor_in_reach, credential_at_the_given_slot, given_slot_anchors, is_leading_title, is_suffix_piece, is_title_piece, is_trailing_title_word, Peel, leading_titles, peel_trailing, peel_walk, tail_reading, @@ -417,19 +417,20 @@ def _release_reads_off(view: Sequence[Sequence[int]], # run only the first time a member's writing leaves the # question open; each call site hands over a lambda, so no # frame is spent building the question either. - anchor_cell: list[list[bool]] = [] + anchors: list[bool] | None = None def anchored_at(q: int) -> bool: - if not anchor_cell: + nonlocal anchors + if anchors is None: # the reach test first, and only before the pass # exists (`_pieces.anchor_in_reach`) if not anchor_in_reach(range(q - 1, -1, -1), view, view_tags, tokens): return False - lead = leading_titles(view, view_tags, tokens) - anchor_cell.append([False] * lead + credential_anchors( - range(lead, len(view)), view, view_tags, tokens)) - return anchor_cell[0][q] + anchors = given_slot_anchors( + view, view_tags, tokens, + leading_titles(view, view_tags, tokens)) + return anchors[q] for q in range(last, -1, -1): piece = view[q] if (is_suffix_piece(piece, view_tags[q], tokens) @@ -873,10 +874,10 @@ def _maiden_take(pieces: Sequence[Sequence[int]], tokens[head[0]], one_case, lambda: anchor_in_reach( range(at - 1, -1, -1), view, view_tags, tokens) - and credential_anchors( - range(min(leading_titles(view, view_tags, tokens), - at), at + 1), - view, view_tags, tokens)[-1]) + and given_slot_anchors( + view, view_tags, tokens, + leading_titles(view, view_tags, tokens), + at + 1)[at]) start = at + 1 else: # TRAILING: the peel over the view IS the member's diff --git a/nameparser/_pipeline/_pieces.py b/nameparser/_pipeline/_pieces.py index 02d9e6a0..89a3fa59 100644 --- a/nameparser/_pipeline/_pieces.py +++ b/nameparser/_pipeline/_pieces.py @@ -375,6 +375,30 @@ def credential_anchors(order: Sequence[int], return out +def given_slot_anchors(pieces: Sequence[Sequence[int]], + ptags: Sequence[Set[str]], + tokens: Sequence[WorkToken], + start: int, end: int | None = None, + skip: Container[int] = ()) -> list[bool]: + """`credential_anchors` for the given part's slot, indexed by PIECE + rather than by position: the pass over `start` to `end`, `skip` + spliced out, and False before `start`, from `end` on and at every + skipped piece. `start` is where the leading title run ends, so a + title/suffix dual standing in it anchors nothing ('Smith, MD MA + Ma'), and the piece at `start` is the given name the reserve keeps. + The one home of that query for assign's given slot, group's + GIVEN_SLOT reader, and the maiden clause's take.""" + stop = len(pieces) if end is None else end + order: Sequence[int] = range(start, stop) + if skip: + order = [q for q in order if q not in skip] + out = [False] * len(pieces) + for q, anchored in zip(order, credential_anchors(order, pieces, ptags, + tokens)): + out[q] = anchored + return out + + def anchor_in_reach(back: Iterable[int], pieces: Sequence[Sequence[int]], ptags: Sequence[Set[str]], @@ -508,7 +532,10 @@ class by SHAPE takes the count instead, which is decided at the anchors: list[bool] | None = None # Whether every piece so far stands in the part's leading title # run (titles, and title/suffix duals), and whether a dual has - # stood there: once one has, nothing in the part anchors. + # stood there: once one has, nothing in the part anchors. Tracked + # here rather than read off `leading_titles`, whose period-shape + # inference counts a suffix like 'Esq.' into the run ('Smith, Esq. + # MD Ma' keeps its anchored 'Ma' only by this walk's reading). leading = True dual_led = False for piece, tags in zip(pieces, ptags): @@ -531,21 +558,20 @@ class by SHAPE takes the count instead, which is decided at the == "credential": leading = False out.append(True) - elif (member and not dual_led - and SHAPE_ACRONYM_TAG not in tokens[piece[0]].tags - and (anchors is not None - or (bool(out) and anchor_in_reach( - range(len(out) - 1, -1, -1), pieces, ptags, - tokens))) - and (anchors := anchors if anchors is not None - else credential_anchors( - range(len(pieces)), pieces, ptags, tokens, - first_kept=False))[len(out)]): - leading = False - if anchored is not None: - anchored.append(len(out)) - out.append(True) - elif (lenient and after_suffix + continue + if (member and not dual_led + and SHAPE_ACRONYM_TAG not in tokens[piece[0]].tags): + if anchors is None and out and anchor_in_reach( + range(len(out) - 1, -1, -1), pieces, ptags, tokens): + anchors = credential_anchors(range(len(pieces)), pieces, + ptags, tokens, first_kept=False) + if anchors is not None and anchors[len(out)]: + leading = False + if anchored is not None: + anchored.append(len(out)) + out.append(True) + continue + if (lenient and after_suffix and _numeral_behind_the_initial_veto(piece, tokens)): leading = False out.append(True) diff --git a/nameparser/_pipeline/_segment.py b/nameparser/_pipeline/_segment.py index 4a46627a..e4ed17bc 100644 --- a/nameparser/_pipeline/_segment.py +++ b/nameparser/_pipeline/_segment.py @@ -211,15 +211,10 @@ def class_run(seg: tuple[int, ...]) -> bool: # `ambiguous_class_candidate`: the run is a property of the CAPS # class ALONE (#516 review round, F2), the other two halves being # single-token by construction, so `all()` over more than one of - # THEM asks a question the design never posed. Measured, the - # un-narrowed call per token moved `'John Smith, Ed Ma'`, `'John - # Smith, ma do'` and `'John Smith, X.Y.Z. A.B.'` to a credential - # run through the CAPS switch. Since #544 all three are credential - # runs under every policy that keeps their words in the class - # (`unlisted_dotted_suffixes=False` leaves 'X.Y.Z.' a name word), - # by the listed-and-dotted run test below and - # its name-word count -- a different question, asked with the - # switch off too, so this narrowing still stands. + # THEM asks a question the design never posed: un-narrowed, a + # listed or dotted run ('John Smith, Ed Ma') would pass as a CAPS + # run through this switch, and such runs are the question of the + # listed-and-dotted run test below, asked with the switch off too. # # `one_case=False` asks the case-free question -- "if this name # turned out mixed, would EVERY token in the run join the CAPS @@ -290,9 +285,7 @@ def class_run(seg: tuple[int, ...]) -> bool: # "reject" first: a name word ends the run without the # numeral test's frame, and the numeral cannot be a # "reject" (it is suffix vocabulary, so it folds "defer") - elif fold == "reject": - break - elif is_single_letter_numeral(text): + elif fold == "reject" or is_single_letter_numeral(text): break else: rest.append(text) diff --git a/tests/v2/test_properties.py b/tests/v2/test_properties.py index 53c81f1d..d8e5a12e 100644 --- a/tests/v2/test_properties.py +++ b/tests/v2/test_properties.py @@ -12,8 +12,6 @@ import itertools import re import warnings -from collections.abc import Callable - import pytest from hypothesis import given, settings from hypothesis import strategies as st @@ -29,8 +27,9 @@ from nameparser._pipeline._state import (AMBIGUOUS_ACRONYM_TAG, ParseState) from nameparser._pipeline._vocab import ambiguous_lean, effective_script -from nameparser._types import (UNJOINED_CONJUNCTION_TAG, UNJOINED_TAG, - AmbiguityKind, ParsedName, Role, Token) +from nameparser._types import (SHAPE_ACRONYM_TAG, UNJOINED_CONJUNCTION_TAG, + UNJOINED_TAG, AmbiguityKind, ParsedName, + Role, Token) from .cases import CASES from .conftest import differential_corpus @@ -326,7 +325,7 @@ def _qualifying_front(toks: list[tuple[Token, tuple[int, int]]], i: int, tok, span = toks[i] j = i - 1 while (j >= 0 and AMBIGUOUS_ACRONYM_TAG in toks[j][0].tags - and "shape:acronym" not in toks[j][0].tags): + and SHAPE_ACRONYM_TAG not in toks[j][0].tags): j -= 1 if j < 0: return None @@ -422,7 +421,7 @@ def _outside_its_company(name: ParsedName) -> list[str]: for i, (tok, span) in enumerate(toks): if (tok.role is Role.SUFFIX or AMBIGUOUS_ACRONYM_TAG not in tok.tags - or "shape:acronym" in tok.tags): + or SHAPE_ACRONYM_TAG in tok.tags): continue qualifies = _qualifying_front(toks, i, name.original) if qualifies is None: @@ -435,9 +434,7 @@ def _outside_its_company(name: ParsedName) -> list[str]: return out -def _credential_without_suffix_role( - name: ParsedName, - one_case_thunk: Callable[[], bool | None]) -> list[str]: +def _credential_without_suffix_role(name: ParsedName) -> list[str]: """The converse of `_outside_its_company` (#544): a member read as a credential (role SUFFIX) because a qualifying word stands in front of it must have that FRONT word in the @@ -454,15 +451,10 @@ def _credential_without_suffix_role( defect took: the reserve-kept front is the one whose SUFFIX role went missing. - `one_case_thunk` computes `_one_case(name.original)` -- a full - extra pipeline run -- and is forced only for a SURVIVING - candidate: a member with a qualifying front that is itself - outside both TITLE and SUFFIX, i.e. exactly the shape that is - about to be reported. Every other member is decided by role tests - alone (cheap), so the vast majority of parses -- which have no - such front at all -- never force the thunk; a rare text with more - than one surviving candidate forces it once per candidate rather - than caching, since that shape is itself the exception. + `_one_case(name.original)` is a full extra pipeline run, asked + only for a SURVIVING candidate: a member whose qualifying front is + outside both TITLE and SUFFIX. Every other member is decided by + role tests alone. Skips a member whose OWN case-based lean (rules.md#S2, #289 -- an ALL-CAPS member of a mixed-case name) already reads it as a @@ -481,14 +473,14 @@ def _credential_without_suffix_role( front, _ = qualifies if front.role is Role.TITLE or front.role is Role.SUFFIX: continue - # a surviving candidate: only now is the thunk worth its cost - one_case = one_case_thunk() + # a surviving candidate: only now is the extra run worth its cost + one_case = _one_case(name.original) # inlined `_pieces.listed_lean`'s own gate, over the public # `Token` rather than the pipeline's `WorkToken` -- the two # types share `.text`/`.tags`, and this is a second # implementation on purpose, like the rest of this function own_lean = (None if (one_case is None - or "shape:acronym" in tok.tags) + or SHAPE_ACRONYM_TAG in tok.tags) else ambiguous_lean(tok.text, one_case)) if own_lean == "credential": continue @@ -593,6 +585,13 @@ def side(name: ParsedName, word: str) -> str: # SUFFIX role, never handed back to a given/family reserve while # still lending its credential-ness behind it orphaned: list[str] = [] + + def audit(label: str, text: str, name: ParsedName) -> None: + company.extend(f"[{label}] {text!r}: {bad}" + for bad in _outside_its_company(name)) + orphaned.extend(f"[{label}] {text!r}: {bad}" + for bad in _credential_without_suffix_role(name)) + failures = [] for label, policy in policies: parser = Parser(policy=policy) @@ -604,25 +603,13 @@ def side(name: ParsedName, word: str) -> str: for word in (base.lower(), base.title(), base.upper()): plain = f"{head} {word}" plain_name = parser.parse(plain) - company += [f"[{label}] {plain!r}: {bad}" for bad - in _outside_its_company(plain_name)] - orphaned += [f"[{label}] {plain!r}: {bad}" for bad - in _credential_without_suffix_role( - plain_name, - lambda: _one_case(plain))] + audit(label, plain, plain_name) plain_side = side(plain_name, word) for body in bodies: for marker in markers: clause = f"{head} {marker} {body} {word}" clause_name = parser.parse(clause) - company += [ - f"[{label}] {clause!r}: {bad}" for bad - in _outside_its_company(clause_name)] - orphaned += [ - f"[{label}] {clause!r}: {bad}" for bad - in _credential_without_suffix_role( - clause_name, - lambda: _one_case(clause))] + audit(label, clause, clause_name) clause_side = side(clause_name, word) pairs += 1 if clause_side == plain_side: