Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 2 additions & 2 deletions AGENTS.md

Large diffs are not rendered by default.

15 changes: 15 additions & 0 deletions docs/design/decisions.md

Large diffs are not rendered by default.

168 changes: 139 additions & 29 deletions docs/design/rules.md

Large diffs are not rendered by default.

8 changes: 6 additions & 2 deletions docs/release_log.rst

Large diffs are not rendered by default.

62 changes: 56 additions & 6 deletions nameparser/_pipeline/_assign.py
Original file line number Diff line number Diff line change
Expand Up @@ -45,13 +45,16 @@
particles_ambiguous token with more pieces following ("Van Johnson",
and since #367 "Dr. Van Johnson" too, a title no longer displacing the
particle out of that position) -- whatever role name_order assigns.
Emits SUFFIX_OR_NAME at FIVE sites: the trailing roman numeral, each
Emits SUFFIX_OR_NAME at SIX sites: the trailing roman numeral, each
ambiguous acronym the trailing peel had to resolve, the bare-suffix
carve-out where an input that is nothing but post-nominal vocabulary
gets its first word made into the name (H4's suffix half, #491),
-- since #289 -- the FAMILY-COMMA path's own read of the first
post-comma piece, and -- since #531 -- the class member ENDING that
path's given part, which the first-piece emitter could never reach.
post-comma piece, -- since #531 -- the class member ENDING that
path's given part, which the first-piece emitter could never reach,
and -- since #544 -- each member of a post-comma part read wholly as
credentials that was read as one only because a credential in front
of it anchors it.
Further emitters of the same kind live in
`_segment.py`, `_group.py` and `_post_rules.py`; they are not
assign's and are not counted here. And
Expand All @@ -71,7 +74,7 @@
effective_script, is_suffix_lenient, resolve_script_set,
)
from nameparser._pipeline._pieces import (
credential_at_the_given_slot,
anchor_in_reach, credential_at_the_given_slot, given_slot_anchors,
is_suffix_piece, leading_titles, peel_walk,
segment_suffix_reading, tail_reading, trailing_titles,
)
Expand Down Expand Up @@ -539,9 +542,11 @@ def assign(state: ParseState) -> ParseState:
# positional read peels a trailing suffix first: 'Smith Jr.,
# Mr.' has two pieces and one name, and read positionally lost
# its family (the code review).
anchored_picks: list[int] = []
reading = segment_suffix_reading(
state.pieces[1], state.piece_tags[1], tokens,
state.policy.lenient_comma_suffixes, state.one_case)
state.policy.lenient_comma_suffixes, state.one_case,
anchored_picks)
# rules.md#C1's exception, scoped to the ambiguous credential
# class: this is the first report of the comma's OWN decision
# (listing or credential run), where the writing left the
Expand Down Expand Up @@ -674,6 +679,31 @@ def previous_kept(m: int, titled: tuple[int, ...]) -> int:
#: which is a `copy_with(role=...)` and leaves
#: text and tags identical.
floors: dict[tuple[int, ...], tuple[int, bool]] = {}
#: #544's anchors, per `titled` value like `floors` and for
#: the same reason: `given_slot_anchors` over the pieces the
#: chain kept, computed once, the first time a member's own
#: writing declines, so a run of members is read in one
#: forward pass rather than one look-behind per member.
anchor_memo: dict[tuple[int, ...], list[bool]] = {}

def anchored(m: int, titled: tuple[int, ...]) -> bool:
memo = anchor_memo.get(titled)
if memo is None:
# nothing in front the pass could read as an
# anchor: the ordinary 'Smith, John Ma' stops here.
# Asked only before the pass exists -- once it
# does it answers in one lookup, where the reach
# test walks back through the run
if not anchor_in_reach(range(m - 1, -1, -1), pieces,
ptags, tokens, titled):
return False
# from past the leading title run, which `n` already
# counts: this closure is reached only from the walk
# below `_peel_leading_titles` sets it on
memo = given_slot_anchors(pieces, ptags, tokens, n,
skip=titled)
anchor_memo[titled] = memo
return memo[m]

def trailing_floor(m: int, titled: tuple[int, ...]) -> int:
"""Where the trailing suffix run starts, walked as far
Expand Down Expand Up @@ -833,8 +863,13 @@ def reads_as_a_suffix(m: int, titled: tuple[int, ...]) -> bool:
# unchanged). Measured 2026-09-19 per
# `Parser.parse`; Derek took that trade
# deliberately.
# #544: an unambiguous credential in front
# of it in the same run anchors it -- asked
# through a thunk, so only a member the
# writing declines pays for the pass
if credential_at_the_given_slot(
tok, state.one_case):
tok, state.one_case,
lambda: anchored(m, titled)):
return True
prev = previous_kept(m, titled)
# trailing piece of a two-part name is unambiguously
Expand Down Expand Up @@ -875,6 +910,21 @@ def reads_as_a_suffix(m: int, titled: tuple[int, ...]) -> bool:
_set_roles(tokens, piece,
Role.SUFFIX if reading[k] else Role.TITLE)
n = len(pieces)
# rules.md#S2's company, reported where it decided: a
# member the anchor read as a credential after its own
# writing declined is a pick, as the peel's are (#544).
# Never piece 0, which nothing stands in front of, so
# never the first-piece report's word above; a member
# whose capitals lean credential is not among them
# ('Smith, PhD MA' stays silent).
for k in anchored_picks:
i2 = pieces[k][0]
ambiguities.append(PendingAmbiguity(
AmbiguityKind.SUFFIX_OR_NAME,
f"{tokens[i2].text!r} behind a credential after "
f"the comma is also an ordinary name word; read "
f"as a credential",
(i2,)))
else:
n = _peel_leading_titles(pieces, ptags, tokens)
# rules.md#H5: "the title is TRANSPARENT to the suffix
Expand Down
5 changes: 3 additions & 2 deletions nameparser/_pipeline/_classify.py
Original file line number Diff line number Diff line change
Expand Up @@ -68,8 +68,9 @@
# rules.md#S2: "a trailing word of the suffix vocabulary reads as a
# suffix — generational forms and credential acronyms alike, and an
# ambiguous acronym written with its periods, one after each
# letter, counts unambiguously; a single trailing period is the
# abbreviation shape any word can wear and does not. A
# letter or one after each of two or more letter chunks, counts
# unambiguously; a single trailing period is the abbreviation shape
# any word can wear and does not. A
# bare ambiguous acronym is consumed only when the name has words to
# spare"
def _tags_for(token: WorkToken, n: str, state: ParseState,
Expand Down
75 changes: 58 additions & 17 deletions nameparser/_pipeline/_group.py
Original file line number Diff line number Diff line change
Expand Up @@ -44,7 +44,7 @@

from nameparser._lexicon import _run_addresses_by_given
from nameparser._pipeline._pieces import (
credential_at_the_given_slot,
anchor_in_reach, credential_at_the_given_slot, given_slot_anchors,
is_leading_title, is_suffix_piece, is_title_piece,
is_trailing_title_word,
Peel, leading_titles, peel_trailing, peel_walk, tail_reading,
Expand Down Expand Up @@ -410,6 +410,27 @@ def _release_reads_off(view: Sequence[Sequence[int]],
last = len(view) - 1
chain_ok = [False] * len(view)
members_ok = running = True
# #544: the anchors `credential_at_the_given_slot` may ask for,
# over the view as the take would leave it -- the reading
# assign's given slot makes of the same pieces, from past the
# leading title run. One forward pass for both loops below,
# run only the first time a member's writing leaves the
# question open; each call site hands over a lambda, so no
# frame is spent building the question either.
anchors: list[bool] | None = None

def anchored_at(q: int) -> bool:
nonlocal anchors
if anchors is None:
# the reach test first, and only before the pass
# exists (`_pieces.anchor_in_reach`)
if not anchor_in_reach(range(q - 1, -1, -1), view,
view_tags, tokens):
return False
anchors = given_slot_anchors(
view, view_tags, tokens,
leading_titles(view, view_tags, tokens))
return anchors[q]
for q in range(last, -1, -1):
piece = view[q]
if (is_suffix_piece(piece, view_tags[q], tokens)
Expand All @@ -420,8 +441,9 @@ def _release_reads_off(view: Sequence[Sequence[int]],
continue
if (members_ok and len(piece) == 1
and AMBIGUOUS_ACRONYM_TAG in tokens[piece[0]].tags
and credential_at_the_given_slot(tokens[piece[0]],
one_case)):
and credential_at_the_given_slot(
tokens[piece[0]], one_case,
lambda: anchored_at(q))):
chain_ok[q] = running
continue
members_ok = False
Expand All @@ -448,8 +470,9 @@ def _release_reads_off(view: Sequence[Sequence[int]],
return False
if (len(piece) == 1
and AMBIGUOUS_ACRONYM_TAG in tokens[piece[0]].tags
and credential_at_the_given_slot(tokens[piece[0]],
one_case)):
and credential_at_the_given_slot(
tokens[piece[0]], one_case,
lambda: anchored_at(q))):
continue
return False
elif reader is TailReader.TRAILING:
Expand Down Expand Up @@ -844,9 +867,17 @@ def _maiden_take(pieces: Sequence[Sequence[int]],
# 'Doe, Jane MA do' does with them, middle 'MA' and
# family 'do Doe'). The member is asked here, the span
# behind it and the name word ahead of it by the
# shared check below.
takes = credential_at_the_given_slot(tokens[head[0]],
one_case)
# shared check below. Anchored (#544) as assign's given
# slot anchors it: over the view up to the member, from
# past the leading title run.
takes = credential_at_the_given_slot(
tokens[head[0]], one_case,
lambda: anchor_in_reach(
range(at - 1, -1, -1), view, view_tags, tokens)
and given_slot_anchors(
view, view_tags, tokens,
leading_titles(view, view_tags, tokens),
at + 1)[at])
start = at + 1
else:
# TRAILING: the peel over the view IS the member's
Expand Down Expand Up @@ -1298,15 +1329,17 @@ def _group_segment(seg: tuple[int, ...], additional: int,
# segment's structure, and a default would be this module guessing
# what that caller already knows. group() passes `None` on the
# first for the chain emitter after a family comma -- the comma
# fixed the family, so that fork is settled -- and #533's is not
# that fork: a credential ending a maiden clause is a question the
# fixed the family, so that fork is settled -- and in a tail
# segment, which assign reads wholly as suffixes outside a maiden
# clause standing in it. #533's fork is
# neither: a credential ending a maiden clause is a question the
# comma settles nothing about, which is why the two channels are
# two parameters. They are given the SAME list wherever nothing is
# suppressed, which is every segment that is NOT after a family
# comma; what the split buys is the other case, where `None` on
# the first must not reach the second -- a maiden channel
# defaulting to whatever the first was would let a caller passing
# `ambiguities=None` silence both (the review's finding).
# suppressed, which is every segment that is neither after a
# family comma nor a tail; what the split buys is the other case,
# where `None` on the first must not reach the second -- a maiden
# channel defaulting to whatever the first was would let a caller
# passing `ambiguities=None` silence both.

def title(k: int) -> bool:
return is_title_piece(pieces[k], ptags[k], tokens)
Expand Down Expand Up @@ -2015,7 +2048,15 @@ def group(state: ParseState) -> ParseState:
bound_join = BoundJoin.STRICT
# Suppressed after a family comma for the same reason _assign
# suppresses it there: the family name is already fixed, so
# there is no fork left to report.
# there is no fork left to report. Suppressed in a tail segment
# as well, after either comma: assign reads that segment,
# outside a maiden clause standing in it ('Jane Doe, PhD, Jr
# nee van Ma' keeps maiden 'van Ma'), wholly as suffixes, so a
# chain report there -- a particle
# chained onto a name piece, or an acronym taken into the name
# -- names a reading the parse never takes. rules.md#C2: "a part
# the parse consumes wholly as suffixes raises no report about
# reading a word of it as a name"
tail = tail_start is not None and seg_idx >= tail_start
seg_cores = cores if tail else frozenset()
# #533: which rule reads what the maiden walk would leave, off
Expand All @@ -2031,7 +2072,7 @@ def group(state: ParseState) -> ParseState:
reader = TailReader.TRAILING
pieces, ptags, taken = _group_segment(
seg, additional, tokens, bound_join,
None if family_comma else ambiguities,
None if (family_comma or tail) else ambiguities,
seg_cores,
state.lexicon.given_name_titles,
opens_the_name=(seg_idx == 0 and not family_comma),
Expand Down
Loading
Loading