1"""Stage: group.
2
3Consumes: tokens (classified), segments, structure, one_case, extracted
4(the role + inner span per delimited region, for the #329 pass below --
5the only stage after tokenize that reads it).
6Produces: pieces + piece_tags per segment (runs of token indices --
7tokens are NEVER joined into strings: the anti-#100 invariant); maiden
8tail tokens get role=MAIDEN; marker tokens land in dropped.
9Reads: token tags (from classify), Lexicon.given_name_titles (the
10P5 licence, #369) and Policy.extra_suffix_delimiters, whose
11delimiter-core tokens tail segments drop (v1 suffix_delimiter parity)
12-- no other Policy field. Policy.lenient_comma_suffixes left this list
13with #436: it reached here only through segment_suffix_reading, whose
14render consumer was this stage's one-entry join and now lives in
15post_rules. The v1 "derived titles/prefixes"
16registration becomes piece_tags entries -- per-parse state that
17dissolves with the state (v1 kept per-parse sets for the same reason).
18
19Implements rules P2, P3, P4 and M2, and the
20group half of M1 (#329: the marker dropped inside EXTRACTED maiden
21content, which M2's pieces walk cannot reach because extract's
22content never enters pieces); each is cited at its code below. Also
23implements rule P5 (cited below at the bound-given join) and ports
24the "Ph. D."-split merge (v1 fix_phd; decisions.md#phd-merge).
25
26The piece-level predicates moved to _pieces in #439 -- the S2
27trailing peel, the leading-title and title-piece tests, the
28suffix-piece test, the no-name-segment test. Most are shared with
29assign; is_title_piece and trailing_start are group's alone and
30travelled because the shared ones call them. They had collected here
31by import direction rather than by topic (assign imported group and
32could not be imported back), which is the accumulation
33mechanisms.md#ONE-PREDICATE-PER-QUESTION describes; group imports them
34back like any other caller, and still does the work H3 and S2 describe
35with them. What remains defined here is group's own: _is_prefix_piece,
36_is_conj_piece, _is_rootname and _is_maiden_marker_piece.
37"""
38from __future__ import annotations
39
40import bisect
41import dataclasses
42from collections.abc import Iterable, Sequence, Set
43from enum import IntEnum
44from typing import assert_never
45
46from nameparser._lexicon import _run_addresses_by_given
47from nameparser._pipeline._pieces import (
48 credential_at_the_given_slot,
49 is_leading_title, is_suffix_piece, is_title_piece,
50 leading_titles, peel_trailing, peel_walk, tail_reading,
51 trailing_start, trailing_start_past_titles,
52)
53from nameparser._pipeline._state import (
54 AMBIGUOUS_ACRONYM_TAG, ParseState, PendingAmbiguity, Structure,
55 WorkToken, _AMBIGUOUS_CREDENTIAL_TAGS,
56)
57from nameparser._pipeline._vocab import D, PH
58from nameparser._pipeline._vocab import delimiter_cores
59from nameparser._types import AmbiguityKind, Role
60
61# the credential-pair regexes live in _vocab, whose own
62# is_wholly_suffix merges the same pair -- and since #319 that
63# predicate has TWO callers to stay in sync with, segment's
64# suffix-comma structure test and script_segment's decline of a
65# wholly-suffix post-comma run, both of which see the merged reading
66
67Piece = list[int]
68#: What the marker pass took out of a segment: the marker's TOKEN
69#: indices (more than one where the entry is a phrase, 'z domu') and
70#: the maiden-name pieces, themselves token indices, to take
71#: Role.MAIDEN. `_group_segment` produces it.
72MaidenTake = tuple[Piece, list[Piece]]
73#: What `_maiden_take` answers with, one index space further out: PIECE
74#: indices into its own `pieces` argument -- the marker's pieces, then
75#: the maiden name's. `_group_segment` resolves them to the token
76#: indices MaidenTake declares, which is the only reason the two are
77#: different types rather than one name used twice.
78MaidenIndices = tuple[list[int], list[int]]
79
80
81class TailReader(IntEnum):
82 """Which rule reads the words the maiden walk would leave standing
83 at the end of this segment -- the reader the acronym fork's second
84 check has to ask, since a stop is only right where that reader
85 takes the word (rules.md#M2, #533).
86
87 NONE is a statement and not a default: before a family comma the
88 words are the family the comma already named, and a tail segment
89 is read as credentials whole, so no trailing rule is consulted
90 there and the clause keeps what it has -- a stop would hand a word
91 to `family` rather than to `suffix`.
92
93 A CLOSED set: `_maiden_take` dispatches on it exhaustively
94 (`typing.assert_never`), so a fourth member is a type error at
95 every reader until it is given a reading. group() is the one
96 place (structure, segment index) is mapped onto it, and
97 tests/v2/pipeline/test_group.py's
98 `test_the_reader_is_pinned_to_the_structure_it_is_read_from`
99 pins that mapping -- by watching the call, not by restating it --
100 along with this enum's size. Change either and that test says so.
101 """
102
103 NONE = 0 # FAMILY_COMMA segment 0, and every tail segment
104 TRAILING = 1 # the S2 peel: NO_COMMA, SUFFIX_COMMA segment 0
105 GIVEN_SLOT = 2 # #531's reading: FAMILY_COMMA segment 1
106
107
108class BoundJoin(IntEnum):
109 """v1 _join_bound_first_name's reserve_last, as the three states it
110 actually has. IntEnum: the value IS the number of name pieces
111 assign's peel must leave in the JOINED view for the join to stand
112 (#425), so the >= comparison below reads unchanged. Post-comma no
113 peel is run -- the pair alone is that one piece -- and DISABLED
114 is a mode, never compared: as a threshold 0 would join everything,
115 which is why the block is entered on identity first."""
116
117 DISABLED = 0 # the FAMILY_COMMA family segment (v1 never joined it)
118 LENIENT = 1 # FAMILY_COMMA's post-comma segment (reserve_last=False)
119 STRICT = 2 # main segments (reserve_last=True: keep a family piece)
120
121
122# rules.md#S2: "a trailing word of the suffix vocabulary reads as a
123# suffix" -- group does not decide that; it stops before whatever
124# trailing_start says the run is, so the chain and the maiden walk
125# end where assign's peel begins (#424).
126# rules.md#P2: "a particle joins the words after it into one name
127# part, the join running until the next particle starts a group of
128# its own, a trailing suffix begins" -- and on to the maiden marker
129# (M2) or the name's end; the final group reads as the family name,
130# earlier groups by position. (history: decisions.md#P2)
131# rules.md#P4: "a particle in the name's leading position chains
132# nothing: the words stay separate" (history: decisions.md#P2)
133def _is_prefix_piece(piece: Sequence[int], ptags: Set[str],
134 tokens: Sequence[WorkToken]) -> bool:
135 if "prefix" in ptags:
136 return True
137 return len(piece) == 1 and "particle" in tokens[piece[0]].tags
138
139
140# rules.md#M2: "a recognized maiden marker standing after at least one
141# name word takes the words after it" -- up to any suffix word, or the
142# trailing numeral assign reads as the suffix, as the maiden name, the
143# marker itself dropped (history: decisions.md#M2)
144#
145# A marker piece is a LONE marker -- M2's own "standing as a word of
146# its own". The consumer runs before every join but the Ph. D. merge
147# (see _group_segment), so a marker inside a wider piece is one the
148# consumer left there: declined ("Jane van der Berg née" reads family
149# 'van der Berg née'), or never examined (a leading marker, or a second
150# marker behind the span it took). Matching inside joined pieces would
151# re-read those.
152#
153# With the pass ahead of the joins no default-vocabulary input reaches
154# the lone-piece half through the one caller left that sees joined
155# pieces (P5's marker decline, marker(fk + 1)): a marker-headed wider
156# piece needs a connective right after a declined marker, and a
157# connective after a marker is a word the consumer takes. Measured at
158# #420's review --
159# dropping `len(piece) == 1` leaves the suite and a 337k-name sweep
160# identical -- so it stays as the rule's definition, not as a guard a
161# pin holds.
162def _is_maiden_marker_piece(piece: Sequence[int],
163 tokens: Sequence[WorkToken]) -> bool:
164 return (len(piece) == 1
165 and "vocab:maiden-marker" in tokens[piece[0]].tags)
166
167
168# mechanisms.md#ONE-PREDICATE-PER-QUESTION: "where the reader comes
169# AFTER the decider, record the answer on the state instead" -- which is
170# what the marker tags are. classify owns the phrase lookahead; group,
171# running later, reads what it wrote.
172def marker_run_length(following: Iterable[Set[str]]) -> int:
173 """How many tokens a marker run spans, given the tag sets of what
174 FOLLOWS its head, in order: 1 plus the leading continuations.
175
176 One walk for both places group asks -- over pieces below and over
177 token indices in the clause drop at the end of this module, 700
178 lines apart and differing only in index space. `following` is
179 consumed lazily and only until the run ends, so a caller passes a
180 generator over the rest of its sequence rather than materializing
181 one."""
182 run = 1
183 for tags in following:
184 if "vocab:maiden-marker-cont" not in tags:
185 break
186 run += 1
187 return run
188
189
190def _marker_run_pieces(seen: Sequence[int], pieces: Sequence[Sequence[int]],
191 tokens: Sequence[WorkToken], m: int) -> int:
192 """How many of `seen`'s pieces the marker run at seen[m] spans: 1
193 for a single-word marker, more for a phrase entry ('z domu').
194
195 classify already decided where the run ends and recorded it on the
196 tokens -- "vocab:maiden-marker" on the head,
197 "vocab:maiden-marker-cont" on the rest.
198
199 Each continuation is the NEXT piece and is always in `seen`, and
200 that holds because classify REFUSES to tag a run whose tokens are
201 not structurally contiguous. It is not a property of this walk, and
202 the reasons `seen` can skip a token are wider than they look. No
203 join has run yet, so a piece is one token. `seen` itself skips a
204 tail segment's delimiter cores, and a core between two marker words
205 would be a token between them, which classify would not have tagged
206 as a run. But `pieces` comes from a SEGMENT, and segment keeps only
207 the tokens no stage has given a role, bucketed by the commas before
208 them -- so a run half inside a bracketed clause, or split across a
209 structure comma, is one no segment holds whole. Walking cont tags
210 without classify's refusal read a proper PREFIX of the phrase as
211 the whole marker, and 'Anna z (domu) Nowak' lost its given name to
212 a bare preposition. Any future stage that removes a token from a
213 segment owes this the same refusal.
214
215 What classify guarantees is exactly the main stream's half of that
216 and no more: a run of ROLE-LESS tokens stays inside one segment. It
217 can also tag a run across two adjacent clauses of the same role,
218 which this walk never sees because a role-bearing token is in no
219 segment at all -- see _vocab.tag_marker_runs, which states the limit.
220 """
221 return marker_run_length(
222 tokens[pieces[seen[k]][0]].tags for k in range(m + 1, len(seen)))
223
224
225# rules.md#M2: "a word the clause gives up reads as a post-nominal or
226# the clause keeps it" -- the two halves of that, asked of the VIEW
227# the take would leave. Split out of `_maiden_take` because each
228# models a LATER decision about the released word, and the stop is
229# only right where that decision goes the way the release assumed
230# (#533 review).
231def _a_name_word_ahead(view: Sequence[Sequence[int]],
232 view_tags: Sequence[Set[str]],
233 tokens: Sequence[WorkToken],
234 at: int) -> bool:
235 """Whether the view holds a name word BEFORE the member at `at`.
236
237 #531's reading is the given part's own trailing slot, and a slot
238 needs a part: the take removes the clause, so where every piece
239 ahead of the member is a title or a suffix the member is the only
240 name word left and there is no trailing slot for it to end.
241 Asked of the view rather than of the segment because the segment
242 still has the maiden name in it -- that is exactly the word the
243 release takes away ('Doe, Prof. nee Smith A.B.' left 'Prof. A.B.',
244 whose A.B. is the GIVEN name and no credential at all)."""
245 return any(not is_title_piece(view[q], view_tags[q], tokens)
246 and not is_suffix_piece(view[q], view_tags[q], tokens)
247 for q in range(at))
248
249
250def _join_takes_the_member(view: Sequence[Sequence[int]],
251 view_tags: Sequence[Set[str]],
252 tokens: Sequence[WorkToken],
253 at: int) -> bool:
254 """Whether a join BELOW this pass would absorb the member at `at`.
255
256 A released word only reads as the credential while it is still the
257 lone piece assign's peel looks at; inside a joined piece it is a
258 word of the name the join built, and the release has moved it from
259 the birth name into the CURRENT one -- #424's class of failure,
260 now from the other side ('Berg, Jane van der nee Smith DO' read
261 family 'van der DO Berg').
262
263 Two joins can reach it, and each is modelled by the shape it needs
264 rather than by running it: P2's chain, where the member is
265 particle vocabulary standing behind a particle piece or beside
266 another released one ('Jane Doe nee Smith DO DO' read family
267 'DO DO'), and P5's bound-given join, where the member is the word
268 after the bound one ('Berg, abdul nee Jones MA' read given
269 'abdul MA'). Both over-decline rather than predict: a shape that
270 only MIGHT join keeps its word in the maiden name, which is the
271 conservative direction M2's invariant asks for."""
272 member = tokens[view[at][0]]
273 if "particle" in member.tags:
274 if at and _is_prefix_piece(view[at - 1], view_tags[at - 1], tokens):
275 return True
276 for q in range(at + 1, len(view)):
277 if len(view[q]) == 1 and "particle" in tokens[view[q][0]].tags:
278 return True
279 # P5 joins the first non-title piece to the one after it, so the
280 # member is at risk exactly where it IS the one after it. The tag
281 # pair is tested before `leading_titles` is asked, which is what
282 # keeps the ordinary credential release from paying that frame.
283 return (at > 0 and len(view[at - 1]) == 1
284 and "vocab:bound-given" in tokens[view[at - 1][0]].tags
285 and leading_titles(view, view_tags, tokens) == at - 1)
286
287
288# rules.md#M2: "a link inside the birth name does not end it" -- the
289# one shape the walk below steps over rather than stopping at.
290# rules.md#P3: "A connective that is also generational vocabulary
291# joins only where a name word stands on each side of it" is the
292# reason, and the class test is that rule's own. The take runs BEFORE
293# every join, so the link is still a piece of its own here and the
294# question is asked of the pieces as classify left them -- the same
295# inputs `_group_segment`'s `frozen` loop gives
296# `_between_name_words`, which is why this calls that predicate
297# rather than restating the class
298# (mechanisms.md#ONE-PREDICATE-PER-QUESTION).
299def _link_joins_inside_the_clause(k: int, lo: int, hi: int,
300 pieces: Sequence[Sequence[int]],
301 ptags: Sequence[Set[str]],
302 tokens: Sequence[WorkToken],
303 beside: list[_Beside]) -> bool:
304 """Whether the suffix piece at `k` is a connective PLACED TO JOIN
305 between two name words of the clause `lo`..`hi`.
306
307 The caller asks `is_suffix_piece` first and consults this only
308 where the answer was yes, so the generational half of "also
309 generational vocabulary" is settled and the connective half is
310 what is left to ask. A lone link is therefore no link at all
311 ('Jane Doe nee Puig i' keeps maiden 'Puig' and suffix 'i'), and
312 neither is one standing before the generation or the credential a
313 clause ends with ('... nee Puig i III', '... i MA'): `hi` is where
314 assign's peel begins, so those stand at or past it and
315 `_between_name_words` refuses them by bound.
316
317 Defined here, beside its one caller, and forward-referencing the
318 two predicates it is built out of: `_is_conj_piece` and
319 `_between_name_words` are the JOIN's, further down this module, and
320 moving them up to meet this would say they belonged to the clause.
321
322 `beside` is the caller's memo cell, filled on the first CONNECTIVE
323 piece the walk reaches rather than on the first suffix piece, so
324 a clause ending at an ordinary credential never builds it at all
325 and a clause holding a RUN of links builds it once ('Jane Doe nee
326 Puig i i i ... Soler'). Filled here rather than at the call site
327 so the laziness costs no frame of its own, and one fill serves
328 the whole walk because that walk mutates neither `pieces` nor
329 `ptags` (#397 second review, the run fix)."""
330 if not _is_conj_piece(pieces[k], ptags[k], tokens):
331 return False
332 if not beside:
333 beside.append(_run_neighbours(pieces, ptags, tokens))
334 return _between_name_words(k, lo, hi, pieces, ptags, tokens, beside[0])
335
336
337def _maiden_take(pieces: Sequence[Sequence[int]],
338 ptags: Sequence[Set[str]],
339 tokens: Sequence[WorkToken],
340 cores: Set[str],
341 one_case: bool | None,
342 reader: TailReader,
343 ambiguities: list[PendingAmbiguity],
344 ) -> MaidenIndices | None:
345 """The PIECE indices the marker pass removes, split the way
346 MaidenIndices declares them: the MARKER's pieces (one, or several
347 for a phrase entry like 'z domu') and the maiden name's. None when
348 the pass declines. Piece indices into `pieces`, not the TOKEN
349 indices of the MaidenTake `_group_segment` builds out of them --
350 the two are different index spaces and both are named types so a
351 reader never has to guess which one an annotation means.
352
353 Split here rather than by the caller because `run` is known here
354 and nowhere else -- it comes off the continuation tags, which are
355 read to find the walk's starting point in the first place.
356
357 Computed before any join (the Ph. D. merge aside), so "up to any
358 trailing suffix" means the first suffix WORD after the marker: a
359 connective beside that suffix cannot un-suffix it first. 'Jane
360 Smith née Jr y Jones' declines where the joined reading took 'Jr y
361 Jones' as the maiden name, and 'Jane Smith née Jones Jr y Smith'
362 takes only 'Jones'. The one reading the order costs; M2's Accepted
363 row and decisions.md#M2 (#420) record it.
364
365 "Up to any trailing suffix" also means up to a trailing CREDENTIAL
366 since #533, where the rule that reads the name left standing reads
367 the word as one -- the TRAILING peel with no comma, the given
368 part's own slot after a family comma, and nobody before that comma
369 or past a second one, which is what `reader` says. A member
370 standing alone after the marker is never given up: the marker
371 announces a name, and the rule gives a word up only where a maiden
372 name is left standing. Nor is one given up to a reader that will
373 not be there to read it, or to a JOIN that runs before the reader
374 does: rules.md#M2's invariant is that a released word ends the
375 parse suffix-roled, and the two view checks below are what makes
376 the stop conservative enough to hold it (#533 review).
377
378 A tail segment's delimiter cores (`cores`, empty elsewhere) are
379 structure, not words, and group() drops them after the pass --
380 before the pass moved ahead of the joins it dropped them first.
381 So the walk steps over them: a core is neither a word the marker
382 can take ('PhD née - Jones' read maiden '- Jones') nor the name
383 word M2 needs ahead of the marker ('- née Jones' took 'Jones',
384 leaving the core alone, which the drop then kept as the segment's
385 only piece). They stay in place for the drop, which still sees
386 the segment as written, which is why this returns indices rather
387 than a slice.
388 """
389 seen = [k for k in range(len(pieces))
390 if not (len(pieces[k]) == 1
391 and tokens[pieces[k][0]].text in cores)]
392 m = next((v for v in range(1, len(seen))
393 if _is_maiden_marker_piece(pieces[seen[v]], tokens)), None)
394 if m is None:
395 return None
396 # the marker may be a phrase, in which case it is several pieces
397 run = _marker_run_pieces(seen, pieces, tokens, m)
398 # "up to any trailing suffix": a suffix WORD anywhere after the
399 # marker ends the maiden name, and so does the trailing numeral as
400 # assign will read it, which the suffix-piece test does not see
401 # (#424): 'John née Jones Smith V' took the V as maiden text. Both
402 # forks now -- the acronym one since #533, asked the same double
403 # way. Read from the MARKER, not after it: a
404 # numeral straight after the marker then has the piece before it
405 # the fork wants, and 'Jane Smith née V' declines like 'Jane Smith
406 # née PhD' -- nothing after the marker but a suffix, so the marker
407 # stays a word -- as 1.4.0 read it.
408 #
409 # `one_case` is LIVE at these sites since #533 and was not before:
410 # the numeral fork is decided before the peel ever reads a lean,
411 # so the fact reached only the bare-acronym fork, which the
412 # numeral-only reading discarded. The acronym fork is asked now,
413 # so the writing decides here as it decides at the trailing slot
414 # of a name. Measured 2026-09-19 with a runtime wrapper forcing
415 # this function's `one_case` argument to None, over the population
416 # and the six policies decisions.md#S2's 2026-09-18 recipe names
417 # -- 36 of 10,752 parses move (1,792 names), on 6 distinct names
418 # ('Doe, Jane nee Smith DO', 'Doe, Jane nee Smith Ma', 'Jane Doe
419 # nee Smith Ma', 'Jane Doe nee Smith Ma JD', 'Jane Doe nee Yo-Yo
420 # Ma', 'John née Jones Smith MA'). THE PAIR IS THE FINDING: over
421 # the same corpus WITHOUT this change's own rows it moves 0, so
422 # the plumbing was live either way and the corpus simply held no
423 # name that could show it -- the blindness mechanisms.md's corpus
424 # field note asks to be measured before any "N names move" is
425 # written down.
426 #
427 # The chain-tail measure below (`tail`, and the re-peel after the
428 # chain) is the opposite, and the 2026-09-18 sweep over the
429 # pre-change 9,852 says so: dropping it moves 18 of them, on
430 # 'John van der Berg Ma', 'John de Ma' and 'Freiherr von Berg MA'
431 # under every one of the six policies.
432 skip = frozenset(range(len(pieces))) - frozenset(seen)
433 rest = peel_walk(seen[m], ptags, skip)
434 peeled = peel_trailing(rest, pieces, ptags, tokens, one_case)
435 trailing = rest[-1] if peeled.numeral is not None else len(pieces)
436 # The fork reads the piece before the numeral, and the take
437 # REMOVES that piece: afterwards assign sees the piece before the
438 # marker there, and if that is initial-shaped the fork will not
439 # fire -- a walk that stopped anyway handed the V to the family
440 # ('J. née Jones Smith V'). So the numeral must read as the suffix
441 # as the take would leave the name too, and the question is asked
442 # the way P5's reserve asks it (#425): the peel is run over the
443 # VIEW the take would leave, not one condition of it. Checking the
444 # preceding piece alone misses a title before the marker, where
445 # 'Dr. née Jones Smith V' leaves the numeral as assign's whole
446 # rest and no fork fires at all.
447 if trailing < len(pieces):
448 left = [i for i in seen if i < seen[m] or i >= trailing]
449 view = [pieces[i] for i in left]
450 view_tags = [ptags[i] for i in left]
451 # The same peel pair as above, read over the view -- and only
452 # `Peel.numeral` off it, because the bare-acronym fork COUNTS
453 # pieces and this view no longer holds the pieces it counted
454 # ('John née Jones Smith Ma' peeled over the pieces as written
455 # reads the acronym as a credential with words to spare, and
456 # once 'Jones Smith' has left it is the family of what
457 # remains). The acronym fork builds a view of its own below.
458 view_rest = peel_walk(leading_titles(view, view_tags, tokens),
459 view_tags)
460 if peel_trailing(view_rest, view, view_tags, tokens,
461 one_case).numeral is None:
462 trailing = len(pieces)
463 # #533: the ACRONYM fork, asked the way the numeral is -- the peel
464 # over the pieces as they stand, then again over the name the take
465 # would leave, and a stop only where both read the word as the
466 # credential. `peeled.names` is a COUNT of positions in `rest`, so
467 # `rest[peeled.names]` is the first piece the peel took and the
468 # class member it stopped at; no walk of our own is needed. One
469 # peel, one view built once per take: O(pieces) for the take, not
470 # per member, and no re-entrancy -- the predicate never calls the
471 # walk that calls it.
472 if (reader is not TailReader.NONE and peeled.names < len(rest)
473 and m + run + 1 < len(seen)):
474 # THE FIRST-WORD FLOOR: the stop never takes the FIRST word
475 # after the marker -- a class member standing alone there
476 # stays the maiden name. A clamp rather than a veto: where the
477 # peel consumed that word AND words behind it, only the first
478 # stays ('Doe, J. nee MA ba' keeps maiden 'MA' and reads
479 # suffix 'ba'; a veto handed 'ba' back to the clause too). The
480 # clamped piece may then be no member at all, and the test
481 # below declines -- which changes nothing, the walk stopping
482 # at that suffix word of its own accord.
483 stop = max(rest[peeled.names], seen[m + run + 1])
484 head = pieces[stop]
485 # `len(head) == 1` is DEFENSIVE and measured inert
486 # (2026-09-19) rather than unreachable: it asks a LONE piece's
487 # question, and the answer below reads `head[0]` as if the
488 # piece were the word. The multi-token piece it keeps out is
489 # the Ph. D. merge above, whose `suffix` ptag stops `peel_walk`
490 # ever returning it -- so the peel half of `stop` cannot be it,
491 # but the FIRST-WORD FLOOR is an index rather than a walk and
492 # does reach the merged piece whenever that piece is the second
493 # word after the marker ('BERG, ABDUL Z DOMU MA PH. D.').
494 #
495 # THE NEGATIVE CONTROL, measured 2026-09-19 over a population
496 # built to HOLD that shape -- 4,224 names (twelve heads x four
497 # markers x 23 bodies, each also with a ', MD' tail and in
498 # upper and lower case) under six policies and two lexicons,
499 # the default and one listing `ph` ambiguous, 50,688 parses. A
500 # probe that fires wherever the tag test admits a head this
501 # length test then DECLINES -- the only sites where dropping it
502 # could matter -- fires 1,440 times, against 25,920 reaches of
503 # this site and 19,296 tag admissions. Dropping it is
504 # byte-identical all the same -- fields, ambiguities and every
505 # token's role and tags -- over 905,796 parses (that population
506 # plus the review's 142,518-name corpus under the six
507 # policies). What it buys is the price, which is real where the
508 # count is not: with `ph` listed, dropping it takes 'BERG,
509 # ABDUL Z DOMU MA PH. D.' from 438 frames to 453. Kept for the
510 # reason `_assign.previous_kept` is: an inert branch is cheaper
511 # than a question asked of the wrong shape, and the three
512 # sibling sites (`_pieces.segment_suffix_reading`, the
513 # GIVEN_SLOT branch below, and the emitter at the end of this
514 # function) each pair a length test with a tag test the same
515 # way.
516 #
517 # The tag is the CLASS the rule is stated in terms of, and it
518 # is not redundant with the walk the way the length test is:
519 # 'J. née Jones Smith V' reaches here on a piece
520 # `is_suffix_piece` REFUSES for being initial-shaped, and what
521 # declines it is the view check rather than the walk. Its own
522 # control is a price too, not a count: dropping it runs the
523 # view machinery over every ordinary credential the peel took
524 # -- 'Jane Doe nee Smith PhD' 341 -> 353 frames and 'Doe, Jane
525 # nee Smith PhD' 367 -> 372, counted per `Parser.parse` the way
526 # tests/v2/test_benchmark counts them. No test pins those
527 # numbers: `_CALL_BASELINE` is per-interpreter and per entry
528 # point, and a row for one name would have to be guessed for
529 # the four interpreters only CI runs.
530 #
531 # A THIRD condition stood here and is gone: `stop < trailing`
532 # guarded nothing, `stop` being the larger of a walked piece
533 # and the floor and both bounded by `trailing`, so at worst
534 # `stop == trailing` and the assignment below sets `trailing`
535 # to what it already is. Measured over 1,760,904 parses:
536 # `stop > trailing` never once, `stop == trailing` 15,696
537 # times, and the tag test declined every one of those.
538 if (len(head) == 1
539 and AMBIGUOUS_ACRONYM_TAG in tokens[head[0]].tags):
540 left = [i for i in seen if i < seen[m] or i >= stop]
541 view = [pieces[i] for i in left]
542 view_tags = [ptags[i] for i in left]
543 # where the member stands in that view: everything before
544 # the marker, then the run the take would leave behind.
545 # The question is whether the reader takes THIS piece, not
546 # whether it takes something -- a suffix word behind the
547 # member answers yes to the weaker question while the
548 # member itself reads as the family name ('JOHN NEE JONES
549 # SMITH MA PHD' left 'JOHN MA PHD', whose MA is the family).
550 at = left.index(stop)
551 if reader is TailReader.GIVEN_SLOT:
552 # after a family comma the words to spare are there by
553 # construction, so the reader is #531's -- the member's
554 # own reading, asked through the one predicate that
555 # owns it, and that rule's FLOOR: the member ends the
556 # given part only where every piece behind it reads as
557 # a suffix too ('Doe, Jane nee Smith MA do' keeps
558 # maiden 'Smith MA do', the trailing particle not
559 # being a credential, so the clause keeps both words
560 # rather than handing one of them to the current
561 # name's middle -- which is what the clause-less
562 # 'Doe, Jane MA do' does with them, middle 'MA' and
563 # family 'do Doe'). And a slot the take would
564 # leave nobody to read is no slot: `_a_name_word_ahead`
565 # is that half, asked first because it is the cheaper
566 # question and because with no name word ahead the
567 # answer below is about a name that would not exist.
568 takes = _a_name_word_ahead(view, view_tags, tokens, at) and all(
569 is_suffix_piece(view[q], view_tags[q], tokens)
570 or (len(view[q]) == 1
571 and AMBIGUOUS_ACRONYM_TAG in tokens[view[q][0]].tags
572 and credential_at_the_given_slot(
573 tokens[view[q][0]], one_case))
574 for q in range(at, len(view)))
575 elif reader is TailReader.TRAILING:
576 takes = trailing_start(
577 leading_titles(view, view_tags, tokens),
578 view, view_tags, tokens,
579 one_case=one_case) <= at
580 else:
581 assert_never(reader)
582 # rules.md#M2: "a word the clause gives up reads as a
583 # post-nominal or the clause keeps it". Both readers above
584 # ask what a TRAILING rule makes of the member, and a join
585 # below this pass runs first and can take the word out of
586 # that rule's reach entirely, so the release is withdrawn
587 # where one would (#533 review).
588 if takes and not _join_takes_the_member(
589 view, view_tags, tokens, at):
590 trailing = stop
591 # The walk starts past the WHOLE marker: a phrase's second word is
592 # the marker, not the first word it takes. With nothing behind the
593 # marker at all there is no clause to walk and no first word to
594 # bound it with, so the decline the `j <= m + run` test below
595 # reaches is taken here instead -- `lo` would have no piece to name
596 # ('Jane van der Berg née').
597 if m + run >= len(seen):
598 return None
599 # rules.md#M2: "a link inside the birth name does not end it" --
600 # the clause's OWN bounds for the link exception, which are not the
601 # segment's. `lo` is the first piece after the marker run, so the
602 # marker is never the name word on a link's left, and a delimiter
603 # core between the marker and that first word is below `lo` by
604 # construction and cannot pass for one either. A core is the TAIL
605 # segment's alone (`extra_suffix_delimiters`, empty by default),
606 # and a dash standing where no tail segment can hold it is an
607 # ordinary word at EITHER policy: 'PhD née - i Jones' keeps maiden
608 # '- i Jones' configured and unconfigured alike, there being no
609 # comma to make a tail out of. It takes the tail a suffix comma
610 # builds for the dash to be a core at all, and then the two
611 # policies part company -- 'Smith, John, PhD née - i Jones' keeps
612 # maiden '- i Jones' by default and declines under a configured
613 # ' - ', which is the pair
614 # test_a_core_between_the_marker_and_the_first_word_is_below_lo
615 # holds (all four readings measured 2026-09-22).
616 # NOT theoretical and not a whole claim about cores, both settled
617 # by measurement 2026-09-21 over corpus u cases.py u the property
618 # grids u a 50,925-name generated set with cores, under thirteen
619 # core-bearing policies: 25,536 of 596,392 maiden takes had a core
620 # standing exactly there, so the bound is load-bearing -- and PAST
621 # `lo` a core is no longer below it, is an ordinary index to
622 # `_run_neighbours` (which steps over connectives and nothing
623 # else), and DOES pass for the name word on a link's side. That
624 # reading is pinned as it stands rather than repaired here
625 # (test_a_core_beside_a_link_wrongly_passes_for_a_word_until_538):
626 # `_between_name_words` is asked about a core in 51,072 of 900,023
627 # calls, the answer differs from a core-skipping reading in 8,094
628 # parses over 1,278 texts, and 1,824 of those move `maiden` on 288
629 # texts -- none of them reachable at the default policy, which is
630 # why rules.md#M2 states it with a policy annotation beside the
631 # marker. The repair is `cores` threaded through
632 # three call sites into `_run_neighbours`, which is its own change
633 # (#538, and rules.md#M2 carries it as a `deviates:` example).
634 # `peel_start` is where assign's trailing run begins over
635 # the pieces as WRITTEN, so the generation or credential a clause
636 # ends with is never the name word on a link's right ('... nee Puig
637 # i III', '... i MA', whose MA carries no `vocab:suffix` tag for
638 # the piece test to refuse it by).
639 #
640 # `peel_start` is `trailing_start`'s whole answer, read off the
641 # peel pair above rather than re-running it, which is the reading
642 # that function's own docstring sends this caller here for. It is
643 # never past the walk's own stop, `trailing`: the numeral fork's
644 # `trailing` is the walk's LAST piece and the acronym fork's `stop`
645 # is a max over this one, so the two never disagree about where the
646 # clause ends, only about what the exception may reach across.
647 # Measured 2026-09-20 with a probe here over the whole suite --
648 # 93,408 reaches of this site, `peel_start > trailing` 0 of them.
649 lo = seen[m + run]
650 peel_start = (rest[peeled.names] if peeled.names < len(rest)
651 else len(pieces))
652 j = m + run
653 # The link exception's memo cell, filled inside the predicate on
654 # the first connective it is asked about (see its docstring).
655 beside: list[_Beside] = []
656 while (j < len(seen) and seen[j] < trailing
657 and (not is_suffix_piece(pieces[seen[j]], ptags[seen[j]],
658 tokens)
659 or _link_joins_inside_the_clause(seen[j], lo, peel_start,
660 pieces, ptags, tokens,
661 beside))):
662 j += 1
663 # j == m + run means nothing followed the marker but a suffix, so
664 # the pass declines and the marker stays ordinary words
665 # (rules.md#M2).
666 if j <= m + run:
667 return None
668 # #533, mechanisms.md#AMBIGUITY-AT-THE-DECISION-SITE: "Emit at the
669 # site that takes the branch, not where an ambiguous tag sits" --
670 # the walk that KEPT the word is where the fork was called, so it
671 # is where the report is raised. The LAST word of the maiden name
672 # is the one the trailing rule was asked about: everything behind
673 # it read as a suffix (that is what let the peel reach it), and a
674 # member the reading TOOK is not in the maiden name any more --
675 # assign reports that one where it peels it, so no token is ever
676 # reported twice. A member with a name word behind it was never
677 # asked and stays silent, which rules.md#A1's hesitating reader is
678 # the reason for rather than the accident of.
679 #
680 # Gated on the reader for the same reason the walk is: where no
681 # trailing rule reads these words, nothing was decided and nothing
682 # may report. Gated on EITHER tag, as the chain emitter's is, so a
683 # by-shape member reports with the dotted switch off -- classify
684 # writes the shape tag there while the class does not admit it,
685 # which is the one place a declined fork can be recorded.
686 #
687 # `last[0]` is the whole word and no `len(last) == 1` guards it,
688 # unlike the three sibling sites: here the invariant IS structural
689 # and a test would be inert by construction rather than by
690 # measurement. The only multi-token piece this pass can see is the
691 # Ph. D. merge, `merge` gives that piece the `suffix` ptag
692 # unconditionally, and `is_suffix_piece` answers yes off that ptag
693 # alone -- so the walk just above ENDS before it, and the last
694 # maiden piece is never it whatever a caller's vocabulary says.
695 # Measured too, both ways: a probe on `len(last) != 1` here fired
696 # 0 times over 1,760,904 parses, and deleting the test is
697 # byte-identical (fields, ambiguities, token roles and tags) over
698 # the 905,796-parse oracle at the same frame counts.
699 last = pieces[seen[j - 1]]
700 word = tokens[last[0]]
701 if (reader is not TailReader.NONE
702 and not word.tags.isdisjoint(_AMBIGUOUS_CREDENTIAL_TAGS)):
703 ambiguities.append(PendingAmbiguity(
704 AmbiguityKind.SUFFIX_OR_NAME,
705 f"{word.text!r} ending the maiden name is also "
706 f"a post-nominal; the maiden marker's clause keeps it "
707 f"rather than reading it as one",
708 tuple(last)))
709 return seen[m:m + run], seen[m + run:j]
710
711
712#: What `_between_name_words` reads instead of walking: two arrays over
713#: the segment's pieces, giving for each index the nearest piece on
714#: its left and on its right that is NOT a connective -- `-1` and
715#: `len(pieces)` where the run reaches the end. Built by
716#: `_run_neighbours`, and an ALIAS rather than a NamedTuple because a
717#: NamedTuple's __new__ is a frame of its own (decisions.md#parse-cost).
718_Beside = tuple[list[int], list[int]]
719
720
721def _run_neighbours(pieces: Sequence[Sequence[int]],
722 ptags: Sequence[Set[str]],
723 tokens: Sequence[WorkToken]) -> _Beside:
724 """The nearest non-connective piece on each side of every index.
725
726 EVERY MEMBER OF ONE RUN HAS THE SAME ANSWER, which is the whole
727 of the fix: `_between_name_words` used to walk the run itself, so a
728 name holding a run of n connectives walked it n times and the
729 stage went quadratic in the run's length -- measured 2026-09-20,
730 `"Josep " + "i " * n + "Rovira"` grew 3.8x per doubling against
731 the 2.0x every other shape holds, 59ms at n=800. Two linear
732 passes answer the same question once for the whole segment.
733 `tests/v2/test_benchmark.py`'s `link_run` shape is the guard.
734
735 `_is_conj_piece` is asked ONCE per piece, into a list the second
736 pass then reads: asking it in both passes would double the calls,
737 and it is the only per-piece call this builder makes. Recorded in
738 the FIRST pass rather than by a comprehension of its own -- which
739 is a code object on 3.11 and a bytecode saving here rather than a
740 frame one, measured: the profile hook emits no call event for it.
741
742 WHAT IT COSTS A SHORT NAME, because answering for the whole
743 segment is not free where the walk would have stopped at once:
744 this call, plus `_is_conj_piece` for the pieces the walk never
745 reached. Re-measured 2026-09-21 on py3.11 against b9ed1429, the
746 whole pair through `tests/v2/test_benchmark.py`'s own
747 `_frames_for` shape -- `Josep Carod i Rovira` 311 -> 313 frames
748 (one call and two more `_is_conj_piece` over its four pieces,
749 less the frame the fold below saved) and `Jane Doe nee Puig i
750 Soler` 315 -> 319. An O(1) rise per link-bearing name against an
751 unbounded saving: the same name with a run of 64 links goes
752 6,741 -> 2,587. (The pair first written here read 304 -> 307 and
753 307 -> 312, with 6,741 -> 2,652 for the run of 64. Re-measured
754 2026-09-22 on py3.11, WHICH OF THOSE REPRODUCE is: the two
755 short-name pairs, neither of them; the run-of-64 pair, its left
756 half only -- b9ed1429 reads 6,741 exactly, while 6048eb5d, the
757 tree the 2,652 was taken on, reads 2,651 here. So the interpreter
758 splice ran through a single arrow, which is what the table
759 `tools/perf/call_count.py`'s own docstring warns about. Every
760 figure above is one interpreter, stated.)
761 `tools/perf/call_count.py` is unmoved (parse=406.00,
762 facade=443.00) -- its reference name carries no link -- and so are
763 `John Smith`, `Smith, John`, `Juan Garcia y Lopez` and `Jane Doe
764 nee Smith`, none of which reaches this at all.
765
766 Called only where a generational connective was found in the
767 segment (the `frozen` loop) or where a clause's walk reached a
768 CONNECTIVE suffix piece (`_link_joins_inside_the_clause`), so no
769 ordinary name pays for it at all.
770
771 Dropping either pass's `not` fails
772 test_a_connective_piece_counts_toward_the_carve_outs_total
773 (mutation-checked 2026-09-20; how the two arrays are READ is
774 checked in `_between_name_words`, which reads them).
775 """
776 n = len(pieces)
777 left = [-1] * n
778 right = [n] * n
779 conj = [False] * n
780 prev = -1
781 for k in range(n):
782 left[k] = prev
783 is_conj = _is_conj_piece(pieces[k], ptags[k], tokens)
784 conj[k] = is_conj
785 if not is_conj:
786 prev = k
787 nxt = n
788 for k in range(n - 1, -1, -1):
789 right[k] = nxt
790 if not conj[k]:
791 nxt = k
792 return left, right
793
794
795# rules.md#P3: "a recognized connective joins its neighbors into one
796# name part, connective runs included — except a single-letter
797# connective in a three-word name, which stays a name word, and a
798# single-letter connective that reads as an initial instead, which
799# never joins" (history: decisions.md#P3)
800def _is_conj_piece(piece: Sequence[int], ptags: Set[str],
801 tokens: Sequence[WorkToken]) -> bool:
802 if "conjunction" in ptags:
803 return True
804 return len(piece) == 1 and "conjunction" in tokens[piece[0]].tags
805
806
807def _is_rootname(piece: Sequence[int], ptags: Set[str],
808 tokens: Sequence[WorkToken]) -> bool:
809 if len(piece) == 1 and "initial" in tokens[piece[0]].tags:
810 return False
811 # rules.md#P3: "A connective counts as a name word wherever this
812 # rule counts them, whatever else the vocabulary says the word is"
813 # (#397). The order against the `initial` refusal above decides
814 # nothing, and what makes that safe lives in classify rather than
815 # here: for a single letter it writes `initial` or `conjunction`
816 # and never both, so a one-case `I`/`i` arrives with no conjunction
817 # tag whichever test runs first -- measured, hoisting this arm
818 # above the refusal moves no field, report or initial on any
819 # corpus name under three name orders, and no test. INLINE rather
820 # than a call to _is_conj_piece: this runs once per piece of every
821 # name (frame budget).
822 if ("conjunction" in ptags
823 or (len(piece) == 1 and "conjunction" in tokens[piece[0]].tags)):
824 return True
825 return not (is_title_piece(piece, ptags, tokens)
826 or _is_prefix_piece(piece, ptags, tokens)
827 or is_suffix_piece(piece, ptags, tokens))
828
829
830# rules.md#P3: "a word the rest of the parse reads as a name word
831# rather than as a generation, a credential or an honorific, looked
832# for past any run of connectives standing between. A connective with
833# nothing to its right is connecting nothing, and a word of that
834# vocabulary ending a name, or standing before the credential a name
835# ends with, is the generation it also spells" (#397) -- the WHOLE
836# clause, its second sentence included, because that sentence is what
837# this predicate answering `False` means. Reading it as a generation
838# is the CALLER's half: `_group_segment`'s `frozen` set is where a
839# connective this refuses is placed as the generation, and
840# `_link_joins_inside_the_clause` is the maiden walk's.
841# `Sequence[Sequence[int]]` rather than `Sequence[Piece]`, widened
842# when the maiden walk became a second caller: this reads a piece and
843# never edits one, and `_maiden_take` holds its pieces at the wider
844# type the stage's entry point hands it.
845def _between_name_words(k: int, lo: int, hi: int,
846 pieces: Sequence[Sequence[int]],
847 ptags: Sequence[Set[str]],
848 tokens: Sequence[WorkToken],
849 beside: _Beside) -> bool:
850 """Whether such a word stands on EACH side of the connective piece
851 at `k`.
852
853 Both sides in one call because neither caller ever wants one: a
854 connective is placed to join only where a name word stands on both
855 sides of it, so the two answers were always ANDed at the call site
856 and the left one short-circuits the right either way. One frame per
857 generational connective rather than two, which is the whole of the
858 saving -- the left arm below is the old left call and the right arm
859 the old right one, unchanged (#397 follow-up).
860
861 `lo` and `hi` bound the name's own words: assign peels the pieces
862 below `lo` as its leading titles and those from `hi` up as its
863 trailing suffix run, so a piece outside that span is a credential
864 or an honorific however it is spelled, and the numeral or bare
865 acronym the peel takes ('i V', 'i MA') is inside `hi` by
866 construction rather than by a second reading of the vocabulary
867 (mechanisms.md#ONE-PREDICATE-PER-QUESTION).
868
869 Inside the span the suffix and title tests still run, because
870 neither bound reaches a credential or an honorific standing in
871 the MIDDLE of a name ('Josep Jr. i Rovira', 'Josep Dr. i
872 Rovira'): the peel walks from the end and stops at the first name
873 word, the title run from the front.
874
875 The answer steps over connectives because a RUN of them joins as
876 one ('Carod i y Rovira'), so the word this rule is about is the
877 first one past the run -- and where the run runs out ('Juan i e')
878 there is no name word on that side at all. `beside` is where that
879 stepping already happened: `_run_neighbours` walked every run once
880 for the whole segment, so this reads an index rather than walking
881 to it. A SENTINEL OUT OF RANGE is how "the run ran out" arrives --
882 -1 on the left, `len(pieces)` on the right -- and each bound test
883 below turns it into False, exactly as the walk did when it ran
884 here and stopped at the same place. `lo` is never negative and
885 `hi` never past `len(pieces)`, so neither sentinel can pass the
886 bound, and the piece tests are never asked about an index that is
887 not one.
888
889 Mutation-checked 2026-09-21, every arm of both sides by a NAMED
890 test. Reading the LEFT array for the right arm too fails
891 test_a_connective_with_nothing_to_its_right_does_not_join;
892 the RIGHT array for the left arm too fails
893 test_a_leading_title_on_the_left_is_no_name_word.
894 The bounds, WIDENED to the whole segment rather than dropped --
895 dropping the right one indexes past the pieces on the sentinel and
896 raises instead of failing a test -- fail
897 test_the_marker_is_not_the_name_word_on_the_links_left (left) and
898 test_the_right_hand_test_reads_the_peel_not_the_suffix_piece
899 (right). Either piece test on the LEFT fails
900 test_a_credential_or_honorific_mid_name_is_no_name_word_either;
901 either on the RIGHT fails
902 test_a_credential_or_honorific_mid_name_on_the_right_too, which is
903 the row the fold asked for: while this was two per-side calls a
904 mutation hit both sides at once and the left-hand rows covered for
905 the right-hand ones, whose own rows stand at the END of the name
906 where `hi` refuses them first. SWAPPING the two arrays outright is
907 an EQUIVALENT mutant and no test fails: the two arms are ANDed, so
908 which array answers which side is not a question the conjunction
909 can see.
910 """
911 left = beside[0][k]
912 if not (lo <= left < hi
913 and not is_suffix_piece(pieces[left], ptags[left], tokens)
914 and not is_title_piece(pieces[left], ptags[left], tokens)):
915 return False
916 right = beside[1][k]
917 return (lo <= right < hi
918 and not is_suffix_piece(pieces[right], ptags[right], tokens)
919 and not is_title_piece(pieces[right], ptags[right], tokens))
920
921
922def _group_segment(seg: tuple[int, ...], additional: int,
923 tokens: Sequence[WorkToken],
924 bound_join: BoundJoin = BoundJoin.STRICT,
925 ambiguities: list[PendingAmbiguity] | None = None,
926 cores: Set[str] = frozenset(),
927 given_name_titles: Set[str] = frozenset(),
928 opens_the_name: bool = False,
929 *,
930 one_case: bool | None,
931 reader: TailReader,
932 maiden_ambiguities: list[PendingAmbiguity],
933 ) -> tuple[list[Piece], list[set[str]], MaidenTake | None]:
934 pieces: list[Piece] = [[i] for i in seg]
935 ptags: list[set[str]] = [set() for _ in seg]
936 # Out-parameter, same shape as _assign_main: forks are reported
937 # where they are decided. A caller that passes None (or a throwaway
938 # list) suppresses reporting -- see group() for when that applies.
939 if ambiguities is None:
940 ambiguities = []
941 # The maiden walk's own channel, and REQUIRED beside `reader` for
942 # the same reason: the one production caller answers both off the
943 # segment's structure, and a default would be this module guessing
944 # what that caller already knows. group() passes `None` on the
945 # first for the chain emitter after a family comma -- the comma
946 # fixed the family, so that fork is settled -- and #533's is not
947 # that fork: a credential ending a maiden clause is a question the
948 # comma settles nothing about, which is why the two channels are
949 # two parameters. They are given the SAME list wherever nothing is
950 # suppressed, which is every segment that is NOT after a family
951 # comma; what the split buys is the other case, where `None` on
952 # the first must not reach the second -- a maiden channel
953 # defaulting to whatever the first was would let a caller passing
954 # `ambiguities=None` silence both (the review's finding).
955
956 def title(k: int) -> bool:
957 return is_title_piece(pieces[k], ptags[k], tokens)
958
959 def prefix(k: int) -> bool:
960 return _is_prefix_piece(pieces[k], ptags[k], tokens)
961
962 def suffix(k: int) -> bool:
963 return is_suffix_piece(pieces[k], ptags[k], tokens)
964
965 def conj(k: int) -> bool:
966 return _is_conj_piece(pieces[k], ptags[k], tokens)
967
968 def marker(k: int) -> bool:
969 return _is_maiden_marker_piece(pieces[k], tokens)
970
971 def joined_tags(lo: int, hi: int, add: Set[str] = frozenset(),
972 drop: Set[str] = frozenset()) -> set[str]:
973 # the ONE definition of a merged piece's tags: merge() applies
974 # it, and P5's reserve reads it to model the join it is
975 # weighing (#425) -- so the view cannot drift from the merge.
976 # A merged piece inherits every part's tags, so a site whose
977 # product is not what its parts were drops what no longer
978 # applies: the particle chain drops `prefix`, the bound join
979 # `title` (a derived title tag on the pair would have assign
980 # peel the given name as a leading title).
981 return (set().union(*ptags[lo:hi]) | add) - drop
982
983 def merge(lo: int, hi: int, add: Set[str] = frozenset(),
984 drop: Set[str] = frozenset()) -> None:
985 # pieces/ptags are parallel arrays; every merge must update
986 # both in lockstep.
987 #
988 # Extend the first piece IN PLACE rather than rebuilding the
989 # merged list. The obvious spelling --
990 # pieces[lo:hi] = [[i for p in pieces[lo:hi] for i in p]]
991 # -- re-flattens everything accumulated so far on every call, so
992 # a chain that merges into the same piece n times copies
993 # 1+2+...+n and the stage goes quadratic in the length of the
994 # chain. A conjunction run ("and " * n) does exactly that: it
995 # measured 2.4x-2.9x per doubling against the 2.0x every other
996 # shape holds. No piece list is aliased outside this function
997 # (each starts as a fresh [i], and the callers only read
998 # pieces[k] before a merge), so mutating is safe; verified
999 # identical token/role/tag/span/ambiguity output over 54,877
1000 # names. tests/v2/test_benchmark.py's "and " shape is the guard.
1001 #
1002 # Every call site passes lo < hi, and this REQUIRES it: with
1003 # lo >= hi the slice assignment would insert rather than
1004 # replace, putting a second reference to pieces[lo] into the
1005 # array, and the next merge to touch either index would extend
1006 # the same list twice. The old rebuild-a-fresh-list spelling
1007 # was harmless there. Keep the bound if you add a caller.
1008 combined = pieces[lo]
1009 for piece in pieces[lo + 1:hi]:
1010 combined.extend(piece)
1011 pieces[lo:hi] = [combined]
1012 ptags[lo:hi] = [joined_tags(lo, hi, add, drop)]
1013
1014 # ph-d merge first: "Ph." "D." adjacent -> one suffix piece
1015 # (decisions.md#phd-merge; v1 fix_phd did this by regex on the
1016 # raw string)
1017 # A suffix never BEGINS a name: position outranks the vocabulary
1018 # match (#371). This merge is what makes a leading "Ph." "D." a
1019 # credential at all -- every other suffix-shaped word standing
1020 # first already falls out as a title (H2's abbreviation clause, or
1021 # TITLES membership) or as a name word, so the pair is the only
1022 # shape that reaches the defect, and it reached it by emptying the
1023 # family: "Ph. D. Van Johnson" read given 'Van Johnson' with no
1024 # surname at all.
1025 #
1026 # WHERE THE INPUT BEGINS, and that is v1's own boundary rather
1027 # than a new one: fix_phd was a regex requiring a preceding space
1028 # (`\s(ph\.?\s+d\.?)`), so it could never fire at the head of the
1029 # string and fired everywhere else -- mid-name as readily as
1030 # trailing. Three tests, because "the head" has three meanings
1031 # here and only one of them is v1's:
1032 # `k == 0` the first PIECE, which is not the first word --
1033 # extract_delimited removes a bracketed or quoted
1034 # clause before segment runs, so `"Bob" Ph. D. John
1035 # Smith` reaches this with the pair at k == 0 and a
1036 # word standing before it in the input. v1 merges
1037 # there; declining broke parity on 112 measured
1038 # names, every one opening with a quote or bracket.
1039 # `a[0] == 0` the first TOKEN, which is v1's boundary.
1040 # seg_idx 0 and not after a family comma: a credential run
1041 # legitimately opens segment 1 ("Smith, Ph. D.
1042 # Jr.", C1's listing form), and segment 0 before a
1043 # comma is the family the comma named, where v1
1044 # merges too ("Ph. D., John" reads last 'Ph. D.').
1045 #
1046 # A leading TITLE is therefore NOT stepped over, unlike the #367
1047 # scan below, and the two disagree on purpose: that scan asks
1048 # where the NAME begins, this asks where the STRING does. `Sir
1049 # Ph. D. Van Johnson` keeps suffix 'Ph. D.' with an empty family,
1050 # which is #371's symptom surviving one word to the left -- 1.4.0
1051 # parity, stated as a boundary in rules.md#S2 rather than fixed
1052 # here, because "the first piece of the name" cannot be computed
1053 # before this merge: `is_leading_title` is true of `Ph.` itself,
1054 # so the scan would step over the very piece being judged.
1055 k = 0
1056 while k < len(pieces) - 1:
1057 a, b = pieces[k], pieces[k + 1]
1058 if (not (opens_the_name and k == 0 and a[0] == 0)
1059 and len(a) == 1 and len(b) == 1
1060 and PH.fullmatch(tokens[a[0]].text)
1061 and D.fullmatch(tokens[b[0]].text)):
1062 merge(k, k + 2, add={"suffix"})
1063 else:
1064 k += 1
1065
1066 # rules.md#M2: "a recognized maiden marker standing after at least
1067 # one name word takes the words after it" -- up to any suffix
1068 # word, or the trailing numeral assign reads as the suffix, as the
1069 # maiden name, the marker itself dropped
1070 # (history: decisions.md#M2) -- the marker pass (#274), and it runs
1071 # BEFORE every join below. Each join rule asks a question about the
1072 # name -- how many words it has (P3's carve-out, P5's reserve),
1073 # what the word after a particle or a bound word is -- and the
1074 # marker and the maiden name are not part of that name: they leave.
1075 # Asked while they were still pieces, the count answered for a name
1076 # that would not exist ('juan y garcia nee jones' counted five, 'y'
1077 # joined, and the family was empty once the clause left: #418), and
1078 # a join could absorb the marker before the pass looked for a lone
1079 # one ('Jane van der Berg née y Jones' kept the marker in the
1080 # family: #412). Removing the pieces first makes both impossible by
1081 # construction, with no stop on the chain (#399, #417) and no
1082 # exclusion in the reserve (#411) left to keep in step.
1083 #
1084 # The tokens are not touched here: this function reads them and
1085 # returns what it took, and group() records the drop and the roles.
1086 taken: MaidenTake | None = None
1087 take = _maiden_take(pieces, ptags, tokens, cores, one_case,
1088 reader, maiden_ambiguities)
1089 if take is not None:
1090 marker_ks, maiden_ks = take
1091 taken = ([i for k in marker_ks for i in pieces[k]],
1092 [pieces[k] for k in maiden_ks])
1093 for k in reversed(marker_ks + maiden_ks):
1094 del pieces[k]
1095 del ptags[k]
1096
1097 if len(pieces) + additional >= 3:
1098 # rules.md#P3: "A connective that is also generational
1099 # vocabulary joins only where a name word stands on each side
1100 # of it" (#397). `frozen` holds the TOKEN index of every such
1101 # connective missing that name word on one side or the other.
1102 # It is joining nothing, so it is the generation it also
1103 # spells: no join of its own reaches it, it may not merge into
1104 # a run, and it counts toward the carve-out total the way the
1105 # generation counted -- not at all, a suffix piece being no
1106 # rootname. A TOKEN index for the same reason the chain's
1107 # trailing run is a length from the end: the merges below move
1108 # piece indices and cannot move this one.
1109 #
1110 # "No join of its own" is the whole claim, and a NEIGHBOUR's
1111 # join can still absorb it: the two loops below skip a frozen
1112 # piece as the join's SUBJECT and nothing keeps it out of the
1113 # span another connective's join takes. 'Josep Carod Rovira
1114 # Puig y i' freezes the trailing 'i' -- nothing stands on its
1115 # right -- and the 'y' beside it joins across it all the same,
1116 # for family 'Puig y i', which is what the parent read there
1117 # too. Pinned by test_a_frozen_link_is_still_absorbed_by_a_
1118 # neighbours_join.
1119 #
1120 # Asked HERE, of the pieces as classify left them, and of the
1121 # NEIGHBOURS' class rather than of the connective's position:
1122 # any piece on each side passes a position test, so a
1123 # generational suffix standing behind the link was swallowed
1124 # into the name ("Josep Lluis Carod i III" read family 'Carod
1125 # i III'). It cannot be re-asked further down either, because
1126 # a merge answers it -- in the part a join produced, the
1127 # absorbed suffix IS the word standing on the right.
1128 #
1129 # Nothing but a connective of the suffix vocabulary reaches
1130 # the body, so a name that has none pays tag lookups and no
1131 # call at all.
1132 #
1133 # NO LENGTH TEST, which is the rule's own scope rather than an
1134 # omission: the clause quoted above names a CLASS and says
1135 # nothing about how the word is spelled. Narrowing it to
1136 # one-letter connectives was caller-reachable -- under
1137 # `Lexicon.default().add(conjunctions={"og"}, suffix_words=
1138 # {"og"})`, 'John Quincy Smith og' reads family 'Smith' plus
1139 # suffix 'og', the answer the rule states, and read family
1140 # 'Smith og' while a `len(tok.text) != 1` stood here (#397
1141 # second review). Nothing SHIPPED can witness it, `i` being
1142 # the only member of the class in the default vocabulary and
1143 # in every locale pack and one letter long -- measured rather
1144 # than assumed: byte-identical (fields, reports, `initials()`,
1145 # `capitalized()` plain and forced, every token's role) across
1146 # 359,053 parses on 2026-09-20, the 351,400-parse review grid
1147 # under eight lexicon/policy/locale configurations, the
1148 # 2,175-parse sweep of tests/v2/cases.py under three orders
1149 # and the 5,478-parse sweep of the differential corpora under
1150 # six. Pinned by test_a_multi_letter_link_of_the_suffix_
1151 # vocabulary_joins_by_the_same_rule.
1152 #
1153 # The three-word carve-out below stays single-letter, because
1154 # THAT is what its own sentence says ("a single-letter
1155 # connective in a three-word name").
1156 frozen: set[int] = set()
1157 lo = hi = -1
1158 beside: _Beside = ([], [])
1159 for k, piece in enumerate(pieces):
1160 tok = tokens[piece[0]]
1161 if (len(piece) != 1
1162 or "conjunction" not in tok.tags
1163 or "vocab:suffix" not in tok.tags):
1164 continue
1165 # The bounds and the run memo, computed once per segment
1166 # and only where such a connective was found, so the cost
1167 # is the link's own and no ordinary name pays any of it.
1168 if hi < 0:
1169 lo = leading_titles(pieces, ptags, tokens)
1170 # H5's reading and not the peel over the pieces as
1171 # WRITTEN: a title standing behind the suffix run
1172 # hides it from `trailing_start`, which then answers
1173 # `len(pieces)` and hands this loop a credential as
1174 # the name word on the link's right. 'John Quincy
1175 # Adams i MA Prof.' joined to family 'Adams i MA' with
1176 # no report at all, where 'John Quincy Adams i MA' --
1177 # the same name, one title shorter -- reads family
1178 # 'Adams', suffix 'i MA' and reports the acronym
1179 # (#397 second review).
1180 hi = trailing_start_past_titles(lo, pieces, ptags,
1181 tokens,
1182 one_case=one_case)
1183 # ONE ANSWER PER RUN: every member of a contiguous run
1184 # of connectives has the same nearest name word on
1185 # each side, and asking per member walked the run once
1186 # per member -- quadratic in its length, 3.8x per
1187 # doubling measured at `b9ed1429`.
1188 beside = _run_neighbours(pieces, ptags, tokens)
1189 if not _between_name_words(k, lo, hi, pieces, ptags, tokens,
1190 beside):
1191 frozen.add(piece[0])
1192 total = sum(_is_rootname(p, t, tokens)
1193 for p, t in zip(pieces, ptags)
1194 if p[0] not in frozen) + additional
1195 # contiguous conjunction runs merge first (v1: "of the")
1196 #
1197 # `pieces[k][0] in frozen` and not `frozen.isdisjoint(...)`:
1198 # the piece this loop extends GROWS with every merge, so a
1199 # test over its tokens costs 1+2+...+n and the stage goes
1200 # quadratic in the length of a connective run -- measured,
1201 # 'and ' x3200 took 41.8ms against 21.7ms, 6.2x per 4x input
1202 # where the shape reads 4.1x, and tests/v2/test_benchmark.py's
1203 # "and " shape is the guard that caught it. Reading the first
1204 # token alone is exact rather than an approximation: a frozen
1205 # piece is one token, nothing merges it (this branch declines,
1206 # and the join below skips it), so a piece holding a frozen
1207 # token IS that token.
1208 k = 0
1209 while k < len(pieces) - 1:
1210 if (conj(k) and conj(k + 1)
1211 and pieces[k][0] not in frozen
1212 and pieces[k + 1][0] not in frozen):
1213 merge(k, k + 2, add={"conjunction"})
1214 else:
1215 k += 1
1216 # each conjunction joins its neighbors, rules.md#P3: "except a
1217 # single-letter connective in a three-word name, which stays a
1218 # name word" (v1's Google Code issue 11 carve-out, the
1219 # "john e smith" bug). The threshold reads the ROOTNAME count,
1220 # and since #397 a connective counts ITSELF toward that count
1221 # WHERE IT IS JOINING, so a connective that is also suffix
1222 # vocabulary no longer raises the bar for its own join and no
1223 # longer lowers it for an unrelated one ("Carod y Rovira i"
1224 # counted the trailing generation and let the `y` join).
1225 k = 0
1226 while k < len(pieces):
1227 # first token again, and here it is exact for the second
1228 # reason as well: the piece a join produces is left BEHIND
1229 # `k`, so no merged piece is ever tested twice.
1230 if not conj(k) or pieces[k][0] in frozen:
1231 k += 1
1232 continue
1233 text = " ".join(tokens[i].text for i in pieces[k])
1234 if len(text) == 1 and text.isalpha() and total < 4:
1235 k += 1
1236 continue
1237 start = max(0, k - 1)
1238 end = min(len(pieces), k + 2)
1239 neighbor = start if start < k else end - 1
1240 derived = set()
1241 if title(neighbor):
1242 derived.add("title")
1243 if prefix(neighbor):
1244 derived.add("prefix")
1245 merge(start, end, add=derived)
1246 k = start + 1
1247 # prefix chains: a non-leading prefix run absorbs everything to
1248 # the next prefix or suffix (v1's leading_first_name rule keeps
1249 # the first piece a name: "Van Johnson")
1250 #
1251 # "Leading" means the first piece of the NAME, not of the input
1252 # (#367): a title is not part of the name, so it must not decide
1253 # whether the name begins with a particle. Keyed on index 0, a
1254 # title displaced the particle and the chain fired, so identical
1255 # name text parsed two ways ("Van Johnson" -> given Van, family
1256 # Johnson; "Dr. Van Johnson" -> family "Van Johnson").
1257 #
1258 # "Title AND NOT prefix" rather than the plain "not a title" the
1259 # rule is stated as, and the difference is not academic: `st`,
1260 # `do` and `freiherr` are each BOTH a title and an ambiguous
1261 # particle, so the plain test skipped over the very piece the
1262 # exception exists to protect and "St John Smith" -- no title in
1263 # front of it at all -- collapsed from title St, given John,
1264 # family Smith into one given "St John Smith". A piece that
1265 # could be the name's own first piece stops the scan; only a
1266 # piece that can ONLY be a title is stepped over.
1267 #
1268 # Computed once, before the loop: every merge below starts at
1269 # some k at or past this index, so no merge can move it.
1270 #
1271 # Suffix pieces are deliberately NOT skipped, and the reason is
1272 # what skipping them WOULD do rather than what it would cost.
1273 # (The "Ph. D. Van Johnson" example that opened this argument
1274 # left it with #371: the pair no longer merges at the head of
1275 # the input, so there is no suffix-shaped leading piece there
1276 # to skip or not. "Dr. Ph. D. Van Johnson" is the replacement
1277 # -- family 'Van Johnson' as shipped, family 'Johnson' with
1278 # the skip.)
1279 # The shapes that decide it are the ones whose leading piece
1280 # lands in `given` instead: "Ph.D. Van Johnson", "II Van
1281 # Johnson" and "Msc.Ed. Van Johnson" each read given
1282 # 'Ph.D.'/'II'/'Msc.Ed.' with family 'Van Johnson', and
1283 # skipping the piece moves `Van` out of the family and into the
1284 # middle name (given 'Ph.D.', middle 'Van', family 'Johnson')
1285 # -- a worse reading, on three shapes, to fix none. "Jr. Van
1286 # Johnson", the shape that looks like it needs the skip,
1287 # classifies its leading piece as a TITLE and is already
1288 # covered here.
1289 #
1290 # A maiden marker the consumer took is already gone (the pass
1291 # above), so the chain cannot carry a maiden name into the
1292 # family the way "Ursula von der Leyen geb. Albrecht" once read
1293 # family 'von der Leyen geb. Albrecht' (#399). A marker still
1294 # here is one the consumer DECLINED -- nothing after it but a
1295 # suffix, or nothing at all -- and M2 says that is just a word,
1296 # so the chain takes it like any other: "Jane van der Berg née"
1297 # reads family 'van der Berg née'. Stopping at it instead
1298 # stranded it as a lone trailing piece that took the family
1299 # field (#399's review), and a stop gated on "will the consumer
1300 # take it" had to restate the consumer's condition and got it
1301 # wrong one suffix later (#417).
1302 #
1303 # The `, 0` fallback is inert by construction rather than a
1304 # default worth testing: it is reached only when every piece is
1305 # a title and none is a prefix, and the loop below merges
1306 # nothing unless some piece is a prefix.
1307 # `title(k)` alone missed H2's unlisted abbreviations, which
1308 # assign peels as titles all the same, so 'Xyz. van Johnson'
1309 # chained where 'Dr. van Johnson' did not (#424 found it
1310 # through the acronym fork: the chain had swallowed the given
1311 # word and left assign two pieces where the fork counted
1312 # three). The scan asks assign's own test.
1313 leading = next((k for k in range(len(pieces))
1314 if not is_leading_title(pieces[k], ptags[k],
1315 tokens)
1316 or prefix(k)), 0)
1317 # rules.md#P2: "a trailing suffix begins" -- where it begins
1318 # is read by assign's peel over the pieces as they stand
1319 # (#424), once per segment and kept as a length from the end,
1320 # which the chain's merges ahead of it do not move -- except
1321 # where a particle that is suffix vocabulary too (vd, mc, do)
1322 # starts the run: the prefix run below takes it as a particle,
1323 # as P6 reads it after a comma, and 'John van Mc' keeps family
1324 # 'van Mc' (every baseline's reading). A suffix WORD stops the
1325 # chain wherever it stands, and the trailing
1326 # run -- the numeral, or the bare acronym with words to spare
1327 # -- stops it where the suffix-piece test alone did not ('John
1328 # van der Berg V' read family 'van der Berg V'). The chain
1329 # takes both forks, and asks again after its merges whether
1330 # the acronym still has the pieces the fork counted (below).
1331 name_start = leading_titles(pieces, ptags, tokens)
1332 tail = len(pieces) - trailing_start(name_start, pieces, ptags,
1333 tokens, one_case=one_case)
1334 def chain(tail: int) -> None:
1335 k = 0
1336 while k < len(pieces):
1337 if k == leading or not prefix(k):
1338 k += 1
1339 continue
1340 j = k + 1
1341 while j < len(pieces) and prefix(j):
1342 j += 1
1343 while (j < len(pieces) - tail and not prefix(j)
1344 and not suffix(j)):
1345 j += 1
1346 # The other half of PARTICLE_OR_GIVEN. _assign reports the
1347 # fork when an ambiguous particle stays a lone leading piece
1348 # ("Van Johnson" -> given under the default order, family
1349 # under FAMILY_FIRST); the chain here takes the opposite
1350 # branch when the particle is not the name's leading piece.
1351 # A fork whose two sides are decided in different stages
1352 # needs an emitter in each.
1353 #
1354 # Narrow, and #367 is why. `all(is_leading_title(...))`
1355 # says every piece ahead of this one is a title, and the
1356 # loop skipped k == leading, so `leading` is STRICTLY
1357 # before k -- and being before k it is one of those titles,
1358 # while being `leading` it satisfies `not title or prefix`.
1359 # For both, it must be a prefix as well: a word in both
1360 # vocabularies (`st`, `do`, `freiherr` by default, or any
1361 # overlap a caller configures). A plain title alone can no
1362 # longer put a particle off the name's leading piece; it is
1363 # stepped over and _assign reports the fork instead.
1364 #
1365 # What that leaves is wider than one shape: any number of
1366 # plain title pieces, then a piece in BOTH vocabularies,
1367 # then any number of further titles, then the ambiguous
1368 # particle whose chain claims something. "Freiherr von
1369 # Richthofen" is the canonical spelling and the one
1370 # tests/v2/cases.py and tests/v2/test_parser.py lead with,
1371 # but "St Van Johnson", "Do St Johnson" (the chained
1372 # particle itself in both vocabularies) and "Dr. Do van
1373 # Johnson" (a plain title AHEAD of the both-vocabulary
1374 # word) all reach here too. What none of them can do is
1375 # dispense with the both-vocabulary WORD. The conjunction
1376 # merge is the only other way a piece acquires `title` or
1377 # `prefix`, and it cannot manufacture the pair: it derives
1378 # from ONE neighbor, which is the left one whenever there
1379 # is a left one, and its right operands are always fresh
1380 # pieces (the loop runs left to right, so nothing to the
1381 # right has been merged yet). Both tags therefore have to
1382 # come from the piece it extends, which bottoms out at a
1383 # lone token in both vocabularies.
1384 #
1385 # j > k + 1 is what makes this a DECISION rather than a
1386 # shape: when the next piece is a suffix the inner scan
1387 # never advances, merge(k, k+1) folds a piece into itself,
1388 # and the particle stays a lone leading piece -- nothing
1389 # was chained, and _assign reports that case instead.
1390 # Without this the two emitters both fire on the same token.
1391 # (Tag test first: it is a set lookup and almost no name has
1392 # an ambiguous particle, while title() is a call per piece.)
1393 if (j > k + 1
1394 and "vocab:particle-ambiguous"
1395 in tokens[pieces[k][0]].tags
1396 and all(is_leading_title(pieces[x], ptags[x],
1397 tokens)
1398 for x in range(k))):
1399 i = pieces[k][0]
1400 ambiguities.append(PendingAmbiguity(
1401 AmbiguityKind.PARTICLE_OR_GIVEN,
1402 f"{tokens[i].text!r} was chained onto the following "
1403 f"name piece; it is also a given name in other "
1404 f"names",
1405 (i,)))
1406 # rules.md#S2: "A BARE ambiguous acronym is consumed
1407 # only when the name has words to spare — as the second
1408 # of two words it stays the family name — and at the
1409 # slots that report, either reading carries the
1410 # ambiguity flag"
1411 #
1412 # The other half of SUFFIX_OR_NAME's declined branch,
1413 # and it is here for the reason PARTICLE_OR_GIVEN's
1414 # second emitter is: a fork whose branches are taken in
1415 # different stages needs an emitter in each
1416 # (mechanisms.md#AMBIGUITY-AT-THE-DECISION-SITE).
1417 # `assign` reports from `peel_trailing`'s picks, and a
1418 # pick reaches it only as a LONE piece -- the peel's own
1419 # `len(piece) == 1` test -- so the moment this chain
1420 # takes the acronym into the particle run, the token
1421 # assign would have reported on no longer exists as a
1422 # piece and NOBODY reports. Measured: 'John van der Berg
1423 # Ma', 'John de Ma', 'Dr. John van Smith Ma', 'John van
1424 # Smith Ma Jr.' and 'John Smith Mc Ma' each lost the
1425 # report #289's own lean had just made true of them,
1426 # while 'John Smith Ma' -- the same fork with no
1427 # particle to chain -- kept it (review round).
1428 #
1429 # The DECLINED reading is what this reports, because
1430 # this branch only runs over pieces the peel left
1431 # standing: `tail` is the peel's own answer, the inner
1432 # scan stops at `len(pieces) - tail`, and a piece the
1433 # peel TOOK is behind that bound. Where the chain runs
1434 # again with a smaller tail (the re-peel below), the
1435 # first pass's appends are truncated with the pieces,
1436 # so the surviving report is the surviving reading's.
1437 #
1438 # `j > k + 1` above is this test's floor too: a merge
1439 # that folds a piece into itself chained nothing.
1440 # Written inline against the tags and with no new walk
1441 # -- `pieces[j - 1]` is the last piece the merge is
1442 # about to claim, one index and one frozenset test --
1443 # so the ordinary chained name ('de la Vega') pays no
1444 # frame for it.
1445 #
1446 # `not prefix(j - 1)` is the other floor, and it names
1447 # WHICH of the two scans above claimed the piece. The
1448 # first extends the PARTICLE run and the second takes
1449 # name words up to the trailing suffix; only the second
1450 # is taking a word the peel had looked at. A word in
1451 # both vocabularies ('do', 'mc', 'vd') ends a particle
1452 # run as a particle, which is P4's reading and P6's
1453 # fork, not this one -- 'anh van do' has read family
1454 # 'van do' silently since 1.4.0 and its case row says
1455 # so. It is tested LAST, and measured: `prefix` is a
1456 # closure over `_is_prefix_piece`, so asking it is TWO
1457 # frames, and asking it ahead of the tag test moved the
1458 # reference name from 412 to 414 -- every chained name
1459 # in the library paying for a question only an
1460 # ambiguous acronym can make interesting. Behind the
1461 # `isdisjoint` (a C call, no frame) almost nothing
1462 # reaches it.
1463 last = pieces[j - 1]
1464 if (j > k + 1 and len(last) == 1
1465 and not tokens[last[0]].tags.isdisjoint(
1466 _AMBIGUOUS_CREDENTIAL_TAGS)
1467 and not prefix(j - 1)):
1468 ambiguities.append(PendingAmbiguity(
1469 AmbiguityKind.SUFFIX_OR_NAME,
1470 f"{tokens[last[0]].text!r} is both a post-nominal "
1471 f"and an ordinary name; the particle chain took "
1472 f"it into the name rather than reading it as a "
1473 f"post-nominal",
1474 tuple(last)))
1475 merge(k, j, drop={"prefix"})
1476 k += 1
1477
1478 # The peel was read over the pieces as they stand, and the
1479 # chain's own merges can change what it counts: behind a word
1480 # in both the title and particle vocabularies the scan above
1481 # stops where assign's title peel does not (P4, #367), so the
1482 # chain takes the name's first word, and the acronym the fork
1483 # counted with three pieces meets assign with two -- 'Freiherr
1484 # von Berg Ma' read given 'von Berg', family 'Ma' (1.4.0's
1485 # reading; the reviews found it behind the claim that the
1486 # merges leave the count alone). So the peel is asked again
1487 # over the pieces the chain leaves, and where it no longer
1488 # takes what the chain stopped before, the chain runs again
1489 # without that stop: what assign will not peel is a name word,
1490 # and the chain takes it. The numeral cannot flip (a chain
1491 # group is never initial-shaped), so the second run is the
1492 # acronym's alone, and rare; the snapshot is one copy per
1493 # segment with a trailing run, linear like the rest.
1494 if tail:
1495 kept = ([list(q) for q in pieces], [set(t) for t in ptags],
1496 len(ambiguities))
1497 chain(tail)
1498 left = len(pieces) - trailing_start(
1499 leading_titles(pieces, ptags, tokens), pieces, ptags,
1500 tokens, one_case=one_case)
1501 if left < tail:
1502 pieces[:], ptags[:] = kept[0], kept[1]
1503 del ambiguities[kept[2]:]
1504 chain(left)
1505 else:
1506 chain(0)
1507 # rules.md#P5: "a recognized bound given-name word joins the
1508 # word after it into one given name" (history: decisions.md#P5)
1509 # -- bound given names: the first non-title piece joins the next
1510 # ONCE (pairwise, v1 parity: 'Salem, Abdul Rahman Ahmed' keeps
1511 # Ahmed a middle name). BoundJoin encodes v1's reserve_last.
1512 # "the first non-title piece" by assign's count (#424): group's
1513 # title test does not see H2's unlisted abbreviations, and
1514 # 'Xyz. abdul John Smith' joined nothing where 'Dr. abdul John
1515 # Smith' read given 'abdul John'.
1516 fk = leading_titles(pieces, ptags, tokens)
1517 if (bound_join is not BoundJoin.DISABLED
1518 and fk + 1 < len(pieces)
1519 and len(pieces[fk]) == 1
1520 and "vocab:bound-given" in tokens[pieces[fk][0]].tags):
1521 # P5 joins the bound word to "the word after it", and a
1522 # marker is not a name word -- it is the announcement that
1523 # another name follows. The only marker left by now is one
1524 # the consumer declined (nothing after it, or nothing but
1525 # a suffix), and the join must still not take it: 'Berg,
1526 # abdul nee PhD' read given 'abdul nee', and 'Berg, abdul
1527 # nee' clears the LENIENT reserve the same way. Measured
1528 # rather than assumed -- dropping this undid #411 on
1529 # exactly that row. Nor is a suffix piece (#421) --
1530 # rules.md#P5: "nor a word of the unambiguous suffix
1531 # vocabulary (S2), wherever position will then place it"
1532 # -- and declining it is also what keeps merge()'s tag
1533 # union from making the joined piece a suffix piece.
1534 if marker(fk + 1) or suffix(fk + 1):
1535 pass
1536 elif bound_join is BoundJoin.LENIENT:
1537 # post-comma the family is fixed and the pair is the
1538 # given whatever follows, so no peel is read
1539 # (decisions.md#P5, #423)
1540 merge(fk, fk + 2, drop={"title"})
1541 else:
1542 # rules.md#P5: "the join is tried on the pieces as it
1543 # would leave them, and the same reading assign runs
1544 # over them — its trailing peel (S2) and its trailing
1545 # title run (H5), each read over what the other leaves
1546 # until neither takes anything more — is read over that
1547 # view, the name words it leaves being the words to
1548 # spare"
1549 # (history: decisions.md#P5). The view is what
1550 # merge() builds -- the same slice assignment, the same
1551 # joined_tags -- and the reading is assign's own, the
1552 # ONE function that runs the peel and the H5 chain to
1553 # their fixed point (_pieces.tail_reading), so the
1554 # reserve and the assignment cannot drift. Modelling it
1555 # here as a subtraction instead is what let them: 'abdul
1556 # rahman MA' declined the join and 'abdul rahman MA
1557 # Prof.' took it, where H5 says the title changes
1558 # nothing (decisions.md#H5, 2026-09-09). And the join
1559 # changes no suffix reading -- rules.md#P5: "a word the
1560 # peel reads as a suffix unjoined must read so joined,
1561 # or the join declines" -- compared as the peeled
1562 # pieces themselves: 'abdul V' peels the V unjoined and
1563 # nothing joined, 'abdul Smith Ma' peels the acronym
1564 # unjoined and keeps it joined. Shapes pinned in
1565 # test_group.py.
1566 rest, chain_took, before = tail_reading(
1567 peel_walk(fk, ptags), pieces, ptags, tokens, one_case)
1568 view, view_tags = list(pieces), list(ptags)
1569 view[fk:fk + 2] = [pieces[fk] + pieces[fk + 1]]
1570 view_tags[fk:fk + 2] = [joined_tags(fk, fk + 2,
1571 drop={"title"})]
1572 view_rest, _, after = tail_reading(
1573 peel_walk(fk, view_tags), view, view_tags, tokens,
1574 one_case)
1575 same_suffixes = (
1576 [tuple(view[j]) for j in view_rest[after.names:]]
1577 == [tuple(pieces[j]) for j in rest[before.names:]])
1578 # rules.md#P5: "a trailing roman numeral, or a bare
1579 # acronym the peel takes, or a trailing title word the
1580 # run takes, is no word to spare" -- and the join joins
1581 # two NAME words, so a piece the chain takes is no more
1582 # joinable than a marker or a suffix piece is: 'Sir
1583 # abdul Prof.' reads title 'Sir Prof.', given 'abdul'.
1584 # Read off the UNJOINED view, the one that still has
1585 # the title as a piece of its own -- the join would
1586 # swallow it, and a swallowed title is a title the
1587 # joined view can no longer see. The suffix comparison
1588 # alone said this while the counts were subtractions,
1589 # by leaving the chained piece in the tail it compared;
1590 # under the shared reading the chain takes it out of
1591 # both views, so the rule is asked as the rule.
1592 chained = fk + 1 in chain_took
1593 # A given-name title ahead of the bound word asserts
1594 # that a given name follows -- the assertion H1 reads
1595 # when it keeps "Sir John" a given name -- so behind
1596 # one there is no family to spare (#369). Asked of the
1597 # title run through the ONE predicate post_rules asks
1598 # for H1, so the two rules cannot disagree about what
1599 # one run asserts; what it reads is the whole run's key
1600 # or that key's LAST word (#489). And it is the same
1601 # RUN on both sides: `range(fk)` is the pieces AHEAD of
1602 # the bound word, and H1 asks its own question of the
1603 # leading run too (_post_rules._addressing_run). They
1604 # did disagree for one commit -- H1 keyed every TITLE
1605 # token, so the trailing title in 'Sir abdul rahman
1606 # Prof.' joined this run and flipped the join's own
1607 # premise, handing the licensed pair to the family.
1608 # Reading the leading run at both sites is what makes
1609 # that unreachable rather than merely unlikely: a
1610 # trailing title is behind the word, and neither site
1611 # can see it. What H2's unlisted abbreviations do
1612 # inside such a run is the predicate's own business,
1613 # and its docstring is where they are worked through.
1614 # The licence lifts the reserve for two name
1615 # WORDS: the piece the join would take must be one word
1616 # -- a particle chain is the family name P2 built ('Sir
1617 # abdul van der Berg' keeps family 'van der Berg').
1618 licensed = (fk > 0 and len(pieces[fk + 1]) == 1
1619 and _run_addresses_by_given(
1620 (tokens[i].text
1621 for k in range(fk)
1622 for i in pieces[k]),
1623 given_name_titles))
1624 reserve = BoundJoin.LENIENT if licensed else BoundJoin.STRICT
1625 if (not chained and same_suffixes
1626 and after.names >= reserve):
1627 # the pair is a given name whatever tag the word
1628 # carried (rules.md#P5); joined_tags says why the
1629 # title tag is dropped. Pinned in test_group.py.
1630 merge(fk, fk + 2, drop={"title"})
1631 return pieces, ptags, taken
1632
1633
1634def group(state: ParseState) -> ParseState:
1635 tokens = list(state.tokens)
1636 dropped = list(state.dropped)
1637 ambiguities = list(state.ambiguities)
1638 all_pieces: list[tuple[tuple[int, ...], ...]] = []
1639 all_ptags: list[tuple[frozenset[str], ...]] = []
1640 # v1 parity: additional_parts_count=1 applies only to FAMILY_COMMA
1641 # parts; the SUFFIX_COMMA pre-comma segment gets 0.
1642 additional = 1 if state.structure is Structure.FAMILY_COMMA else 0
1643 # v1 expand_suffix_delimiter parity (#206): tail segments (wholly
1644 # consumed as suffixes by assign) drop delimiter-core tokens, the
1645 # same structural mechanism as the maiden marker (taken out in
1646 # _group_segment, recorded in `dropped` just below)
1647 cores = delimiter_cores(state.policy.extra_suffix_delimiters)
1648 tail_start = {Structure.SUFFIX_COMMA: 1,
1649 Structure.FAMILY_COMMA: 2}.get(state.structure)
1650 family_comma = state.structure is Structure.FAMILY_COMMA
1651 for seg_idx, seg in enumerate(state.segments):
1652 if family_comma:
1653 bound_join = (BoundJoin.LENIENT if seg_idx == 1
1654 else BoundJoin.DISABLED)
1655 else:
1656 bound_join = BoundJoin.STRICT
1657 # Suppressed after a family comma for the same reason _assign
1658 # suppresses it there: the family name is already fixed, so
1659 # there is no fork left to report.
1660 tail = tail_start is not None and seg_idx >= tail_start
1661 seg_cores = cores if tail else frozenset()
1662 # #533: which rule reads what the maiden walk would leave, off
1663 # the three facts already in hand here. A tail segment is read
1664 # as credentials whole and segment 0 of a family comma is the
1665 # family the comma named, so neither consults a trailing rule.
1666 if tail:
1667 reader = TailReader.NONE
1668 elif family_comma:
1669 reader = (TailReader.GIVEN_SLOT if seg_idx == 1
1670 else TailReader.NONE)
1671 else:
1672 reader = TailReader.TRAILING
1673 pieces, ptags, taken = _group_segment(
1674 seg, additional, tokens, bound_join,
1675 None if family_comma else ambiguities,
1676 seg_cores,
1677 state.lexicon.given_name_titles,
1678 opens_the_name=(seg_idx == 0 and not family_comma),
1679 one_case=state.one_case,
1680 reader=reader,
1681 maiden_ambiguities=ambiguities)
1682 # the marker is dropped and the maiden name's tokens become
1683 # MAIDEN (#274); which pieces those are was settled in
1684 # _group_segment, before the joins
1685 if taken is not None:
1686 marker_piece, maiden_pieces = taken
1687 dropped.extend(marker_piece)
1688 for piece in maiden_pieces:
1689 for i in piece:
1690 tokens[i] = dataclasses.replace(
1691 tokens[i], role=Role.MAIDEN)
1692 # rules.md#C1: "a part that is nothing but suffix words is the
1693 # credential run and reads as suffixes, whole" -- WHOLE is this
1694 # block's half of the rule, the routing being assign's.
1695 #
1696 # v1 expand_suffix_delimiter parity (#206): a delimiter core
1697 # inside a segment separates suffix entries and is dropped, but
1698 # a segment that IS only the core stays whole (v1 expand()
1699 # splits within a part, never erases a lone part). Keyed on
1700 # `tail` through `seg_cores`, which is empty off a tail
1701 # segment, because the #206 parity is a TAIL rule.
1702 #
1703 # What this block decides is the #206 core DROP and nothing
1704 # else: a delimiter core inside a tail segment leaves the
1705 # pieces, and `dropped` is where that fact is recorded -- for
1706 # the render, and for post_rules' entry pass, which reads it
1707 # back as the one dropped token that separates two entries.
1708 #
1709 # It used to decide the ENTRY too, marking a continuation
1710 # token "joined" between pieces off segment SHAPE -- `tail` by
1711 # index, ORed since #429 with segment_suffix_reading's
1712 # per-piece content verdict, gated per piece and sticky across
1713 # an interleaved title. post_rules derives that from the
1714 # commas the writer typed instead (#436/#437, rules.md#R1),
1715 # which is what the input answers directly. The stickiness
1716 # survives by construction rather than by code:
1717 # 'Smith, MD Dr. PhD' has no comma between MD and PhD and a
1718 # title between them renders elsewhere, so they join, while
1719 # 'Smith Jr., Mr. Jr.' has the writer's own comma and they do
1720 # not.
1721 if seg_cores:
1722 kept: list[int] = []
1723 for k in range(len(pieces)):
1724 is_core = (len(pieces[k]) == 1
1725 and tokens[pieces[k][0]].text in seg_cores
1726 and len(pieces) > 1)
1727 if is_core:
1728 dropped.extend(pieces[k])
1729 continue
1730 kept.append(k)
1731 if len(kept) != len(pieces):
1732 pieces = [pieces[k] for k in kept]
1733 ptags = [ptags[k] for k in kept]
1734 # continuation tokens of a suffix-merged piece (the ph-d merge)
1735 # carry the stable "joined" tag: the suffix string view joins
1736 # SUFFIX tokens with ", ", and the tag lets it heal the split
1737 for piece, piece_tags_ in zip(pieces, ptags):
1738 if "suffix" in piece_tags_ and len(piece) > 1:
1739 for i in piece[1:]:
1740 tokens[i] = dataclasses.replace(
1741 tokens[i], tags=tokens[i].tags | {"joined"})
1742 all_pieces.append(tuple(tuple(p) for p in pieces))
1743 all_ptags.append(tuple(frozenset(t) for t in ptags))
1744 # rules.md#M1: "a leading recognized marker being dropped where
1745 # the clause holds a word past it; a clause of nothing but its
1746 # marker keeps its words" — a marker inside EXTRACTED maiden
1747 # content (#329).
1748 # classify tags
1749 # such a marker like any other token -- what the #274 rule above
1750 # lacks is not the TAG but the token: extract claims a delimited
1751 # clause and tokenize gives its tokens Role.MAIDEN up front, so
1752 # segment (main stream = role is None) leaves them out of every
1753 # segment, they never enter `pieces`, and a rule that walks pieces
1754 # cannot reach them.
1755 #
1756 # Scoped to the CLAUSE, via state.extracted (one role + inner span
1757 # per delimited region), rather than to a maiden token's
1758 # neighbours. Both reasons are load-bearing:
1759 # * Role.MAIDEN is not proof of extraction -- the #274 rule above
1760 # sets it too, on the bare form, earlier in this same function.
1761 # A neighbour test would fire there and eat the 'Nee' out of
1762 # "Jane Smith nee Nee Jones". Keying on extracted spans puts
1763 # the bare path out of reach by construction.
1764 # * Separate clauses are separate content. In "(Nee) (Jones)" the
1765 # two land as one contiguous run of maiden tokens, so only the
1766 # clause bound keeps the lone "(Nee)" intact.
1767 # Drop the clause's LEADING MARKER only when the clause holds a
1768 # word past it: `Nee` is a real surname (Irish Ni/Nee, and a
1769 # Chinese romanization), so a one-token "(Nee)" is a maiden name,
1770 # not a marker. The marker and no more, whatever the clause holds
1771 # past it: cases.py's maiden_marker_delimited_three_token_clause is
1772 # the row that bounds this in both directions, every other
1773 # delimited row having a two-token clause where the two readings
1774 # agree. The marker is one token for a word entry and the whole run
1775 # for a phrase one ('z domu'), which is why the loop below counts
1776 # continuation tags rather than assuming one.
1777 # Spans index the original string by the anti-#100
1778 # invariant, and script_segment only ever splits a token into
1779 # sub-slices, so containment stays exact.
1780 #
1781 # Bisect rather than scan the token list per clause: that is
1782 # quadratic in the number of delimited pairs, and "(a) " * 3200 --
1783 # 4x test_benchmark's base, NOT a doubling -- measured a 14.1x cost
1784 # against the 4.1x the same shape holds under a policy with no
1785 # maiden_delimiters. The control is what says which unit a ratio is
1786 # in: linear is ~4 for 4x the input and ~2 for a doubling, so a 4.1x
1787 # control cannot be per doubling. Re-measured 2026-08-03 at 11.2x
1788 # against 4.2x (3.2x against 2.1x per doubling) -- the separation
1789 # replicates, the exact ratio moves with the runner. Same idiom,
1790 # and the same reason, as _extract._overlaps and _tokenize's origin
1791 # resolution. test_benchmark's maiden_pairs shape is the guard.
1792 if any(role is Role.MAIDEN for role, _ in state.extracted):
1793 starts = [t.span.start for t in tokens]
1794 for role, clause in state.extracted:
1795 if role is not Role.MAIDEN:
1796 continue
1797 # first token starting at or after the clause opens; tokens
1798 # are span-sorted and group never reorders or resizes them
1799 first = bisect.bisect_left(starts, clause.start)
1800 if (first >= len(tokens)
1801 or "vocab:maiden-marker" not in tokens[first].tags):
1802 continue
1803 # How many tokens the marker is: one, or the whole run of a
1804 # phrase entry, as classify recorded it. Same walk the
1805 # piece test above makes, in token-index space.
1806 run = marker_run_length(
1807 tokens[k].tags for k in range(first + 1, len(tokens)))
1808 # Testing the end of the token AFTER the run proves the run
1809 # AND that token are all inside: tokens do not overlap and
1810 # are index-ordered, so every earlier one ends no later,
1811 # and bisect already put first.start at or after
1812 # clause.start. That is also the "more than the marker"
1813 # test, since the tokens inside a clause are contiguous in
1814 # index order -- a clause holding nothing but its marker
1815 # keeps its words, exactly as a one-token "(Nee)" does.
1816 if (first + run < len(tokens)
1817 and tokens[first + run].span.end <= clause.end):
1818 dropped.extend(range(first, first + run))
1819 return dataclasses.replace(
1820 state, tokens=tuple(tokens), pieces=tuple(all_pieces),
1821 piece_tags=tuple(all_ptags), dropped=tuple(dropped),
1822 ambiguities=tuple(ambiguities))