1"""Stage: assign.
2
3Consumes: pieces + piece_tags (grouped), segments, structure, tokens,
4one_case.
5Produces: tokens with roles set on every main-stream token.
6Reads: Policy.name_order (#270), is_suffix_lenient on the trailing
7piece of a two-part comma name, and Policy.script_orders (#271, which
8overrides it when every name piece is written wholly in one script, or
9in the Han/Hiragana/Katakana repertoire the #272 kana license shares
10across pieces); token/piece tags; Lexicon only through tags already
11applied by classify (plus the leading-title period rule).
12
13Implements rules H2, H4, H5, N3, O4, O5 and W4 of docs/design/rules.md,
14each cited at its code below. Ports v1's assignment loops.
15NO_COMMA (per name_order):
16leading title pieces chain while no given-position name has been seen
17(a title needs a following piece, unless the whole name is one title);
18then positional assignment per name_order with the trailing-suffix
19rule: the piece from which everything after is a strict suffix is the
20last name-position piece, the rest are suffixes. Behind that peel a
21trailing run of period-marked title words chains into the title from
22the end, leaving one name piece standing; where the run TAKES
23something the peel was only provisional and runs again over the pieces
24with the titled ones spliced out, the two alternating until the run
25takes nothing, so a trailing title is transparent to the suffix
26reading however many titles are written (_pieces.tail_reading). Where
27the run takes nothing -- almost every name -- the first peel is the
28only one and its answer stands.
29The v1 single-name+nickname rule lives here (decisions.md#N3): a
30nonempty nickname beside exactly one piece in total puts that piece
31in FAMILY.
32FAMILY_COMMA: segment 0 wholly FAMILY (v1 parity) UNLESS segment 1
33holds no name word (titles and suffixes only), which fixed no family
34boundary -- there segment 0 takes the NO_COMMA positional read instead,
35order and all ('John Smith, Dr.', 'John Smith, Mr. Jr.'); segment
361 is wholly SUFFIX when it is nothing but suffix pieces ('Smith, Jr.',
37'Smith, Ph. D. Jr.' -- the credential run C1 describes, in the listing
38form), else gets leading titles, then the same trailing title run
39over the pieces this segment does not read as suffixes ('Smith, John
40Prof.'), then given, then middles with those suffix-reading pieces to
41suffix -- one predicate for both; segments 2+ are suffixes (lenient --
42segment already flagged non-suffixy ones COMMA_STRUCTURE).
43SUFFIX_COMMA: segment 0 as NO_COMMA; segments 1+ wholly SUFFIX.
44Emits PARTICLE_OR_GIVEN when the leading name piece is a lone
45particles_ambiguous token with more pieces following ("Van Johnson",
46and since #367 "Dr. Van Johnson" too, a title no longer displacing the
47particle out of that position) -- whatever role name_order assigns.
48Emits SUFFIX_OR_NAME at FIVE sites: the trailing roman numeral, each
49ambiguous acronym the trailing peel had to resolve, the bare-suffix
50carve-out where an input that is nothing but post-nominal vocabulary
51gets its first word made into the name (H4's suffix half, #491),
52-- since #289 -- the FAMILY-COMMA path's own read of the first
53post-comma piece, and -- since #531 -- the class member ENDING that
54path's given part, which the first-piece emitter could never reach.
55Further emitters of the same kind live in
56`_segment.py`, `_group.py` and `_post_rules.py`; they are not
57assign's and are not counted here. And
58at the one site that places a LONE name word, GIVEN_OR_FAMILY for the
59field the convention picked (O5, #449) and TITLE_OR_NAME for the two
60shapes where the doubt is whether a word is a title instead (H4,
61#491): the peel leaving one title-vocabulary word standing, and a
62joined unit carrying title vocabulary.
63"""
64from __future__ import annotations
65
66import dataclasses
67from collections.abc import Sequence, Set
68from typing import NamedTuple
69
70from nameparser._lexicon import Lexicon
71from nameparser._pipeline._vocab import (
72 effective_script, is_suffix_lenient, resolve_script_set,
73)
74from nameparser._pipeline._pieces import (
75 credential_at_the_given_slot,
76 is_suffix_piece, leading_titles, peel_walk,
77 segment_suffix_reading, tail_reading, trailing_titles,
78)
79from nameparser._pipeline._state import (
80 AMBIGUOUS_ACRONYM_TAG, ParseState, PendingAmbiguity, Structure,
81 WorkToken, _AMBIGUOUS_CREDENTIAL_TAGS, _NEVER_FLIPPED,
82)
83from nameparser._policy import Policy, Script
84from nameparser._types import AmbiguityKind, Role
85
86def _set_roles(tokens: list[WorkToken], piece: tuple[int, ...],
87 role: Role) -> None:
88 for i in piece:
89 tokens[i] = dataclasses.replace(tokens[i], role=role)
90
91
92#: Tags that say the word's own reading was claimed before position
93#: could speak, so O5's convention decided nothing. Built from M4's
94#: `_NEVER_FLIPPED` pair rather than respelling it: the two sets answer
95#: different questions -- may M4 retag the word, and did anything
96#: decide the field -- and share the pair because a word vocabulary
97#: claimed as a given name, or wrote as an initial, answers both. The
98#: third, `particle`, is here alone because a lone particle's reading
99#: is P4's. None is a predicate this emitter owns: they are read off
100#: the tags classify already recorded (mechanisms.md#TWO-LAYER-ASSIGN).
101_WORD_ALREADY_CLAIMED = _NEVER_FLIPPED | frozenset({"particle"})
102
103
104# rules.md#H2: "an abbreviation opening the part of the name that
105# carries the given name — the whole name, or the part after a
106# family comma — reads as a title even when unlisted" -- the count is
107# _pieces.leading_titles since #424 (its test, is_leading_title, is
108# the leading-particle scan's too); the roles are set here.
109def _peel_leading_titles(pieces: tuple[tuple[int, ...], ...],
110 ptags: tuple[frozenset[str], ...],
111 tokens: list[WorkToken]) -> int:
112 """Assign TITLE to the leading title pieces and return the first
113 non-title index."""
114 n = leading_titles(pieces, ptags, tokens)
115 for k in range(n):
116 _set_roles(tokens, pieces[k], Role.TITLE)
117 return n
118
119
120class EffectiveOrder(NamedTuple):
121 """What _effective_order made of a name's scripts. `order` is the
122 order the positional read uses; `by_script` says a script_orders
123 entry RESOLVED it, which is not the same as `order` happening to
124 equal the declared name_order -- under a declared family-first
125 order a Han name's W4 entry and the declaration agree, and only
126 this flag distinguishes the script rule DECIDING the reading from
127 the caller's declaration standing unopposed. O5's convention
128 report reads it (#449)."""
129
130 order: tuple[Role, Role, Role]
131 by_script: bool
132
133
134# rules.md#W4: "a name written wholly in one East Asian script, or in
135# the kana-licensed Japanese repertoire, reads family-first whatever
136# order the caller declared; a wholly-katakana name keeps the declared
137# order" (history: decisions.md#W4)
138def _effective_order(policy: Policy,
139 pieces: Sequence[tuple[int, ...]],
140 name_pieces: Sequence[int],
141 tokens: list[WorkToken],
142 *, dot_divided: bool) -> EffectiveOrder:
143 """script_orders resolution (#271): when every name piece is
144 written wholly in ONE script that has an entry, that script's
145 order governs the positional read; anything else -- Latin, mixed
146 scripts, no entry -- falls back to name_order. A 间隔号-divided
147 name (`dot_divided`, #298) suppresses the whole lookup first: the
148 dot marks a transcription -- playing the role pure katakana plays
149 in the kana license, orthography naming the convention -- so the
150 license yields to name_order. Piece-level, after
151 title/suffix peeling: 'Dr. 毛泽东' is a wholly-Han NAME under a
152 Latin title. Kana-licensed tokens (高橋みなみ, #272) resolve to
153 HIRAGANA the same way a wholly-Han or wholly-Hangul token resolves
154 to its own script -- and so does a kana-licensed NAME split across
155 separately single-script PIECES ('高橋 みなみ', Han piece plus
156 Hiragana piece): resolve_script_set generalizes the license from
157 one token's characters to the whole found-script set below, which
158 is why Han+Hangul ('毛 김') still declines even though both
159 individually read family-first -- the license is specific to the
160 Han/Hiragana/Katakana repertoire, not "the entries happen to
161 agree".
162
163 Naming note, since the two are easy to conflate: THIS function
164 resolves the ORDER for a whole name; `_vocab.effective_script`
165 resolves the SCRIPT for a single token. This function calls that
166 one per token below.
167
168 Takes the segment's pieces and WHICH of them the name kept, the
169 shape every piece-layer predicate takes: a caller holding those
170 indices had to build a second list of the same pieces to hand
171 over otherwise, which on 3.11 is a comprehension frame on every
172 parse (decisions.md#parse-cost).
173
174 Returns an EffectiveOrder: the order triple, and `by_script` set only on
175 the one path where an entry answered. Every fallback below is a
176 script rule DECLINING, and reports it as such.
177 """
178 declared = EffectiveOrder(policy.name_order, by_script=False)
179 # #298 transcription marker -- see the docstring; codepoint-scoped
180 # (only U+00B7 records; decisions.md#T3)
181 if dot_divided:
182 return declared
183 if not policy.script_orders:
184 return declared
185 # Collect every token's script rather than comparing pairwise as
186 # tokens are seen: the kana license needs the WHOLE set (a Han
187 # piece and a Hiragana piece only license together, never one at a
188 # time), so resolution is deferred to resolve_script_set below.
189 found: set[Script] = set()
190 for piece_idx in name_pieces:
191 for i in pieces[piece_idx]:
192 script = effective_script(tokens[i].text)
193 if script is None:
194 # Latin, mixed, or a script with no entry: never a key
195 return declared
196 found.add(script)
197 resolved = resolve_script_set(found)
198 if resolved is None:
199 # e.g. Han+Hangul: two scripts, neither the kana license's
200 # Han/Hiragana/Katakana repertoire -- no single tradition
201 return declared
202 for script, order in policy.script_orders:
203 if script is resolved:
204 return EffectiveOrder(order, by_script=True)
205 # the resolved script has no entry: the declaration stands, and
206 # nothing about the writing system decided the reading
207 return declared
208
209
210# rules.md#O4: "words no vocabulary has claimed read by position. In
211# the default given-first order the first name word is the given name,
212# the last is the family name, and everything between is middle names"
213def _name_positions(order: tuple[Role, Role, Role],
214 count: int) -> list[Role]:
215 """Roles for `count` name pieces (titles/suffixes already peeled),
216 per name_order. GIVEN_FIRST: given, middles..., family.
217 FAMILY_FIRST: family, given, middles... FAMILY_FIRST_GIVEN_LAST:
218 family, middles..., given. One piece takes order[0]'s role; two
219 pieces take order[0] and the other primary."""
220 first, second = order[0], order[1]
221 # rules.md#O5: "a name of one name word that nothing else has
222 # decided reads that word as the given name under the default
223 # given-first order, and as the family name under a declared
224 # family-first one" -- a convention, not a determination: O4 has
225 # no positions to compare at one word, so this line is where the
226 # library picks one of two equally consistent readings and picks
227 # it the same way every time. The rules that DO decide such a
228 # name (H1, N3, M4) run after this and retag.
229 if count == 1:
230 return [first]
231 if first is Role.GIVEN: # GIVEN_FIRST
232 return ([Role.GIVEN] + [Role.MIDDLE] * (count - 2)
233 + [Role.FAMILY])
234 if second is Role.GIVEN: # FAMILY_FIRST
235 return ([Role.FAMILY, Role.GIVEN]
236 + [Role.MIDDLE] * (count - 2))
237 return ([Role.FAMILY] + [Role.MIDDLE] * (count - 2) # F_F_GIVEN_LAST
238 + [Role.GIVEN])
239
240
241def _assign_main(seg_idx: int, state: ParseState,
242 tokens: list[WorkToken],
243 ambiguities: list[PendingAmbiguity],
244 ) -> tuple[Role, Role, Role] | None:
245 """Returns the order the positional read used, for ParseState.order
246 -- None on every path that returns before resolving one."""
247 pieces = state.pieces[seg_idx]
248 ptags = state.piece_tags[seg_idx]
249 n = _peel_leading_titles(pieces, ptags, tokens)
250 if n == len(pieces):
251 return None
252 # group-flagged suffix pieces (the ph-d merge) are suffixes at ANY
253 # position -- v1's fix_phd extracted the credential from the string
254 # before parsing, so position never mattered (PR review I3).
255 # Walked rather than collected first: the list was never read
256 # again, and on 3.11 a comprehension is a frame of its own, which
257 # is one frame back against the one the H5 reading below costs. A
258 # 3.11 FACT and not a portable one: PEP 709 inlines comprehensions
259 # from 3.12, where this tree's reference parse costs 395 written
260 # either way (measured 2026-09-09, against 416 and 417 on 3.11).
261 # 3.11 is the interpreter the band is quoted for, so 3.11 is what
262 # the shape is chosen on (decisions.md#parse-cost).
263 for k in range(n, len(pieces)):
264 if "suffix" in ptags[k]:
265 _set_roles(tokens, pieces[k], Role.SUFFIX)
266 rest = peel_walk(n, ptags)
267 if not rest:
268 return None
269 # rules.md#N3: "a name that is only a nickname and one name word
270 # reads that word as the family name" (history: decisions.md#N3)
271 # -- v1's p_len == 1 counted
272 # the WHOLE segment before any title peeling -- 'Xyz. (Bud) Smith'
273 # has two pieces, so the title peel wins and Smith stays the given
274 # name (pinned live 2026-07-17)
275 # The nickname scan sits LAST in the test: it is a generator, which
276 # every interpreter resumes once per token (seven call events on
277 # the reference name), and only a one-piece segment ever reads its
278 # answer. Hoisting it to the top of the read, where it once stood,
279 # is what put 3.12-3.15 one call over the band (decisions.md#parse-cost).
280 if (len(pieces) == 1 and len(rest) == 1
281 and any(t.role is Role.NICKNAME for t in tokens)):
282 _set_roles(tokens, pieces[rest[0]], Role.FAMILY)
283 return None
284 # rules.md#S2's trailing peel and rules.md#H5's title chain, read
285 # together to their fixed point by _pieces.tail_reading -- one
286 # function since the /simplify round, shared with the bound-given
287 # reserve (P5), which must count the name words this leaves.
288 #
289 # rules.md#H5: "successive single words that wear the abbreviation
290 # shape and are title vocabulary chain into the title from the end,
291 # leaving one name word standing" -- the titles are set BEFORE
292 # _name_positions, so the shortened list is what the positional
293 # read and the script test both see (a trailing Latin title must
294 # not make a wholly-CJK name look mixed-script, the same reason the
295 # leading peel runs first). Read ahead of the bare-suffix carve-out
296 # because with no name piece left there is nothing for the chain to
297 # read: its floor keeps 0 pieces of an empty list, so the branch
298 # below is reached exactly as before.
299 #
300 # Every bare ambiguous acronym the FINAL peel had to resolve is one
301 # coin-flip each, in either direction, so the report collects
302 # rather than overwrites. Deferred to after assignment because the
303 # wording reads the role back, and which role "not peeled" means
304 # depends on name_order. (The roman-numeral fork needs no such
305 # deferral and is reported here.)
306 rest, titled_tail, peeled = tail_reading(rest, pieces, ptags, tokens,
307 state.one_case)
308 for piece_idx in titled_tail:
309 _set_roles(tokens, pieces[piece_idx], Role.TITLE)
310 if peeled.numeral is not None:
311 # a trailing single letter is a name part unless it happens
312 # to be a roman numeral -- and V/X/I are ordinary middle
313 # initials, so taking it as a suffix is a call, not a fact
314 ambiguities.append(PendingAmbiguity(
315 AmbiguityKind.SUFFIX_OR_NAME,
316 f"{tokens[peeled.numeral[0]].text!r} is a roman numeral, so "
317 f"it reads as a generational suffix; any other single "
318 f"letter there would be a middle initial",
319 peeled.numeral))
320 name_pieces, suffix_pieces = rest[:peeled.names], rest[peeled.names:]
321 if peeled.names == 0:
322 # everything suffix-shaped after titles: first one is the name
323 name_pieces, suffix_pieces = suffix_pieces[:1], suffix_pieces[1:]
324 # AFTER the whole tail reading, and load-bearing: the script test
325 # sees the NAME pieces only, so a Latin title or suffix ('Dr. 毛
326 # 泽东', '毛 泽东, PhD') cannot make a wholly-CJK name look
327 # mixed-script.
328 resolved = _effective_order(state.policy, pieces, name_pieces, tokens,
329 dot_divided=bool(state.interpunct_offsets))
330 order = resolved.order
331 roles = _name_positions(order, len(name_pieces))
332 for pos, piece_idx in enumerate(name_pieces):
333 _set_roles(tokens, pieces[piece_idx], roles[pos])
334 for piece_idx in suffix_pieces:
335 _set_roles(tokens, pieces[piece_idx], Role.SUFFIX)
336 # Both emitters below turn on the PEEL's count, so neither can
337 # reach a name that kept two name pieces, and the two scans are
338 # nested under the count rather than run beside it: they cost the
339 # call budget on every parse otherwise (decisions.md#parse-cost).
340 if peeled.names <= 1:
341 head = pieces[name_pieces[0]]
342 token = tokens[head[0]]
343 assert token.role is not None
344 # Both conventions here turn on a lone name word, so both
345 # report at the site that places one
346 # (mechanisms.md#AMBIGUITY-AT-THE-DECISION-SITE), and one
347 # predicate carries what says nothing else decided the FIELD:
348 # a title (segment 0's own peel, or one a family comma left in
349 # the next segment -- either is a Role.TITLE by now, and
350 # either makes the reading H1's), a maiden name (M4's), or a
351 # script order convention. `resolved.by_script` rather than a
352 # comparison against name_order: a declared family-first order
353 # agreeing with a Han name's entry is agreement, not
354 # authorship, and for a lone CJK honorific ('さん', '씨') it
355 # scopes W2's reading out rather than excusing it (#271/#308).
356 titled = any(t.role is Role.TITLE for t in tokens)
357 field_undecided = (
358 not resolved.by_script and not titled
359 and not any(t.role is Role.MAIDEN for t in tokens))
360 # rules.md#H4: "an input whose every word is post-nominal
361 # vocabulary reads its first word as a name and reports
362 # `suffix-or-name`" (history: decisions.md#H4) -- only the
363 # word made into a name reports, and no name piece survived
364 # the peel, so the carve-out above made the first post-nominal
365 # the name. The role comes off the token for the reason stated
366 # at the particle emitter below.
367 if peeled.names == 0 and field_undecided:
368 text = " ".join(tokens[i].text for i in head)
369 ambiguities.append(PendingAmbiguity(
370 AmbiguityKind.SUFFIX_OR_NAME,
371 f"{text!r} is post-nominal vocabulary with no name word "
372 f"beside it; read as a {token.role.value} name rather than "
373 f"a post-nominal, nothing else being left to be the name",
374 tuple(head)))
375 # One name piece off the peel: the convention placed a lone
376 # name word. A suffix beside it is not a decision -- 'Smith
377 # Jr.' and "'Smitty' Jones Jr." ARE this convention -- and the
378 # count comes off the peel rather than off name_pieces, which
379 # is what keeps the carve-out above out of these branches.
380 # Title vocabulary in the unit makes the doubt H4's (is the
381 # WORD a title), anything else O5's (which FIELD it took), so
382 # only the second reads `field_undecided`.
383 if peeled.names == 1 and not resolved.by_script:
384 # rules.md#H4: "an input whose only remaining name word
385 # after the title peel is itself title vocabulary reads
386 # that word as the name by convention and reports
387 # `title-or-name`" (history: decisions.md#H4), and H4's
388 # join clause, stated at rules.md#O5 as its exception --
389 # one branch, its detail naming a field only in the join
390 # shape ('John of Prince', 'Attorney General of
391 # Minnesota'), where the unit is more than the title word.
392 # A LONE title word never reaches here ('Dr.', 'Prince of
393 # Wales'): the leading-title peel took the whole name.
394 if any("vocab:title" in tokens[i].tags for i in head):
395 text = " ".join(tokens[i].text for i in head)
396 ambiguities.append(PendingAmbiguity(
397 AmbiguityKind.TITLE_OR_NAME,
398 f"{text!r} is title vocabulary and the only name word "
399 f"the title peel left standing; read as the name by "
400 f"convention rather than as more title"
401 if len(head) == 1 else
402 f"{text!r} is the only name unit and joins title "
403 f"vocabulary to a name word; read as a "
404 f"{token.role.value} name by convention",
405 tuple(head)))
406 # rules.md#O5: "a name of one name word that nothing else has
407 # decided reads that word as the given name under the default
408 # given-first order, and as the family name under a declared
409 # family-first one" (history: decisions.md#O5). Past
410 # `field_undecided`, two clauses of O5's own: the word's
411 # own reading may have claimed it, and A2's content test
412 # -- a piece with no alphanumeric character is no name
413 # word, so parse("(") keeps its unbalanced-delimiter
414 # report and gains nothing here.
415 elif (field_undecided
416 and all(tokens[i].tags.isdisjoint(_WORD_ALREADY_CLAIMED)
417 for i in head)
418 and any(c.isalnum() for i in head
419 for c in tokens[i].text)):
420 text = " ".join(tokens[i].text for i in head)
421 ambiguities.append(PendingAmbiguity(
422 AmbiguityKind.GIVEN_OR_FAMILY,
423 f"{text!r} is the only name word and nothing else "
424 f"decides it; read as a {token.role.value} name by "
425 f"convention, which follows the read order",
426 tuple(head)))
427 for piece in peeled.picks:
428 # every pick is in rest, so the loops above just gave it a role
429 token = tokens[piece[0]]
430 assert token.role is not None
431 taken, declined = (
432 ("a suffix", "a name part") if token.role is Role.SUFFIX
433 else (f"a {token.role.value} name", "a post-nominal"))
434 ambiguities.append(PendingAmbiguity(
435 AmbiguityKind.SUFFIX_OR_NAME,
436 f"{token.text!r} written without periods is both a "
437 f"post-nominal and an ordinary name; read as {taken} "
438 f"rather than {declined}",
439 piece))
440 # leading ambiguous particle read as a name (#121 surfaced)
441 if name_pieces:
442 head = pieces[name_pieces[0]]
443 if (len(head) == 1 and len(name_pieces) > 1
444 and "vocab:particle-ambiguous" in tokens[head[0]].tags):
445 # the loops above gave the head piece its role from
446 # `order`, which is _effective_order's answer and not
447 # necessarily name_order's -- a script_orders entry
448 # overrides it. So read the role off the token rather than
449 # assume given, or re-derive it here; same reason as
450 # SUFFIX_OR_NAME just above.
451 token = tokens[head[0]]
452 assert token.role is not None
453 ambiguities.append(PendingAmbiguity(
454 AmbiguityKind.PARTICLE_OR_GIVEN,
455 f"leading {token.text!r} may be a family-name "
456 f"particle; read as a {token.role.value} name",
457 tuple(head)))
458 return order
459
460
461def _reads_as_a_trailing_suffix(piece: Sequence[int],
462 prev_piece: Sequence[int],
463 prev_ptags: Set[str],
464 tokens: Sequence[WorkToken],
465 lexicon: Lexicon) -> bool:
466 """The lenient tail test for a trailing one-token piece after a
467 family comma, and #432's carve-out from it.
468
469 A word that could be a middle initial, written with the period that
470 marks an abbreviation, is name material -- so 'Smith, John V.' is
471 middle 'V.' where 'Smith, John V' stays suffix 'V' (v1 parity,
472 #144). Only behind a NAME word: behind a suffix the credential run
473 owns it, and 'Smith, John PhD I.' keeps suffix 'PhD, I.'.
474
475 The period is the whole carve-out. is_initial_shaped would be
476 redundant beside it: the caller only consults this where
477 is_suffix_piece said no, and a lenient single token it refuses is
478 one carrying the `initial` tag, so the shape is already implied.
479
480 NOT is_trailing_numeral_suffix, though it answers the period half:
481 it also refuses a numeral behind an initial-shaped piece, which is
482 a no-comma rule and the opposite of this path's v1 parity --
483 'Chang, Andy C I' is first Andy, middle C, suffix I, and asking
484 that predicate here made the numeral a middle. The #401/#421 entry
485 under decisions.md#P5 records the same fork declining to transfer
486 to this walk under LENIENT.
487 """
488 text = tokens[piece[0]].text
489 if text.endswith(".") and not is_suffix_piece(
490 prev_piece, prev_ptags, tokens):
491 return False
492 return is_suffix_lenient(text, lexicon)
493
494
495def assign(state: ParseState) -> ParseState:
496 tokens = list(state.tokens)
497 ambiguities = list(state.ambiguities)
498 if not state.segments:
499 return state
500 order: tuple[Role, Role, Role] | None = None
501 if state.structure is Structure.NO_COMMA:
502 order = _assign_main(0, state, tokens, ambiguities)
503 tail = len(state.segments)
504 elif state.structure is Structure.SUFFIX_COMMA:
505 order = _assign_main(0, state, tokens, ambiguities)
506 tail = 1
507 else: # FAMILY_COMMA
508 # PARTICLE_OR_GIVEN is deliberately not emitted on the
509 # wholly-family read: after a comma that fixed the family, a
510 # leading given-position particle is not meaningfully
511 # ambiguous, and script_orders is not consulted for the parallel
512 # reason. Scoped, not silent: the comma fixed WHICH PIECE is the
513 # family and said nothing about a particle trailing the given
514 # name, so P6's attachment in post_rules decides that fork and
515 # reports it there, at the site that takes the branch (#405).
516 # The positional read below (a comma followed by no
517 # name word) emits and consults both, being the no-comma read
518 # of segment 0; group's chain emitter still does not, since
519 # group runs before assign decides which read applies (recorded
520 # at decisions.md#C1).
521 # v1: "lastname part may have suffixes in it" -- the first
522 # piece is always the family even if suffix-shaped; any later
523 # strict-suffix piece goes to SUFFIX per piece ('Smith Jr.,
524 # John' -> family=Smith, suffix=Jr.)
525 fam_pieces = state.pieces[0]
526 fam_tags = state.piece_tags[0]
527 # A comma followed by no name word fixed nothing, so segment 0
528 # keeps its positional read -- including script_orders, the
529 # particle fork, and the ORDER, which post_rules' family-first
530 # fold (P1) and its leading-piece scan key on: "assign records
531 # no order after a family comma" is the invariant those rules
532 # rest on, and this is the path that gives one, so they read
533 # segment 0 as the name it is ('de Mesnil Jean, Dr.' under a
534 # family-first order keeps family 'de Mesnil'; the test review
535 # found the fold missing it). The wholly-family branch below
536 # suppresses all three precisely because the comma HAD fixed
537 # the family. Needs two NAME pieces: with one, the positional
538 # read would make it a lone GIVEN, which is worse than what it
539 # replaces -- and the count is of name pieces, since the
540 # positional read peels a trailing suffix first: 'Smith Jr.,
541 # Mr.' has two pieces and one name, and read positionally lost
542 # its family (the code review).
543 reading = segment_suffix_reading(
544 state.pieces[1], state.piece_tags[1], tokens,
545 state.policy.lenient_comma_suffixes, state.one_case)
546 # rules.md#C1's exception, scoped to the ambiguous credential
547 # class: this is the first report of the comma's OWN decision
548 # (listing or credential run), where the writing left the
549 # fork open and the comma stayed quiet by design until now.
550 # P6's attachment fork already reports on a family-comma path
551 # from post_rules, since 2.3 ("Berg, Jan vd") -- a different
552 # fork. Emitted on the family-comma path only -- the structure
553 # decision reports itself in `segment`, where that branch is
554 # taken, so this DECISION is never reported twice; a second ambiguous
555 # token elsewhere in the name is a second fork and reports on
556 # its own (#289, mechanisms.md#AMBIGUITY-AT-THE-DECISION-SITE).
557 #
558 # The report tracks the FORK BEING CONSULTED, not the lean --
559 # exactly as the trailing slot has always done (`Jack MA`
560 # reported before #289 too, even where the pick was declined
561 # for want of words to spare). So membership alone gates it:
562 # a caseless script or an all-lower spelling still called this
563 # fork and read it positionally.
564 #
565 # The gate reads EITHER tag (`_AMBIGUOUS_CREDENTIAL_TAGS`, the
566 # same pair `_group`'s chain emitter asks), and the shape one
567 # is what reaches a by-shape member -- under EITHER 2.4
568 # switch, the dotted and the caps alike, since classify writes
569 # it from both branches. It reaches one with the dotted switch
570 # OFF as well: classify writes the shape tag whether or not
571 # the switch admits the token to the class, which is what lets
572 # a declined fork be REPORTED without being taken ('Smith,
573 # A.B.' under `unlisted_dotted_suffixes=False` reports and
574 # keeps its given). An earlier wording said the class reaches
575 # a by-shape member "once `Policy.unlisted_dotted_suffixes`
576 # admits it", which is true of the CLASS and false of this
577 # report.
578 #
579 # Read off the FIRST post-comma piece only --
580 # `segment_suffix_reading` decides piece by piece, and this is
581 # the one piece the lean can reach at one word before the
582 # comma. So `"Smith, MA PhD"` reports ONCE, for 'MA' alone:
583 # 'PhD' is settled vocabulary and carries neither tag, and
584 # even a second CLASS member there would not be read here
585 # (test_assign.py asserts the count).
586 #
587 # The #531 emitter at the far end of this branch counts
588 # differently, and the difference is the slot rather than a
589 # second policy: it reports once per member of the trailing
590 # run it reads, so 'Doe, John MA JD' reports TWICE, matching
591 # the comma-less 'John Smith MA JD'. A member the writing
592 # keeps as a name stops that run, which is why
593 # 'Doe, John MA Ma' reports once and for 'Ma' alone -- 'MA'
594 # then has a name word behind it and is never asked
595 # (rules.md#S2).
596 if state.pieces[1] and len(state.pieces[1][0]) == 1:
597 i = state.pieces[1][0][0]
598 if not tokens[i].tags.isdisjoint(_AMBIGUOUS_CREDENTIAL_TAGS):
599 chose = ("a credential" if reading and reading[0]
600 else "the given name")
601 ambiguities.append(PendingAmbiguity(
602 AmbiguityKind.SUFFIX_OR_NAME,
603 f"{tokens[i].text!r} after the comma is also an "
604 f"ordinary name word; read as {chose}",
605 (i,)))
606 # Segment 1 is read FIRST, ahead of either branch below. It
607 # consumes `reading`, piece tags and text only -- nothing
608 # segment 0's read writes -- and running it first is what puts
609 # a TITLE role on the title standing after the comma ('John V,
610 # Dr.') before _assign_main scans for one, so the positional
611 # read below sees that title the way it sees its own peeled
612 # ones (#449 review round; it replaced a `titled` keyword).
613 if len(state.segments) > 1:
614 pieces = state.pieces[1]
615 ptags = state.piece_tags[1]
616 # Both are the walk's, and both are empty on the gate's
617 # path below, which reads the whole segment as a
618 # credential run and leaves no piece for the walk to
619 # place.
620 titled_idx: tuple[int, ...] = ()
621 walkable: list[int] = []
622 #: Where the trailing suffix run starts, for the #531
623 #: report below: `trailing_floor`'s answer, read once on
624 #: the first member the loop meets and -1 until then (no
625 #: piece index can be negative, and the loop's own floor is
626 #: n + 1). Once is enough for the same reason one number is
627 #: enough inside that walk: the question is MONOTONE, the
628 #: loop below ascends, and a later member can only ask for
629 #: LESS of the walk than the first one did.
630 run_floor = -1
631
632 def previous_kept(m: int, titled: tuple[int, ...]) -> int:
633 """The piece before `m` that the H5 chain did NOT
634 take. Both readings this segment needs are that one:
635 the piece the lenient tail test measures against, and
636 -- asked of one past the end -- where the segment's
637 name ENDS, since a title the chain took is not where a
638 name ends (#144). One walk for both, so 'as if the
639 titled pieces were absent' cannot come to mean two
640 things.
641
642 DEFENSIVE, and measured inert: over 191,146 generated
643 inputs the skip fired on 153 of the 140,227 walks the
644 lenient test's `prev` asked for, and deleting it
645 changed no parse among them (2026-09-09, on the shape
646 this replaced, where the name's END walked past the
647 same pieces in a copy of this loop). Kept because "as
648 if the titled pieces were absent" is the rule the
649 predicate below implements, and a caller's vocabulary
650 reaches shapes the sweep's word list does not -- an
651 inert branch is cheaper than a rule with a hole in it.
652 """
653 m -= 1
654 while m in titled:
655 m -= 1
656 return m
657
658 #: State of the #531 walk below, per `titled` value: how
659 #: far down the trailing suffix run has been walked, and
660 #: whether that walk has SETTLED (it stopped on a piece
661 #: that refuses, so no lower piece can end the given part
662 #: either and the refusal is never re-asked).
663 #:
664 #: The key carries `titled` for the same reason the
665 #: predicate takes it as a parameter: the two passes ask
666 #: about the same pieces with different ones spliced out.
667 #: It CANNOT go stale. Everything the recorded verdict
668 #: rests on is tags and text -- `is_suffix_piece` reads
669 #: ptags and token tags, `_reads_as_a_trailing_suffix`
670 #: reads text plus `is_suffix_piece`, `listed_lean` reads
671 #: tags, text and `state.one_case` -- and none of them
672 #: reads `.role`, verified by reading all three
673 #: (2026-09-19). The one thing this segment's code rewrites
674 #: between the two passes is the role, through `_set_roles`,
675 #: which is a `dataclasses.replace(role=...)` and leaves
676 #: text and tags identical.
677 floors: dict[tuple[int, ...], tuple[int, bool]] = {}
678
679 def trailing_floor(m: int, titled: tuple[int, ...]) -> int:
680 """Where the trailing suffix run starts, walked as far
681 down as `m` needs it: `m >= trailing_floor(m, titled)`
682 is exactly "every kept piece behind `m` reads as a
683 suffix", which is what ENDING the given part means
684 (rules.md#S2, #531).
685
686 ONE walk per `titled` value, shared by the predicate
687 below and by that rule's report at the foot of this
688 segment, and the reason both are LINEAR in the run's
689 length. The question is MONOTONE -- a piece ends the
690 given part whenever the piece behind it does -- so a
691 single descent answers for every member, each piece
692 read at most once. Asked member by member instead, the
693 reading is recursive (a member ends the given part iff
694 everything kept behind it reads as a suffix, and a
695 piece behind it is a member asking the same of its own
696 tail): a walk per member, which cost O(run**2) with the
697 per-member memo this replaced and 2**run without one.
698 Measured on `'Doe, John ' + 'MA '*k`, k doubling from
699 200: 2.1/4.1/8.3/17.1ms here, against 8.6/31.4/120/463
700 with the memo and 1.5/3.2/7.2/17.3 at cc78c960, where
701 no such walk existed at all (2026-09-19).
702
703 `low` descends only to `m`, so what comes back is a
704 floor FOR `m` rather than the run's own first piece
705 whenever the run reaches past it; that is all either
706 caller asks, and stopping there is what keeps the
707 member's own frame count where it was. `final` carries
708 the other half: a piece that refuses settles the floor
709 for everything in front of it.
710
711 Re-entrant by construction, and it has to be: the
712 descent asks the predicate below about a piece that is
713 itself often a member, which asks this back. `floors`
714 names the piece under test BEFORE that call, so the
715 re-entrant reading is "this piece ends the given part",
716 which is what the descent has just established of it.
717 """
718 entry = floors.get(titled)
719 if entry is None:
720 # previous_kept() of one past the end, spelled out
721 # here rather than called: the frame budget again,
722 # this walk being asked of every family-comma name
723 # with a member in the given part, and the skip is
724 # two lines. Keep the two in step.
725 low = len(pieces) - 1
726 while low in titled:
727 low -= 1
728 entry = (low, False)
729 floors[titled] = entry
730 low, final = entry
731 while not final and low > m:
732 if reads_as_a_suffix(low, titled):
733 low -= 1
734 while low in titled:
735 low -= 1
736 else:
737 final = True
738 floors[titled] = (low, final)
739 return low
740
741 def reads_as_a_suffix(m: int, titled: tuple[int, ...]) -> bool:
742 """Does this segment's walk read piece `m` as a suffix?
743
744 Asked twice, and by one predicate rather than by two
745 conditions written to match
746 (mechanisms.md#ONE-PREDICATE-PER-QUESTION): once to
747 find the pieces the H5 title chain must not reach
748 past, and once by the walk order below, which is the
749 site that places them.
750
751 `titled` is a PARAMETER because the two passes hand it
752 different values -- () on the first, the chain's own
753 pieces on the second, which is what makes that second
754 reading the one 'as if the titled pieces were absent'.
755 A closure over the caller's local said the same thing,
756 but only by WHEN it was rebound.
757 """
758 if is_suffix_piece(pieces[m], ptags[m], tokens):
759 return True
760 # rules.md#S2, the given part's trailing slot (#531).
761 # INLINE, and that is the frame budget talking rather
762 # than taste: the walkable pass asks this closure once
763 # per piece of every family-comma segment, so a helper
764 # call would cost a frame on every non-member piece and
765 # a generator expression would cost its own on 3.11.
766 # Membership is therefore a bare `in` on tags already
767 # in hand, after a `len` -- 'Doe, John Q.' and
768 # 'Smith, John V' measure +0 with this shape and +1
769 # with a helper (2026-09-18).
770 #
771 # That "once per piece" is the NON-MEMBER cost, and
772 # only it. A member is asked a second time by
773 # trailing_floor()'s descent above, and twice is the
774 # whole of it: the descent takes each piece once and
775 # the walkable pass asks each piece once, which is what
776 # makes the run linear. What the slot costs, measured
777 # against cc78c960: 'Smith, John', 'Doe, John Q.',
778 # 'Smith, John V', 'Smith, MA' and 'Berg, Jan vd' are
779 # all +0 frames, while 'Doe, John MA' is +6 and 'Doe,
780 # John MA PhD' +23 -- the member's descent, its lean,
781 # and the report below (2026-09-19).
782 #
783 # The comma has already named the family and the first
784 # piece after it is the given name, so the words to
785 # spare S2's count asks about are there by
786 # construction and the count says nothing at this
787 # slot. What is left is the writing, which is the same
788 # evidence the comma-less spelling of the same name
789 # reads. #144's two-segment restriction below is NOT
790 # inherited: it exists because a trailing 'V' before a
791 # third comma part is likely a middle initial, and a
792 # class member is not initial-shaped while a
793 # credential list behind it makes the credential
794 # reading likelier rather than less.
795 piece = pieces[m]
796 if len(piece) == 1:
797 tok = tokens[piece[0]]
798 if AMBIGUOUS_ACRONYM_TAG in tok.tags:
799 # 'ending the given part' reaches past the
800 # credentials behind it and past a trailing
801 # title, which trailing_floor() skips the way
802 # previous_kept() does -- so 'Doe, John MA
803 # Prof.' and 'Doe, John Prof. MA' land on one
804 # answer without a second notion of trailing.
805 # A name word behind the member ends the
806 # reach, and the member is an ordinary middle
807 # name read in silence.
808 if m >= trailing_floor(m, titled):
809 # A member that is ALSO particle
810 # vocabulary reads as the credential only
811 # on a POSITIVE credential lean: P6's
812 # attachment outranks this reading in
813 # every other spelling, and 'Doe, John do'
814 # leans nothing, so the positional reading
815 # would take it -- the wrong answer there,
816 # not merely a stray report
817 # (decisions.md#S2, 2026-09-18).
818 #
819 # A FUNCTION since #533, not two conditions
820 # written to match: the maiden walk's
821 # second check asks this same question of
822 # the name a take would leave, and the
823 # drift would have been silent -- each
824 # site's own tests would have gone on
825 # passing (mechanisms.md
826 # #ONE-PREDICATE-PER-QUESTION). The call
827 # costs one frame PER MEMBER asked at this
828 # slot, not one per name: against
829 # 2f57ff21, 'Doe, John MA' is 310 -> 311
830 # and 'Doe, John MA Ma MA', which asks
831 # four times, 439 -> 443, while a name
832 # with no member here never reaches it
833 # ('Smith, John' 206, 'MA JD' 185, both
834 # unchanged). Measured 2026-09-19 per
835 # `Parser.parse`; Derek took that trade
836 # deliberately.
837 if credential_at_the_given_slot(
838 tok, state.one_case):
839 return True
840 prev = previous_kept(m, titled)
841 # trailing piece of a two-part name is unambiguously
842 # positioned: v1 accepts the lenient test there
843 # ('Smith, John V' -> suffix='V', #144); with a third
844 # comma part the trailing token is more likely a middle
845 # initial, so strict only
846 return (m == previous_kept(len(pieces), titled)
847 and len(state.segments) == 2
848 and len(pieces[m]) == 1
849 and _reads_as_a_trailing_suffix(
850 pieces[m], pieces[prev], ptags[prev],
851 tokens, state.lexicon))
852
853 # rules.md#C1: "a credential run after the comma means the
854 # name is in natural order with suffixes appended" -- and
855 # with one word before the comma the listing form holds,
856 # the family is that word, and the run is still the
857 # credential run. A no-name segment is read piece by
858 # piece, the suffix vocabulary's verdict BEFORE the title
859 # reading of the same word: the slot after a family comma
860 # is postnominal position, so a segment of nothing but
861 # suffix pieces is the credential run, whole (#296:
862 # 'Smith, Jr.' read title 'Jr.' through the period-
863 # abbreviation inference; #325: 'Smith, Ph. D. Jr.' put
864 # the split credential in the given name, the lone-piece
865 # route not applying), and a mixed run is a title and a
866 # postnominal, each where it stands ('Smith, Mr. Jr.').
867 # Vocabulary decides which words qualify -- 'Smith, Dr.'
868 # reads the title, 'dr' not being suffix vocabulary since
869 # the audit -- and position breaks the tie for the genuine
870 # duals ('Smith, Sr.' is Senior, 'Sr. Garcia' Señor). A
871 # name word in the segment makes it v1's walk ('Smith,
872 # John Jr.').
873 if reading is not None:
874 # the gate's own reading (#430)
875 for k, piece in enumerate(pieces):
876 _set_roles(tokens, piece,
877 Role.SUFFIX if reading[k] else Role.TITLE)
878 n = len(pieces)
879 else:
880 n = _peel_leading_titles(pieces, ptags, tokens)
881 # rules.md#H5: "the title is TRANSPARENT to the suffix
882 # reading: where two or more name words stand, what
883 # stands once the chain is taken reads exactly as it
884 # would read written without the title, plus the title"
885 # -- the trailing title run, on this walk
886 # too. A name word in segment 1 is what keeps the gate
887 # above from reading the segment as a credential run,
888 # so 'Smith, John Prof.' had no route to title at all
889 # and read middle 'Prof.' at every baseline; the
890 # 'Smith, Dr.' family of rows that already route are
891 # the gate's doing, a different mechanism.
892 #
893 # The candidates are the pieces this walk would NOT
894 # read as a suffix, which is this path's answer to the
895 # peel the no-comma path runs first -- 'Smith, John
896 # Prof. Jr.' must reach `Prof.` past the postnominal
897 # behind it, and 'Smith, John Prof. V' past the
898 # numeral the lenient tail test claims (#144), which
899 # is why the filter is the walk's own predicate and
900 # not the strict suffix test alone. Piece `n` is
901 # always the given below, whatever that predicate
902 # would say of it, so it is always a candidate. That
903 # `k == n` is LOAD-BEARING, not defensive: it is what
904 # the walk's floor stands on when the given piece
905 # itself reads as a suffix, and dropping it leaves
906 # 'Smith, II Mr. V' a middle 'Mr.' where the title is
907 # (24 inputs of that shape move, of 191,146 generated,
908 # measured 2026-09-09).
909 walkable = [k for k in range(n, len(pieces))
910 if k == n or not reads_as_a_suffix(k, ())]
911 kept = trailing_titles(walkable, pieces, ptags, tokens)
912 titled_idx = tuple(walkable[kept:])
913 for k in titled_idx:
914 _set_roles(tokens, pieces[k], Role.TITLE)
915 # v1 walk order: the first non-title piece is ALWAYS the
916 # given, before any suffix check -- 'Hardman, RN - CRNA'
917 # keeps first='RN'. The one deliberate 2.0 deviation,
918 # classified fix(comma-family) -- a last piece that is
919 # unambiguously suffix-shaped is a suffix, where v1 made
920 # it the given ('Andrews, M.D.', 'Smith, Dr. Jr.') -- is
921 # the no-name read above now: a segment whose only
922 # non-title piece is a suffix piece holds no name word,
923 # so the walk here never meets the case.
924 if n < len(pieces):
925 _set_roles(tokens, pieces[n], Role.GIVEN)
926 # The chain's floor leaves a name piece standing, so the
927 # given above is never one of the pieces it took. What the
928 # chain DOES move is where this segment's name ends, which
929 # the lenient tail test turns on (#144) -- so where it took
930 # something the question is re-asked with the pieces it
931 # left. Where it took nothing, `walkable` already IS this
932 # predicate's answer for every piece: the first pass's own
933 # memo, not a second spelling of the question, and it
934 # cannot have gone stale because nothing the chain does
935 # moved the end of the name.
936 for m in range(n + 1, len(pieces)):
937 if m in titled_idx:
938 continue
939 suffix_here = (reads_as_a_suffix(m, titled_idx)
940 if titled_idx else m not in walkable)
941 _set_roles(tokens, pieces[m],
942 Role.SUFFIX if suffix_here else Role.MIDDLE)
943 # #531's report, and the FIFTH SUFFIX_OR_NAME site in
944 # this module. The gate reads EITHER tag, as the
945 # first-post-comma emitter's does: classify writes the
946 # shape tag whether or not the dotted switch admits
947 # the token, which is what lets a declined fork be
948 # reported without being taken.
949 #
950 # The report tracks the FORK CONSULTED, not the lean
951 # (#289's rule), so a member the writing kept as a
952 # name reports too -- 'Doe, John Ma' stays a middle
953 # name and says so. It reports only in the TRAILING
954 # RUN: with a name word behind it no fork was
955 # consulted, and AGENTS.md's "a kind is worth adding
956 # only if a reader would hesitate too" is why that
957 # must stay silent rather than why it happens to (the
958 # sentence is the 2.0-conventions section's, not
959 # rules.md#A1's, which this cited until 2026-09-19).
960 #
961 # The `do` carve-out is a REPORT carve-out on top of
962 # the reading one above: a particle-tagged member this
963 # walk did not itself take is P6's fork, and P6
964 # reports it in its own kind, which is
965 # mechanisms.md#AMBIGUITY-AT-THE-DECISION-SITE read
966 # strictly. Without it 'Doe, John do' reported both
967 # kinds.
968 piece = pieces[m]
969 if (len(piece) == 1
970 and not tokens[piece[0]].tags.isdisjoint(
971 _AMBIGUOUS_CREDENTIAL_TAGS)
972 and (suffix_here
973 or "particle" not in tokens[piece[0]].tags)):
974 # The SAME floor the predicate measures members
975 # against, so this report has no walk of its own to
976 # regress: it had one, and on a run the predicate's
977 # memo had already made quadratic the report was
978 # CUBIC, because its per-member walk re-scanned
979 # `walkable` -- a list -- at every step. Measured
980 # then on `'Doe, ' + 'John '*r + 'MA '*r`, r
981 # doubling from 100: 5.8/28/175/1229ms, 4.9x then
982 # 6.2x then 7.0x per doubling and heading for the
983 # 8x a cubic gives, against 1.8/3.7/8.0ms here
984 # (2026-09-19). Nothing guarded it: the scan was a
985 # C-level `in` over a list and emitted no frame, so
986 # the frame-ratio test in tests/v2/test_benchmark.py
987 # was structurally blind to it, and the clock-based
988 # shapes beside it repeat ONE unit where this cost
989 # needs a name holding two runs. What keeps it gone
990 # is the structure: one walk, read by both callers,
991 # so a second would have to be written on purpose.
992 #
993 # Read once for the whole loop, `run_floor` being
994 # where the FIRST member's walk stopped and the
995 # floor for every later member too: if the walk
996 # reached that member, nothing behind it refuses
997 # and no later member can be refused either; if it
998 # stopped short, it stopped at the LAST piece that
999 # refuses, which is exactly what a later member's
1000 # own walk would have found (2026-09-19).
1001 if run_floor < 0:
1002 run_floor = trailing_floor(m, titled_idx)
1003 if m >= run_floor:
1004 i2 = piece[0]
1005 ambiguities.append(PendingAmbiguity(
1006 AmbiguityKind.SUFFIX_OR_NAME,
1007 f"{tokens[i2].text!r} ending the given "
1008 f"part is also an ordinary name word; "
1009 f"read as "
1010 f"{'a credential' if suffix_here else 'a name'}",
1011 (i2,)))
1012 if reading is not None and sum(
1013 1 for k, piece in enumerate(fam_pieces)
1014 if not is_suffix_piece(piece, fam_tags[k], tokens)) > 1:
1015 order = _assign_main(0, state, tokens, ambiguities)
1016 else:
1017 for k, piece in enumerate(fam_pieces):
1018 if k > 0 and is_suffix_piece(piece, fam_tags[k], tokens):
1019 _set_roles(tokens, piece, Role.SUFFIX)
1020 else:
1021 _set_roles(tokens, piece, Role.FAMILY)
1022 tail = 2
1023 # segments past the structure's name segments are wholly suffixes
1024 for seg_idx in range(tail, len(state.segments)):
1025 for piece in state.pieces[seg_idx]:
1026 _set_roles(tokens, piece, Role.SUFFIX)
1027 return dataclasses.replace(state, tokens=tuple(tokens),
1028 order=order,
1029 ambiguities=tuple(ambiguities))