1"""Stage: segment.
2
3Consumes: tokens (role-None main stream), comma_offsets, one_case
4(where an earlier stage recorded it).
5Produces: segments (runs of main-token indices; interior segments may
6be EMPTY -- doubled commas keep their structural position), structure,
7one_case where the comma form asked for it, COMMA_STRUCTURE
8ambiguities for unrecognized extra segments, and SUFFIX_OR_NAME where
9the comma FLIPPED the structure for a member of the ambiguous
10credential class. The flip and nothing else: where the structure did
11not move, the word's reading is still open and `assign` takes it on
12the family-comma path, so it reports there.
13Reads: Lexicon suffix vocabulary and Policy, both through
14_vocab.is_wholly_suffix -- the suffix-comma decision is definitionally
15vocabulary-dependent (decisions.md#C1), and the predicate
16owns the rest (Policy.lenient_comma_suffixes picks the lenient or
17strict token test; Policy.extra_suffix_delimiters gives v1
18suffix_delimiter parity, a delimiter-core token being transparent);
19Lexicon.maiden_markers DIRECTLY, for the own-words span the lazy case
20gate takes (_pieces.own_words); Lexicon title and suffix vocabulary
21plus Policy.lenient_comma_suffixes again through _vocab.
22name_word_count, which counts NAME words for the class's own comma
23rule; and, since 2.4, Policy.unlisted_dotted_suffixes through
24_vocab.ambiguous_class_candidate, and Policy.unlisted_caps_suffixes
25DIRECTLY as this stage's own gate before the run test calls
26_vocab.caps_shape_candidate -- which reads every Lexicon vocabulary
27field in turn (_lexicon._VOCAB_FIELDS) to decide that an all-caps
28word is UNLISTED. `is_wholly_suffix` deliberately sees neither
29by-shape half, dotted or caps (#516). An unlisted dotted or all-caps
30token joins the ambiguous credential class by SHAPE at this stage's
31own candidate tests the same way a listed member does; the caps half
32additionally needs `one_case` to decide membership at all, which this
33stage's own lazy gate supplies.
34
35Implements rules C1 and C2 of docs/design/rules.md, cited at the
36decision site below; history in decisions.md#C1.
37"""
38from __future__ import annotations
39
40import dataclasses
41
42from nameparser._pipeline._pieces import own_words
43from nameparser._pipeline._state import (
44 ParseState, PendingAmbiguity, Structure, comma_bucket,
45)
46from nameparser._pipeline._vocab import (
47 ambiguous_class_candidate, ambiguous_class_member, caps_shape_candidate,
48 is_one_case, is_wholly_suffix, name_word_count,
49)
50from nameparser._types import AmbiguityKind
51
52
53
54
55
56def segment(state: ParseState) -> ParseState:
57 main = [i for i, t in enumerate(state.tokens) if t.role is None]
58 if not main:
59 return dataclasses.replace(state, segments=(),
60 structure=Structure.NO_COMMA)
61 if not state.comma_offsets:
62 return dataclasses.replace(state, segments=(tuple(main),),
63 structure=Structure.NO_COMMA)
64 buckets: list[list[int]] = [[] for _ in range(len(state.comma_offsets) + 1)]
65 for i in main:
66 # _state.comma_bucket, not a local bisect: classify asks the
67 # same question of the same offsets to keep a marker run inside
68 # one segment, and the two must be one expression rather than
69 # two that agree
70 buckets[comma_bucket(state.tokens[i].span.start,
71 state.comma_offsets)].append(i)
72 groups = [tuple(b) for b in buckets]
73 # v1 strips exactly ONE trailing comma as cosmetic (parser.py's
74 # collapse_whitespace); every other empty bucket is STRUCTURAL and
75 # keeps its position -- in 'Doe,, Jr.' the given segment is empty,
76 # so 'Jr.' stays a tail suffix instead of masquerading as a lone
77 # post-comma title (v1 parity, pinned live 2026-07-16)
78 if len(groups) > 1 and not groups[-1]:
79 groups.pop()
80 if len(groups) <= 1:
81 segs = tuple(groups) if groups and groups[0] else (tuple(main),)
82 return dataclasses.replace(state, segments=segs,
83 structure=Structure.NO_COMMA)
84
85 # The case fact, asked LAZILY: only a comma form can turn on it
86 # here, and only where the part after the first comma is a single
87 # token -- the shape the ambiguous class comes in. A comma-less
88 # name never reaches this and pays nothing; classify asks for
89 # itself later where this did not (decisions.md#S2, and #429's
90 # precedent for paying a predicate twice rather than plumbing a
91 # field two sites would not otherwise share).
92 one_case = state.one_case
93
94 def case_class() -> bool:
95 nonlocal one_case
96 if one_case is None:
97 # `own, _` rather than `[0]`: the second element is the
98 # maiden clause's start index, which this stage has no use
99 # for, and saying so by name is what stops a reader having
100 # to go and look up what a bare subscript dropped.
101 own, _ = own_words(state.tokens, state.comma_offsets,
102 state.lexicon.maiden_markers)
103 one_case = is_one_case(own)
104 return one_case
105
106 def texts(seg: tuple[int, ...]) -> list[str]:
107 return [state.tokens[i].text for i in seg]
108
109 # Inlined rather than built on `texts` (measured, #289/#516's
110 # eager-gate fix round): every comma parse calls `suffixy` at
111 # least once, and a `texts(seg)` indirection costs a SECOND frame
112 # on top of the comprehension's own -- on 3.11 a list comprehension
113 # IS a frame (PEP 709 inlines it only from 3.12 on; see
114 # tools/perf/call_count.py's docstring), so wrapping it in another
115 # call doubles the cost every comma name pays, member or not.
116 # `texts` still serves the two call sites that are not on this
117 # path (name_word_count's pre-comma texts, and a flagged tail
118 # segment's joined display).
119 #
120 # `case` is ParseState.one_case: None is the case-FREE reading,
121 # which is what every caller on this path wants, and the tail
122 # walk below passes `case_class()` for the leaning one. One
123 # closure with a default rather than two whose bodies differ only
124 # in that argument.
125 def suffixy(seg: tuple[int, ...], case: bool | None = None) -> bool:
126 return is_wholly_suffix([state.tokens[i].text for i in seg],
127 state.lexicon, state.policy, one_case=case)
128
129 def class_run(seg: tuple[int, ...]) -> bool:
130 # Every token of the run joins the ambiguous credential class
131 # BY SHAPE -- a candidate that is not a LISTED member, which is
132 # the one half `is_wholly_suffix` cannot see (its own docstring
133 # says so, and the blindness is what keeps a by-shape token out
134 # of C1's legacy token-count disjunct). A tail segment is
135 # consumed as suffix either way, so what this decides is only
136 # whether the parse says it did not RECOGNIZE the segment -- and
137 # a run the parse itself reads as a credential run by shape is
138 # recognized. Without it, the narrow roman retirement
139 # (rules.md#S3) moved 'R.A.I.', 'X.Y.I.' and 'J.u.n.i.o.r.' out
140 # of the vocabulary verdict and into the shape class, and every
141 # one of them gained a COMMA_STRUCTURE flag in a third segment
142 # that 2.3 did not raise -- a report about the parser's own new
143 # reading rather than about the name (review round, #289/#516).
144 #
145 # The LISTED half is excluded rather than folded in: a listed
146 # member reaches this reading through the LEAN (the leaning
147 # `suffixy` call below), and that is the whole of what quiets
148 # it -- 'STEVEN HARDMAN, MD, DO, DDS' is written in one case,
149 # leans nothing, and keeps its flag, the recorded negative
150 # control for the lean's own effect here. Admitting membership
151 # alone would silence it and leave the control measuring
152 # nothing.
153 #
154 # Policy-sensitive by construction: with
155 # `unlisted_dotted_suffixes` off the token is name material, is
156 # no candidate, and the flag stands.
157 return all(ambiguous_class_candidate(state.tokens[i].text,
158 state.lexicon, state.policy)
159 and not ambiguous_class_member(state.tokens[i].text,
160 state.lexicon)
161 for i in seg)
162
163 # rules.md#C1: "the name reads as trailing suffixes when the part
164 # after the first comma is entirely suffix words and more than one
165 # word precedes the comma; otherwise it reads as the listing form"
166 # (v1 parity: only parts[1] decides, parser.py:1318; history:
167 # decisions.md#C1)
168 #
169 # And for the AMBIGUOUS class the count is of NAME words, not of
170 # words: 'Smith Jr., MA' is two tokens and one name, and the token
171 # count hands its family to `given` (#289/#516). One rule for the
172 # whole class: the listed and dotted halves are asked only of a
173 # single-token part, a member of either being always exactly one
174 # token (a bare word, or one glued acronym), while the caps half
175 # may come as a RUN ('LEED AP', below).
176 #
177 # Membership is tested CASE-FREE first (`ambiguous_class_candidate`,
178 # the listed set OR -- since 2.4 -- a by-shape member Policy
179 # admits, #516): the structure decision below is itself
180 # case-independent (item 5's count decides "whatever case the name
181 # is written in"), so the case fact is worth forcing only once a
182 # genuine candidate is found. Passing `case_class()` as an
183 # ARGUMENT instead ran the `own_words` -> `tag_marker_runs` walk
184 # for every comma name whose post-comma part is one token --
185 # `"Smith, John"`, `"John Smith, Jr."`, neither able to reach the
186 # class at all (measured regression).
187 candidate = (len(groups[1]) == 1
188 and ambiguous_class_candidate(
189 state.tokens[groups[1][0]].text, state.lexicon,
190 state.policy))
191 # The all-caps half (Policy.unlisted_caps_suffixes, #516) is the
192 # FIRST shape this class can wear across more than one token --
193 # 'LEED AP' is two separate all-caps words, not one glued acronym
194 # -- and the one membership test in this class that NEEDS the case
195 # fact to answer membership at all: 'XYZ' is only credential-shaped
196 # where the name contrasts it. Gated on the switch (default off, so
197 # a non-candidate comma name never enters this branch) and tried
198 # only where the single-token test above already declined.
199 #
200 # `caps_shape_candidate` directly, not the un-narrowed
201 # `ambiguous_class_candidate`: the run is a property of the CAPS
202 # class ALONE (#516 review round, F2), the other two halves being
203 # single-token by construction, so `all()` over more than one of
204 # THEM asks a question the design never posed. Measured, the
205 # un-narrowed call per token moved `'John Smith, Ed Ma'`, `'John
206 # Smith, ma do'` and `'John Smith, X.Y.Z. A.B.'` to a credential
207 # run neither the spec nor any case row wants.
208 #
209 # `one_case=False` asks the case-free question -- "if this name
210 # turned out mixed, would EVERY token in the run join the CAPS
211 # shape" -- the same trick the single-token test above uses,
212 # generalized over `all()`. Every other conjunct of
213 # `caps_shape_candidate` is case-independent, so for a run already
214 # confirmed shape-eligible the fact itself is the only unknown
215 # left, and the verdict is `case_class() is False` directly rather
216 # than a second walk that could only reach the same answer (a
217 # quality-review finding: the walk was provably redundant).
218 if (not candidate and state.policy.unlisted_caps_suffixes and groups[1]
219 and all(caps_shape_candidate(state.tokens[i].text,
220 state.lexicon, state.policy,
221 one_case=False)
222 for i in groups[1])):
223 candidate = case_class() is False
224 # Computed only where `candidate` is true, alongside `case_class()`
225 # -- the same lazy gate: a non-candidate comma name never counts
226 # its pre-comma words either. Hoisted to a local because the
227 # report below quotes the exact count rather than a hardcoded
228 # "two" ('John Q. Public, MA' has three).
229 pre_comma_names = None
230 if candidate:
231 # forced here, downstream of the structure decision that does
232 # not need it, because assign's post-comma slot and its report do
233 case_class()
234 pre_comma_names = name_word_count(texts(groups[0]), state.lexicon,
235 state.policy)
236 structure = (
237 Structure.SUFFIX_COMMA
238 if ((suffixy(groups[1]) and len(groups[0]) > 1)
239 or (pre_comma_names is not None and pre_comma_names >= 2))
240 else Structure.FAMILY_COMMA)
241 ambiguities = list(state.ambiguities)
242 if candidate and structure is Structure.SUFFIX_COMMA:
243 # The first report of the comma's OWN structure call in the
244 # library, and it is emitted for the branch taken HERE only --
245 # the flip. Where the structure did not move, that token's
246 # reading is still open and `assign` takes it on the
247 # family-comma path, so it reports there; this DECISION is
248 # never reported twice, and a second ambiguous token elsewhere
249 # is a second fork reporting on its own
250 # (mechanisms.md#AMBIGUITY-AT-THE-DECISION-SITE -- emitted
251 # where the branch is taken, and this branch is taken here).
252 # The neighbouring reports are other forks: C2's structural
253 # flag below says what the parse could not RECOGNIZE, and P6's
254 # attachment fork ("Berg, Jan vd", post_rules, since 2.3) is
255 # the attachment. rules.md#C1's comma-quiet policy gains its
256 # exception for this class and no other.
257 #
258 # `disp`/the index tuple cover the WHOLE post-comma part --
259 # the same text as before for the single-token listed and
260 # dotted halves (a join of one element is that element), while
261 # the caps half's run ('LEED AP') is the first time this class
262 # reaches the comma form as more than one token (#516). The
263 # single-token case skips the join: measured, a generator
264 # expression is its own frame on 3.11 regardless of element
265 # count (unlike a list comprehension, which PEP 709 inlines
266 # only from 3.12), so the join alone cost every REPORTING
267 # comma name (`John Smith, MA`, `John Smith, A.B.`, `Davis
268 # Royce, Ed`) +2 frames at the DEFAULT policy, a path this
269 # switch must not touch at all (#516 review round, F4).
270 disp = (state.tokens[groups[1][0]].text if len(groups[1]) == 1
271 else " ".join(state.tokens[i].text for i in groups[1]))
272 ambiguities.append(PendingAmbiguity(
273 AmbiguityKind.SUFFIX_OR_NAME,
274 f"{disp!r} after the comma is also an "
275 f"ordinary name word; the part before the comma holds "
276 f"{pre_comma_names} name words, so it is read as a "
277 f"credential run",
278 groups[1]))
279 # rules.md#C2: "a non-empty extra part that is not entirely suffix
280 # words is flagged as a structural ambiguity rather than rejected"
281 # -- parts[2:] are consumed as suffixes unconditionally either
282 # way, so a non-suffix tail segment gets the COMMA_STRUCTURE
283 # flag, not a structure veto. The lean reaches this reading too:
284 # a tail of leaning credentials is a credential run, which is the
285 # one place this design quiets a report rather than adding one.
286 for seg in groups[2:]:
287 # empty segments are consumed silently (v1 skips them without
288 # comment); only non-empty non-suffix tails get flagged.
289 # The case-FREE `suffixy(seg)` first: by construction the
290 # leaning call can only ADD a disjunct to it
291 # (`is_wholly_suffix`'s credential-lean branch), never remove
292 # one -- so a seg that already reads wholly suffix case-free
293 # reads so case-aware too, and the case fact is worth forcing
294 # only when the case-free answer was False (measured
295 # regression: the leaning call forced the fact for every tail
296 # segment, suffix or not).
297 # `class_run` last, for the same lazy reason and one step
298 # further out: it is the only one of the three that walks the
299 # by-shape class, and it is asked only of a run BOTH suffix
300 # readings have already declined.
301 if (seg and not suffixy(seg) and not suffixy(seg, case_class())
302 and not class_run(seg)):
303 texts_joined = " ".join(texts(seg))
304 ambiguities.append(PendingAmbiguity(
305 AmbiguityKind.COMMA_STRUCTURE,
306 f"segment {texts_joined!r} beyond the recognized comma "
307 f"structures; consumed as suffix best-effort",
308 tuple(seg)))
309 return dataclasses.replace(state, segments=tuple(groups),
310 structure=structure,
311 ambiguities=tuple(ambiguities),
312 one_case=one_case)