Coverage for /pythoncovmergedfiles/medio/medio/usr/local/lib/python3.11/site-packages/nameparser/_pipeline/_segment.py: 98%

Shortcuts on this page

r m x   toggle line displays

j k   next/prev highlighted chunk

0   (zero) top of page

1   (one) first highlighted chunk

51 statements  

1"""Stage: segment. 

2 

3Consumes: tokens (role-None main stream), comma_offsets, one_case 

4(where an earlier stage recorded it). 

5Produces: segments (runs of main-token indices; interior segments may 

6be EMPTY -- doubled commas keep their structural position), structure, 

7one_case where the comma form asked for it, COMMA_STRUCTURE 

8ambiguities for unrecognized extra segments, and SUFFIX_OR_NAME where 

9the comma FLIPPED the structure for a member of the ambiguous 

10credential class. The flip and nothing else: where the structure did 

11not move, the word's reading is still open and `assign` takes it on 

12the family-comma path, so it reports there. 

13Reads: Lexicon suffix vocabulary and Policy, both through 

14_vocab.is_wholly_suffix -- the suffix-comma decision is definitionally 

15vocabulary-dependent (decisions.md#C1), and the predicate 

16owns the rest (Policy.lenient_comma_suffixes picks the lenient or 

17strict token test; Policy.extra_suffix_delimiters gives v1 

18suffix_delimiter parity, a delimiter-core token being transparent); 

19Lexicon.maiden_markers DIRECTLY, for the own-words span the lazy case 

20gate takes (_pieces.own_words); Lexicon title and suffix vocabulary 

21plus Policy.lenient_comma_suffixes again through _vocab. 

22name_word_count, which counts NAME words for the class's own comma 

23rule; and, since 2.4, Policy.unlisted_dotted_suffixes through 

24_vocab.ambiguous_class_candidate, and Policy.unlisted_caps_suffixes 

25DIRECTLY as this stage's own gate before the run test calls 

26_vocab.caps_shape_candidate -- which reads every Lexicon vocabulary 

27field in turn (_lexicon._VOCAB_FIELDS) to decide that an all-caps 

28word is UNLISTED. `is_wholly_suffix` deliberately sees neither 

29by-shape half, dotted or caps (#516). An unlisted dotted or all-caps 

30token joins the ambiguous credential class by SHAPE at this stage's 

31own candidate tests the same way a listed member does; the caps half 

32additionally needs `one_case` to decide membership at all, which this 

33stage's own lazy gate supplies. 

34 

35Implements rules C1 and C2 of docs/design/rules.md, cited at the 

36decision site below; history in decisions.md#C1. 

37""" 

38from __future__ import annotations 

39 

40import dataclasses 

41 

42from nameparser._pipeline._pieces import own_words 

43from nameparser._pipeline._state import ( 

44 ParseState, PendingAmbiguity, Structure, comma_bucket, 

45) 

46from nameparser._pipeline._vocab import ( 

47 ambiguous_class_candidate, ambiguous_class_member, caps_shape_candidate, 

48 is_one_case, is_wholly_suffix, name_word_count, 

49) 

50from nameparser._types import AmbiguityKind 

51 

52 

53 

54 

55 

56def segment(state: ParseState) -> ParseState: 

57 main = [i for i, t in enumerate(state.tokens) if t.role is None] 

58 if not main: 

59 return dataclasses.replace(state, segments=(), 

60 structure=Structure.NO_COMMA) 

61 if not state.comma_offsets: 

62 return dataclasses.replace(state, segments=(tuple(main),), 

63 structure=Structure.NO_COMMA) 

64 buckets: list[list[int]] = [[] for _ in range(len(state.comma_offsets) + 1)] 

65 for i in main: 

66 # _state.comma_bucket, not a local bisect: classify asks the 

67 # same question of the same offsets to keep a marker run inside 

68 # one segment, and the two must be one expression rather than 

69 # two that agree 

70 buckets[comma_bucket(state.tokens[i].span.start, 

71 state.comma_offsets)].append(i) 

72 groups = [tuple(b) for b in buckets] 

73 # v1 strips exactly ONE trailing comma as cosmetic (parser.py's 

74 # collapse_whitespace); every other empty bucket is STRUCTURAL and 

75 # keeps its position -- in 'Doe,, Jr.' the given segment is empty, 

76 # so 'Jr.' stays a tail suffix instead of masquerading as a lone 

77 # post-comma title (v1 parity, pinned live 2026-07-16) 

78 if len(groups) > 1 and not groups[-1]: 

79 groups.pop() 

80 if len(groups) <= 1: 

81 segs = tuple(groups) if groups and groups[0] else (tuple(main),) 

82 return dataclasses.replace(state, segments=segs, 

83 structure=Structure.NO_COMMA) 

84 

85 # The case fact, asked LAZILY: only a comma form can turn on it 

86 # here, and only where the part after the first comma is a single 

87 # token -- the shape the ambiguous class comes in. A comma-less 

88 # name never reaches this and pays nothing; classify asks for 

89 # itself later where this did not (decisions.md#S2, and #429's 

90 # precedent for paying a predicate twice rather than plumbing a 

91 # field two sites would not otherwise share). 

92 one_case = state.one_case 

93 

94 def case_class() -> bool: 

95 nonlocal one_case 

96 if one_case is None: 

97 # `own, _` rather than `[0]`: the second element is the 

98 # maiden clause's start index, which this stage has no use 

99 # for, and saying so by name is what stops a reader having 

100 # to go and look up what a bare subscript dropped. 

101 own, _ = own_words(state.tokens, state.comma_offsets, 

102 state.lexicon.maiden_markers) 

103 one_case = is_one_case(own) 

104 return one_case 

105 

106 def texts(seg: tuple[int, ...]) -> list[str]: 

107 return [state.tokens[i].text for i in seg] 

108 

109 # Inlined rather than built on `texts` (measured, #289/#516's 

110 # eager-gate fix round): every comma parse calls `suffixy` at 

111 # least once, and a `texts(seg)` indirection costs a SECOND frame 

112 # on top of the comprehension's own -- on 3.11 a list comprehension 

113 # IS a frame (PEP 709 inlines it only from 3.12 on; see 

114 # tools/perf/call_count.py's docstring), so wrapping it in another 

115 # call doubles the cost every comma name pays, member or not. 

116 # `texts` still serves the two call sites that are not on this 

117 # path (name_word_count's pre-comma texts, and a flagged tail 

118 # segment's joined display). 

119 # 

120 # `case` is ParseState.one_case: None is the case-FREE reading, 

121 # which is what every caller on this path wants, and the tail 

122 # walk below passes `case_class()` for the leaning one. One 

123 # closure with a default rather than two whose bodies differ only 

124 # in that argument. 

125 def suffixy(seg: tuple[int, ...], case: bool | None = None) -> bool: 

126 return is_wholly_suffix([state.tokens[i].text for i in seg], 

127 state.lexicon, state.policy, one_case=case) 

128 

129 def class_run(seg: tuple[int, ...]) -> bool: 

130 # Every token of the run joins the ambiguous credential class 

131 # BY SHAPE -- a candidate that is not a LISTED member, which is 

132 # the one half `is_wholly_suffix` cannot see (its own docstring 

133 # says so, and the blindness is what keeps a by-shape token out 

134 # of C1's legacy token-count disjunct). A tail segment is 

135 # consumed as suffix either way, so what this decides is only 

136 # whether the parse says it did not RECOGNIZE the segment -- and 

137 # a run the parse itself reads as a credential run by shape is 

138 # recognized. Without it, the narrow roman retirement 

139 # (rules.md#S3) moved 'R.A.I.', 'X.Y.I.' and 'J.u.n.i.o.r.' out 

140 # of the vocabulary verdict and into the shape class, and every 

141 # one of them gained a COMMA_STRUCTURE flag in a third segment 

142 # that 2.3 did not raise -- a report about the parser's own new 

143 # reading rather than about the name (review round, #289/#516). 

144 # 

145 # The LISTED half is excluded rather than folded in: a listed 

146 # member reaches this reading through the LEAN (the leaning 

147 # `suffixy` call below), and that is the whole of what quiets 

148 # it -- 'STEVEN HARDMAN, MD, DO, DDS' is written in one case, 

149 # leans nothing, and keeps its flag, the recorded negative 

150 # control for the lean's own effect here. Admitting membership 

151 # alone would silence it and leave the control measuring 

152 # nothing. 

153 # 

154 # Policy-sensitive by construction: with 

155 # `unlisted_dotted_suffixes` off the token is name material, is 

156 # no candidate, and the flag stands. 

157 return all(ambiguous_class_candidate(state.tokens[i].text, 

158 state.lexicon, state.policy) 

159 and not ambiguous_class_member(state.tokens[i].text, 

160 state.lexicon) 

161 for i in seg) 

162 

163 # rules.md#C1: "the name reads as trailing suffixes when the part 

164 # after the first comma is entirely suffix words and more than one 

165 # word precedes the comma; otherwise it reads as the listing form" 

166 # (v1 parity: only parts[1] decides, parser.py:1318; history: 

167 # decisions.md#C1) 

168 # 

169 # And for the AMBIGUOUS class the count is of NAME words, not of 

170 # words: 'Smith Jr., MA' is two tokens and one name, and the token 

171 # count hands its family to `given` (#289/#516). One rule for the 

172 # whole class: the listed and dotted halves are asked only of a 

173 # single-token part, a member of either being always exactly one 

174 # token (a bare word, or one glued acronym), while the caps half 

175 # may come as a RUN ('LEED AP', below). 

176 # 

177 # Membership is tested CASE-FREE first (`ambiguous_class_candidate`, 

178 # the listed set OR -- since 2.4 -- a by-shape member Policy 

179 # admits, #516): the structure decision below is itself 

180 # case-independent (item 5's count decides "whatever case the name 

181 # is written in"), so the case fact is worth forcing only once a 

182 # genuine candidate is found. Passing `case_class()` as an 

183 # ARGUMENT instead ran the `own_words` -> `tag_marker_runs` walk 

184 # for every comma name whose post-comma part is one token -- 

185 # `"Smith, John"`, `"John Smith, Jr."`, neither able to reach the 

186 # class at all (measured regression). 

187 candidate = (len(groups[1]) == 1 

188 and ambiguous_class_candidate( 

189 state.tokens[groups[1][0]].text, state.lexicon, 

190 state.policy)) 

191 # The all-caps half (Policy.unlisted_caps_suffixes, #516) is the 

192 # FIRST shape this class can wear across more than one token -- 

193 # 'LEED AP' is two separate all-caps words, not one glued acronym 

194 # -- and the one membership test in this class that NEEDS the case 

195 # fact to answer membership at all: 'XYZ' is only credential-shaped 

196 # where the name contrasts it. Gated on the switch (default off, so 

197 # a non-candidate comma name never enters this branch) and tried 

198 # only where the single-token test above already declined. 

199 # 

200 # `caps_shape_candidate` directly, not the un-narrowed 

201 # `ambiguous_class_candidate`: the run is a property of the CAPS 

202 # class ALONE (#516 review round, F2), the other two halves being 

203 # single-token by construction, so `all()` over more than one of 

204 # THEM asks a question the design never posed. Measured, the 

205 # un-narrowed call per token moved `'John Smith, Ed Ma'`, `'John 

206 # Smith, ma do'` and `'John Smith, X.Y.Z. A.B.'` to a credential 

207 # run neither the spec nor any case row wants. 

208 # 

209 # `one_case=False` asks the case-free question -- "if this name 

210 # turned out mixed, would EVERY token in the run join the CAPS 

211 # shape" -- the same trick the single-token test above uses, 

212 # generalized over `all()`. Every other conjunct of 

213 # `caps_shape_candidate` is case-independent, so for a run already 

214 # confirmed shape-eligible the fact itself is the only unknown 

215 # left, and the verdict is `case_class() is False` directly rather 

216 # than a second walk that could only reach the same answer (a 

217 # quality-review finding: the walk was provably redundant). 

218 if (not candidate and state.policy.unlisted_caps_suffixes and groups[1] 

219 and all(caps_shape_candidate(state.tokens[i].text, 

220 state.lexicon, state.policy, 

221 one_case=False) 

222 for i in groups[1])): 

223 candidate = case_class() is False 

224 # Computed only where `candidate` is true, alongside `case_class()` 

225 # -- the same lazy gate: a non-candidate comma name never counts 

226 # its pre-comma words either. Hoisted to a local because the 

227 # report below quotes the exact count rather than a hardcoded 

228 # "two" ('John Q. Public, MA' has three). 

229 pre_comma_names = None 

230 if candidate: 

231 # forced here, downstream of the structure decision that does 

232 # not need it, because assign's post-comma slot and its report do 

233 case_class() 

234 pre_comma_names = name_word_count(texts(groups[0]), state.lexicon, 

235 state.policy) 

236 structure = ( 

237 Structure.SUFFIX_COMMA 

238 if ((suffixy(groups[1]) and len(groups[0]) > 1) 

239 or (pre_comma_names is not None and pre_comma_names >= 2)) 

240 else Structure.FAMILY_COMMA) 

241 ambiguities = list(state.ambiguities) 

242 if candidate and structure is Structure.SUFFIX_COMMA: 

243 # The first report of the comma's OWN structure call in the 

244 # library, and it is emitted for the branch taken HERE only -- 

245 # the flip. Where the structure did not move, that token's 

246 # reading is still open and `assign` takes it on the 

247 # family-comma path, so it reports there; this DECISION is 

248 # never reported twice, and a second ambiguous token elsewhere 

249 # is a second fork reporting on its own 

250 # (mechanisms.md#AMBIGUITY-AT-THE-DECISION-SITE -- emitted 

251 # where the branch is taken, and this branch is taken here). 

252 # The neighbouring reports are other forks: C2's structural 

253 # flag below says what the parse could not RECOGNIZE, and P6's 

254 # attachment fork ("Berg, Jan vd", post_rules, since 2.3) is 

255 # the attachment. rules.md#C1's comma-quiet policy gains its 

256 # exception for this class and no other. 

257 # 

258 # `disp`/the index tuple cover the WHOLE post-comma part -- 

259 # the same text as before for the single-token listed and 

260 # dotted halves (a join of one element is that element), while 

261 # the caps half's run ('LEED AP') is the first time this class 

262 # reaches the comma form as more than one token (#516). The 

263 # single-token case skips the join: measured, a generator 

264 # expression is its own frame on 3.11 regardless of element 

265 # count (unlike a list comprehension, which PEP 709 inlines 

266 # only from 3.12), so the join alone cost every REPORTING 

267 # comma name (`John Smith, MA`, `John Smith, A.B.`, `Davis 

268 # Royce, Ed`) +2 frames at the DEFAULT policy, a path this 

269 # switch must not touch at all (#516 review round, F4). 

270 disp = (state.tokens[groups[1][0]].text if len(groups[1]) == 1 

271 else " ".join(state.tokens[i].text for i in groups[1])) 

272 ambiguities.append(PendingAmbiguity( 

273 AmbiguityKind.SUFFIX_OR_NAME, 

274 f"{disp!r} after the comma is also an " 

275 f"ordinary name word; the part before the comma holds " 

276 f"{pre_comma_names} name words, so it is read as a " 

277 f"credential run", 

278 groups[1])) 

279 # rules.md#C2: "a non-empty extra part that is not entirely suffix 

280 # words is flagged as a structural ambiguity rather than rejected" 

281 # -- parts[2:] are consumed as suffixes unconditionally either 

282 # way, so a non-suffix tail segment gets the COMMA_STRUCTURE 

283 # flag, not a structure veto. The lean reaches this reading too: 

284 # a tail of leaning credentials is a credential run, which is the 

285 # one place this design quiets a report rather than adding one. 

286 for seg in groups[2:]: 

287 # empty segments are consumed silently (v1 skips them without 

288 # comment); only non-empty non-suffix tails get flagged. 

289 # The case-FREE `suffixy(seg)` first: by construction the 

290 # leaning call can only ADD a disjunct to it 

291 # (`is_wholly_suffix`'s credential-lean branch), never remove 

292 # one -- so a seg that already reads wholly suffix case-free 

293 # reads so case-aware too, and the case fact is worth forcing 

294 # only when the case-free answer was False (measured 

295 # regression: the leaning call forced the fact for every tail 

296 # segment, suffix or not). 

297 # `class_run` last, for the same lazy reason and one step 

298 # further out: it is the only one of the three that walks the 

299 # by-shape class, and it is asked only of a run BOTH suffix 

300 # readings have already declined. 

301 if (seg and not suffixy(seg) and not suffixy(seg, case_class()) 

302 and not class_run(seg)): 

303 texts_joined = " ".join(texts(seg)) 

304 ambiguities.append(PendingAmbiguity( 

305 AmbiguityKind.COMMA_STRUCTURE, 

306 f"segment {texts_joined!r} beyond the recognized comma " 

307 f"structures; consumed as suffix best-effort", 

308 tuple(seg))) 

309 return dataclasses.replace(state, segments=tuple(groups), 

310 structure=structure, 

311 ambiguities=tuple(ambiguities), 

312 one_case=one_case)