Coverage for /pythoncovmergedfiles/medio/medio/usr/local/lib/python3.11/site-packages/nameparser/_render.py: 32%

Shortcuts on this page

r m x   toggle line displays

j k   next/prev highlighted chunk

0   (zero) top of page

1   (one) first highlighted chunk

78 statements  

1"""Rendering for the 2.0 API: ParsedName -> display strings. 

2 

3Layering: imports nameparser._types, and nameparser._lexicon for 

4Lexicon.default() (capitalized() with lexicon=None) and _normalize 

5(enforced by tests/v2/test_layering.py). Parsing code never imports 

6this module; ParsedName's rendering methods delegate here via 

7call-time imports. 

8 

9Malformed str.format specs beyond unknown keys (positional fields, 

10bad conversions) surface the raw str.format error; only unknown KEYS 

11get the enriched KeyError. 

12""" 

13from __future__ import annotations 

14 

15import re 

16 

17from nameparser._lexicon import Lexicon, _normalize 

18from nameparser._types import (FOLDED_TAG, UNCLASSIFIED_TAG, 

19 UNJOINED_CONJUNCTION_TAG, UNJOINED_TAG, 

20 Ambiguity, ParsedName, Role, Token) 

21 

22_SPACES = re.compile(r"\s+") 

23_SPACE_BEFORE_COMMA = re.compile(r"\s+,") 

24_COMMA_CHAR = re.compile(r"[,،,]") # ASCII, Arabic, fullwidth 

25_MAC = re.compile(r"^(ma?c)(\w{2,})", re.IGNORECASE) 

26_WORD = re.compile(r"(\w|\.)+") 

27 

28#: str.format keys render() accepts: the seven role fields in canonical 

29#: order (derived from Role -- never restated) plus the derived views. 

30_DERIVED_VIEWS = ("family_base", "family_particles", "surnames", "given_names") 

31_RENDER_KEYS = tuple(r.value for r in Role) + _DERIVED_VIEWS 

32 

33#: str.format keys initials() accepts: the three name-bearing roles. 

34_INITIALS_KEYS = (Role.GIVEN.value, Role.MIDDLE.value, Role.FAMILY.value) 

35 

36#: Tags whose tokens contribute no initial outside the given group -- 

37#: unless the token also carries UNJOINED_TAG, i.e. the whole part is 

38#: particles, in which case they are the part's only words and do 

39#: contribute (rules.md#R3, #404). The mark readmits a token carrying 

40#: EITHER tag: a conjunction with nothing to join is not acting as a 

41#: conjunction any more than a particle with nothing to join is acting 

42#: as a particle, so it is a name word of the part like the rest. 

43#: Not STABLE_TAGS -- that also contains "initial", which must contribute. 

44_SKIP_TAGS = frozenset({"particle", "conjunction"}) 

45#: The given group's own skip set (rules.md#R3, #461): a connective 

46#: contributes no initial in ANY group where it is joining words, so 

47#: the given group no longer exempts it -- "one rule for every group". 

48#: The particle exemption stays: a given-group particle is a name word 

49#: there, which is what the whole-group exemption was for. 

50_SKIP_TAGS_GIVEN = frozenset({"conjunction"}) 

51#: Either unjoined mark readmits the word it sits on. 

52_UNJOINED_MARKS = frozenset({UNJOINED_TAG, UNJOINED_CONJUNCTION_TAG}) 

53 

54# Ported verbatim from v1 (nameparser/config/regexes.py "initial", minus 

55# the empty alternative) -- layering forbids importing the pipeline here; 

56# keep in sync with _pipeline/_vocab.py by hand. 

57# Its one reader is _reads_as_conjunction below, and that reader only 

58# ever sees a bare string with no token attached to it -- a spliced 

59# field (replace()), restored state (__setstate__: a pickle load, 

60# copy.copy, or copy.deepcopy), a direct string call, or -- since 

61# 84d9000 -- a widen-only _process_initial override that drops the 

62# token it was handed on its way to the string path. The first two 

63# never had a token to begin with: for anything the parser classified 

64# AND the caller passed the token along, the tag is the answer and 

65# this pattern is not asked. The last two might have been classified 

66# and the reader cannot tell -- it only knows no token was passed. 

67# So the two copies no longer decide the same question about the same 

68# token -- _vocab's says what the parse decided, this one says what it 

69# WOULD have decided about text handed over with no token -- which is 

70# why they must keep answering alike, and why test_regex_sync pins the 

71# patterns against each other and against config. 

72# Deliberately NOT composed with _vocab's repertoire test (#320): 

73# layering forbids the import. The divergence is reachable only for a 

74# caller-added CJK conjunction spliced into a field, since no shipped 

75# vocabulary carries one, and it costs nothing there: CJK is caseless, 

76# so the carve-out's lower() and the fall-through's capitalize() return 

77# the same string. Since #528 this pattern is read by more than case 

78# repair: through _reads_as_conjunction below, whose callers are case 

79# repair's spliced field and the v1 facade's initials view, the latter 

80# in three shapes -- a token carrying UNCLASSIFIED_TAG 

81# (_facade._token_is_conjunction), a direct _process_initial call with 

82# no tokens at all, and, since 84d9000, a widen-only _process_initial 

83# override that drops a token the parse DID classify. For initials the 

84# CJK divergence would decide whether such a spliced, dropped or 

85# never-parsed connective contributes a letter rather than which case 

86# it renders in -- still unreachable from any shipped vocabulary, and 

87# still not worth the import layering forbids. 

88_INITIAL = re.compile(r"^(\w\.|[A-Z])$") 

89 

90 

91def _reads_as_conjunction(word: str, lex: Lexicon) -> bool: 

92 """v1's is_conjunction, asked only where the CALLER supplies no 

93 token. 

94 

95 A token the parse classified carries its reading in its tags, and 

96 the library's own token path -- _facade._token_is_conjunction -- 

97 consults that tag first and never reaches here for a token it 

98 holds. This function is reached only where there is no token to 

99 consult: a field spliced in as raw text (replace()) or restored 

100 state (__setstate__: a pickle load, copy.copy, or copy.deepcopy), 

101 both carrying UNCLASSIFIED_TAG; a direct call with no parse behind 

102 it; or, since 84d9000, a widen-only _process_initial override that 

103 drops the token it was handed on its way to the string path -- 

104 text the parse DID classify, whose reading the override chose not 

105 to forward. This function cannot tell any of those apart from one 

106 another; it can only give the answer the parser would have given 

107 from the word alone, the initial carve-out included ('E.' assigned 

108 to middle is an initial, not the Italian conjunction). What it 

109 cannot give is an answer the parse reached by looking at the whole 

110 NAME -- rules.md#P3's one-case fork is the live example -- which 

111 is why it is the fallback and the tags are the rule. 

112 """ 

113 return bool(_normalize(word) in lex.conjunctions 

114 and not _INITIAL.fullmatch(word)) 

115 

116 

117def _collapse(rendered: str) -> str: 

118 """The #254 collapse: empty fields substitute '' and every artifact 

119 of that is removed -- dangling empty-nickname wrappers, space runs, 

120 space-before-comma, one trailing comma character (any script), 

121 leading/trailing ', ' debris.""" 

122 rendered = (rendered.replace(" ()", "") 

123 .replace(" ''", "") 

124 .replace(' ""', "")) 

125 rendered = _SPACE_BEFORE_COMMA.sub(",", rendered) 

126 rendered = _SPACES.sub(" ", rendered.strip()) 

127 if rendered and _COMMA_CHAR.fullmatch(rendered[-1]): 

128 rendered = rendered[:-1] 

129 return rendered.strip(", ") 

130 

131 

132def _format_spec(spec: str, values: dict[str, str], noun: str, 

133 keys: tuple[str, ...]) -> str: 

134 """Shared tail of render()/initials(): fill the spec, enrich 

135 unknown-KEY errors with the valid key list, collapse.""" 

136 if not isinstance(spec, str): 

137 raise TypeError(f"spec must be a str, got {spec!r}") 

138 try: 

139 rendered = spec.format(**values) 

140 except KeyError as exc: 

141 raise KeyError( 

142 f"unknown {noun} field {exc.args[0]!r}; valid fields: " 

143 f"{', '.join(keys)}" 

144 ) from None 

145 return _collapse(rendered) 

146 

147 

148def render(name: ParsedName, spec: str) -> str: 

149 """Fill the str.format spec from the seven role fields and the 

150 derived views (empty fields substitute ''), then apply the #254 

151 collapse. Unknown keys raise KeyError naming the valid fields.""" 

152 values = {key: getattr(name, key) for key in _RENDER_KEYS} 

153 return _format_spec(spec, values, "render", _RENDER_KEYS) 

154 

155 

156# rules.md#R3: "initials take the first letter of each given, middle, 

157# and base family word; titles, suffixes, particles and nicknames 

158# contribute nothing" 

159def initials(name: ParsedName, spec: str, delimiter: str, separator: str) -> str: 

160 """First letter of each contributing token per group, v1 semantics: 

161 delimiter follows each initial, separator sits between initials 

162 within a group. Each group is ordered the way its FIELD is 

163 ordered -- written order, except folded words, which initial 

164 before the rest of the group (#408). 

165 A token tagged conjunction contributes no initial in ANY group, 

166 and one tagged particle contributes none in middle/family 

167 (given-group particles always contribute); either unjoined mark 

168 readmits the word it sits on, so the words of an all-particle part 

169 and a connective with nothing in its part to join both count; 

170 tags come from the pipeline -- 

171 hand-built untagged tokens all contribute, and so do the words of 

172 a field spliced in by replace(), which the parse never read. 

173 This view takes NO lexicon, so it has none to fall back to for 

174 that text: `replace(family='de la vega')` initials every word of 

175 that field where the same name parsed gives 'j. v.' 

176 (rules.md#R3's Accepted 

177 clause, and decisions.md#R4 for why the fallback was tried 

178 and dropped -- #464 is the crossing that would make it 

179 answerable). Valid spec keys: given, middle, family.""" 

180 if not isinstance(delimiter, str): 

181 raise TypeError(f"delimiter must be a str, got {delimiter!r}") 

182 if not isinstance(separator, str): 

183 raise TypeError(f"separator must be a str, got {separator!r}") 

184 values: dict[str, str] = {} 

185 for key in _INITIALS_KEYS: 

186 role = Role(key) 

187 tokens = name.tokens_for(role) 

188 skip = _SKIP_TAGS_GIVEN if role is Role.GIVEN else _SKIP_TAGS 

189 tokens = tuple(t for t in tokens 

190 if not (skip & t.tags) 

191 or _UNJOINED_MARKS & t.tags) 

192 # mechanisms.md#FOLDED_TAG: "a rule that needs different 

193 # rendering order tags the token, and the rendering views 

194 # consult the tag" -- this is a rendering view, so it reads 

195 # the tag the same way _types._text_for does, and for the same 

196 # reason: the fold is an ORDER the parse recorded, not one the 

197 # view is free to take again 

198 # (mechanisms.md#RENDER-HONORS-THE-PARSE: "the parse decides 

199 # it; the render views honor those decisions and never 

200 # re-evaluate them"). Applied to every role this view renders, 

201 # exactly as _text_for applies it -- the pipeline puts the tag 

202 # on FAMILY tokens alone today, so GIVEN and MIDDLE are 

203 # uniformity with the mechanism rather than reachable 

204 # behavior; a producer that ever folds into another part would 

205 # otherwise reopen #408 there. 

206 tokens = (tuple(t for t in tokens if FOLDED_TAG in t.tags) 

207 + tuple(t for t in tokens if FOLDED_TAG not in t.tags)) 

208 values[key] = separator.join( 

209 t.text[0] + delimiter for t in tokens) 

210 return _format_spec(spec, values, "initials", _INITIALS_KEYS) 

211 

212 

213def _cap_word(word: str, role: Role, tags: frozenset[str], 

214 lex: Lexicon) -> str: 

215 # v1 cap_word order: particle/conjunction rule first, then the 

216 # exceptions map, then Mac/Mc, then str.capitalize 

217 normalized = _normalize(word) 

218 # rules.md#R4: "a part whose every word is particle vocabulary is 

219 # repaired as ordinary name words, since none of them is doing a 

220 # particle's work there" -- UNJOINED_TAG is that mark (#407). 

221 # Only the PARTICLE conjunct is gated on it, and that is the rule 

222 # rather than an omission -- rules.md#R4: "A CONNECTIVE the parse 

223 # placed among the name words keeps its lowercase wherever it 

224 # stands there, including inside a part whose other words the 

225 # unjoined mark has turned into ordinary name words" -- so a 

226 # conjunction keeps conjunction treatment even inside a part the 

227 # mark has turned into ordinary name words. 

228 # The `generation` guard below is the other half of that sentence: 

229 # a word this vocabulary holds can ALSO be the generation it 

230 # spells ('i' is the Catalan link and the roman numeral), and 

231 # where the parse read the generation the token still carries the 

232 # `conjunction` tag classify gave it -- so without the guard, 

233 # `parse("John Quincy Smith i").capitalized(force=True)` gave 

234 # 'John Quincy Smith i' where every release through 2.3 gave 

235 # 'John Quincy Smith I' (#397 review). Such a token is repaired 

236 # as the suffix it was read as, which is the rest of R4's 

237 # sentence: "one the parse read as the generation it also spells 

238 # is not a connective of this name at all". 

239 # BOTH HALVES, and the role alone is not enough -- the role says 

240 # where the word landed and the vocabulary says whether landing 

241 # there made it a generation. A plain connective can land in the 

242 # suffix field without being generational vocabulary at all (a 

243 # third comma part: `Smith, John, and`), and on the role test 

244 # alone every one of them was repaired as a name word -- 

245 # 'John Smith And' where 1.4.0, 2.0 through 2.3 and the parent 

246 # commit all gave 'John Smith and', and 'John Smith De, Y' for a 

247 # field spliced to suffix='de y' where R4's own Accepted 

248 # paragraph says the vocabulary answers and the 'y' keeps its 

249 # lowercase (#397 second review). `vocab:suffix` is classify's 

250 # record of the vocabulary half, so the pair reads two decisions 

251 # the parse already made and re-derives neither 

252 # (mechanisms.md#RENDER-HONORS-THE-PARSE). 

253 # It guards the whole test rather than the two conjunction arms 

254 # alone, which reads as the wider claim and is not one: the 

255 # particle arm asks for role MIDDLE or FAMILY, so a SUFFIX-roled 

256 # token can never reach it either way. 

257 # No SHIPPED name witnesses the difference: `particles` and 

258 # `conjunctions` are disjoint in the default vocabulary and in 

259 # every locale pack, so no shipped conjunction can sit in an 

260 # all-particle part and carry the mark. That is a property of the 

261 # shipped DATA, not an invariant -- both sets are public, 

262 # configurable API, and a caller's Lexicon may put one word in 

263 # both, the way _pipeline/_post_rules.py's arms allow for. Measured: 

264 # under `Lexicon.default().add(particles={'y'})`, `anh y van` has 

265 # an all-particle family whose `y` carries both tags and the mark, 

266 # and gives 'Anh y Van'; gating this conjunct too would give 

267 # 'Anh Y Van'. That is pinned by test_repair_keeps_a_conjunction_ 

268 # lowercase_in_a_particle_part -- until which gating it passed the 

269 # whole suite. 

270 # initials() does NOT match this carve-out, and since #461 that 

271 # is a DECIDED disagreement rather than a recorded one: a 

272 # connective that initials because it joins nothing is still not 

273 # written the way a name is written, which is the sentence quoted 

274 # above and this rule's own reason rather than a borrowing from 

275 # R3. Under that same lexicon `Anh y Van` repairs to 'Anh y Van' 

276 # and initials 'A. y. V.' -- the two views agreeing on this row 

277 # because R2's mark readmits the word for both -- while 

278 # `parse("Juan de y")` repairs to 'Juan de y' and initials 

279 # 'J. y.', where they part. Pinned by 

280 # test_initials_readmits_a_conjunction_in_a_particle_part and 

281 # test_repair_keeps_a_lone_connective_lowercase_where_it_initials. 

282 # That conjunct reads the TAG, not the word (#458). classify takes 

283 # the conjunction-versus-initial decision once, over the whole 

284 # token -- v1's is_conjunction excludes initials, so 'E.' in 

285 # 'Scott E. Werner' is an initial and is never tagged (pinned live 

286 # 2026-07-17) -- and a view honors that decision rather than 

287 # taking it again from the spelling 

288 # (mechanisms.md#RENDER-HONORS-THE-PARSE: "the render views honor 

289 # those decisions and never re-evaluate them"), the tags being 

290 # classify's record of it (mechanisms.md#VOCAB-TAGS: "later stages 

291 # test tags"). Asking again was not even the same question: 

292 # the copy of the initial pattern that stood here was the SHAPE 

293 # half alone, and it re-decided per WORD of a token's text, so 

294 # 'juan e-f smith' capitalized to 'Juan e-F Smith'. 

295 # mechanisms.md#RENDER-HONORS-THE-PARSE: "a token the parse never 

296 # saw carries no decision to honor, so a view falls back to the 

297 # vocabulary" -- _reads_as_conjunction above, which is v1's 

298 # predicate applied over TODAY's vocabulary rather than 1.4.0's. 

299 # That is the honest claim and it is narrower than parity: the two 

300 # vocabularies differ, so an assigned field can repair differently 

301 # from 1.4.0 without this predicate differing at all. Measured on 

302 # the released wheel: `h.last = "хосе и мария сантос"` gives 

303 # 'Хосе И Мария Сантос' on 1.4.0 and 'Хосе и Мария Сантос' here, 

304 # the Cyrillic `и` being a 2.x conjunction and not a 1.4.0 one; 

305 # `h.last = "de la vega"` gives 'de la Vega' there and here. 

306 # The mark, not the SPAN, is what says the text was never read: 

307 # Parser.revise() also builds span-less tokens, from a sub-parse 

308 # whose tags it keeps on purpose, and keying this on `span is None` 

309 # overrode them -- `revise(middle='e-f')` repaired to 'e-F' where 

310 # the same words parsed gave 'E-F' (#463 review). 

311 generation = role is Role.SUFFIX and "vocab:suffix" in tags 

312 if not generation and ( 

313 (normalized in lex.particles 

314 and role in (Role.MIDDLE, Role.FAMILY) 

315 and UNJOINED_TAG not in tags) 

316 or "conjunction" in tags 

317 or (UNCLASSIFIED_TAG in tags 

318 and _reads_as_conjunction(word, lex))): 

319 return word.lower() 

320 # v1 cap_word tries the edge-stripped form, then the period-free 

321 # form ('Ph.D.' -> 'ph.d' -> 'phd' hits the exceptions map) 

322 for key in (normalized, normalized.replace(".", "")): 

323 exception = lex.capitalization_exceptions_map.get(key) 

324 if exception is not None: 

325 return exception 

326 # A credential acronym the exceptions map doesn't carry (mba, jd, 

327 # qc, mp, ...) is an initialism, not a word to title-case: a one- 

328 # case name repairs to the acronym's caps instead of 'Mba' (#459). 

329 # The exceptions map is consulted first and holds the entries that 

330 # spell differently -- md -> M.D. and phd -> Ph.D. (the generational 

331 # ii/iii/iv are suffix_words, not acronyms, and ride the map because 

332 # str.capitalize() would give 'Ii'). The all-caps default is the 

333 # right call for an initialism; its cost is that an acronym 

334 # conventionally written mixed-case (bsc, msc) reads all-caps here 

335 # (BSc -> BSC under force) rather than mixed, which the letter-mask 

336 # design deferred to #459 is meant to recover. Gated on the SUFFIX 

337 # role so a word that is a family name only happens to be in the 

338 # vocabulary (anh van DO) still repairs as an ordinary name word. 

339 if role is Role.SUFFIX and normalized.replace(".", "") in lex.suffix_acronyms: 

340 return word.upper() 

341 if _MAC.match(word): 

342 return _MAC.sub( 

343 lambda m: m.group(1).capitalize() + m.group(2).capitalize(), 

344 word) 

345 return word.capitalize() 

346 

347 

348def _cap_text(text: str, role: Role, tags: frozenset[str], 

349 lex: Lexicon) -> str: 

350 # word-by-word within the token text: hyphenated names capitalize 

351 # both sides ("macdole-eisenhower" -> "MacDole-Eisenhower"). The 

352 # per-word walk is also why an UNCLASSIFIED token gets the 

353 # vocabulary asked per word: the parse would have made one token 

354 # per word of that text, so this is the granularity its answer 

355 # would have had. 

356 return _WORD.sub(lambda m: _cap_word(m.group(0), role, tags, lex), text) 

357 

358 

359# rules.md#R4: "case repair returns a repaired copy and never mutates 

360# the parse" 

361def capitalized(name: ParsedName, lexicon: Lexicon | None, *, 

362 force: bool) -> ParsedName: 

363 """Case-fixing transform -> new ParsedName, same spans, new token 

364 texts. Gate (v1 parity): only single-case input is 

365 touched unless force=True; the gate reads the joined token texts 

366 (not render() output -- the case gate stays decoupled from spec 

367 formatting and the #254 collapse). 

368 The repair reads token TAGS as well as texts: a part whose every 

369 word is particle vocabulary is repaired as ordinary name words, 

370 and the mark saying so comes from the pipeline, as does the 

371 reading that a word is a conjunction rather than an initial. A 

372 token carrying UNCLASSIFIED_TAG -- replace() splices those in, and 

373 so does the facade's v1 pickle load -- was never read: the 

374 vocabulary answers the per-word conjunction question for it, and 

375 the per-part particle question is left to plain particle treatment, 

376 since re-deriving the part answer needs a tag on every word of the 

377 part and these have none. A family set that way to 'de la' stays 

378 'de la' where the same words parsed give 'De La'; one set to 

379 'de y' keeps the 'y' lowercase, as the parse does and as 1.4.0 

380 did. Parser.revise() is the edit that classifies the value, and 

381 gives 'De La' (rules.md#R4's Accepted boundary). 

382 Idempotent: without force, a capitalized result is mixed-case and 

383 the gate returns it unchanged; with force, every _cap_word rule is 

384 a fixpoint on its own output.""" 

385 if lexicon is not None and not isinstance(lexicon, Lexicon): 

386 # eager, before the gate: a garbage argument must not become a 

387 # silent no-op on mixed-case input or a deep AttributeError 

388 raise TypeError(f"lexicon must be a Lexicon or None, got {lexicon!r}") 

389 lex = Lexicon.default() if lexicon is None else lexicon 

390 joined = " ".join(t.text for t in name.tokens) 

391 # rules.md#R5: "case repair acts only on a name written entirely 

392 # in one case" 

393 if not force and joined not in (joined.upper(), joined.lower()): 

394 return name 

395 new_tokens = tuple( 

396 Token(_cap_text(t.text, t.role, t.tags, lex), t.span, t.role, t.tags) 

397 for t in name.tokens) 

398 # equal tokens (possible only for synthetic span=None duplicates) 

399 # collapse to one mapping entry -- benign: the rebuilt ambiguity 

400 # references an equal token, so the subset invariant still holds 

401 replacement = dict(zip(name.tokens, new_tokens)) 

402 new_ambiguities = tuple( 

403 Ambiguity(a.kind, a.detail, 

404 tuple(replacement[t] for t in a.tokens)) 

405 for a in name.ambiguities) 

406 return ParsedName(original=name.original, tokens=new_tokens, 

407 ambiguities=new_ambiguities)