Coverage for /pythoncovmergedfiles/medio/medio/usr/local/lib/python3.11/site-packages/nameparser/_render.py: 32%
Shortcuts on this page
r m x toggle line displays
j k next/prev highlighted chunk
0 (zero) top of page
1 (one) first highlighted chunk
Shortcuts on this page
r m x toggle line displays
j k next/prev highlighted chunk
0 (zero) top of page
1 (one) first highlighted chunk
1"""Rendering for the 2.0 API: ParsedName -> display strings.
3Layering: imports nameparser._types, and nameparser._lexicon for
4Lexicon.default() (capitalized() with lexicon=None) and _normalize
5(enforced by tests/v2/test_layering.py). Parsing code never imports
6this module; ParsedName's rendering methods delegate here via
7call-time imports.
9Malformed str.format specs beyond unknown keys (positional fields,
10bad conversions) surface the raw str.format error; only unknown KEYS
11get the enriched KeyError.
12"""
13from __future__ import annotations
15import re
17from nameparser._lexicon import Lexicon, _normalize
18from nameparser._types import (FOLDED_TAG, UNCLASSIFIED_TAG,
19 UNJOINED_CONJUNCTION_TAG, UNJOINED_TAG,
20 Ambiguity, ParsedName, Role, Token)
22_SPACES = re.compile(r"\s+")
23_SPACE_BEFORE_COMMA = re.compile(r"\s+,")
24_COMMA_CHAR = re.compile(r"[,،,]") # ASCII, Arabic, fullwidth
25_MAC = re.compile(r"^(ma?c)(\w{2,})", re.IGNORECASE)
26_WORD = re.compile(r"(\w|\.)+")
28#: str.format keys render() accepts: the seven role fields in canonical
29#: order (derived from Role -- never restated) plus the derived views.
30_DERIVED_VIEWS = ("family_base", "family_particles", "surnames", "given_names")
31_RENDER_KEYS = tuple(r.value for r in Role) + _DERIVED_VIEWS
33#: str.format keys initials() accepts: the three name-bearing roles.
34_INITIALS_KEYS = (Role.GIVEN.value, Role.MIDDLE.value, Role.FAMILY.value)
36#: Tags whose tokens contribute no initial outside the given group --
37#: unless the token also carries UNJOINED_TAG, i.e. the whole part is
38#: particles, in which case they are the part's only words and do
39#: contribute (rules.md#R3, #404). The mark readmits a token carrying
40#: EITHER tag: a conjunction with nothing to join is not acting as a
41#: conjunction any more than a particle with nothing to join is acting
42#: as a particle, so it is a name word of the part like the rest.
43#: Not STABLE_TAGS -- that also contains "initial", which must contribute.
44_SKIP_TAGS = frozenset({"particle", "conjunction"})
45#: The given group's own skip set (rules.md#R3, #461): a connective
46#: contributes no initial in ANY group where it is joining words, so
47#: the given group no longer exempts it -- "one rule for every group".
48#: The particle exemption stays: a given-group particle is a name word
49#: there, which is what the whole-group exemption was for.
50_SKIP_TAGS_GIVEN = frozenset({"conjunction"})
51#: Either unjoined mark readmits the word it sits on.
52_UNJOINED_MARKS = frozenset({UNJOINED_TAG, UNJOINED_CONJUNCTION_TAG})
54# Ported verbatim from v1 (nameparser/config/regexes.py "initial", minus
55# the empty alternative) -- layering forbids importing the pipeline here;
56# keep in sync with _pipeline/_vocab.py by hand.
57# Its one reader is _reads_as_conjunction below, and that reader only
58# ever sees a bare string with no token attached to it -- a spliced
59# field (replace()), restored state (__setstate__: a pickle load,
60# copy.copy, or copy.deepcopy), a direct string call, or -- since
61# 84d9000 -- a widen-only _process_initial override that drops the
62# token it was handed on its way to the string path. The first two
63# never had a token to begin with: for anything the parser classified
64# AND the caller passed the token along, the tag is the answer and
65# this pattern is not asked. The last two might have been classified
66# and the reader cannot tell -- it only knows no token was passed.
67# So the two copies no longer decide the same question about the same
68# token -- _vocab's says what the parse decided, this one says what it
69# WOULD have decided about text handed over with no token -- which is
70# why they must keep answering alike, and why test_regex_sync pins the
71# patterns against each other and against config.
72# Deliberately NOT composed with _vocab's repertoire test (#320):
73# layering forbids the import. The divergence is reachable only for a
74# caller-added CJK conjunction spliced into a field, since no shipped
75# vocabulary carries one, and it costs nothing there: CJK is caseless,
76# so the carve-out's lower() and the fall-through's capitalize() return
77# the same string. Since #528 this pattern is read by more than case
78# repair: through _reads_as_conjunction below, whose callers are case
79# repair's spliced field and the v1 facade's initials view, the latter
80# in three shapes -- a token carrying UNCLASSIFIED_TAG
81# (_facade._token_is_conjunction), a direct _process_initial call with
82# no tokens at all, and, since 84d9000, a widen-only _process_initial
83# override that drops a token the parse DID classify. For initials the
84# CJK divergence would decide whether such a spliced, dropped or
85# never-parsed connective contributes a letter rather than which case
86# it renders in -- still unreachable from any shipped vocabulary, and
87# still not worth the import layering forbids.
88_INITIAL = re.compile(r"^(\w\.|[A-Z])$")
91def _reads_as_conjunction(word: str, lex: Lexicon) -> bool:
92 """v1's is_conjunction, asked only where the CALLER supplies no
93 token.
95 A token the parse classified carries its reading in its tags, and
96 the library's own token path -- _facade._token_is_conjunction --
97 consults that tag first and never reaches here for a token it
98 holds. This function is reached only where there is no token to
99 consult: a field spliced in as raw text (replace()) or restored
100 state (__setstate__: a pickle load, copy.copy, or copy.deepcopy),
101 both carrying UNCLASSIFIED_TAG; a direct call with no parse behind
102 it; or, since 84d9000, a widen-only _process_initial override that
103 drops the token it was handed on its way to the string path --
104 text the parse DID classify, whose reading the override chose not
105 to forward. This function cannot tell any of those apart from one
106 another; it can only give the answer the parser would have given
107 from the word alone, the initial carve-out included ('E.' assigned
108 to middle is an initial, not the Italian conjunction). What it
109 cannot give is an answer the parse reached by looking at the whole
110 NAME -- rules.md#P3's one-case fork is the live example -- which
111 is why it is the fallback and the tags are the rule.
112 """
113 return bool(_normalize(word) in lex.conjunctions
114 and not _INITIAL.fullmatch(word))
117def _collapse(rendered: str) -> str:
118 """The #254 collapse: empty fields substitute '' and every artifact
119 of that is removed -- dangling empty-nickname wrappers, space runs,
120 space-before-comma, one trailing comma character (any script),
121 leading/trailing ', ' debris."""
122 rendered = (rendered.replace(" ()", "")
123 .replace(" ''", "")
124 .replace(' ""', ""))
125 rendered = _SPACE_BEFORE_COMMA.sub(",", rendered)
126 rendered = _SPACES.sub(" ", rendered.strip())
127 if rendered and _COMMA_CHAR.fullmatch(rendered[-1]):
128 rendered = rendered[:-1]
129 return rendered.strip(", ")
132def _format_spec(spec: str, values: dict[str, str], noun: str,
133 keys: tuple[str, ...]) -> str:
134 """Shared tail of render()/initials(): fill the spec, enrich
135 unknown-KEY errors with the valid key list, collapse."""
136 if not isinstance(spec, str):
137 raise TypeError(f"spec must be a str, got {spec!r}")
138 try:
139 rendered = spec.format(**values)
140 except KeyError as exc:
141 raise KeyError(
142 f"unknown {noun} field {exc.args[0]!r}; valid fields: "
143 f"{', '.join(keys)}"
144 ) from None
145 return _collapse(rendered)
148def render(name: ParsedName, spec: str) -> str:
149 """Fill the str.format spec from the seven role fields and the
150 derived views (empty fields substitute ''), then apply the #254
151 collapse. Unknown keys raise KeyError naming the valid fields."""
152 values = {key: getattr(name, key) for key in _RENDER_KEYS}
153 return _format_spec(spec, values, "render", _RENDER_KEYS)
156# rules.md#R3: "initials take the first letter of each given, middle,
157# and base family word; titles, suffixes, particles and nicknames
158# contribute nothing"
159def initials(name: ParsedName, spec: str, delimiter: str, separator: str) -> str:
160 """First letter of each contributing token per group, v1 semantics:
161 delimiter follows each initial, separator sits between initials
162 within a group. Each group is ordered the way its FIELD is
163 ordered -- written order, except folded words, which initial
164 before the rest of the group (#408).
165 A token tagged conjunction contributes no initial in ANY group,
166 and one tagged particle contributes none in middle/family
167 (given-group particles always contribute); either unjoined mark
168 readmits the word it sits on, so the words of an all-particle part
169 and a connective with nothing in its part to join both count;
170 tags come from the pipeline --
171 hand-built untagged tokens all contribute, and so do the words of
172 a field spliced in by replace(), which the parse never read.
173 This view takes NO lexicon, so it has none to fall back to for
174 that text: `replace(family='de la vega')` initials every word of
175 that field where the same name parsed gives 'j. v.'
176 (rules.md#R3's Accepted
177 clause, and decisions.md#R4 for why the fallback was tried
178 and dropped -- #464 is the crossing that would make it
179 answerable). Valid spec keys: given, middle, family."""
180 if not isinstance(delimiter, str):
181 raise TypeError(f"delimiter must be a str, got {delimiter!r}")
182 if not isinstance(separator, str):
183 raise TypeError(f"separator must be a str, got {separator!r}")
184 values: dict[str, str] = {}
185 for key in _INITIALS_KEYS:
186 role = Role(key)
187 tokens = name.tokens_for(role)
188 skip = _SKIP_TAGS_GIVEN if role is Role.GIVEN else _SKIP_TAGS
189 tokens = tuple(t for t in tokens
190 if not (skip & t.tags)
191 or _UNJOINED_MARKS & t.tags)
192 # mechanisms.md#FOLDED_TAG: "a rule that needs different
193 # rendering order tags the token, and the rendering views
194 # consult the tag" -- this is a rendering view, so it reads
195 # the tag the same way _types._text_for does, and for the same
196 # reason: the fold is an ORDER the parse recorded, not one the
197 # view is free to take again
198 # (mechanisms.md#RENDER-HONORS-THE-PARSE: "the parse decides
199 # it; the render views honor those decisions and never
200 # re-evaluate them"). Applied to every role this view renders,
201 # exactly as _text_for applies it -- the pipeline puts the tag
202 # on FAMILY tokens alone today, so GIVEN and MIDDLE are
203 # uniformity with the mechanism rather than reachable
204 # behavior; a producer that ever folds into another part would
205 # otherwise reopen #408 there.
206 tokens = (tuple(t for t in tokens if FOLDED_TAG in t.tags)
207 + tuple(t for t in tokens if FOLDED_TAG not in t.tags))
208 values[key] = separator.join(
209 t.text[0] + delimiter for t in tokens)
210 return _format_spec(spec, values, "initials", _INITIALS_KEYS)
213def _cap_word(word: str, role: Role, tags: frozenset[str],
214 lex: Lexicon) -> str:
215 # v1 cap_word order: particle/conjunction rule first, then the
216 # exceptions map, then Mac/Mc, then str.capitalize
217 normalized = _normalize(word)
218 # rules.md#R4: "a part whose every word is particle vocabulary is
219 # repaired as ordinary name words, since none of them is doing a
220 # particle's work there" -- UNJOINED_TAG is that mark (#407).
221 # Only the PARTICLE conjunct is gated on it, and that is the rule
222 # rather than an omission -- rules.md#R4: "A CONNECTIVE the parse
223 # placed among the name words keeps its lowercase wherever it
224 # stands there, including inside a part whose other words the
225 # unjoined mark has turned into ordinary name words" -- so a
226 # conjunction keeps conjunction treatment even inside a part the
227 # mark has turned into ordinary name words.
228 # The `generation` guard below is the other half of that sentence:
229 # a word this vocabulary holds can ALSO be the generation it
230 # spells ('i' is the Catalan link and the roman numeral), and
231 # where the parse read the generation the token still carries the
232 # `conjunction` tag classify gave it -- so without the guard,
233 # `parse("John Quincy Smith i").capitalized(force=True)` gave
234 # 'John Quincy Smith i' where every release through 2.3 gave
235 # 'John Quincy Smith I' (#397 review). Such a token is repaired
236 # as the suffix it was read as, which is the rest of R4's
237 # sentence: "one the parse read as the generation it also spells
238 # is not a connective of this name at all".
239 # BOTH HALVES, and the role alone is not enough -- the role says
240 # where the word landed and the vocabulary says whether landing
241 # there made it a generation. A plain connective can land in the
242 # suffix field without being generational vocabulary at all (a
243 # third comma part: `Smith, John, and`), and on the role test
244 # alone every one of them was repaired as a name word --
245 # 'John Smith And' where 1.4.0, 2.0 through 2.3 and the parent
246 # commit all gave 'John Smith and', and 'John Smith De, Y' for a
247 # field spliced to suffix='de y' where R4's own Accepted
248 # paragraph says the vocabulary answers and the 'y' keeps its
249 # lowercase (#397 second review). `vocab:suffix` is classify's
250 # record of the vocabulary half, so the pair reads two decisions
251 # the parse already made and re-derives neither
252 # (mechanisms.md#RENDER-HONORS-THE-PARSE).
253 # It guards the whole test rather than the two conjunction arms
254 # alone, which reads as the wider claim and is not one: the
255 # particle arm asks for role MIDDLE or FAMILY, so a SUFFIX-roled
256 # token can never reach it either way.
257 # No SHIPPED name witnesses the difference: `particles` and
258 # `conjunctions` are disjoint in the default vocabulary and in
259 # every locale pack, so no shipped conjunction can sit in an
260 # all-particle part and carry the mark. That is a property of the
261 # shipped DATA, not an invariant -- both sets are public,
262 # configurable API, and a caller's Lexicon may put one word in
263 # both, the way _pipeline/_post_rules.py's arms allow for. Measured:
264 # under `Lexicon.default().add(particles={'y'})`, `anh y van` has
265 # an all-particle family whose `y` carries both tags and the mark,
266 # and gives 'Anh y Van'; gating this conjunct too would give
267 # 'Anh Y Van'. That is pinned by test_repair_keeps_a_conjunction_
268 # lowercase_in_a_particle_part -- until which gating it passed the
269 # whole suite.
270 # initials() does NOT match this carve-out, and since #461 that
271 # is a DECIDED disagreement rather than a recorded one: a
272 # connective that initials because it joins nothing is still not
273 # written the way a name is written, which is the sentence quoted
274 # above and this rule's own reason rather than a borrowing from
275 # R3. Under that same lexicon `Anh y Van` repairs to 'Anh y Van'
276 # and initials 'A. y. V.' -- the two views agreeing on this row
277 # because R2's mark readmits the word for both -- while
278 # `parse("Juan de y")` repairs to 'Juan de y' and initials
279 # 'J. y.', where they part. Pinned by
280 # test_initials_readmits_a_conjunction_in_a_particle_part and
281 # test_repair_keeps_a_lone_connective_lowercase_where_it_initials.
282 # That conjunct reads the TAG, not the word (#458). classify takes
283 # the conjunction-versus-initial decision once, over the whole
284 # token -- v1's is_conjunction excludes initials, so 'E.' in
285 # 'Scott E. Werner' is an initial and is never tagged (pinned live
286 # 2026-07-17) -- and a view honors that decision rather than
287 # taking it again from the spelling
288 # (mechanisms.md#RENDER-HONORS-THE-PARSE: "the render views honor
289 # those decisions and never re-evaluate them"), the tags being
290 # classify's record of it (mechanisms.md#VOCAB-TAGS: "later stages
291 # test tags"). Asking again was not even the same question:
292 # the copy of the initial pattern that stood here was the SHAPE
293 # half alone, and it re-decided per WORD of a token's text, so
294 # 'juan e-f smith' capitalized to 'Juan e-F Smith'.
295 # mechanisms.md#RENDER-HONORS-THE-PARSE: "a token the parse never
296 # saw carries no decision to honor, so a view falls back to the
297 # vocabulary" -- _reads_as_conjunction above, which is v1's
298 # predicate applied over TODAY's vocabulary rather than 1.4.0's.
299 # That is the honest claim and it is narrower than parity: the two
300 # vocabularies differ, so an assigned field can repair differently
301 # from 1.4.0 without this predicate differing at all. Measured on
302 # the released wheel: `h.last = "хосе и мария сантос"` gives
303 # 'Хосе И Мария Сантос' on 1.4.0 and 'Хосе и Мария Сантос' here,
304 # the Cyrillic `и` being a 2.x conjunction and not a 1.4.0 one;
305 # `h.last = "de la vega"` gives 'de la Vega' there and here.
306 # The mark, not the SPAN, is what says the text was never read:
307 # Parser.revise() also builds span-less tokens, from a sub-parse
308 # whose tags it keeps on purpose, and keying this on `span is None`
309 # overrode them -- `revise(middle='e-f')` repaired to 'e-F' where
310 # the same words parsed gave 'E-F' (#463 review).
311 generation = role is Role.SUFFIX and "vocab:suffix" in tags
312 if not generation and (
313 (normalized in lex.particles
314 and role in (Role.MIDDLE, Role.FAMILY)
315 and UNJOINED_TAG not in tags)
316 or "conjunction" in tags
317 or (UNCLASSIFIED_TAG in tags
318 and _reads_as_conjunction(word, lex))):
319 return word.lower()
320 # v1 cap_word tries the edge-stripped form, then the period-free
321 # form ('Ph.D.' -> 'ph.d' -> 'phd' hits the exceptions map)
322 for key in (normalized, normalized.replace(".", "")):
323 exception = lex.capitalization_exceptions_map.get(key)
324 if exception is not None:
325 return exception
326 # A credential acronym the exceptions map doesn't carry (mba, jd,
327 # qc, mp, ...) is an initialism, not a word to title-case: a one-
328 # case name repairs to the acronym's caps instead of 'Mba' (#459).
329 # The exceptions map is consulted first and holds the entries that
330 # spell differently -- md -> M.D. and phd -> Ph.D. (the generational
331 # ii/iii/iv are suffix_words, not acronyms, and ride the map because
332 # str.capitalize() would give 'Ii'). The all-caps default is the
333 # right call for an initialism; its cost is that an acronym
334 # conventionally written mixed-case (bsc, msc) reads all-caps here
335 # (BSc -> BSC under force) rather than mixed, which the letter-mask
336 # design deferred to #459 is meant to recover. Gated on the SUFFIX
337 # role so a word that is a family name only happens to be in the
338 # vocabulary (anh van DO) still repairs as an ordinary name word.
339 if role is Role.SUFFIX and normalized.replace(".", "") in lex.suffix_acronyms:
340 return word.upper()
341 if _MAC.match(word):
342 return _MAC.sub(
343 lambda m: m.group(1).capitalize() + m.group(2).capitalize(),
344 word)
345 return word.capitalize()
348def _cap_text(text: str, role: Role, tags: frozenset[str],
349 lex: Lexicon) -> str:
350 # word-by-word within the token text: hyphenated names capitalize
351 # both sides ("macdole-eisenhower" -> "MacDole-Eisenhower"). The
352 # per-word walk is also why an UNCLASSIFIED token gets the
353 # vocabulary asked per word: the parse would have made one token
354 # per word of that text, so this is the granularity its answer
355 # would have had.
356 return _WORD.sub(lambda m: _cap_word(m.group(0), role, tags, lex), text)
359# rules.md#R4: "case repair returns a repaired copy and never mutates
360# the parse"
361def capitalized(name: ParsedName, lexicon: Lexicon | None, *,
362 force: bool) -> ParsedName:
363 """Case-fixing transform -> new ParsedName, same spans, new token
364 texts. Gate (v1 parity): only single-case input is
365 touched unless force=True; the gate reads the joined token texts
366 (not render() output -- the case gate stays decoupled from spec
367 formatting and the #254 collapse).
368 The repair reads token TAGS as well as texts: a part whose every
369 word is particle vocabulary is repaired as ordinary name words,
370 and the mark saying so comes from the pipeline, as does the
371 reading that a word is a conjunction rather than an initial. A
372 token carrying UNCLASSIFIED_TAG -- replace() splices those in, and
373 so does the facade's v1 pickle load -- was never read: the
374 vocabulary answers the per-word conjunction question for it, and
375 the per-part particle question is left to plain particle treatment,
376 since re-deriving the part answer needs a tag on every word of the
377 part and these have none. A family set that way to 'de la' stays
378 'de la' where the same words parsed give 'De La'; one set to
379 'de y' keeps the 'y' lowercase, as the parse does and as 1.4.0
380 did. Parser.revise() is the edit that classifies the value, and
381 gives 'De La' (rules.md#R4's Accepted boundary).
382 Idempotent: without force, a capitalized result is mixed-case and
383 the gate returns it unchanged; with force, every _cap_word rule is
384 a fixpoint on its own output."""
385 if lexicon is not None and not isinstance(lexicon, Lexicon):
386 # eager, before the gate: a garbage argument must not become a
387 # silent no-op on mixed-case input or a deep AttributeError
388 raise TypeError(f"lexicon must be a Lexicon or None, got {lexicon!r}")
389 lex = Lexicon.default() if lexicon is None else lexicon
390 joined = " ".join(t.text for t in name.tokens)
391 # rules.md#R5: "case repair acts only on a name written entirely
392 # in one case"
393 if not force and joined not in (joined.upper(), joined.lower()):
394 return name
395 new_tokens = tuple(
396 Token(_cap_text(t.text, t.role, t.tags, lex), t.span, t.role, t.tags)
397 for t in name.tokens)
398 # equal tokens (possible only for synthetic span=None duplicates)
399 # collapse to one mapping entry -- benign: the rebuilt ambiguity
400 # references an equal token, so the subset invariant still holds
401 replacement = dict(zip(name.tokens, new_tokens))
402 new_ambiguities = tuple(
403 Ambiguity(a.kind, a.detail,
404 tuple(replacement[t] for t in a.tokens))
405 for a in name.ambiguities)
406 return ParsedName(original=name.original, tokens=new_tokens,
407 ambiguities=new_ambiguities)