Coverage for /pythoncovmergedfiles/medio/medio/usr/local/lib/python3.11/site-packages/nameparser/_facade.py: 48%
Shortcuts on this page
r m x toggle line displays
j k next/prev highlighted chunk
0 (zero) top of page
1 (one) first highlighted chunk
Shortcuts on this page
r m x toggle line displays
j k next/prev highlighted chunk
0 (zero) top of page
1 (one) first highlighted chunk
1"""The 2.0 ``HumanName`` facade (mechanisms.md#FACADE-CONTRACT): a
2mutable wrapper
3over a frozen ParsedName, delegating parsing to the core Parser resolved
4from the bound Constants shim. Keeps every v1 spelling. Deleted in 3.0.
6Layering: facade layer -- may import anything public plus _render.
7"""
8from __future__ import annotations
10import dataclasses
11import warnings
12from collections.abc import Iterator, Mapping
13from typing import Any
15# Import order matters here -- breaks a real import cycle. nameparser.
16# config's package __init__ re-exports CONSTANTS/Constants/etc. from
17# _config_shim (the v1 nameparser.config.Constants compat path), while
18# _config_shim's own default CONSTANTS singleton needs nameparser.config's
19# DATA submodules (titles, prefixes, ...), imported lazily -- see
20# _config_shim.py's module docstring. If _config_shim were the first of
21# the two ever touched, building its CONSTANTS would need to import
22# nameparser.config, whose __init__ would in turn need _config_shim's
23# (not-yet-built) CONSTANTS: ImportError. Importing the config package
24# here first lets its __init__ run to completion; when IT then imports
25# _config_shim to build the default CONSTANTS, nameparser.config is
26# already registered in sys.modules, so its data-submodule imports
27# resolve directly instead of re-entering (and failing on) its own
28# still-executing __init__.
29import nameparser.config # noqa: F401
31import nameparser._render as _render
32from nameparser._config_shim import CONSTANTS, Constants, _cached_parser
33from nameparser._lexicon import _normalize
34from nameparser._parser import Parser
35from nameparser._types import (FOLDED_TAG, UNCLASSIFIED_TAG,
36 UNJOINED_CONJUNCTION_TAG, ParsedName,
37 Role, Token)
39_V2_FIELD = {"first": "given", "last": "family"} # v1 name -> v2 name
40_V1_SPELLING = {v2: v1 for v1, v2 in _V2_FIELD.items()}
41# derived from Role: declaration order IS the canonical field order
42# (never restated), rendered in the v1 spellings
43_MEMBERS = tuple(_V1_SPELLING.get(r.value, r.value) for r in Role)
47#: v1 parsing hooks the facade never calls
48#: (mechanisms.md#FACADE-CONTRACT / #280).
49_V1_HOOKS = (
50 "pre_process", "post_process", "parse_full_name", "parse_pieces",
51 "parse_nicknames", "join_on_conjunctions", "squash_emoji",
52 "handle_firstnames", "handle_middle_name_as_last",
53 "is_title", "is_conjunction", "is_prefix", "is_roman_numeral",
54 "is_suffix", "is_suffix_lenient", "is_an_initial",
55)
56# Module-level mutable state (sanctioned exception, AGENTS.md "One
57# sanctioned global"): strong-references every distinct HumanName
58# subclass for process lifetime so the hook warning fires once per
59# class. Fine in practice -- subclasses are statically defined, and the
60# whole module is deleted with the facade layer in 3.0.
61_WARNED_SUBCLASSES: set[type] = set()
64def _empty_parsed() -> ParsedName:
65 return ParsedName(original="", tokens=(), ambiguities=())
68class HumanName:
69 """v1 ``HumanName`` facade: a mutable wrapper over a frozen
70 ``ParsedName``, delegating all parsing to the core ``Parser``
71 resolved from the bound ``Constants`` shim (dirty-tracked via its
72 ``_generation``). Keeps every v1 spelling (``first``/``last`` over
73 the core ``given``/``family``); the v1 parsing hooks are never
74 called (#280). Deleted with the facade layer in 3.0.
75 """
77 def __init__(
78 self,
79 full_name: str = "",
80 constants: Constants | None = CONSTANTS,
81 string_format: str | None = None,
82 initials_format: str | None = None,
83 initials_delimiter: str | None = None,
84 initials_separator: str | None = None,
85 suffix_delimiter: str | None = None,
86 first: str | list[str] | None = None,
87 middle: str | list[str] | None = None,
88 last: str | list[str] | None = None,
89 title: str | list[str] | None = None,
90 suffix: str | list[str] | None = None,
91 nickname: str | list[str] | None = None,
92 maiden: str | list[str] | None = None,
93 ) -> None:
94 if constants is None:
95 raise TypeError(
96 "constants=None was removed in 2.0 (#261): pass a "
97 "Constants instance, or use the new Parser/Lexicon/"
98 "Policy API for per-call configuration"
99 )
100 if not isinstance(constants, Constants):
101 raise TypeError(
102 f"constants must be a Constants instance, got {constants!r}"
103 )
104 self._warn_overridden_hooks()
105 self._C = constants
106 self._snapshot_gen = -1 # forces first resolve
107 _, _, defaults = constants._snapshot()
108 self.string_format = (string_format if string_format is not None
109 else defaults.string_format)
110 self.initials_format = (initials_format if initials_format is not None
111 else defaults.initials_format)
112 self.initials_delimiter = (
113 initials_delimiter if initials_delimiter is not None
114 else defaults.initials_delimiter)
115 self.initials_separator = (
116 initials_separator if initials_separator is not None
117 else defaults.initials_separator)
118 # These five assignments route through the validating properties
119 # below. The suffix_delimiter setter resets _snapshot_gen to -1
120 # on every assignment (including this one, harmlessly -- it's
121 # already -1 above), so reassigning it post-construction (e.g.
122 # n.suffix_delimiter = " - ") correctly forces the next
123 # _resolve() to rebuild the Policy with the new delimiter.
124 self.suffix_delimiter = (suffix_delimiter if suffix_delimiter is not None
125 else defaults.suffix_delimiter)
126 self._full_name = ""
127 self._parsed = _empty_parsed()
128 if first or middle or last or title or suffix or nickname or maiden:
129 # These route through the field-setter properties (None
130 # clears the field); no full-string parse, full_name stays "".
131 self.first = first
132 self.middle = middle
133 self.last = last
134 self.title = title
135 self.suffix = suffix
136 self.nickname = nickname
137 self.maiden = maiden
138 else:
139 self._apply_full_name(full_name)
141 @classmethod
142 def _warn_overridden_hooks(cls) -> None:
143 if cls is HumanName or cls in _WARNED_SUBCLASSES:
144 return
145 overridden = [h for h in _V1_HOOKS
146 if getattr(cls, h, None) is not getattr(
147 HumanName, h, None)]
148 _WARNED_SUBCLASSES.add(cls)
149 if overridden:
150 warnings.warn(
151 f"{cls.__name__} overrides v1 parsing hooks "
152 f"({', '.join(overridden)}) that the 2.0 facade never "
153 f"calls; parsing is delegated to the core Parser. "
154 f"Migrate to the Lexicon/Policy API. See "
155 f"https://github.com/derek73/python-nameparser/issues/280",
156 DeprecationWarning, stacklevel=3)
158 # -- render defaults -----------------------------------------------------
159 # One-line validating setters (mechanisms.md#FACADE-CONTRACT):
160 # assigning a non-str (or, for
161 # the two fields that allow it, non-str-non-None) raises TypeError at
162 # assignment time instead of failing later inside .format().
164 @property
165 def string_format(self) -> str | None:
166 return self._string_format
168 @string_format.setter
169 def string_format(self, value: str | None) -> None:
170 if value is not None and not isinstance(value, str):
171 raise TypeError(
172 f"string_format must be a str or None, got {value!r}")
173 self._string_format = value
175 @property
176 def initials_format(self) -> str:
177 return self._initials_format
179 @initials_format.setter
180 def initials_format(self, value: str) -> None:
181 if not isinstance(value, str):
182 raise TypeError(
183 f"initials_format must be a str, got {value!r}")
184 self._initials_format = value
186 @property
187 def initials_delimiter(self) -> str:
188 return self._initials_delimiter
190 @initials_delimiter.setter
191 def initials_delimiter(self, value: str) -> None:
192 if not isinstance(value, str):
193 raise TypeError(
194 f"initials_delimiter must be a str, got {value!r}")
195 self._initials_delimiter = value
197 @property
198 def initials_separator(self) -> str:
199 return self._initials_separator
201 @initials_separator.setter
202 def initials_separator(self, value: str) -> None:
203 if not isinstance(value, str):
204 raise TypeError(
205 f"initials_separator must be a str, got {value!r}")
206 self._initials_separator = value
208 @property
209 def suffix_delimiter(self) -> str | None:
210 return self._suffix_delimiter
212 @suffix_delimiter.setter
213 def suffix_delimiter(self, value: str | None) -> None:
214 if value is not None and not isinstance(value, str):
215 raise TypeError(
216 f"suffix_delimiter must be a str or None, got {value!r}")
217 self._suffix_delimiter = value
218 # Invalidate the cached Policy: _resolve() layers suffix_delimiter
219 # onto extra_suffix_delimiters, so a stale snapshot would keep
220 # parsing against the old delimiter.
221 self._snapshot_gen = -1
223 # -- config / parsing ---------------------------------------------------
225 def _resolve(self) -> Parser:
226 """Dirty-tracked parser resolution
227 (mechanisms.md#CONFIG-SHIM-SNAPSHOT): rebuild the
228 snapshot only when the bound Constants' generation moved."""
229 gen = self._C._generation
230 if self._snapshot_gen != gen:
231 lexicon, policy, _ = self._C._snapshot()
232 if self.suffix_delimiter:
233 policy = dataclasses.replace(
234 policy,
235 extra_suffix_delimiters=frozenset(
236 {self.suffix_delimiter}))
237 self._lexicon, self._policy = lexicon, policy
238 self._parser = _cached_parser(lexicon, policy)
239 self._snapshot_gen = gen
240 # the fast path is a plain attribute return: hashing the two
241 # value objects for the lru lookup is the whole fast-path cost
242 return self._parser
244 def parse_full_name(self) -> None:
245 """Re-parse the stored ``full_name`` (v1's documented re-parse
246 trigger, docs/customize.rst): mutate ``name.C`` then call this to
247 force a re-parse without reassigning ``full_name``. The v1
248 parsing INTERNALS this name evokes live in the core ``Parser``,
249 not here; a subclass overriding this method still triggers the
250 #280 hook-override warning, and full_name assignment never
251 consults it."""
252 self._apply_full_name(self._full_name)
254 def _apply_full_name(self, value: str) -> None:
255 if isinstance(value, bytes):
256 raise TypeError(
257 "bytes input was removed in 2.0 (#245): decode first, "
258 "e.g. HumanName(raw.decode('utf-8'))"
259 )
260 if not isinstance(value, str):
261 raise TypeError(f"full_name must be a str, got {value!r}")
262 # parse FIRST: if snapshot resolution raises, the instance must
263 # not be left with a new full_name over the old parsed fields
264 parsed = self._resolve().parse(value)
265 self._full_name = value
266 self._parsed = parsed
267 if self._C.capitalize_name:
268 self.capitalize() # v1 parser.py:1653 parity
270 def capitalize(self, force: bool | None = None) -> None:
271 """Re-capitalize the current parse against the bound lexicon.
272 force=None reads the bound Constants' render default
273 (force_mixed_case_capitalization); the core's capitalized()
274 implements the single-case gate (v1 parity) -- not
275 re-implemented here."""
276 self._resolve()
277 if force is None:
278 force = self._C.force_mixed_case_capitalization
279 self._parsed = self._parsed.capitalized(self._lexicon, force=force)
281 @property
282 def full_name(self) -> str:
283 return self._full_name
285 @full_name.setter
286 def full_name(self, value: str) -> None:
287 self._apply_full_name(value)
289 @property
290 def original(self) -> str:
291 return self._parsed.original or self._full_name
293 @property
294 def C(self) -> Constants:
295 return self._C
297 @C.setter
298 def C(self, constants: Constants | None) -> None:
299 # v1.4 closed #239 by making C a validating setter that ONLY
300 # stores the new value -- no re-parse (checked against v1.4:
301 # `git show 2d5d8c2:nameparser/parser.py` lines ~204-206, the C
302 # setter body is exactly `self._C = self._validate_constants(...)`).
303 # A caller who wants the new config reflected must still trigger
304 # a re-parse, e.g. via parse_full_name() or a full_name
305 # reassignment -- matched here rather than re-parsing eagerly.
306 if constants is None:
307 raise TypeError(
308 "assigning constants=None to C was removed in 2.0 (#261): "
309 "pass a Constants instance, or use the new Parser/Lexicon/"
310 "Policy API for per-call configuration"
311 )
312 if not isinstance(constants, Constants):
313 raise TypeError(
314 f"constants must be a Constants instance, got {constants!r}"
315 )
316 self._C = constants
317 self._snapshot_gen = -1 # invalidate: next _resolve() rebuilds
319 @property
320 def has_own_config(self) -> bool:
321 """True when this instance is not using the shared module-level
322 CONSTANTS."""
323 return self._C is not CONSTANTS
325 # -- fields ---------------------------------------------------------
327 def _get_field(self, member: str) -> str:
328 return getattr(self._parsed, _V2_FIELD.get(member, member))
330 def _set_field(self, member: str, value: str | list[str] | None) -> None:
331 if value is None:
332 joined = ""
333 elif isinstance(value, list):
334 for element in value:
335 if not isinstance(element, str):
336 raise TypeError(
337 f"name parts must be strings, got {element!r}")
338 joined = " ".join(value)
339 elif isinstance(value, str):
340 joined = value
341 else:
342 raise TypeError(
343 f"{member} must be a str, list, or None, got {value!r}")
344 # v1 setters stay on replace(): revise()'s vocabulary tags would
345 # change v1 parity
346 self._parsed = self._parsed.replace(
347 **{_V2_FIELD.get(member, member): joined})
349 def _list_tokens_for(self, member: str) -> list[tuple[Token, ...]]:
350 # A "joined" continuation token ("Ph." + "D.") belongs to its
351 # predecessor's part, matching v1's fix_phd (suffix_list had ONE
352 # "Ph. D." element). ParsedName._text_for heals only the suffix
353 # string view (the ", " join); the facade list view heals for
354 # every role -- a continuation is never its own list element.
355 role = Role(_V2_FIELD.get(member, member))
356 parts: list[list[Token]] = []
357 folded: list[list[Token]] = []
358 for tok in self._parsed.tokens_for(role):
359 if "joined" in tok.tags and parts:
360 parts[-1].append(tok)
361 elif FOLDED_TAG in tok.tags:
362 # middle_as_family fold: v1 PREPENDED middle_list to
363 # last_list -- keep the list view consistent with the
364 # string view (_text_for orders folded-first too)
365 folded.append([tok])
366 else:
367 parts.append([tok])
368 return [tuple(group) for group in folded + parts]
370 def _list_for(self, member: str) -> list[str]:
371 # The STRING view of the walk above, which is the v1 `*_list`
372 # shape. Two views off one walk rather than two walks: the
373 # initials view needs each element's backing tokens (#528) and
374 # every other reader needs its text, and a second walk could
375 # drift from this one on exactly the element boundaries the
376 # comment above exists to hold. Not on the parse path --
377 # measured 2026-09-13, HumanName(name) never reaches it -- so
378 # the tuple building is off the benchmarked budget
379 # (tests/v2/test_benchmark.py budgets parse() and HumanName()).
380 return [" ".join(tok.text for tok in group)
381 for group in self._list_tokens_for(member)]
383 @property
384 def title(self) -> str:
385 return self._get_field("title")
387 @title.setter
388 def title(self, value: str | list[str] | None) -> None:
389 self._set_field("title", value)
391 @property
392 def title_list(self) -> list[str]:
393 return self._list_for("title")
395 @property
396 def first(self) -> str:
397 return self._get_field("first")
399 @first.setter
400 def first(self, value: str | list[str] | None) -> None:
401 self._set_field("first", value)
403 @property
404 def first_list(self) -> list[str]:
405 return self._list_for("first")
407 @property
408 def middle(self) -> str:
409 return self._get_field("middle")
411 @middle.setter
412 def middle(self, value: str | list[str] | None) -> None:
413 self._set_field("middle", value)
415 @property
416 def middle_list(self) -> list[str]:
417 return self._list_for("middle")
419 @property
420 def last(self) -> str:
421 return self._get_field("last")
423 @last.setter
424 def last(self, value: str | list[str] | None) -> None:
425 self._set_field("last", value)
427 @property
428 def last_list(self) -> list[str]:
429 return self._list_for("last")
431 @property
432 def suffix(self) -> str:
433 return self._get_field("suffix")
435 @suffix.setter
436 def suffix(self, value: str | list[str] | None) -> None:
437 self._set_field("suffix", value)
439 @property
440 def suffix_list(self) -> list[str]:
441 return self._list_for("suffix")
443 @property
444 def nickname(self) -> str:
445 return self._get_field("nickname")
447 @nickname.setter
448 def nickname(self, value: str | list[str] | None) -> None:
449 self._set_field("nickname", value)
451 @property
452 def nickname_list(self) -> list[str]:
453 return self._list_for("nickname")
455 @property
456 def maiden(self) -> str:
457 return self._get_field("maiden")
459 @maiden.setter
460 def maiden(self, value: str | list[str] | None) -> None:
461 self._set_field("maiden", value)
463 @property
464 def maiden_list(self) -> list[str]:
465 return self._list_for("maiden")
467 # -- derived views ----------------------------------------------------
469 @property
470 def surnames_list(self) -> list[str]:
471 return self.middle_list + self.last_list
473 @property
474 def surnames(self) -> str:
475 return " ".join(self.surnames_list)
477 @property
478 def given_names_list(self) -> list[str]:
479 return self.first_list + self.middle_list
481 @property
482 def given_names(self) -> str:
483 return " ".join(self.given_names_list)
485 def _is_particle(self, text: str) -> bool:
486 self._resolve()
487 return _normalize(text) in self._lexicon.particles
489 def _token_is_conjunction(self, tok: Token) -> bool:
490 # #528: the PARSE's answer, not the vocabulary's. A token the
491 # parser classified carries its reading in its tags, which is
492 # the source the core's initials() has always read
493 # (mechanisms.md#RENDER-HONORS-THE-PARSE), so a bare capital
494 # 'Y' in a one-case name contributes no initial where its tag
495 # says connective, and a one-case 'e' contributes one where its
496 # tag says initial. Before #528 this was computed from the raw
497 # word instead -- "in the conjunctions set AND NOT
498 # _render._INITIAL" (v1's is_conjunction, restored by #462) --
499 # a shape test standing in for a tag, which stopped agreeing
500 # with the parse the moment #383/#479 gave the classifier a
501 # fork the shape cannot see.
502 #
503 # UNCLASSIFIED_TAG is the one case with no parse to honor: the
504 # words were spliced into a field as raw text, by `hn.middle =
505 # ...` (ParsedName.replace) or restored via __setstate__ -- a
506 # pickle load, or a copy.copy/copy.deepcopy, which go through
507 # the same state hooks. They carry no reading, so the vocabulary
508 # answers -- the same fallback _render._cap_word takes for the
509 # same tokens and through the same helper, which is why the
510 # helper is imported rather than the predicate rewritten
511 # (decisions.md#R4 for why one question and not two: whether a
512 # word is a connective is a fact the word can answer alone,
513 # whether a particle is acting as one is a fact about the part).
514 #
515 # _resolve() first, as _is_particle above does: an unpickled or
516 # copied instance has no _lexicon until resolved -- and neither
517 # does a keyword-constructed one (`HumanName(first=..., middle=
518 # "y", last=...)`), which never runs the full-string parse path
519 # other callers rely on to have called _resolve() already.
520 # LOAD-BEARING for exactly those caller shapes: do not delete
521 # this call as dead just because most callers arrive resolved.
522 self._resolve()
523 if UNCLASSIFIED_TAG in tok.tags:
524 return _render._reads_as_conjunction(tok.text, self._lexicon)
525 # #461: a connective with nothing in its part to join is not
526 # acting as one. The parse decided that and marked the token,
527 # and this reads the same mark the core's initials() reads, so
528 # the two views cannot disagree about it.
529 return ("conjunction" in tok.tags
530 and UNJOINED_CONJUNCTION_TAG not in tok.tags)
532 def _split_last(self) -> tuple[list[str], list[str]]:
533 # rules.md#R2: "a name part whose every word is particle
534 # vocabulary is a part where none of them is doing a
535 # particle's work" -- the all-particle guard
536 # below is this rule, and predates its statement: v1 assumed a
537 # family name does not consist entirely of particles, e.g. the
538 # surname "Do" which also appears in PARTICLES. v1
539 # parser.py _split_last otherwise verbatim, vocabulary lookup
540 # at ACCESS time so assigned last names split too.
541 words = " ".join(self.last_list).split()
542 i = 0
543 while i < len(words) and self._is_particle(words[i]):
544 i += 1
545 if i == len(words):
546 return [], words
547 return words[:i], words[i:]
549 @property
550 def last_prefixes_list(self) -> list[str]:
551 return self._split_last()[0]
553 @property
554 def last_prefixes(self) -> str:
555 return " ".join(self._split_last()[0])
557 @property
558 def last_base_list(self) -> list[str]:
559 return self._split_last()[1]
561 @property
562 def last_base(self) -> str:
563 return " ".join(self._split_last()[1])
565 # -- initials -------------------------------------------------------------
567 def _process_initial(self, name_part: str,
568 firstname: bool = False,
569 tokens: tuple[Token, ...] | None = None) -> str:
570 # after v1 parser.py:427, not verbatim: particles and
571 # conjunctions are filtered from initials unless the part is a
572 # first name.
573 #
574 # TWO WAYS IN. `tokens` is the part's backing tokens, which
575 # _initials_lists always has and passes; `name_part` is v1's
576 # signature, kept because subclasses and tests call this
577 # directly with a string (tests/test_initials.py), and there
578 # the words come from splitting it. `tokens` supersedes
579 # `name_part` entirely when given -- the words are the tokens'
580 # own text rather than a re-split of the joined element, so a
581 # two-word element and its tokens cannot fall out of step.
582 # STATED BREAK: _initials_lists always calls with `tokens=`, so
583 # a subclass overriding with v1's two-argument signature
584 # (name_part, firstname=False) now raises TypeError the first
585 # time initials() runs, rather than being silently skipped. The
586 # alternative -- a string wrapper kept over a token core --
587 # would make such an override silently ineffective instead,
588 # which hides the override rather than breaking it loudly.
589 # Such an override must ACCEPT `tokens` AND FORWARD it:
590 # super()._process_initial(name_part, firstname, tokens=tokens).
591 # That is still the only way to RECEIVE #528's fix. Widening
592 # the signature alone -- `**kwargs`, or a `tokens=None` the
593 # super() call drops -- does not raise and does not go silent
594 # either: `name_part` on this path is the group's own text
595 # (Derek, 2026-09-14), so the override takes the STRING path
596 # below and keeps working, just without the fix -- the
597 # PRE-#528 answer, computed from the vocabulary fallback
598 # instead of the parse (measured 2026-09-14:
599 # `WidensOnly("john e smith").initials()` is "j. s.", where a
600 # forwarding override and the library itself give "j. e. s.").
601 # Real text was chosen over the empty placeholder precisely so
602 # an override that ignores the keyword behaves as it did
603 # before the upgrade, rather than going quietly blank.
604 #
605 # An overridden PUBLIC `first_list`/`middle_list`/`last_list`
606 # property lands here the same way: _initials_lists (above)
607 # detects the override and calls this method for that member
608 # with no `tokens=` at all, so it takes the STRING path below
609 # regardless of what `tokens=` a WidensOnly-style override
610 # might otherwise forward -- the override supplies strings,
611 # not tokens, so there is nothing to forward. Same degradation,
612 # same pre-#528 vocabulary answer, for the same reason: no
613 # tokens exist to read a parse's tag from.
614 #
615 # Particles are NOT decided per token: _is_particle stays a
616 # live vocabulary lookup, as _render._cap_word keeps it --
617 # rules.md#R4 draws this boundary per question, not per field.
618 self._resolve()
619 if tokens is None:
620 # STRING PATH. split() rather than split(" ") because
621 # split(" ") yields '' between repeated spaces and
622 # `word[0]` below would raise IndexError on it (#232). v1
623 # stated the reason as `*_list` attributes bypassing
624 # whitespace normalization, which no longer holds -- the
625 # `*_list` properties are read-only in 2.x, and assignment
626 # through `hn.middle = ...` normalizes -- but a doubled
627 # space in a bare string handed directly to this method
628 # (`_process_initial(name_part, ...)` with no `tokens=`)
629 # still reaches here.
630 words: tuple[str, ...] = tuple(name_part.split())
631 # No parse read this text, so every word takes the same
632 # fallback _token_is_conjunction takes for a spliced one.
633 conjunctions: tuple[bool, ...] = tuple(
634 _render._reads_as_conjunction(word, self._lexicon)
635 for word in words)
636 else:
637 words = tuple(tok.text for tok in tokens)
638 conjunctions = tuple(self._token_is_conjunction(tok)
639 for tok in tokens)
640 initials = []
641 for word, conjunction in zip(words, conjunctions, strict=True):
642 # #461: the conjunction filter reaches EVERY group; only
643 # the particle filter is exempted for the given group.
644 if not conjunction and (firstname or not self._is_particle(word)):
645 initials.append(word[0])
646 if len(initials) > 0:
647 return self.initials_separator.join(initials)
648 # Return '' (never empty_attribute_default, which may be None)
649 # when a part has no initialable words. group_initials below
650 # decides what that means: one such element among others is
651 # dropped (`Alex van Johnson`'s `van`); a group that yields
652 # nothing AND is wholly particles initials its words; and a
653 # group that yields nothing for any other reason is still
654 # dropped.
655 #
656 # No group of a PARSED name reaches that third case any more,
657 # and #461 is why: any word that is neither a particle nor a
658 # connective initials, so a group reaching it is all particles
659 # and connectives; not being wholly particles it holds a
660 # connective; and that connective's part holds nothing but
661 # particles and connectives for it to join, so the mark
662 # readmits it and the group yields it. Measured 2026-09-20,
663 # zero such groups over 95,119 names -- every corpus and case
664 # text plus the review's generated grid -- where "Vega, Santa
665 # de y" was the example until #461 and now initials 'S. y. V.'.
666 #
667 # What still reaches it is the paths with no parse to read,
668 # where a connective is answered from the vocabulary and
669 # carries no mark: HumanName(first="Santa", middle="de y",
670 # last="Vega").initials() gives 'S. V.', the pre-#461 answer,
671 # and so does a subclass overriding middle_list with the same
672 # words (tests/v2/test_facade.py pins both).
673 return ""
675 def _initials_lists(self) -> tuple[list[str], list[str], list[str]]:
676 """Initials for the first, middle and last name groups. Parts
677 that yield no initials are dropped rather than kept as empty
678 strings -- except a part that is wholly PARTICLES, whose words
679 initial as ordinary name words since #404, so the prefix-only
680 middle name "de la" is no longer an example of the dropping.
682 Each group is walked as TOKENS rather than as the strings of
683 the `*_list` view (#528), so every word carries the reading the
684 parse gave it; the elements are the list view's own, folded
685 first and continuations merged, because one walk builds both.
686 A member whose PUBLIC `first_list`/`middle_list`/`last_list`
687 property is overridden is the one exception: `_list_tokens_for`
688 (the private token walk this method otherwise uses) does not
689 consult that override, so honoring it means reading the
690 override's own strings instead -- the pre-#528 walk, degraded
691 to the vocabulary fallback exactly as a widen-only
692 `_process_initial` override is (STATED BREAK, below): the
693 override supplies strings, not tokens, so there is nothing to
694 forward even if it accepted `tokens=`.
695 """
696 def all_particle_guard(got: list[str], words: list[str]) -> list[str]:
697 if got or not words or not all(self._is_particle(w)
698 for w in words):
699 return got
700 # rules.md#R3: "except the particles of a part whose every
701 # word is one, which are not acting as particles there"
702 # -- nothing survived
703 # the filter, so the whole group is particles. The
704 # facade's twin of the core's
705 # UNJOINED_TAG. NOT pinned against it by the case runners,
706 # which compare the seven role fields only (Case carries
707 # no initials column); the covering test is
708 # tests/test_initials.py::test_initials_middle_name_all_prefixes,
709 # and since #484 the differential compares initials() on
710 # both surfaces for names whose roles agree. _split_last
711 # already applies the same guard to the base, which is why
712 # last_base was never empty here.
713 return [w[0] for w in words]
715 def group_initials(groups: list[tuple[Token, ...]],
716 firstname: bool = False) -> list[str]:
717 got = [i for i in (
718 self._process_initial(
719 " ".join(tok.text for tok in group),
720 firstname=firstname, tokens=group)
721 for group in groups) if i]
722 words = [tok.text for group in groups for tok in group]
723 return all_particle_guard(got, words)
725 def group_initials_from_list(names: list[str],
726 firstname: bool = False) -> list[str]:
727 # PRE-#528 walk (`git show 338daf7:nameparser/_facade.py`),
728 # reproduced exactly: no tokens exist to walk here, only
729 # the override's own strings, so each element goes through
730 # `_process_initial` with no `tokens=` -- the STRING path,
731 # answered from the vocabulary rather than the parse.
732 got = [i for i in (self._process_initial(n, firstname=firstname)
733 for n in names if n) if i]
734 words = [w for n in names if n for w in n.split()]
735 return all_particle_guard(got, words)
737 def initials_for(member: str, firstname: bool) -> list[str]:
738 # One class-attribute identity check per member (cheap: no
739 # parse involved) decides which walk honors that member's
740 # public property.
741 overridden = (getattr(type(self), f"{member}_list")
742 is not getattr(HumanName, f"{member}_list"))
743 if overridden:
744 return group_initials_from_list(
745 getattr(self, f"{member}_list"), firstname)
746 return group_initials(self._list_tokens_for(member), firstname)
748 return (initials_for("first", True),
749 initials_for("middle", False),
750 initials_for("last", False))
752 def initials_list(self) -> list[str]:
753 first, middle, last = self._initials_lists()
754 return first + middle + last
756 def initials(self) -> str:
757 first, middle, last = self._initials_lists()
758 joiner = self.initials_delimiter + self.initials_separator
760 def group(items: list[str]) -> str:
761 return joiner.join(items) + self.initials_delimiter \
762 if items else ""
764 # A fully-empty result renders as "" -- the v1 fallback to
765 # C.empty_attribute_default (which may be None) is dropped per
766 # #255.
767 _s = self.initials_format.format(
768 first=group(first), middle=group(middle), last=group(last))
769 return self.collapse_whitespace(_s)
771 # -- comparison -----------------------------------------------------------
773 def matches(self, other: str | HumanName) -> bool:
774 """Component-wise case-insensitive comparison (v1 parity); a
775 str argument is parsed with this instance's resolved parser."""
776 if not isinstance(other, (str, HumanName)):
777 # pre-check so the error names the facade type a caller
778 # actually passed HumanName.matches(), not the core
779 # ParsedName it delegates to below
780 raise TypeError(
781 f"matches() takes a str or HumanName, got {other!r}")
782 target = other._parsed if isinstance(other, HumanName) else other
783 return self._parsed.matches(target, parser=self._resolve())
785 def comparison_key(self) -> tuple[str, ...]:
786 """One casefolded component per field in canonical order -- the
787 v1 replacement for ==/hash (#223); see ParsedName.comparison_key."""
788 return self._parsed.comparison_key()
790 # -- dunders ------------------------------------------------------------
792 def collapse_whitespace(self, string: str) -> str:
793 # v1 parser.py:976 verbatim, over _render's regexes (the #254
794 # collapse owns them; this public method keeps v1's narrower
795 # two-step contract for initials() and direct callers)
796 string = _render._SPACES.sub(" ", string.strip())
797 if string and _render._COMMA_CHAR.fullmatch(string[-1]):
798 string = string[:-1]
799 return string
801 def __str__(self) -> str:
802 if self.string_format is not None:
803 rendered = self.string_format.format(
804 **{k: v or "" for k, v in self.as_dict().items()})
805 # the full #254 collapse is _render._collapse -- one owner
806 # for the cleanup chain the v1 __str__ spelled inline
807 return _render._collapse(rendered)
808 return " ".join(self)
810 def __repr__(self) -> str:
811 attrs = (
812 f" title: {self.title or ''!r}\n"
813 f" first: {self.first or ''!r}\n"
814 f" middle: {self.middle or ''!r}\n"
815 f" last: {self.last or ''!r}\n"
816 f" suffix: {self.suffix or ''!r}\n"
817 f" nickname: {self.nickname or ''!r}\n"
818 f" maiden: {self.maiden or ''!r}"
819 )
820 return f"<{self.__class__.__name__} : [\n{attrs}\n]>"
822 def __iter__(self) -> Iterator[str]:
823 return (value for member in _MEMBERS
824 if (value := getattr(self, member)))
826 def __len__(self) -> int:
827 return sum(1 for member in _MEMBERS if getattr(self, member))
829 def __getitem__(self, key: str) -> str:
830 if isinstance(key, slice):
831 raise TypeError(
832 "slicing a HumanName was removed in 2.0 (#258); access "
833 "the named attributes instead"
834 )
835 # Role is a StrEnum, so Role members (and the plain 'given'/
836 # 'family' strings) reach here too -- translate to the v1
837 # spelling the facade actually exposes as attributes.
838 return getattr(self, _V1_SPELLING.get(key, key))
840 def __setattr__(self, name: str, value: object) -> None:
841 # "given"/"family" are the 2.0 spellings of first/last; the
842 # facade has no such attributes, so plain assignment creates a
843 # stray instance attribute while the parse (and .first/.last)
844 # keeps the old value -- a silently forked name. Warn but
845 # still set: ad-hoc attribute stashing is a legal v1 pattern,
846 # so any code that worked keeps working. Only these two names
847 # warn -- the other five 2.0 field names are real properties
848 # whose setters work, and Role members reach here as their
849 # string values (StrEnum).
850 if name in _V1_SPELLING:
851 warnings.warn(
852 f"assigning HumanName.{name} creates an inert attribute; "
853 f"the parse is unchanged -- use .{_V1_SPELLING[name]} "
854 f"(the v1 spelling) to update the name",
855 UserWarning, stacklevel=2)
856 super().__setattr__(name, value)
858 def as_dict(self, include_empty: bool = True) -> dict[str, str]:
859 """The seven v1-named components as a dict; include_empty=False
860 drops empty fields."""
861 d = {member: getattr(self, member) for member in _MEMBERS}
862 if include_empty:
863 return d
864 return {k: v for k, v in d.items() if v}
866 # -- pickle (v1-shaped state; one path for 1.4 and 2.x blobs) -----------
868 def __getstate__(self) -> dict[str, Any]:
869 # The emitted key set matches v1.4's pickle shape (minus
870 # encoding/_had_comma/_derived_*, which are v1-internal and
871 # ignored on read), so one __setstate__ path serves both eras.
872 state: dict[str, Any] = {
873 "_full_name": self._full_name,
874 "original": self.original,
875 "C": None if self._C is CONSTANTS else self._C,
876 "string_format": self.string_format,
877 "initials_format": self.initials_format,
878 "initials_delimiter": self.initials_delimiter,
879 "initials_separator": self.initials_separator,
880 "suffix_delimiter": self.suffix_delimiter,
881 }
882 for member in _MEMBERS:
883 state[f"{member}_list"] = getattr(self, f"{member}_list")
884 return state
886 def __setstate__(self, state: dict[str, Any]) -> None:
887 c = state.get("C")
888 self._C = CONSTANTS if c is None else c
889 self._snapshot_gen = -1
890 defaults = self._C._snapshot()[2]
891 self._string_format = state.get("string_format",
892 defaults.string_format)
893 self._initials_format = state.get("initials_format",
894 defaults.initials_format)
895 self._initials_delimiter = state.get("initials_delimiter",
896 defaults.initials_delimiter)
897 self._initials_separator = state.get("initials_separator",
898 defaults.initials_separator)
899 self._suffix_delimiter = state.get("suffix_delimiter",
900 defaults.suffix_delimiter)
901 self._full_name = state.get("_full_name", "")
902 # Components come back exactly as pickled
903 # (mechanisms.md#FACADE-CONTRACT): synthetic
904 # tokens, never a re-parse. Build them per *_list ENTRY rather
905 # than from one joined string -- an entry may hold several words
906 # ("Ph. D.", "Q.C. M.P."), and re-splitting the joined string on
907 # whitespace would promote each word to its own entry, which the
908 # suffix view then renders comma-separated ("Ph., D."). Marking
909 # continuation words "joined" is the inverse of _list_tokens_for's
910 # heal, so list -> pickle -> list is the identity v1 gave us.
911 tokens: list[Token] = []
912 for member in _MEMBERS:
913 role = Role(_V2_FIELD.get(member, member))
914 entries = state.get(f"{member}_list") or []
915 # Everything here is iterable but shreds differently: a str
916 # yields characters ("John" -> first "J o h n"), a Mapping
917 # yields keys only, bytes yields ints. v1 stored lists, so
918 # this only guards foreign or hand-built state -- but it
919 # names the field at the load site instead of failing
920 # opaquely later, or not at all.
921 if isinstance(entries, (str, bytes, Mapping)):
922 raise TypeError(
923 f"{member}_list must be a list of strings, not "
924 f"{type(entries).__name__} ({entries!r}); this "
925 f"pickle was not written by nameparser"
926 )
927 for entry in entries:
928 if not isinstance(entry, str):
929 raise TypeError(
930 f"{member}_list entries must be strings, got "
931 f"{entry!r}; this pickle was not written by "
932 f"nameparser"
933 )
934 for position, word in enumerate(entry.split()):
935 # UNCLASSIFIED_TAG for the same reason replace()
936 # stamps it: a pickle carries the *_list STRINGS
937 # and no tags, so nothing here was read by a parse
938 # and case repair must ask the vocabulary rather
939 # than read an absent conjunction tag. Without it a
940 # restored "juan ortega y gasset" repairs to
941 # "Ortega Y Gasset", which is neither v1's answer
942 # nor the same name's unpickled one.
943 tags = {UNCLASSIFIED_TAG}
944 if position:
945 tags.add("joined")
946 tokens.append(Token(word, None, role, frozenset(tags)))
947 self._parsed = ParsedName(
948 original=str(state.get("original", "")), tokens=tuple(tokens))