Coverage for /pythoncovmergedfiles/medio/medio/usr/local/lib/python3.11/site-packages/nameparser/_facade.py: 48%

Shortcuts on this page

r m x   toggle line displays

j k   next/prev highlighted chunk

0   (zero) top of page

1   (one) first highlighted chunk

385 statements  

1"""The 2.0 ``HumanName`` facade (mechanisms.md#FACADE-CONTRACT): a 

2mutable wrapper 

3over a frozen ParsedName, delegating parsing to the core Parser resolved 

4from the bound Constants shim. Keeps every v1 spelling. Deleted in 3.0. 

5 

6Layering: facade layer -- may import anything public plus _render. 

7""" 

8from __future__ import annotations 

9 

10import dataclasses 

11import warnings 

12from collections.abc import Iterator, Mapping 

13from typing import Any 

14 

15# Import order matters here -- breaks a real import cycle. nameparser. 

16# config's package __init__ re-exports CONSTANTS/Constants/etc. from 

17# _config_shim (the v1 nameparser.config.Constants compat path), while 

18# _config_shim's own default CONSTANTS singleton needs nameparser.config's 

19# DATA submodules (titles, prefixes, ...), imported lazily -- see 

20# _config_shim.py's module docstring. If _config_shim were the first of 

21# the two ever touched, building its CONSTANTS would need to import 

22# nameparser.config, whose __init__ would in turn need _config_shim's 

23# (not-yet-built) CONSTANTS: ImportError. Importing the config package 

24# here first lets its __init__ run to completion; when IT then imports 

25# _config_shim to build the default CONSTANTS, nameparser.config is 

26# already registered in sys.modules, so its data-submodule imports 

27# resolve directly instead of re-entering (and failing on) its own 

28# still-executing __init__. 

29import nameparser.config # noqa: F401 

30 

31import nameparser._render as _render 

32from nameparser._config_shim import CONSTANTS, Constants, _cached_parser 

33from nameparser._lexicon import _normalize 

34from nameparser._parser import Parser 

35from nameparser._types import (FOLDED_TAG, UNCLASSIFIED_TAG, 

36 UNJOINED_CONJUNCTION_TAG, ParsedName, 

37 Role, Token) 

38 

39_V2_FIELD = {"first": "given", "last": "family"} # v1 name -> v2 name 

40_V1_SPELLING = {v2: v1 for v1, v2 in _V2_FIELD.items()} 

41# derived from Role: declaration order IS the canonical field order 

42# (never restated), rendered in the v1 spellings 

43_MEMBERS = tuple(_V1_SPELLING.get(r.value, r.value) for r in Role) 

44 

45 

46 

47#: v1 parsing hooks the facade never calls 

48#: (mechanisms.md#FACADE-CONTRACT / #280). 

49_V1_HOOKS = ( 

50 "pre_process", "post_process", "parse_full_name", "parse_pieces", 

51 "parse_nicknames", "join_on_conjunctions", "squash_emoji", 

52 "handle_firstnames", "handle_middle_name_as_last", 

53 "is_title", "is_conjunction", "is_prefix", "is_roman_numeral", 

54 "is_suffix", "is_suffix_lenient", "is_an_initial", 

55) 

56# Module-level mutable state (sanctioned exception, AGENTS.md "One 

57# sanctioned global"): strong-references every distinct HumanName 

58# subclass for process lifetime so the hook warning fires once per 

59# class. Fine in practice -- subclasses are statically defined, and the 

60# whole module is deleted with the facade layer in 3.0. 

61_WARNED_SUBCLASSES: set[type] = set() 

62 

63 

64def _empty_parsed() -> ParsedName: 

65 return ParsedName(original="", tokens=(), ambiguities=()) 

66 

67 

68class HumanName: 

69 """v1 ``HumanName`` facade: a mutable wrapper over a frozen 

70 ``ParsedName``, delegating all parsing to the core ``Parser`` 

71 resolved from the bound ``Constants`` shim (dirty-tracked via its 

72 ``_generation``). Keeps every v1 spelling (``first``/``last`` over 

73 the core ``given``/``family``); the v1 parsing hooks are never 

74 called (#280). Deleted with the facade layer in 3.0. 

75 """ 

76 

77 def __init__( 

78 self, 

79 full_name: str = "", 

80 constants: Constants | None = CONSTANTS, 

81 string_format: str | None = None, 

82 initials_format: str | None = None, 

83 initials_delimiter: str | None = None, 

84 initials_separator: str | None = None, 

85 suffix_delimiter: str | None = None, 

86 first: str | list[str] | None = None, 

87 middle: str | list[str] | None = None, 

88 last: str | list[str] | None = None, 

89 title: str | list[str] | None = None, 

90 suffix: str | list[str] | None = None, 

91 nickname: str | list[str] | None = None, 

92 maiden: str | list[str] | None = None, 

93 ) -> None: 

94 if constants is None: 

95 raise TypeError( 

96 "constants=None was removed in 2.0 (#261): pass a " 

97 "Constants instance, or use the new Parser/Lexicon/" 

98 "Policy API for per-call configuration" 

99 ) 

100 if not isinstance(constants, Constants): 

101 raise TypeError( 

102 f"constants must be a Constants instance, got {constants!r}" 

103 ) 

104 self._warn_overridden_hooks() 

105 self._C = constants 

106 self._snapshot_gen = -1 # forces first resolve 

107 _, _, defaults = constants._snapshot() 

108 self.string_format = (string_format if string_format is not None 

109 else defaults.string_format) 

110 self.initials_format = (initials_format if initials_format is not None 

111 else defaults.initials_format) 

112 self.initials_delimiter = ( 

113 initials_delimiter if initials_delimiter is not None 

114 else defaults.initials_delimiter) 

115 self.initials_separator = ( 

116 initials_separator if initials_separator is not None 

117 else defaults.initials_separator) 

118 # These five assignments route through the validating properties 

119 # below. The suffix_delimiter setter resets _snapshot_gen to -1 

120 # on every assignment (including this one, harmlessly -- it's 

121 # already -1 above), so reassigning it post-construction (e.g. 

122 # n.suffix_delimiter = " - ") correctly forces the next 

123 # _resolve() to rebuild the Policy with the new delimiter. 

124 self.suffix_delimiter = (suffix_delimiter if suffix_delimiter is not None 

125 else defaults.suffix_delimiter) 

126 self._full_name = "" 

127 self._parsed = _empty_parsed() 

128 if first or middle or last or title or suffix or nickname or maiden: 

129 # These route through the field-setter properties (None 

130 # clears the field); no full-string parse, full_name stays "". 

131 self.first = first 

132 self.middle = middle 

133 self.last = last 

134 self.title = title 

135 self.suffix = suffix 

136 self.nickname = nickname 

137 self.maiden = maiden 

138 else: 

139 self._apply_full_name(full_name) 

140 

141 @classmethod 

142 def _warn_overridden_hooks(cls) -> None: 

143 if cls is HumanName or cls in _WARNED_SUBCLASSES: 

144 return 

145 overridden = [h for h in _V1_HOOKS 

146 if getattr(cls, h, None) is not getattr( 

147 HumanName, h, None)] 

148 _WARNED_SUBCLASSES.add(cls) 

149 if overridden: 

150 warnings.warn( 

151 f"{cls.__name__} overrides v1 parsing hooks " 

152 f"({', '.join(overridden)}) that the 2.0 facade never " 

153 f"calls; parsing is delegated to the core Parser. " 

154 f"Migrate to the Lexicon/Policy API. See " 

155 f"https://github.com/derek73/python-nameparser/issues/280", 

156 DeprecationWarning, stacklevel=3) 

157 

158 # -- render defaults ----------------------------------------------------- 

159 # One-line validating setters (mechanisms.md#FACADE-CONTRACT): 

160 # assigning a non-str (or, for 

161 # the two fields that allow it, non-str-non-None) raises TypeError at 

162 # assignment time instead of failing later inside .format(). 

163 

164 @property 

165 def string_format(self) -> str | None: 

166 return self._string_format 

167 

168 @string_format.setter 

169 def string_format(self, value: str | None) -> None: 

170 if value is not None and not isinstance(value, str): 

171 raise TypeError( 

172 f"string_format must be a str or None, got {value!r}") 

173 self._string_format = value 

174 

175 @property 

176 def initials_format(self) -> str: 

177 return self._initials_format 

178 

179 @initials_format.setter 

180 def initials_format(self, value: str) -> None: 

181 if not isinstance(value, str): 

182 raise TypeError( 

183 f"initials_format must be a str, got {value!r}") 

184 self._initials_format = value 

185 

186 @property 

187 def initials_delimiter(self) -> str: 

188 return self._initials_delimiter 

189 

190 @initials_delimiter.setter 

191 def initials_delimiter(self, value: str) -> None: 

192 if not isinstance(value, str): 

193 raise TypeError( 

194 f"initials_delimiter must be a str, got {value!r}") 

195 self._initials_delimiter = value 

196 

197 @property 

198 def initials_separator(self) -> str: 

199 return self._initials_separator 

200 

201 @initials_separator.setter 

202 def initials_separator(self, value: str) -> None: 

203 if not isinstance(value, str): 

204 raise TypeError( 

205 f"initials_separator must be a str, got {value!r}") 

206 self._initials_separator = value 

207 

208 @property 

209 def suffix_delimiter(self) -> str | None: 

210 return self._suffix_delimiter 

211 

212 @suffix_delimiter.setter 

213 def suffix_delimiter(self, value: str | None) -> None: 

214 if value is not None and not isinstance(value, str): 

215 raise TypeError( 

216 f"suffix_delimiter must be a str or None, got {value!r}") 

217 self._suffix_delimiter = value 

218 # Invalidate the cached Policy: _resolve() layers suffix_delimiter 

219 # onto extra_suffix_delimiters, so a stale snapshot would keep 

220 # parsing against the old delimiter. 

221 self._snapshot_gen = -1 

222 

223 # -- config / parsing --------------------------------------------------- 

224 

225 def _resolve(self) -> Parser: 

226 """Dirty-tracked parser resolution 

227 (mechanisms.md#CONFIG-SHIM-SNAPSHOT): rebuild the 

228 snapshot only when the bound Constants' generation moved.""" 

229 gen = self._C._generation 

230 if self._snapshot_gen != gen: 

231 lexicon, policy, _ = self._C._snapshot() 

232 if self.suffix_delimiter: 

233 policy = dataclasses.replace( 

234 policy, 

235 extra_suffix_delimiters=frozenset( 

236 {self.suffix_delimiter})) 

237 self._lexicon, self._policy = lexicon, policy 

238 self._parser = _cached_parser(lexicon, policy) 

239 self._snapshot_gen = gen 

240 # the fast path is a plain attribute return: hashing the two 

241 # value objects for the lru lookup is the whole fast-path cost 

242 return self._parser 

243 

244 def parse_full_name(self) -> None: 

245 """Re-parse the stored ``full_name`` (v1's documented re-parse 

246 trigger, docs/customize.rst): mutate ``name.C`` then call this to 

247 force a re-parse without reassigning ``full_name``. The v1 

248 parsing INTERNALS this name evokes live in the core ``Parser``, 

249 not here; a subclass overriding this method still triggers the 

250 #280 hook-override warning, and full_name assignment never 

251 consults it.""" 

252 self._apply_full_name(self._full_name) 

253 

254 def _apply_full_name(self, value: str) -> None: 

255 if isinstance(value, bytes): 

256 raise TypeError( 

257 "bytes input was removed in 2.0 (#245): decode first, " 

258 "e.g. HumanName(raw.decode('utf-8'))" 

259 ) 

260 if not isinstance(value, str): 

261 raise TypeError(f"full_name must be a str, got {value!r}") 

262 # parse FIRST: if snapshot resolution raises, the instance must 

263 # not be left with a new full_name over the old parsed fields 

264 parsed = self._resolve().parse(value) 

265 self._full_name = value 

266 self._parsed = parsed 

267 if self._C.capitalize_name: 

268 self.capitalize() # v1 parser.py:1653 parity 

269 

270 def capitalize(self, force: bool | None = None) -> None: 

271 """Re-capitalize the current parse against the bound lexicon. 

272 force=None reads the bound Constants' render default 

273 (force_mixed_case_capitalization); the core's capitalized() 

274 implements the single-case gate (v1 parity) -- not 

275 re-implemented here.""" 

276 self._resolve() 

277 if force is None: 

278 force = self._C.force_mixed_case_capitalization 

279 self._parsed = self._parsed.capitalized(self._lexicon, force=force) 

280 

281 @property 

282 def full_name(self) -> str: 

283 return self._full_name 

284 

285 @full_name.setter 

286 def full_name(self, value: str) -> None: 

287 self._apply_full_name(value) 

288 

289 @property 

290 def original(self) -> str: 

291 return self._parsed.original or self._full_name 

292 

293 @property 

294 def C(self) -> Constants: 

295 return self._C 

296 

297 @C.setter 

298 def C(self, constants: Constants | None) -> None: 

299 # v1.4 closed #239 by making C a validating setter that ONLY 

300 # stores the new value -- no re-parse (checked against v1.4: 

301 # `git show 2d5d8c2:nameparser/parser.py` lines ~204-206, the C 

302 # setter body is exactly `self._C = self._validate_constants(...)`). 

303 # A caller who wants the new config reflected must still trigger 

304 # a re-parse, e.g. via parse_full_name() or a full_name 

305 # reassignment -- matched here rather than re-parsing eagerly. 

306 if constants is None: 

307 raise TypeError( 

308 "assigning constants=None to C was removed in 2.0 (#261): " 

309 "pass a Constants instance, or use the new Parser/Lexicon/" 

310 "Policy API for per-call configuration" 

311 ) 

312 if not isinstance(constants, Constants): 

313 raise TypeError( 

314 f"constants must be a Constants instance, got {constants!r}" 

315 ) 

316 self._C = constants 

317 self._snapshot_gen = -1 # invalidate: next _resolve() rebuilds 

318 

319 @property 

320 def has_own_config(self) -> bool: 

321 """True when this instance is not using the shared module-level 

322 CONSTANTS.""" 

323 return self._C is not CONSTANTS 

324 

325 # -- fields --------------------------------------------------------- 

326 

327 def _get_field(self, member: str) -> str: 

328 return getattr(self._parsed, _V2_FIELD.get(member, member)) 

329 

330 def _set_field(self, member: str, value: str | list[str] | None) -> None: 

331 if value is None: 

332 joined = "" 

333 elif isinstance(value, list): 

334 for element in value: 

335 if not isinstance(element, str): 

336 raise TypeError( 

337 f"name parts must be strings, got {element!r}") 

338 joined = " ".join(value) 

339 elif isinstance(value, str): 

340 joined = value 

341 else: 

342 raise TypeError( 

343 f"{member} must be a str, list, or None, got {value!r}") 

344 # v1 setters stay on replace(): revise()'s vocabulary tags would 

345 # change v1 parity 

346 self._parsed = self._parsed.replace( 

347 **{_V2_FIELD.get(member, member): joined}) 

348 

349 def _list_tokens_for(self, member: str) -> list[tuple[Token, ...]]: 

350 # A "joined" continuation token ("Ph." + "D.") belongs to its 

351 # predecessor's part, matching v1's fix_phd (suffix_list had ONE 

352 # "Ph. D." element). ParsedName._text_for heals only the suffix 

353 # string view (the ", " join); the facade list view heals for 

354 # every role -- a continuation is never its own list element. 

355 role = Role(_V2_FIELD.get(member, member)) 

356 parts: list[list[Token]] = [] 

357 folded: list[list[Token]] = [] 

358 for tok in self._parsed.tokens_for(role): 

359 if "joined" in tok.tags and parts: 

360 parts[-1].append(tok) 

361 elif FOLDED_TAG in tok.tags: 

362 # middle_as_family fold: v1 PREPENDED middle_list to 

363 # last_list -- keep the list view consistent with the 

364 # string view (_text_for orders folded-first too) 

365 folded.append([tok]) 

366 else: 

367 parts.append([tok]) 

368 return [tuple(group) for group in folded + parts] 

369 

370 def _list_for(self, member: str) -> list[str]: 

371 # The STRING view of the walk above, which is the v1 `*_list` 

372 # shape. Two views off one walk rather than two walks: the 

373 # initials view needs each element's backing tokens (#528) and 

374 # every other reader needs its text, and a second walk could 

375 # drift from this one on exactly the element boundaries the 

376 # comment above exists to hold. Not on the parse path -- 

377 # measured 2026-09-13, HumanName(name) never reaches it -- so 

378 # the tuple building is off the benchmarked budget 

379 # (tests/v2/test_benchmark.py budgets parse() and HumanName()). 

380 return [" ".join(tok.text for tok in group) 

381 for group in self._list_tokens_for(member)] 

382 

383 @property 

384 def title(self) -> str: 

385 return self._get_field("title") 

386 

387 @title.setter 

388 def title(self, value: str | list[str] | None) -> None: 

389 self._set_field("title", value) 

390 

391 @property 

392 def title_list(self) -> list[str]: 

393 return self._list_for("title") 

394 

395 @property 

396 def first(self) -> str: 

397 return self._get_field("first") 

398 

399 @first.setter 

400 def first(self, value: str | list[str] | None) -> None: 

401 self._set_field("first", value) 

402 

403 @property 

404 def first_list(self) -> list[str]: 

405 return self._list_for("first") 

406 

407 @property 

408 def middle(self) -> str: 

409 return self._get_field("middle") 

410 

411 @middle.setter 

412 def middle(self, value: str | list[str] | None) -> None: 

413 self._set_field("middle", value) 

414 

415 @property 

416 def middle_list(self) -> list[str]: 

417 return self._list_for("middle") 

418 

419 @property 

420 def last(self) -> str: 

421 return self._get_field("last") 

422 

423 @last.setter 

424 def last(self, value: str | list[str] | None) -> None: 

425 self._set_field("last", value) 

426 

427 @property 

428 def last_list(self) -> list[str]: 

429 return self._list_for("last") 

430 

431 @property 

432 def suffix(self) -> str: 

433 return self._get_field("suffix") 

434 

435 @suffix.setter 

436 def suffix(self, value: str | list[str] | None) -> None: 

437 self._set_field("suffix", value) 

438 

439 @property 

440 def suffix_list(self) -> list[str]: 

441 return self._list_for("suffix") 

442 

443 @property 

444 def nickname(self) -> str: 

445 return self._get_field("nickname") 

446 

447 @nickname.setter 

448 def nickname(self, value: str | list[str] | None) -> None: 

449 self._set_field("nickname", value) 

450 

451 @property 

452 def nickname_list(self) -> list[str]: 

453 return self._list_for("nickname") 

454 

455 @property 

456 def maiden(self) -> str: 

457 return self._get_field("maiden") 

458 

459 @maiden.setter 

460 def maiden(self, value: str | list[str] | None) -> None: 

461 self._set_field("maiden", value) 

462 

463 @property 

464 def maiden_list(self) -> list[str]: 

465 return self._list_for("maiden") 

466 

467 # -- derived views ---------------------------------------------------- 

468 

469 @property 

470 def surnames_list(self) -> list[str]: 

471 return self.middle_list + self.last_list 

472 

473 @property 

474 def surnames(self) -> str: 

475 return " ".join(self.surnames_list) 

476 

477 @property 

478 def given_names_list(self) -> list[str]: 

479 return self.first_list + self.middle_list 

480 

481 @property 

482 def given_names(self) -> str: 

483 return " ".join(self.given_names_list) 

484 

485 def _is_particle(self, text: str) -> bool: 

486 self._resolve() 

487 return _normalize(text) in self._lexicon.particles 

488 

489 def _token_is_conjunction(self, tok: Token) -> bool: 

490 # #528: the PARSE's answer, not the vocabulary's. A token the 

491 # parser classified carries its reading in its tags, which is 

492 # the source the core's initials() has always read 

493 # (mechanisms.md#RENDER-HONORS-THE-PARSE), so a bare capital 

494 # 'Y' in a one-case name contributes no initial where its tag 

495 # says connective, and a one-case 'e' contributes one where its 

496 # tag says initial. Before #528 this was computed from the raw 

497 # word instead -- "in the conjunctions set AND NOT 

498 # _render._INITIAL" (v1's is_conjunction, restored by #462) -- 

499 # a shape test standing in for a tag, which stopped agreeing 

500 # with the parse the moment #383/#479 gave the classifier a 

501 # fork the shape cannot see. 

502 # 

503 # UNCLASSIFIED_TAG is the one case with no parse to honor: the 

504 # words were spliced into a field as raw text, by `hn.middle = 

505 # ...` (ParsedName.replace) or restored via __setstate__ -- a 

506 # pickle load, or a copy.copy/copy.deepcopy, which go through 

507 # the same state hooks. They carry no reading, so the vocabulary 

508 # answers -- the same fallback _render._cap_word takes for the 

509 # same tokens and through the same helper, which is why the 

510 # helper is imported rather than the predicate rewritten 

511 # (decisions.md#R4 for why one question and not two: whether a 

512 # word is a connective is a fact the word can answer alone, 

513 # whether a particle is acting as one is a fact about the part). 

514 # 

515 # _resolve() first, as _is_particle above does: an unpickled or 

516 # copied instance has no _lexicon until resolved -- and neither 

517 # does a keyword-constructed one (`HumanName(first=..., middle= 

518 # "y", last=...)`), which never runs the full-string parse path 

519 # other callers rely on to have called _resolve() already. 

520 # LOAD-BEARING for exactly those caller shapes: do not delete 

521 # this call as dead just because most callers arrive resolved. 

522 self._resolve() 

523 if UNCLASSIFIED_TAG in tok.tags: 

524 return _render._reads_as_conjunction(tok.text, self._lexicon) 

525 # #461: a connective with nothing in its part to join is not 

526 # acting as one. The parse decided that and marked the token, 

527 # and this reads the same mark the core's initials() reads, so 

528 # the two views cannot disagree about it. 

529 return ("conjunction" in tok.tags 

530 and UNJOINED_CONJUNCTION_TAG not in tok.tags) 

531 

532 def _split_last(self) -> tuple[list[str], list[str]]: 

533 # rules.md#R2: "a name part whose every word is particle 

534 # vocabulary is a part where none of them is doing a 

535 # particle's work" -- the all-particle guard 

536 # below is this rule, and predates its statement: v1 assumed a 

537 # family name does not consist entirely of particles, e.g. the 

538 # surname "Do" which also appears in PARTICLES. v1 

539 # parser.py _split_last otherwise verbatim, vocabulary lookup 

540 # at ACCESS time so assigned last names split too. 

541 words = " ".join(self.last_list).split() 

542 i = 0 

543 while i < len(words) and self._is_particle(words[i]): 

544 i += 1 

545 if i == len(words): 

546 return [], words 

547 return words[:i], words[i:] 

548 

549 @property 

550 def last_prefixes_list(self) -> list[str]: 

551 return self._split_last()[0] 

552 

553 @property 

554 def last_prefixes(self) -> str: 

555 return " ".join(self._split_last()[0]) 

556 

557 @property 

558 def last_base_list(self) -> list[str]: 

559 return self._split_last()[1] 

560 

561 @property 

562 def last_base(self) -> str: 

563 return " ".join(self._split_last()[1]) 

564 

565 # -- initials ------------------------------------------------------------- 

566 

567 def _process_initial(self, name_part: str, 

568 firstname: bool = False, 

569 tokens: tuple[Token, ...] | None = None) -> str: 

570 # after v1 parser.py:427, not verbatim: particles and 

571 # conjunctions are filtered from initials unless the part is a 

572 # first name. 

573 # 

574 # TWO WAYS IN. `tokens` is the part's backing tokens, which 

575 # _initials_lists always has and passes; `name_part` is v1's 

576 # signature, kept because subclasses and tests call this 

577 # directly with a string (tests/test_initials.py), and there 

578 # the words come from splitting it. `tokens` supersedes 

579 # `name_part` entirely when given -- the words are the tokens' 

580 # own text rather than a re-split of the joined element, so a 

581 # two-word element and its tokens cannot fall out of step. 

582 # STATED BREAK: _initials_lists always calls with `tokens=`, so 

583 # a subclass overriding with v1's two-argument signature 

584 # (name_part, firstname=False) now raises TypeError the first 

585 # time initials() runs, rather than being silently skipped. The 

586 # alternative -- a string wrapper kept over a token core -- 

587 # would make such an override silently ineffective instead, 

588 # which hides the override rather than breaking it loudly. 

589 # Such an override must ACCEPT `tokens` AND FORWARD it: 

590 # super()._process_initial(name_part, firstname, tokens=tokens). 

591 # That is still the only way to RECEIVE #528's fix. Widening 

592 # the signature alone -- `**kwargs`, or a `tokens=None` the 

593 # super() call drops -- does not raise and does not go silent 

594 # either: `name_part` on this path is the group's own text 

595 # (Derek, 2026-09-14), so the override takes the STRING path 

596 # below and keeps working, just without the fix -- the 

597 # PRE-#528 answer, computed from the vocabulary fallback 

598 # instead of the parse (measured 2026-09-14: 

599 # `WidensOnly("john e smith").initials()` is "j. s.", where a 

600 # forwarding override and the library itself give "j. e. s."). 

601 # Real text was chosen over the empty placeholder precisely so 

602 # an override that ignores the keyword behaves as it did 

603 # before the upgrade, rather than going quietly blank. 

604 # 

605 # An overridden PUBLIC `first_list`/`middle_list`/`last_list` 

606 # property lands here the same way: _initials_lists (above) 

607 # detects the override and calls this method for that member 

608 # with no `tokens=` at all, so it takes the STRING path below 

609 # regardless of what `tokens=` a WidensOnly-style override 

610 # might otherwise forward -- the override supplies strings, 

611 # not tokens, so there is nothing to forward. Same degradation, 

612 # same pre-#528 vocabulary answer, for the same reason: no 

613 # tokens exist to read a parse's tag from. 

614 # 

615 # Particles are NOT decided per token: _is_particle stays a 

616 # live vocabulary lookup, as _render._cap_word keeps it -- 

617 # rules.md#R4 draws this boundary per question, not per field. 

618 self._resolve() 

619 if tokens is None: 

620 # STRING PATH. split() rather than split(" ") because 

621 # split(" ") yields '' between repeated spaces and 

622 # `word[0]` below would raise IndexError on it (#232). v1 

623 # stated the reason as `*_list` attributes bypassing 

624 # whitespace normalization, which no longer holds -- the 

625 # `*_list` properties are read-only in 2.x, and assignment 

626 # through `hn.middle = ...` normalizes -- but a doubled 

627 # space in a bare string handed directly to this method 

628 # (`_process_initial(name_part, ...)` with no `tokens=`) 

629 # still reaches here. 

630 words: tuple[str, ...] = tuple(name_part.split()) 

631 # No parse read this text, so every word takes the same 

632 # fallback _token_is_conjunction takes for a spliced one. 

633 conjunctions: tuple[bool, ...] = tuple( 

634 _render._reads_as_conjunction(word, self._lexicon) 

635 for word in words) 

636 else: 

637 words = tuple(tok.text for tok in tokens) 

638 conjunctions = tuple(self._token_is_conjunction(tok) 

639 for tok in tokens) 

640 initials = [] 

641 for word, conjunction in zip(words, conjunctions, strict=True): 

642 # #461: the conjunction filter reaches EVERY group; only 

643 # the particle filter is exempted for the given group. 

644 if not conjunction and (firstname or not self._is_particle(word)): 

645 initials.append(word[0]) 

646 if len(initials) > 0: 

647 return self.initials_separator.join(initials) 

648 # Return '' (never empty_attribute_default, which may be None) 

649 # when a part has no initialable words. group_initials below 

650 # decides what that means: one such element among others is 

651 # dropped (`Alex van Johnson`'s `van`); a group that yields 

652 # nothing AND is wholly particles initials its words; and a 

653 # group that yields nothing for any other reason is still 

654 # dropped. 

655 # 

656 # No group of a PARSED name reaches that third case any more, 

657 # and #461 is why: any word that is neither a particle nor a 

658 # connective initials, so a group reaching it is all particles 

659 # and connectives; not being wholly particles it holds a 

660 # connective; and that connective's part holds nothing but 

661 # particles and connectives for it to join, so the mark 

662 # readmits it and the group yields it. Measured 2026-09-20, 

663 # zero such groups over 95,119 names -- every corpus and case 

664 # text plus the review's generated grid -- where "Vega, Santa 

665 # de y" was the example until #461 and now initials 'S. y. V.'. 

666 # 

667 # What still reaches it is the paths with no parse to read, 

668 # where a connective is answered from the vocabulary and 

669 # carries no mark: HumanName(first="Santa", middle="de y", 

670 # last="Vega").initials() gives 'S. V.', the pre-#461 answer, 

671 # and so does a subclass overriding middle_list with the same 

672 # words (tests/v2/test_facade.py pins both). 

673 return "" 

674 

675 def _initials_lists(self) -> tuple[list[str], list[str], list[str]]: 

676 """Initials for the first, middle and last name groups. Parts 

677 that yield no initials are dropped rather than kept as empty 

678 strings -- except a part that is wholly PARTICLES, whose words 

679 initial as ordinary name words since #404, so the prefix-only 

680 middle name "de la" is no longer an example of the dropping. 

681 

682 Each group is walked as TOKENS rather than as the strings of 

683 the `*_list` view (#528), so every word carries the reading the 

684 parse gave it; the elements are the list view's own, folded 

685 first and continuations merged, because one walk builds both. 

686 A member whose PUBLIC `first_list`/`middle_list`/`last_list` 

687 property is overridden is the one exception: `_list_tokens_for` 

688 (the private token walk this method otherwise uses) does not 

689 consult that override, so honoring it means reading the 

690 override's own strings instead -- the pre-#528 walk, degraded 

691 to the vocabulary fallback exactly as a widen-only 

692 `_process_initial` override is (STATED BREAK, below): the 

693 override supplies strings, not tokens, so there is nothing to 

694 forward even if it accepted `tokens=`. 

695 """ 

696 def all_particle_guard(got: list[str], words: list[str]) -> list[str]: 

697 if got or not words or not all(self._is_particle(w) 

698 for w in words): 

699 return got 

700 # rules.md#R3: "except the particles of a part whose every 

701 # word is one, which are not acting as particles there" 

702 # -- nothing survived 

703 # the filter, so the whole group is particles. The 

704 # facade's twin of the core's 

705 # UNJOINED_TAG. NOT pinned against it by the case runners, 

706 # which compare the seven role fields only (Case carries 

707 # no initials column); the covering test is 

708 # tests/test_initials.py::test_initials_middle_name_all_prefixes, 

709 # and since #484 the differential compares initials() on 

710 # both surfaces for names whose roles agree. _split_last 

711 # already applies the same guard to the base, which is why 

712 # last_base was never empty here. 

713 return [w[0] for w in words] 

714 

715 def group_initials(groups: list[tuple[Token, ...]], 

716 firstname: bool = False) -> list[str]: 

717 got = [i for i in ( 

718 self._process_initial( 

719 " ".join(tok.text for tok in group), 

720 firstname=firstname, tokens=group) 

721 for group in groups) if i] 

722 words = [tok.text for group in groups for tok in group] 

723 return all_particle_guard(got, words) 

724 

725 def group_initials_from_list(names: list[str], 

726 firstname: bool = False) -> list[str]: 

727 # PRE-#528 walk (`git show 338daf7:nameparser/_facade.py`), 

728 # reproduced exactly: no tokens exist to walk here, only 

729 # the override's own strings, so each element goes through 

730 # `_process_initial` with no `tokens=` -- the STRING path, 

731 # answered from the vocabulary rather than the parse. 

732 got = [i for i in (self._process_initial(n, firstname=firstname) 

733 for n in names if n) if i] 

734 words = [w for n in names if n for w in n.split()] 

735 return all_particle_guard(got, words) 

736 

737 def initials_for(member: str, firstname: bool) -> list[str]: 

738 # One class-attribute identity check per member (cheap: no 

739 # parse involved) decides which walk honors that member's 

740 # public property. 

741 overridden = (getattr(type(self), f"{member}_list") 

742 is not getattr(HumanName, f"{member}_list")) 

743 if overridden: 

744 return group_initials_from_list( 

745 getattr(self, f"{member}_list"), firstname) 

746 return group_initials(self._list_tokens_for(member), firstname) 

747 

748 return (initials_for("first", True), 

749 initials_for("middle", False), 

750 initials_for("last", False)) 

751 

752 def initials_list(self) -> list[str]: 

753 first, middle, last = self._initials_lists() 

754 return first + middle + last 

755 

756 def initials(self) -> str: 

757 first, middle, last = self._initials_lists() 

758 joiner = self.initials_delimiter + self.initials_separator 

759 

760 def group(items: list[str]) -> str: 

761 return joiner.join(items) + self.initials_delimiter \ 

762 if items else "" 

763 

764 # A fully-empty result renders as "" -- the v1 fallback to 

765 # C.empty_attribute_default (which may be None) is dropped per 

766 # #255. 

767 _s = self.initials_format.format( 

768 first=group(first), middle=group(middle), last=group(last)) 

769 return self.collapse_whitespace(_s) 

770 

771 # -- comparison ----------------------------------------------------------- 

772 

773 def matches(self, other: str | HumanName) -> bool: 

774 """Component-wise case-insensitive comparison (v1 parity); a 

775 str argument is parsed with this instance's resolved parser.""" 

776 if not isinstance(other, (str, HumanName)): 

777 # pre-check so the error names the facade type a caller 

778 # actually passed HumanName.matches(), not the core 

779 # ParsedName it delegates to below 

780 raise TypeError( 

781 f"matches() takes a str or HumanName, got {other!r}") 

782 target = other._parsed if isinstance(other, HumanName) else other 

783 return self._parsed.matches(target, parser=self._resolve()) 

784 

785 def comparison_key(self) -> tuple[str, ...]: 

786 """One casefolded component per field in canonical order -- the 

787 v1 replacement for ==/hash (#223); see ParsedName.comparison_key.""" 

788 return self._parsed.comparison_key() 

789 

790 # -- dunders ------------------------------------------------------------ 

791 

792 def collapse_whitespace(self, string: str) -> str: 

793 # v1 parser.py:976 verbatim, over _render's regexes (the #254 

794 # collapse owns them; this public method keeps v1's narrower 

795 # two-step contract for initials() and direct callers) 

796 string = _render._SPACES.sub(" ", string.strip()) 

797 if string and _render._COMMA_CHAR.fullmatch(string[-1]): 

798 string = string[:-1] 

799 return string 

800 

801 def __str__(self) -> str: 

802 if self.string_format is not None: 

803 rendered = self.string_format.format( 

804 **{k: v or "" for k, v in self.as_dict().items()}) 

805 # the full #254 collapse is _render._collapse -- one owner 

806 # for the cleanup chain the v1 __str__ spelled inline 

807 return _render._collapse(rendered) 

808 return " ".join(self) 

809 

810 def __repr__(self) -> str: 

811 attrs = ( 

812 f" title: {self.title or ''!r}\n" 

813 f" first: {self.first or ''!r}\n" 

814 f" middle: {self.middle or ''!r}\n" 

815 f" last: {self.last or ''!r}\n" 

816 f" suffix: {self.suffix or ''!r}\n" 

817 f" nickname: {self.nickname or ''!r}\n" 

818 f" maiden: {self.maiden or ''!r}" 

819 ) 

820 return f"<{self.__class__.__name__} : [\n{attrs}\n]>" 

821 

822 def __iter__(self) -> Iterator[str]: 

823 return (value for member in _MEMBERS 

824 if (value := getattr(self, member))) 

825 

826 def __len__(self) -> int: 

827 return sum(1 for member in _MEMBERS if getattr(self, member)) 

828 

829 def __getitem__(self, key: str) -> str: 

830 if isinstance(key, slice): 

831 raise TypeError( 

832 "slicing a HumanName was removed in 2.0 (#258); access " 

833 "the named attributes instead" 

834 ) 

835 # Role is a StrEnum, so Role members (and the plain 'given'/ 

836 # 'family' strings) reach here too -- translate to the v1 

837 # spelling the facade actually exposes as attributes. 

838 return getattr(self, _V1_SPELLING.get(key, key)) 

839 

840 def __setattr__(self, name: str, value: object) -> None: 

841 # "given"/"family" are the 2.0 spellings of first/last; the 

842 # facade has no such attributes, so plain assignment creates a 

843 # stray instance attribute while the parse (and .first/.last) 

844 # keeps the old value -- a silently forked name. Warn but 

845 # still set: ad-hoc attribute stashing is a legal v1 pattern, 

846 # so any code that worked keeps working. Only these two names 

847 # warn -- the other five 2.0 field names are real properties 

848 # whose setters work, and Role members reach here as their 

849 # string values (StrEnum). 

850 if name in _V1_SPELLING: 

851 warnings.warn( 

852 f"assigning HumanName.{name} creates an inert attribute; " 

853 f"the parse is unchanged -- use .{_V1_SPELLING[name]} " 

854 f"(the v1 spelling) to update the name", 

855 UserWarning, stacklevel=2) 

856 super().__setattr__(name, value) 

857 

858 def as_dict(self, include_empty: bool = True) -> dict[str, str]: 

859 """The seven v1-named components as a dict; include_empty=False 

860 drops empty fields.""" 

861 d = {member: getattr(self, member) for member in _MEMBERS} 

862 if include_empty: 

863 return d 

864 return {k: v for k, v in d.items() if v} 

865 

866 # -- pickle (v1-shaped state; one path for 1.4 and 2.x blobs) ----------- 

867 

868 def __getstate__(self) -> dict[str, Any]: 

869 # The emitted key set matches v1.4's pickle shape (minus 

870 # encoding/_had_comma/_derived_*, which are v1-internal and 

871 # ignored on read), so one __setstate__ path serves both eras. 

872 state: dict[str, Any] = { 

873 "_full_name": self._full_name, 

874 "original": self.original, 

875 "C": None if self._C is CONSTANTS else self._C, 

876 "string_format": self.string_format, 

877 "initials_format": self.initials_format, 

878 "initials_delimiter": self.initials_delimiter, 

879 "initials_separator": self.initials_separator, 

880 "suffix_delimiter": self.suffix_delimiter, 

881 } 

882 for member in _MEMBERS: 

883 state[f"{member}_list"] = getattr(self, f"{member}_list") 

884 return state 

885 

886 def __setstate__(self, state: dict[str, Any]) -> None: 

887 c = state.get("C") 

888 self._C = CONSTANTS if c is None else c 

889 self._snapshot_gen = -1 

890 defaults = self._C._snapshot()[2] 

891 self._string_format = state.get("string_format", 

892 defaults.string_format) 

893 self._initials_format = state.get("initials_format", 

894 defaults.initials_format) 

895 self._initials_delimiter = state.get("initials_delimiter", 

896 defaults.initials_delimiter) 

897 self._initials_separator = state.get("initials_separator", 

898 defaults.initials_separator) 

899 self._suffix_delimiter = state.get("suffix_delimiter", 

900 defaults.suffix_delimiter) 

901 self._full_name = state.get("_full_name", "") 

902 # Components come back exactly as pickled 

903 # (mechanisms.md#FACADE-CONTRACT): synthetic 

904 # tokens, never a re-parse. Build them per *_list ENTRY rather 

905 # than from one joined string -- an entry may hold several words 

906 # ("Ph. D.", "Q.C. M.P."), and re-splitting the joined string on 

907 # whitespace would promote each word to its own entry, which the 

908 # suffix view then renders comma-separated ("Ph., D."). Marking 

909 # continuation words "joined" is the inverse of _list_tokens_for's 

910 # heal, so list -> pickle -> list is the identity v1 gave us. 

911 tokens: list[Token] = [] 

912 for member in _MEMBERS: 

913 role = Role(_V2_FIELD.get(member, member)) 

914 entries = state.get(f"{member}_list") or [] 

915 # Everything here is iterable but shreds differently: a str 

916 # yields characters ("John" -> first "J o h n"), a Mapping 

917 # yields keys only, bytes yields ints. v1 stored lists, so 

918 # this only guards foreign or hand-built state -- but it 

919 # names the field at the load site instead of failing 

920 # opaquely later, or not at all. 

921 if isinstance(entries, (str, bytes, Mapping)): 

922 raise TypeError( 

923 f"{member}_list must be a list of strings, not " 

924 f"{type(entries).__name__} ({entries!r}); this " 

925 f"pickle was not written by nameparser" 

926 ) 

927 for entry in entries: 

928 if not isinstance(entry, str): 

929 raise TypeError( 

930 f"{member}_list entries must be strings, got " 

931 f"{entry!r}; this pickle was not written by " 

932 f"nameparser" 

933 ) 

934 for position, word in enumerate(entry.split()): 

935 # UNCLASSIFIED_TAG for the same reason replace() 

936 # stamps it: a pickle carries the *_list STRINGS 

937 # and no tags, so nothing here was read by a parse 

938 # and case repair must ask the vocabulary rather 

939 # than read an absent conjunction tag. Without it a 

940 # restored "juan ortega y gasset" repairs to 

941 # "Ortega Y Gasset", which is neither v1's answer 

942 # nor the same name's unpickled one. 

943 tags = {UNCLASSIFIED_TAG} 

944 if position: 

945 tags.add("joined") 

946 tokens.append(Token(word, None, role, frozenset(tags))) 

947 self._parsed = ParsedName( 

948 original=str(state.get("original", "")), tokens=tuple(tokens))