1from nameparser.config._invariants import assert_normalized
2
3SUFFIX_WORDS = frozenset({
4 # #269: Cyrillic мл/ст (junior/senior, the jr/sr analogs) deferred
5 # pending the within-script collision vetting the issue asks for;
6 # 'ст' especially is a plausible false-positive risk (many two-
7 # letter Cyrillic abbreviations exist), and both are short enough
8 # to worry about. Not shipped in this pass.
9 'esq',
10 'esquire',
11 'jr',
12 'jnr',
13 'junior',
14 'sr',
15 'snr',
16 '2',
17 'i',
18 'ii',
19 'iii',
20 'iv',
21 'v',
22 # Bare, not '(ret)'/'(vet)': moved here from literal parenthesized
23 # entries in SUFFIX_ACRONYMS. parse_nicknames()'s handle_match() now
24 # strips parens/quotes before this set is consulted, so the bare form
25 # is correct -- do not re-add the parenthesized form, that would
26 # silently reintroduce the #111 bug (parenthesized "(Ret)" matching
27 # literally instead of going through nickname/suffix disambiguation).
28 'ret',
29 'vet',
30
31 # #269 follow-up: Hebrew post-nominals, both gershayim spellings
32 # (ASCII '"' and U+05F4); the mid-word quote is inert in
33 # extraction, like the ד"ר title. Neither is ever a name.
34 'ז"ל', # "of blessed memory" (deceased), ASCII quote
35 'ז״ל', # same, U+05F4 gershayim
36 'שליט"א', # honorific for a living rabbi, ASCII quote
37 'שליט״א', # same, U+05F4 gershayim
38
39 # #307/#308: CJK postnominal honorifics and degrees. Matched
40 # whole-token here, which reaches the SPACED forms; the GLUED
41 # forms (田中さん, 김민준씨) are reached by the peel #308 adds to
42 # _pipeline/_script_segment.py: it splits a listed tail off the
43 # last name token and hands the piece back to this vocabulary --
44 # so every GLUED_HONORIFICS entry below is also an entry here,
45 # asserted at the foot of this module.
46 # Self-selecting like the Korean surnames: a Han, kana or hangul
47 # entry can only ever match CJK text. Vetting per the мл/ст
48 # standard above, and note the two bars are different -- an entry
49 # can be safe in the spaced position and unsafe glued, never the
50 # other way round, because the spaced position is a token boundary
51 # its writer drew:
52 # - 氏: a rare Japanese surname reading exists, but a bare
53 # trailing 氏 after a name is the news-style honorific, and a
54 # 氏-surnamed person writes it FIRST.
55 # - 양/군: 양 is also a top-tier surname (Yang) -- but a surname
56 # LEADS, so the trailing-only suffix gate never sees it there
57 # (양 미선 keeps family 양); as single-syllable GIVEN names in
58 # final position both are vanishingly rare in practice.
59 # - 博士: an attested Japanese given name (ひろし, Hiroshi) --
60 # the doctorate reading vastly dominates the spaced trailing
61 # position this set matches.
62 # - 殿: some ninety Japanese surnames end in it (鵜殿, 真殿), which
63 # is what keeps it out of GLUED_HONORIFICS below -- but the
64 # surname-LEADS argument that clears 양/군 clears it here too.
65 # - 님: a single hangul syllable, the risk class 양/군 are in, but
66 # it has no hanja reading at all, so it cannot sit inside a
67 # Sino-Korean given name.
68 # Bare hangul 선생/교수 are deliberately absent (only the -님
69 # honorific forms ship): the bare forms read as common nouns as
70 # readily as address terms. 박사 is the exception among the three,
71 # shipping bare as well as with -님, because it names a degree
72 # rather than a role -- but all three -님 forms ship together
73 # (선생님, 교수님, 박사님 are the standard professional honorifics,
74 # and shipping two of the three was an oversight). Further
75 # candidates (여사, 太太) wait on the same case-by-case argument.
76 '씨', # ko Mr./Ms. -- standardly spaced in Korean orthography
77 '박사', # ko doctorate holder ("Dr.")
78 '박사님', # ko the same, honorific form
79 '선생님', # ko teacher/respected elder
80 '교수님', # ko professor (honorific form)
81 '군', # ko young man ("Master")
82 '양', # ko young woman ("Miss")
83 '님', # ko -- the bare honorific of online/formal address
84 '先生', # zh Mr. / ja teacher-master -- honorific in both
85 '女士', # zh Ms./Madam
86 '小姐', # zh Miss
87 '博士', # zh+ja doctorate holder, shared Han
88 '教授', # zh+ja professor, shared Han
89 '様', # ja formal Mr./Ms. (the mail-addressing honorific)
90 '氏', # ja news-style Mr. (田中氏)
91 '殿', # ja formal, official/rank-flavored
92 'さん', # ja the everyday honorific, kana
93 'さま', # ja the kana spelling of 様
94 'くん', # ja the kana spelling of 君
95 'ちゃん', # ja familiar/diminutive
96 # #346/#344/#343: spaced trailing honorifics, South Asian and
97 # Tibetan. SPACED ONLY -- none of these is in GLUED_HONORIFICS
98 # below, and none may be added to it. जी is the reason and the
99 # reason generalizes: Banerjee, Mukherjee and Chatterjee are
100 # written बनर्जी, मुखर्जी, चटर्जी, so a glued peel would strand a
101 # fragment on a bare virama (बनर् + जी) -- the 殿 criterion in a
102 # non-CJK script. गांधीजी therefore stays unpeeled, which is the
103 # accepted trade (decisions.md#indic-honorifics).
104 # NOTE the leading/trailing split across the two scripts, which is
105 # real and must not be "harmonized": Devanagari बाबू is a LEADING
106 # honorific -- a TITLES entry (#344), not a suffix -- while Bengali
107 # বাবু is TRAILING (অমল বাবু). Different codepoints, so the two
108 # entries cannot interact.
109 'rinpoche', # bo postpositional (Sogyal Rinpoche, Lama Zopa Rinpoche)
110 'जी', # hi/mr the universal respect particle (मोदी जी)
111 'साहब', # hi Sahib
112 'साहिब', # hi/pa the Sahib spelling with the i-matra
113 'साहेब', # mr the Marathi Saheb spelling
114 'महाराज', # hi Maharaj, trailing; the leading महाराजा is a TITLES entry (#344)
115 'সাহেব', # bn Saheb (রহমান সাহেব)
116 'বাবু', # bn Babu -- TRAILING; see the note above
117 'মহারাজ', # bn Maharaj
118})
119"""
120
121Post-nominal suffixes matched as WORDS: the lookup uses the normalized token,
122so only EDGE periods come off and interior ones survive -- "Junior." matches
123here, "J.u.n.o.r." does not and stays name text ("John J.u.n.o.r." parses a
124family name, on both APIs). The example is deliberately not "J.u.n.i.o.r.",
125which fails this lookup too: an interior-period token that no whole-token
126set claims goes to ``period_joined_vocab``, which splits it on its periods
127and, no chunk being a title, calls the whole thing a suffix if ANY chunk is
128suffix vocabulary -- except where every chunk it matches is a SINGLE ASCII
129CHARACTER, which since 2.4 retires rather than claims (decisions.md#S2): the
130one-character suffix vocabulary is the Roman numerals and the digit "2" --
131ASCII is load-bearing there, since the seven single-character CJK
132honorifics in THIS set (様, 殿, 氏, 군, 님, 씨, 양 -- three of them also
133glued-honorific heads below) must KEEP their chunk claim ("J.씨" still
134derives from 씨) -- and a word
135built of single ASCII letters is not about generations. So "J.u.n.i.o.r."
136no longer reaches this set's "i" entry as a generational claim; it is
137read by POSITION instead, as an unlisted multi-chunk word
138(``Policy.unlisted_dotted_suffixes``, default on), and with words to
139spare it is still a suffix (rules.md#S3). A multi-character chunk match
140is untouched -- "Msc.Ed." still derives from ``ed`` and reads as a
141suffix outright, and "Lt.Gov." still derives a TITLE from its own
142multi-character chunk match.
143So membership here is not the last word on a dotted token; the sentence is
144about this set's lookup alone. :data:`SUFFIX_ACRONYMS` is the set matched
145with every period removed, so it alone covers a multi-dot spelling: both
146"P.H.D." and the bare "PhD" reach its `phd` entry, the two normalizing to
147the same string once the periods come off. The two sets are asserted
148DISJOINT (see the guard block at the bottom): a post-nominal belongs to
149one of them or the other, and which one holds it is what decides whether
150its multi-dot spelling reaches a whole-token lookup at all --
151``period_joined_vocab`` may still claim the token chunk by chunk, as
152"Msc.Ed." and "Lt.Gov." above show -- or read it by position instead,
153as "J.u.n.i.o.r." now does.
154
155"""
156GLUED_HONORIFICS = frozenset({
157 # #308: the entries above that may also be peeled off the END of a
158 # name token -- 田中さん, 山田太郎様, 김민준씨. A separate set, not
159 # SUFFIX_WORDS reused, because the glued position has no token
160 # boundary to lean on: the vetting question is not "is this a
161 # name?" but "can this END a name?", and only entries that can
162 # never end one belong here.
163 # kana -- name-final never, in any of the four, and the kana/kanji
164 # split is itself a vetting result: くん ships where 君 cannot,
165 # since 王君 is a complete Chinese name while the hiragana spelling
166 # is unavailable to Chinese at all. Scoped to Chinese on purpose --
167 # hiragana of course spells Japanese GIVEN names (高橋みなみ is one,
168 # pinned in the case table), which is why the four entries above
169 # are vetted one at a time as never name-FINAL rather than waved
170 # through as kana.
171 'さん', 'さま', 'くん', 'ちゃん',
172 # Han -- two-character honorifics with no name-final reading in
173 # either language, plus 様. 殿 is deliberately absent though it
174 # ships spaced above; see the exclusions below.
175 '様', '先生', '教授', '女士', '小姐',
176 # hangul -- the -님 compounds too: longest-first peels 선생님 off
177 # 김선생님 whole, rather than leaving 김선생 to segment into a
178 # family 김 and a given 선생, and 박사님 off 김민준박사님 rather
179 # than stranding 박사 in the given name. 박사 is safe glued where
180 # its Han twin 博士 is not: that collision is Japanese (博士 =
181 # ひろし) and the hangul spelling carries none of it.
182 '씨', '님', '선생님', '교수님', '박사', '박사님',
183})
184"""
185
186The subset of :data:`SUFFIX_WORDS` a name token may end WITH, peeled off as
187its own token before segmentation (#308). Deliberately harsher than the
188spaced set, because a glued tail has no writer-drawn token boundary to lean
189on -- these entries are recognized in the SPACED position only:
190
191* 양, 군 -- 김지양 and 김지군 are given names ending in these syllables, and
192 양 is a top-tier surname besides.
193* 氏 -- 王氏 is a historical name form ("the Wang woman").
194* 博士 -- glued 田中博士 IS Tanaka Hiroshi, an attested given name.
195* 殿 -- some ninety Japanese surnames END in it, 鵜殿 (Udono) and 真殿
196 (Madono) with four-figure populations, so peeling it would cut a real
197 family name in two. Spaced 殿 is safe for the reason 양/군 are: a
198 殿-surnamed person's name LEADS, and the suffix gate is trailing-only.
199* The Indic trailing set (जी, साहब, साहिब, साहेब, महाराज, সাহেব, বাবু,
200 মহারাজ) and Latin rinpoche are spaced-only for the same reason 殿 is:
201 जी ends Banerjee/Mukherjee/Chatterjee (बनर्जी, मुखर्जी, चटर्जी) and a
202 glued peel would strand बनर् on a bare virama. The criterion is not
203 CJK-specific, which is the point (decisions.md#indic-honorifics).
204
205Three more are in NEITHER set, so neither spelling is recognized. 君: 王君 is
206a complete Chinese name (君 is a common given-name final), so the honorific
207reading never gets the benefit of the doubt -- while its kana spelling くん
208ships glued, above. Bare 선생 and 교수: they read as common nouns as readily
209as address terms, and only their -님 forms ship.
210
211"""
212SUFFIX_ACRONYMS_AMBIGUOUS = frozenset({
213 # Suffix acronyms that also commonly work as given-name nicknames on
214 # their own (e.g. "Ed", "JD"). Two readers in 2.x, not the single v1
215 # one this comment used to name: _extract._suffix_shaped, deciding
216 # whether parenthesized/quoted content is a nickname or a suffix
217 # (content matching one of these stays a nickname, the more common
218 # reading in ambiguous, delimiter-only context), and _vocab's
219 # suffix_as_written, which excludes the ambiguous subset from plain
220 # acronym membership so the period gate below is not dead code.
221 # _classify also tags membership as "vocab:suffix-ambiguous".
222 #
223 # When adding a new entry to SUFFIX_ACRONYMS, also add it here only if
224 # the exact letter sequence could plausibly be someone's name on its
225 # own -- a given name or nickname (e.g. 'jd', 'ed') or a common
226 # surname (e.g. 'ma', 'do'). Unambiguous certifications/degrees
227 # (e.g. 'mba', 'cpa', 'phd') don't need an entry. In 2.0 this set
228 # also gates bare recognition: an ambiguous acronym counts as a
229 # suffix only when written with periods ('M.A.' yes, 'Ma' no), so
230 # 'Jack Ma' keeps its family name.
231 #
232 # The other half of the criterion, added 2026-09-07 with #342.
233 # Being borne at all is only the entry ticket; what decides among
234 # the three answers is a comparison of FREQUENCIES -- how common
235 # the word is as a borne name in the TRAILING position against how
236 # common it is as a credential. Roughly balanced earns the marking
237 # here, and the parse reports the fork: 'ba' is the entry that
238 # earned it that day, BA being a common credential and Ba a real
239 # surname (Vietnamese; Senegalese Fula) about as common as the
240 # credential, which is the ma/Ma shape exactly. Where the NAME
241 # reading dominates, the entry is REMOVED from SUFFIX_ACRONYMS
242 # instead of marked: 'rai' and 'cha' left the set that day, both
243 # far more common as surnames than their credentials are as
244 # credentials (RAI is "RETA Authorized Instructor", CHA is
245 # Certified Hotel Administrator or Certified Healthcare Auditor,
246 # both tenuous or specialized; Rai is a common surname across
247 # Hindi- and Bengali-speaking regions and Cha the Korean 차).
248 # Where the CREDENTIAL dominates, the entry stays unambiguous.
249 # LENGTH is a correlate and not the test -- a short acronym is
250 # more often a common credential AND more often a name -- so do
251 # not read the letter counts here as a rule.
252 #
253 # Removal takes the DOTTED spelling with it too, except by
254 # accident: "John Smith R.A.I." still reads suffix 'R.A.I.' only
255 # because rules.md#S3 splits an interior-period token on its
256 # periods and the chunk 'i' happens to be a Roman numeral in
257 # SUFFIX_WORDS. "John Smith R.A.X." reads family, and so does
258 # "John Smith C.H.A." after the removal. Do not count on a dotted
259 # spelling surviving a removal. A caller who needs an entry back
260 # adds it -- Lexicon.default().add(suffix_acronyms={"cha"}) --
261 # which is the answer this library gives for every
262 # locale-specific vocabulary. The cost is stated and accepted:
263 # with the entry gone, "John Smith RAI" reads family 'RAI'. See
264 # decisions.md#suffix-acronym-collisions.
265 #
266 # NOT 'ms' or 'sa', though #296's audit table put them here for the
267 # leading-title collision (bare "Ms" the honorific, "M.S." the
268 # degree): the gate is position-blind and the collision is not.
269 # Gated, 'John Smith, MS' lost its suffix-comma route and read
270 # title 'MS', and 'Smith, Ms.' passed the gate on its one period
271 # and read as a credential anyway. Both words are genuine duals --
272 # title and unambiguous suffix -- and position decides, as for
273 # 'sr' and 'lt' (decisions.md#C1).
274 #
275 # NOT 'se' or 'om' either, weighed 2026-09-07 with #342 and left
276 # alone: no surname evidence worth standing behind, and OM is the
277 # Order of Merit. And NOT 'mc' or 'vd', which are also PARTICLES:
278 # #454 closed by design -- neither is a borne name, so a bare
279 # trailing one is the decoration, and rules.md#P6's Accepted
280 # clause names them as the two words whose positional reading
281 # does not hold.
282 'ba',
283 'do',
284 'ed',
285 'jd',
286 'ma',
287})
288"""
289
290Acronym suffixes from SUFFIX_ACRONYMS that also plausibly collide with a
291common given-name nickname. Not a partition of SUFFIX_ACRONYMS -- a small,
292standalone exception list, read by the delimited-content escape in
293``_pipeline/_extract.py`` and by ``_pipeline/_vocab.py``'s period gate.
294
295"""
296SUFFIX_ACRONYMS = frozenset({
297 '8-vsb',
298 'aas',
299 'aba',
300 'abc',
301 # "All But Dissertation". Also the Latin transliteration of the
302 # Arabic bound given-name word in BOUND_GIVEN_NAMES (whose
303 # Arabic-script counterpart عبد is in that set only) -- the one
304 # word in both sets, kept in both deliberately. Position decides
305 # at the two ends but not in a family comma's given slot; see
306 # decisions.md#P5.
307 'abd',
308 'abpp',
309 'abr',
310 'aca',
311 'acas',
312 'ace',
313 'acha',
314 'acp',
315 'ae',
316 'ae',
317 'aem',
318 'afasma',
319 'afc',
320 'afc',
321 'afm',
322 'afm',
323 'agsf',
324 'aia',
325 'aicp',
326 'ala',
327 'alc',
328 'alp',
329 'am',
330 'amd',
331 'ame',
332 'amieee',
333 'ams',
334 'aphr',
335 'apn',
336 'aprn',
337 'apr',
338 'apss',
339 'aqp',
340 'arm',
341 'arrc',
342 'asa',
343 'asc',
344 'asid',
345 'asla',
346 'asp',
347 'atc',
348 'awb',
349 'ba',
350 'bca',
351 'bcl',
352 'bcss',
353 'bds',
354 'bem',
355 'bls-i',
356 'bn',
357 'bpe',
358 'bpi',
359 'bpt',
360 'bsc',
361 'bt',
362 'btcs',
363 'bts',
364 'cacts',
365 'cae',
366 'caha',
367 'caia',
368 'cams',
369 'cap',
370 'capa',
371 'capm',
372 'capp',
373 'caps',
374 'caro',
375 'cas',
376 'casp',
377 'cb',
378 'cbe',
379 'cbm',
380 'cbne',
381 'cbnt',
382 'cbp',
383 'cbrte',
384 'cbs',
385 'cbsp',
386 'cbt',
387 'cbte',
388 'cbv',
389 'cca',
390 'ccc',
391 'ccca',
392 'cccm',
393 'cce',
394 'cchp',
395 'ccie',
396 'ccim',
397 'cciso',
398 'ccm',
399 'ccmt',
400 'ccna',
401 'ccnp',
402 'ccp',
403 'ccp-c',
404 'ccpr',
405 'ccs',
406 'ccufc',
407 'cd',
408 'cdal',
409 'cdfm',
410 'cdmp',
411 'cds',
412 'cdt',
413 'cea',
414 'ceas',
415 'cebs',
416 'ceds',
417 'ceh',
418 'cela',
419 'cem',
420 'cep',
421 'cera',
422 'cet',
423 'cfa',
424 'cfc',
425 'cfcc',
426 'cfce',
427 'cfcm',
428 'cfe',
429 'cfeds',
430 'cfi',
431 'cfm',
432 'cfp',
433 'cfps',
434 'cfr',
435 'cfre',
436 'cga',
437 'cgap',
438 'cgb',
439 'cgc',
440 'cgfm',
441 'cgfo',
442 'cgm',
443 'cgm',
444 'cgma',
445 'cgp',
446 'cgr',
447 'cgsp',
448 'ch',
449 'chba',
450 'chdm',
451 'che',
452 'ches',
453 'chfc',
454 'chfc',
455 'chi',
456 'chmc',
457 'chmm',
458 'chp',
459 'chpa',
460 'chpe',
461 'chpln',
462 'chpse',
463 'chrm',
464 'chsc',
465 'chse',
466 'chse-a',
467 'chsos',
468 'chss',
469 'cht',
470 'cia',
471 'cic',
472 'cie',
473 'cig',
474 'cip',
475 'cipm',
476 'cips',
477 'ciro',
478 'cisa',
479 'cism',
480 'cissp',
481 'cla',
482 'clsd',
483 'cltd',
484 'clu',
485 'cm',
486 'cma',
487 'cmas',
488 'cmc',
489 'cmfo',
490 'cmg',
491 'cmp',
492 'cms',
493 'cmsp',
494 'cmt',
495 'cna',
496 'cnm',
497 'cnp',
498 'cp',
499 'cp-c',
500 'cpa',
501 'cpacc',
502 'cpbe',
503 'cpcm',
504 'cpcu',
505 'cpe',
506 'cpfa',
507 'cpfo',
508 'cpg',
509 'cph',
510 'cpht',
511 'cpim',
512 'cpl',
513 'cplp',
514 'cpm',
515 'cpo',
516 'cpp',
517 'cppm',
518 'cprc',
519 'cpre',
520 'cprp',
521 'cpsc',
522 'cpsi',
523 'cpss',
524 'cpt',
525 'cpwa',
526 'crde',
527 'crisc',
528 'crma',
529 'crme',
530 'crna',
531 'cro',
532 'crp',
533 'crt',
534 'crtt',
535 'csa',
536 'csbe',
537 'csc',
538 'cscp',
539 'cscu',
540 'csep',
541 'csi',
542 'csm',
543 'csp',
544 'cspo',
545 'csre',
546 'csrte',
547 'csslp',
548 'cssm',
549 'cst',
550 'cste',
551 'ctbs',
552 'ctfa',
553 'cto',
554 'ctp',
555 'cts',
556 'cua',
557 'cusp',
558 'cva',
559 'cva[22]',
560 'cvo',
561 'cvp',
562 'cvrs',
563 'cwap',
564 'cwb',
565 'cwdp',
566 'cwep',
567 'cwna',
568 'cwne',
569 'cwp',
570 'cwsp',
571 'cxa',
572 'cyds',
573 'cysa',
574 'dabfm',
575 'dabvlm',
576 'dacvim',
577 'dbe',
578 'dc',
579 'dcb',
580 'dcm',
581 'dcmg',
582 'dcvo',
583 'dd',
584 'dds',
585 'ded',
586 'dep',
587 'dfc',
588 'dfm',
589 'diplac',
590 'diplom',
591 'djur',
592 'dma',
593 'dmd',
594 'dmin',
595 'dnp',
596 'do',
597 'dpm',
598 'dpt',
599 'drb',
600 'drmp',
601 'drph',
602 'dsc',
603 'dsm',
604 'dso',
605 'dss',
606 'dtr',
607 'dvep',
608 'dvm',
609 'ea',
610 'ed',
611 'edd',
612 'ei',
613 'eit',
614 'els',
615 'emd',
616 'emt-b',
617 'emt-i/85',
618 'emt-i/99',
619 'emt-p',
620 'enp',
621 'erd',
622 'evp',
623 'faafp',
624 'faan',
625 'faap',
626 'fac-c',
627 'facc',
628 'facd',
629 'facem',
630 'facep',
631 'facha',
632 'facofp',
633 'facog',
634 'facp',
635 'facph',
636 'facs',
637 'faia',
638 'faicp',
639 'fala',
640 'fashp',
641 'fasid',
642 'fasla',
643 'fasma',
644 'faspen',
645 'fca',
646 'fcas',
647 'fcela',
648 'fd',
649 'fec',
650 'fhames',
651 'fic',
652 'ficf',
653 'fieee',
654 'fmp',
655 'fmva',
656 'fnss',
657 'fp&a',
658 'fp-c',
659 'fpc',
660 'frm',
661 'fsa',
662 'fsdp',
663 'fws',
664 'gaee[14]',
665 'gba',
666 'gbe',
667 'gc',
668 'gcb',
669 'gcb',
670 'gchs',
671 'gcie',
672 'gcmg',
673 'gcmg',
674 'gcsi',
675 'gcvo',
676 'gcvo',
677 'gisp',
678 'git',
679 'gm',
680 'gmb',
681 'gmr',
682 'gphr',
683 'gri',
684 'grp',
685 'gsmieee',
686 'hccp',
687 'hrs',
688 'iaccp',
689 'iaee',
690 'iccm-d',
691 'iccm-f',
692 'idsm',
693 'ifgict',
694 'iom',
695 'ipep',
696 'ipm',
697 'iso',
698 'issp-csp',
699 'issp-sa',
700 'itil',
701 'jd',
702 'jp',
703 'kbe',
704 'kcb',
705 'kchs/dchs',
706 'kcie',
707 'kcie',
708 'kcmg',
709 'kcsi',
710 'kcsi',
711 'kcvo',
712 'kg',
713 'khs/dhs',
714 'kp',
715 'kt',
716 'lac',
717 'lcmt',
718 'lcpc',
719 'lcsw',
720 'lg',
721 'litk',
722 'litl',
723 'litp',
724 'llm',
725 'lm',
726 'lmsw',
727 'lmt',
728 'lp',
729 'lpa',
730 'lpc',
731 'lpn',
732 'lpss',
733 'lsi',
734 'lsit',
735 'lt',
736 'lvn',
737 'lvo',
738 'lvt',
739 'ma',
740 'maaa',
741 'mai',
742 'mba',
743 'mbe',
744 'mbs',
745 'mc',
746 'mcct',
747 'mcdba',
748 'mches',
749 'mcm',
750 'mcp',
751 'mcpd',
752 'mcsa',
753 'mcsd',
754 'mcse',
755 'mct',
756 'md',
757 'mda',
758 'mdb',
759 'mdbb',
760 'mdep',
761 'mdhb',
762 'mdiv',
763 'mdl',
764 'mem',
765 'meng',
766 'mfa',
767 'micp',
768 'mieee',
769 'mirm',
770 'mle',
771 'mls',
772 'mlse',
773 'mlt',
774 'mm',
775 'mmad',
776 'mmas',
777 'mnaa',
778 'mnae',
779 'mp',
780 'mpa',
781 'mph',
782 'mpse',
783 'mra',
784 'ms',
785 'msa',
786 'msc',
787 'mscmsm',
788 'msm',
789 'mt',
790 'mts',
791 'mvo',
792 'nbc-his',
793 'nbcch',
794 'nbcch-ps',
795 'nbcdch',
796 'nbcdch-ps',
797 'nbcfch',
798 'nbcfch-ps',
799 'nbct',
800 'ncarb',
801 'nccp',
802 'ncidq',
803 'ncps',
804 'ncso',
805 'ncto',
806 'nd',
807 'ndtr',
808 'nmd',
809 'np',
810 'np[18]',
811 'nraemt',
812 'nremr',
813 'nremt',
814 'nrp',
815 'obe',
816 'obi',
817 'oca',
818 'ocm',
819 'ocp',
820 'od',
821 'om',
822 'oscp',
823 'ot',
824 'pa-c',
825 'pcc',
826 'pci',
827 'pe',
828 'pfmp',
829 'pg',
830 'pgmp',
831 'pharmd',
832 'phc',
833 'phd',
834 'phr',
835 'phrca',
836 'pla',
837 'pls',
838 'pmc',
839 'pmi-acp',
840 'pmp',
841 'pp',
842 'pps',
843 'prm',
844 'psm',
845 'psp',
846 'psyd',
847 'pt',
848 'pta',
849 'qam',
850 'qc',
851 'qcsw',
852 'qfsm',
853 'qgm',
854 'qpm',
855 'qsd',
856 'qsp',
857 'ra',
858 'rba',
859 'rci',
860 'rcp',
861 'rd',
862 'rdcs',
863 'rdh',
864 'rdms',
865 'rdn',
866 'res',
867 'rfp',
868 'rhca',
869 'rid',
870 'rls',
871 'rmsks',
872 'rn',
873 'rp',
874 'rpa',
875 'rph',
876 'rpl',
877 'rrc',
878 'rrt',
879 'rrt-accs',
880 'rrt-nps',
881 'rrt-sds',
882 'rtrp',
883 'rvm',
884 'rvt',
885 'sa',
886 'same',
887 'sasm',
888 'sccp',
889 'scmp',
890 'se',
891 'secb',
892 'sfp',
893 'sgm',
894 'shrm-cp',
895 'shrm-scp',
896 'si',
897 'siie',
898 'smieee',
899 'sphr',
900 'sscp',
901 'stb',
902 'stmieee',
903 'tbr-ct',
904 'td',
905 'thd',
906 'thm',
907 'ud',
908 'usa',
909 'usaf',
910 'usar',
911 'uscg',
912 'usmc',
913 'usn',
914 'usnr',
915 'uxc',
916 'uxmc',
917 'vc',
918 'vc',
919 'vcp',
920 'vd',
921 'vrd',
922})
923"""
924
925Post-nominal acronyms. Titles, degrees and other things people stick after their name
926that may or may not have periods between the letters. The parser removes periods
927when matching against these pieces.
928
929"""
930
931
932# Guard the invariants the docstrings above promise, so a future edit that
933# breaks them fails at import time instead of silently drifting until a test
934# happens to catch it (same rationale as particles.py). Note `assert` is
935# stripped under `python -O`; Lexicon re-checks the relationships at
936# construction, which is what protects a caller's own vocabulary.
937assert SUFFIX_ACRONYMS_AMBIGUOUS <= SUFFIX_ACRONYMS, \
938 "SUFFIX_ACRONYMS_AMBIGUOUS must stay a subset of SUFFIX_ACRONYMS"
939# The two sets normalize differently -- the word test strips only edge
940# periods, the acronym test strips all of them -- so a word in both is
941# matched twice by two rules, and which one fired is unreadable from the
942# outside. The single overlap was 'esq', dropped 2026-09-08
943# (decisions.md#suffix-acronym-collisions), and the assert is what keeps
944# a bulk import from quietly re-creating one. It guards the SHIPPED sets
945# only: Lexicon has no matching invariant, so a caller who wants the
946# overlap in their own vocabulary may still have it.
947#
948# It carries the AMBIGUOUS set's stake too, which is the sharper one:
949# suffix_as_written ORs the word branch and the acronym branch, so an
950# ambiguous acronym that were also a suffix word would be claimed
951# through the word membership and bypass the period gate the ambiguous
952# set exists to impose. The ambiguous set is a subset of the acronyms
953# (the assert above), so this one covers it -- a separate assert of
954# `SUFFIX_ACRONYMS_AMBIGUOUS & SUFFIX_WORDS` cannot fail while both of
955# these hold.
956assert not (SUFFIX_ACRONYMS & SUFFIX_WORDS), \
957 "a post-nominal belongs to one set or the other, never both (the " \
958 "two normalize differently, and the word branch would bypass an " \
959 "ambiguous acronym's period gate): " \
960 f"{sorted(SUFFIX_ACRONYMS & SUFFIX_WORDS)}"
961# The peel splits its tail off as a TOKEN and suffix classification is
962# what claims it downstream, so a tail that is not also a suffix word
963# would split the name and then leave the piece sitting in it. The
964# reverse direction is deliberately unguarded: a suffix word that is
965# not a tail is the ordinary case, and an empty tail set is inert
966# rather than wrong.
967assert GLUED_HONORIFICS <= SUFFIX_WORDS, \
968 "GLUED_HONORIFICS must stay a subset of SUFFIX_WORDS: " \
969 f"{sorted(GLUED_HONORIFICS - SUFFIX_WORDS)}"
970assert_normalized("suffix", SUFFIX_ACRONYMS | SUFFIX_WORDS)
971
972
973# 1.x name, deprecated in 2.2 and removed in 3.0 (#293). The constant
974# did not change module, so this aliases a name to one of this module's
975# own globals: a module __getattr__ runs only once the body has finished
976# and the module is in sys.modules, so the lookup resolves rather than
977# recursing.
978from typing import TYPE_CHECKING # noqa: E402
979
980from nameparser.config._deprecated import alias_getattr # noqa: E402
981
982# Declared for the type checker, served by __getattr__ at runtime --
983# the split is what keeps mypy checking this module's LIVE names; see
984# the note in titles.py and alias_getattr's docstring.
985if TYPE_CHECKING:
986 SUFFIX_NOT_ACRONYMS: frozenset[str]
987else:
988 __getattr__, __dir__ = alias_getattr(__name__, {
989 "SUFFIX_NOT_ACRONYMS": (
990 "nameparser.config.suffixes", "SUFFIX_WORDS"),
991 })
992
993# Star imports read __all__ and never the module __getattr__ -- see the
994# note in prefixes.py. Live constants listed alongside the retired name
995# for the same reason titles.py lists its own.
996#
997# In SOURCE order, not alphabetical: `automodule :members:` follows
998# __all__ where a module defines one, so an alphabetical list here would
999# silently reorder this module's entries in modules.html. The retired
1000# name goes last because autodoc does not document it, and so it has no
1001# position to preserve. Not for want of SEEING it: the module member
1002# scan walks dir(), which our __dir__ lists the retired name in, and
1003# then calls safe_getattr on it -- so an html build resolves this name
1004# and titles.py's retired one alike, emitting a real DeprecationWarning
1005# for each. (Invisible in a "build succeeded, 0 warnings" line: Sphinx
1006# warnings and Python warnings are different channels. Wrap
1007# sphinx.cmd.build.build_main in warnings.catch_warnings to see them.)
1008# What declines it is the attribute-doc scan: ModuleAnalyzer parses the
1009# SOURCE and finds no assignment statement for a name served by
1010# __getattr__, so autodoc computes is_attr=False, and at module level a
1011# member that is not an attribute and is neither a class nor a callable
1012# matches no object type at all -- no documenter is chosen and the
1013# member is skipped.
1014__all__ = [
1015 "SUFFIX_WORDS",
1016 "GLUED_HONORIFICS",
1017 "SUFFIX_ACRONYMS_AMBIGUOUS",
1018 "SUFFIX_ACRONYMS",
1019 "SUFFIX_NOT_ACRONYMS",
1020]