1from nameparser.config._invariants import assert_normalized
2
3CONJUNCTIONS = frozenset({
4 '&',
5 'and',
6 'et',
7 'e',
8 'of',
9 'the',
10 'und',
11 'y',
12 # #397: the Catalan/Polish surname link ("Carod i Rovira",
13 # "Kowalski i Nowak"). Single-letter like 'y'/'и'/'e', and the one
14 # entry that is ALSO generational vocabulary -- 'i' is the roman
15 # numeral I, a bare entry of SUFFIX_WORDS -- so the carve-out
16 # counts it as a name word for its own sake and it joins only
17 # with a name word on each side (rules.md#P3).
18 'i',
19 # #269: Cyrillic (ru/uk/bg) "and": и, і, та. Ukrainian writes і and
20 # й for the same conjunction, alternating on the surrounding
21 # vowel/consonant for euphony ("Олесь і Олена", "Марія й Петро"),
22 # so real data carries both spellings and neither alone suffices.
23 'и',
24 'і',
25 'й',
26 'та',
27 # #269: Greek "and": και.
28 'και',
29 # #269 follow-up: Arabic "and". Formal script attaches و to the
30 # following word (وفاطمة), so a standalone و token appears only in
31 # informal spacing -- common in real data. Single-character like
32 # 'y'/'и': the single-letter carve-out (Google Code issue 11,
33 # the "john e smith" bug) protects short names (joins
34 # only with enough rootname pieces).
35 'و',
36})
37"""
38Pieces that should join to their neighboring pieces, e.g. "and", "y" and "&".
39"of" and "the" are also include to facilitate joining multiple titles,
40e.g. "President of the United States".
41"""
42
43CONJUNCTIONS_AMBIGUOUS = frozenset({
44 # #383/#479: single letters that read as an INITIAL rather than a
45 # connective when the input carries no case evidence -- a name
46 # written wholly in one case, upper or lower alike. A letter
47 # outside this set joins there, bare capital included.
48 #
49 # 'e' and not 'y', measured: a bare E initial is common (Edward,
50 # Elizabeth) and an 'e' between two surnames is rare outside couple
51 # listings; a bare Y initial is rare and 'y' between two surnames is
52 # the commonest Hispanic compound and this library's oldest fixture
53 # ('Velasquez y Garcia'). Cyrillic и/і/й follow y, not e: #267
54 # blessed their joining and nothing here narrows it.
55 #
56 # 'i' (Catalan/Polish) ships here beside 'e' for the same reason
57 # (#397): a bare I initial is as common as a bare E, so a name
58 # written wholly in one case reads the letter as an initial and
59 # reports the fork rather than joining in silence.
60 #
61 # A caller edits the vocabulary rather than a switch: remove 'e' to
62 # restore joining for Portuguese data, remove 'i' for Catalan or
63 # Polish data, add 'y' for a Dutch-style "every single letter is an
64 # initial".
65 'e',
66 'i',
67})
68"""
69Single-letter entries of :data:`CONJUNCTIONS` that read as an initial,
70not a connective, in a name written wholly in one case.
71"""
72
73
74assert_normalized("CONJUNCTIONS", CONJUNCTIONS)
75
76# Guard the invariant the docstring promises, so a future edit that
77# breaks it fails at import time (same rationale as suffixes.py).
78# This holds the SHIPPED constant and nothing else, and two things
79# follow. `assert` is stripped under `python -O`, so under -O even
80# that much is gone. And unlike the other marker subsets, the pair is
81# deliberately absent from Lexicon's subset checks, so a CALLER'S
82# orphan is never rejected -- it is inert instead, the classify fork
83# and its emitter both requiring the base entry before they read the
84# marker (see decisions.md#P3, the 2026-09-13 entries).
85assert CONJUNCTIONS_AMBIGUOUS <= CONJUNCTIONS, \
86 "CONJUNCTIONS_AMBIGUOUS must stay a subset of CONJUNCTIONS"
87assert all(len(w) == 1 and w.upper() != w.lower()
88 for w in CONJUNCTIONS_AMBIGUOUS), \
89 "CONJUNCTIONS_AMBIGUOUS holds CASED single letters; the fork only " \
90 "reads a cased single-letter token, so a caseless letter (Arabic " \
91 "و) or a multi-letter entry would be silently inert"
92assert_normalized("CONJUNCTIONS_AMBIGUOUS", CONJUNCTIONS_AMBIGUOUS)