1"""Internal pipeline state: WorkToken and ParseState.
2
3WorkTokens are pipeline-internal (no validation -- the tokenizer is the
4only producer) and are addressed BY INDEX in every stage: pieces and
5segments are runs of token indices, never joined strings, so value-based
6lookup (v1's #100 family) is structurally impossible.
7
8Layering: imports _types, _lexicon, _policy only (enforced by
9tests/v2/test_layering.py).
10"""
11from __future__ import annotations
12
13import bisect
14from collections.abc import Sequence
15from dataclasses import dataclass
16from enum import Enum, auto
17
18from nameparser._lexicon import Lexicon
19from nameparser._policy import Policy
20from nameparser._types import AmbiguityKind, Role, Segmenter, Span
21
22
23# The comma characters (ASCII/Arabic/fullwidth, #265). Shared here so
24# tokenize (separators/segmentation) and extract (close-quote
25# boundaries) cannot drift apart.
26COMMA_CHARS = frozenset({",", "\u060c", "\uff0c"})
27
28
29def comma_bucket(start: int, comma_offsets: Sequence[int]) -> int:
30 """Which comma-delimited part of the name a token starting at
31 `start` falls in: the number of commas before it.
32
33 Shared here for the reason COMMA_CHARS is, and the sharing is
34 load-bearing in the same way. segment BUILDS the segments with this
35 (comma_offsets is sorted and no offset ever equals a token start,
36 so bisect_left counts the commas before the token); classify asks
37 it to decide whether two tokens could be in one segment, which is
38 half of what stops a maiden marker run from spanning a boundary no
39 segment holds. Two tokens agree here iff segment would put them in
40 one segment, and that is an identity rather than a resemblance
41 only while both sides ask this function.
42 """
43 return bisect.bisect_left(comma_offsets, start)
44
45@dataclass(frozen=True, slots=True)
46class WorkToken:
47 """One tokenized word. role stays None until assign; extracted
48 nickname/maiden tokens arrive with their role pre-set. text is
49 always the exact original slice (tokenize is the sole producer;
50 the anti-#100 invariant depends on it)."""
51
52 text: str
53 span: Span
54 tags: frozenset[str] = frozenset()
55 role: Role | None = None
56
57
58#: M4's two carve-outs, as the tags classify recorded them: a bound
59#: given-name word is vocabulary claiming the word as a given name,
60#: and `initial` is the initial reading. Neither is a predicate M4 owns.
61#: Shared here beside WorkToken.tags for the reason COMMA_CHARS is:
62#: assign's `_WORD_ALREADY_CLAIMED` is built from this pair, and the
63#: two stages must not drift (post_rules imports _assign, so _assign
64#: cannot reach the other way).
65_NEVER_FLIPPED = frozenset({"vocab:bound-given", "initial"})
66
67#: The by-shape half of #289/#516's ambiguous credential class: a
68#: token classify admits to `vocab:suffix-ambiguous`'s READING by
69#: SHAPE rather than by the listed vocabulary
70#: (`Policy.unlisted_dotted_suffixes` is the first emitter;
71#: `Policy.unlisted_caps_suffixes` is the second, and classify writes
72#: the tag from both branches). One constant, not a string literal at
73#: each site, because the readers that must tell a by-shape member
74#: apart from a listed one -- `_pieces.peel_trailing` and
75#: `_pieces.listed_lean`, which `segment_suffix_reading` asks through,
76#: so the reading it decides is second-hand -- cannot afford to spell
77#: it several ways and have one of them typo silently past the
78#: others. The two sites that want EITHER half read
79#: `_AMBIGUOUS_CREDENTIAL_TAGS` below rather than this constant.
80SHAPE_ACRONYM_TAG = "shape:acronym"
81
82#: The MEMBERSHIP half of the same class: classify's tag for a token
83#: the ambiguous credential vocabulary claims, by listing
84#: (`Lexicon.suffix_acronyms_ambiguous`) or -- where a Policy switch
85#: admits the by-shape half -- beside `SHAPE_ACRONYM_TAG`. Beside that
86#: constant and for its reason: the string was spelled out at eight
87#: sites across four modules, each of them a place for a typo to pass
88#: silently, since a tag that is never written is simply a tag no
89#: reader ever finds.
90AMBIGUOUS_ACRONYM_TAG = "vocab:suffix-ambiguous"
91
92#: EITHER way a token joins the ambiguous credential class -- the
93#: vocabulary's claim and the writing's. The two emitters that report
94#: a fork the peel called and declined ask exactly this: `_group`'s
95#: prefix chain (a listed member the case lean read as a name, 'John
96#: van der Berg Ma'; a by-shape member the count left standing,
97#: 'Freiherr von Berg X.Y.I.') and `_assign`'s family-comma slot. One
98#: constant beside the two above and for their reason -- the pair was
99#: spelled two ways, a frozenset here and an `or` of two `in` tests
100#: there -- and a frozenset so each test is one `isdisjoint`, a C call
101#: with no Python frame, on a branch every chained name reaches.
102_AMBIGUOUS_CREDENTIAL_TAGS = frozenset(
103 {AMBIGUOUS_ACRONYM_TAG, SHAPE_ACRONYM_TAG})
104
105
106class Structure(Enum):
107 """segment's comma-structure decision."""
108
109 NO_COMMA = auto()
110 FAMILY_COMMA = auto() # "Family, Given ..." (v1 lastname-comma)
111 SUFFIX_COMMA = auto() # "Given Family, Suffix ..."
112
113
114@dataclass(frozen=True, slots=True)
115# rules.md#A1: "parsing never fails on any input: where the text's
116# structure or a word's reading is genuinely uncertain, the parse
117# completes on the best reading and carries an ambiguity report
118# naming the doubt"
119class PendingAmbiguity:
120 """An ambiguity recorded mid-pipeline by token INDEX; assemble
121 materializes real Ambiguity objects over the final tokens.
122
123 ``origin`` is for the one stage that runs BEFORE tokens exist:
124 extract_delimited knows only a character offset, so it records that
125 and tokenize resolves it to the containing token's index. Stages
126 after tokenize set ``indices`` directly and leave ``origin`` None.
127 """
128
129 kind: AmbiguityKind
130 detail: str
131 indices: tuple[int, ...] = ()
132 origin: int | None = None
133
134
135@dataclass(frozen=True, slots=True)
136class ParseState:
137 """Carried through the stage fold. Frozen; stages return copies via
138 dataclasses.replace. Fields are filled progressively:
139 extract_delimited -> extracted/masked; tokenize -> tokens (span-
140 sorted)/comma_offsets/interpunct_offsets (the 间隔号 offsets the
141 order and segmentation decisions consult, #298; the nakaguro
142 separators record NOTHING); segment -> segments/structure/one_case
143 (lazily, only where a comma form could turn it on -- #289/#516);
144 script_segment -> tokens and segments again (the one stage that
145 changes the token COUNT: an unspaced CJK token splits into n+1
146 pieces, still as sub-slices of the original, and every later index
147 in the segment runs shifts by n); classify -> token tags AND
148 one_case; group -> pieces/piece_tags/dropped AND maiden token
149 roles;
150 assign -> the remaining token roles AND `order`, the effective
151 order it read them under; post_rules -> roles again, and the
152 ambiguity P6's attachment reports.
153 Ambiguities are recorded by every stage that DECIDES one --
154 extract (resolved to a token index by tokenize), segment,
155 script_segment, classify, group, assign, and post_rules -- since a
156 fork whose branches are taken in different stages needs an emitter
157 in each. Post-group, segments may retain indices of dropped tokens
158 -- assign iterates pieces, never segments. This ownership map is
159 pinned by tests/v2/pipeline/test_state.py.
160
161 segmenter belongs to no stage: like original/lexicon/policy it is
162 passed in at construction by Parser.parse and only ever READ (by
163 script_segment, for a token the vocabulary declined)."""
164
165 original: str
166 lexicon: Lexicon
167 policy: Policy
168 #: The optional Parser(segmenter=...) hook; None = not configured.
169 segmenter: Segmenter | None = None
170 extracted: tuple[tuple[Role, Span], ...] = ()
171 masked: tuple[Span, ...] = ()
172 tokens: tuple[WorkToken, ...] = ()
173 comma_offsets: tuple[int, ...] = ()
174 interpunct_offsets: tuple[int, ...] = ()
175 segments: tuple[tuple[int, ...], ...] = ()
176 structure: Structure = Structure.NO_COMMA
177 # pieces[s][p] = run of token indices: piece p of segment s.
178 # piece_tags[s][p] = derived flags for that piece ("title", "prefix",
179 # "suffix", "conjunction") set by group's joins.
180 pieces: tuple[tuple[tuple[int, ...], ...], ...] = ()
181 piece_tags: tuple[tuple[frozenset[str], ...], ...] = ()
182 dropped: tuple[int, ...] = () # structural tokens (maiden markers)
183 #: The order `assign` actually READ the name under -- name_order,
184 #: or the script_orders entry that overrode it. None wherever no
185 #: positional read happened: after a family comma (which fixes the
186 #: family, so assign consults no order at all), and on every early
187 #: return in `_assign_main`, where a segment holds no name piece
188 #: to position. Recorded rather than recomputed downstream,
189 #: because the two can differ and a post_rules rule keyed on
190 #: `policy.name_order` would then disagree with the roles assign
191 #: already wrote (#395). Reaching that divergence needs a custom
192 #: lexicon -- every shipped particle is Latin, and Latin has no
193 #: script_orders entry -- which is why the test for it builds its
194 #: own (test_post_rules.py).
195 order: tuple[Role, Role, Role] | None = None
196 #: Whether the name's OWN words are written wholly in one case --
197 #: all upper or all lower alike -- and so carry no case EVIDENCE
198 #: about any word in them (rules.md#P3's own-words span,
199 #: _pieces.own_words). None means NOT ASKED YET: the fact is
200 #: computed by whichever of segment and classify needs it first,
201 #: segment only when a comma form could turn on it, so a reader
202 #: between the two stages sees None and must not guess.
203 #: Recorded rather than recomputed, the way `order` above is: the
204 #: trailing suffix slot, the post-comma slot, the tail-segment
205 #: reading, the prefix chain's own tail measure (_group) and the
206 #: glued-honorific peel's decline of a post-comma run
207 #: (_script_segment) all consult it and must not disagree
208 #: (#289/#516, decisions.md#S2). Five stages read it in all --
209 #: segment, script_segment, classify, group and assign -- plus the
210 #: two predicate layers they read it through (_pieces, _vocab).
211 #: A fact segment records SURVIVES script_segment, the one stage
212 #: that changes the token count, and survives it unrecomputed
213 #: because splitting a token cannot change the answer: the verdict
214 #: is `is_one_case` over the name's own words JOINED, and a split
215 #: only moves a space into a string whose upper/lower comparison
216 #: ignores spaces entirely -- '김민준씨' and '김민준 씨' fold alike.
217 #: So the field is carried through rather than invalidated, and
218 #: script_segment reads it (its suffix-run predicate takes the
219 #: lean) rather than asking again.
220 one_case: bool | None = None
221 ambiguities: tuple[PendingAmbiguity, ...] = ()