1"""Stage: classify.
2
3Consumes: tokens, comma_offsets (with token roles, the two halves of
4the structural-boundary test the marker pass applies -- see
5_vocab.tag_marker_runs), and one_case where an earlier stage recorded
6it -- segment writes it lazily where a comma form can turn it on
7(#289/#516), so this read is the fallback for every other name.
8Produces: tokens with vocabulary tags added (text/span/role unchanged),
9plus ambiguities (SUFFIX_OR_NICKNAME, CONJUNCTION_OR_INITIAL) and
10one_case -- whether the name's own words are written in one case,
11recorded for the later stages that read it (#289/#516) and left alone
12where an earlier stage already asked.
13Reads: every Lexicon vocabulary field except surnames and
14honorific_tails, which script_segment consumes upstream; and, since
152.4, Policy.unlisted_dotted_suffixes and Policy.unlisted_caps_suffixes,
16which decide whether an UNLISTED dotted or all-caps token joins the
17ambiguous credential class by SHAPE (#516). is_initial also consults
18the _policy module's _NO_INITIALS constant. It is named apart from the
19fields above because it is not CONFIGURATION -- no Lexicon or Policy
20carries it and no caller can change it -- and not because it decides
21nothing: the tags this stage writes do vary by it ('씨.' is not tagged
22`initial`, which is the whole of #320).
23
24Tags emitted -- stable (API): "particle", "conjunction", "initial";
25namespaced (unstable): "vocab:title", "vocab:given-title",
26"vocab:suffix", "vocab:suffix-word", "vocab:suffix-ambiguous",
27"vocab:particle-ambiguous", "vocab:bound-given", "vocab:maiden-marker",
28"vocab:maiden-marker-cont"; and, in a namespace of its own,
29"shape:acronym".
30"vocab:maiden-marker" tags the HEAD of a maiden marker, which is a
31whole marker whenever the marker is one word; the continuation tag
32carries the rest of a PHRASE marker ("z domu"), so a site asking
33"does a marker start here" reads the same tag it always did and a site
34asking "where does it end" walks the continuations.
35"vocab:suffix" means "counts as a suffix as written": unambiguous
36suffix vocabulary, or an ambiguous acronym written with periods --
37at the TAG level 'M.A.' gets "vocab:suffix" while 'Ma' gets only
38"vocab:suffix-ambiguous"; what assign then does with a trailing
39ambiguous tag is the rest of rule S2's statement (the
40words-to-spare guard) and its Accepted consequences.
41The initial veto is assign's job, not classify's: 'V' carries both
42"vocab:suffix" and "initial".
43"shape:acronym" is the one tag in the shape: namespace and it records
44WHERE a class claim came from rather than what the vocabulary holds:
45an unlisted token the writing makes credential-shaped. It rides
46beside "vocab:suffix-ambiguous" where a Policy switch admits the
47token to that class, and stands alone where the switch is off, which
48is what lets the fork be reported without being taken.
49"""
50from __future__ import annotations
51
52import dataclasses
53
54from nameparser._lexicon import _normalize
55from nameparser._pipeline._state import (
56 AMBIGUOUS_ACRONYM_TAG, SHAPE_ACRONYM_TAG, ParseState, PendingAmbiguity,
57 WorkToken,
58)
59from nameparser._types import AmbiguityKind, Role
60from nameparser._pipeline._vocab import (
61 caps_shape_candidate, is_initial, is_one_case, period_joined_vocab,
62 suffix_as_written, tag_marker_runs,
63)
64from nameparser._pipeline._pieces import own_words
65
66
67
68
69# rules.md#S2: "a trailing word of the suffix vocabulary reads as a
70# suffix — generational forms and credential acronyms alike, and an
71# ambiguous acronym written with its periods, one after each
72# letter, counts unambiguously; a single trailing period is the
73# abbreviation shape any word can wear and does not. A
74# bare ambiguous acronym is consumed only when the name has words to
75# spare"
76def _tags_for(token: WorkToken, n: str, state: ParseState,
77 marker_tag: str | None, one_case_own: bool,
78 one_case: bool) -> frozenset[str]:
79 """`n` is _normalize(token.text), folded once by the caller and
80 shared with the marker pass; `marker_tag` is what that pass decided
81 for this token, or None. The marker DECISION is entirely
82 `_vocab.tag_marker_runs`'; only the writing happens here, so the
83 two tokens of a phrase are built once rather than replaced twice.
84
85 `one_case_own` is true when the name's OWN words are written in one
86 case AND this token is one of the name's own words -- a maiden clause
87 and any delimited (nickname) content are not, so the fork never
88 reads them either (rules.md#P3): a clause's words are not the
89 name's own words, and appending one must not change how THIS token
90 reads.
91
92 `one_case` is the bare NAME-level fact alone -- P3's own-words
93 span is not this question's business. #516's caps branch reads
94 THIS, not `one_case_own`: a maiden clause's own words are outside
95 `one_case_own`'s span by construction (`i < clause_at` fails for
96 every one of them), so a token past the clause cut reads
97 `one_case_own` as False regardless of whether the WHOLE name is
98 written in one case -- 'JOHN SMITH NEE' flipped 'NEE' to a
99 credential reading with the switch on, one case and all, because
100 `not one_case_own` was true for it purely from being past the
101 clause cut, never from the name's own writing (#516 review round,
102 a second reviewer's finding). `single_letter_connective` below
103 keeps `one_case_own`: that fork is genuinely about the OWN-WORDS
104 span, and reads a clause's word as no evidence on purpose."""
105 lex = state.lexicon
106 tags = set(token.tags)
107 if marker_tag is not None:
108 tags.add(marker_tag)
109 if n in lex.titles:
110 tags.add("vocab:title")
111 if n in lex.given_name_titles:
112 tags.add("vocab:given-title")
113 if suffix_as_written(n, token.text, lex):
114 tags.add("vocab:suffix")
115 if n in lex.suffix_words:
116 tags.add("vocab:suffix-word")
117 if n in lex.suffix_acronyms_ambiguous:
118 tags.add(AMBIGUOUS_ACRONYM_TAG)
119 if n in lex.particles:
120 tags.add("particle")
121 if n in lex.particles_ambiguous:
122 tags.add("vocab:particle-ambiguous")
123 # rules.md#P3: "a single-letter connective reads as an initial
124 # where the writing says so: written as a bare Latin capital in a
125 # name that is not written wholly in one case, or — in a name
126 # written wholly in one case, where nothing says so — where the
127 # letter is one the vocabulary marks as reading both ways"
128 # (#383/#479; history: decisions.md#P3)
129 single_letter_connective = (len(token.text) == 1
130 and token.text.upper() != token.text.lower()
131 and n in lex.conjunctions)
132 if single_letter_connective and one_case_own:
133 # No case evidence, so the vocabulary decides. No namespaced
134 # tag beside it: the emitted ambiguity IS the record of the
135 # decision (mechanisms.md#MARK-DONT-STRIP is satisfied by the
136 # report) -- mechanisms.md#AMBIGUITY-AT-THE-DECISION-SITE says
137 # emit where the branch is taken, not where an ambiguous tag
138 # sits, and a vocab: tag records MEMBERSHIP, not the branch
139 # taken, so it is the wrong shape of record here.
140 if n in lex.conjunctions_ambiguous:
141 tags.add("initial")
142 else:
143 # a bare capital y joins here, where mixed case vetoes it
144 tags.add("conjunction")
145 else:
146 # the mixed-case rule, unchanged. v1's is_conjunction excludes
147 # initials: 'e.' in 'john e. smith' is a middle initial, not
148 # the Spanish conjunction 'e'
149 initial = is_initial(token.text)
150 if n in lex.conjunctions and not initial:
151 tags.add("conjunction")
152 if initial:
153 tags.add("initial")
154 if n in lex.bound_given_names:
155 tags.add("vocab:bound-given")
156 # maiden markers are NOT tagged here: an entry may be a phrase whose
157 # words are not markers on their own, and this function sees one
158 # token with no neighbours. `_vocab.tag_marker_runs` does the whole
159 # field, single words included, so there is one place that decides
160 # it (mechanisms.md#ONE-PREDICATE-PER-QUESTION).
161 # The block a WHOLE-TOKEN vocabulary match skips, and four
162 # readings inside it, in precedence order. The first two are v1's
163 # period-joined derivation (parse_pieces): a token with a period
164 # not at the end, ANY of whose period chunks is a title, is a
165 # title as a whole ('Lt.Gov.', and by the ANY rule 'Mr.Smith');
166 # else ANY suffix chunk makes it a suffix ('JD.CPA'). Title wins
167 # (v1's continue). The third and fourth are 2.4's by-shape halves
168 # (#516) and reach only a token NO vocabulary claimed: a dotted
169 # token of two or more alphabetic chunks, and -- where its switch
170 # is on -- an unlisted all-caps word. They are an `elif` chain
171 # with the dotted one first, so a dotted token never reaches the
172 # caps test and the two shapes stay disjoint.
173 #
174 # Both by-shape branches carry SHAPE_ACRONYM_TAG BESIDE the
175 # membership tag rather than instead of it: `vocab:` records
176 # membership (the roster above) and the peel reads membership,
177 # while the shape tag records that the claim came from the
178 # WRITING. The dotted branch writes it even with its switch off,
179 # which is what lets a declined fork be reported without being
180 # taken. `role is None` guards both: a token already carrying a
181 # role is delimited content, decided by extract's escape and never
182 # at the trailing slot -- 'Bridge (A.B)' is the control that
183 # proves the guard load-bearing (without it the nickname reading
184 # of 'A.B' gains a spurious SUFFIX_OR_NICKNAME report below), and
185 # 'Bridge (1.4)' cannot exercise it, a digit chunk never reaching
186 # the shape verdict at all (period_joined_vocab's alphabetic
187 # gate).
188 if "vocab:title" not in tags and "vocab:suffix" not in tags:
189 derived = period_joined_vocab(token.text, lex)
190 if derived == "title":
191 tags.add("vocab:title")
192 elif derived == "suffix":
193 tags.add("vocab:suffix")
194 elif (derived == "shape" and token.role is None
195 and n not in lex.suffix_acronyms_ambiguous):
196 # This branch and `_vocab.ambiguous_class_candidate` ask
197 # the SAME question twice, of necessity -- `segment` runs
198 # before `classify` and has no tags to read yet -- kept
199 # from drifting by `test_classify.
200 # test_ambiguous_class_candidate_agrees_with_the_tag`
201 # rather than by this sentence alone.
202 #
203 # `n not in suffix_acronyms_ambiguous` is the third guard,
204 # and it is about a LISTED member rather than a role: a
205 # caller may list a dotted entry ('a.b'), which the whole-
206 # token membership test above matches while
207 # `suffix_as_written`'s period-free acronym lookup ('ab')
208 # misses, so the chunk view reached here and called the
209 # word by-shape -- silencing `_pieces.listed_lean`, which
210 # declines wherever SHAPE_ACRONYM_TAG rides, and costing
211 # the caller's own listing its case lean ('Jack A.B.' read
212 # family where 'Jack MA' reads suffix). One frozenset
213 # lookup on a branch only a dotted token reaches; the
214 # shipped ambiguous vocabulary carries no periods, so
215 # nothing default changes.
216 tags.add(SHAPE_ACRONYM_TAG)
217 if state.policy.unlisted_dotted_suffixes:
218 tags.add(AMBIGUOUS_ACRONYM_TAG)
219 elif (state.policy.unlisted_caps_suffixes and token.role is None
220 and caps_shape_candidate(token.text, lex, state.policy,
221 one_case)):
222 # #516's all-caps half, OPT-IN: an unlisted word written
223 # in capitals inside a mixed-case name. The policy conjunct
224 # comes FIRST and stays a plain attribute read -- False by
225 # default, so `caps_shape_candidate` is never CALLED at the
226 # default and sharing its body costs the default nothing
227 # (that is why this half is a call where the dotted branch
228 # above stays inline: the dotted caller has no such cheap
229 # first conjunct to hide behind). The predicate's own
230 # docstring carries the whole-vocabulary roster and what
231 # each measured entry would have cost unfixed.
232 tags.add(SHAPE_ACRONYM_TAG)
233 tags.add(AMBIGUOUS_ACRONYM_TAG)
234 return frozenset(tags)
235
236
237def classify(state: ParseState) -> ParseState:
238 # One fold per token, shared by the marker pass and the vocabulary
239 # tags -- the shape suffix_as_written already asks for ("n is
240 # _normalize(text), passed in so callers normalize once").
241 folded = [_normalize(t.text) for t in state.tokens]
242 marker_tags = tag_marker_runs(state.tokens, state.comma_offsets,
243 state.lexicon.maiden_markers, folded)
244 # rules.md#P3 says a maiden marker, taken as one, and the words it
245 # takes, are not among the name's own words -- so the span and its
246 # clause cut are _pieces.own_words', shared with the site that
247 # needs the same answer two stages earlier (#289/#516). The marker
248 # map goes with it: this stage has already decided which tokens
249 # are run HEADS, so the helper reads that decision rather than
250 # walking the texts again, and classify's answer is the one it was
251 # before the helper existed.
252 own, clause_at = own_words(state.tokens, state.comma_offsets,
253 state.lexicon.maiden_markers, marker_tags)
254 # ONE fact per parse, and it is recorded now (ParseState.one_case):
255 # segment writes it first where a comma form could turn on it, and
256 # a fact two stages decide apart is what recording it prevents
257 # (decisions.md#S2).
258 one_case = state.one_case
259 if one_case is None:
260 one_case = is_one_case(own)
261 # The fork itself must not read a clause's words either, so the
262 # `one_case and ...` argument below repeats `own`'s membership test
263 # per token, and the fork and its emitter then agree with the case
264 # class they consult. No extra frame -- it is one more boolean in a
265 # comprehension that already walks every token.
266 tokens = tuple(
267 dataclasses.replace(
268 t, tags=_tags_for(t, folded[i], state, marker_tags.get(i),
269 one_case_own=one_case and i < clause_at
270 and t.role is None, one_case=one_case))
271 for i, t in enumerate(state.tokens))
272 # Delimited content whose vocabulary cannot settle it: extract's
273 # escape sends an UNambiguous suffix straight through ("(MBA)" ->
274 # suffix) and keeps everything else as a nickname, so an AMBIGUOUS
275 # acronym in there was a coin the parser had to call. Reported here
276 # rather than at the escape itself, which runs before tokenize and
277 # so has no token index to point at.
278 ambiguities = list(state.ambiguities)
279 for i, token in enumerate(tokens):
280 if (token.role is Role.NICKNAME
281 and AMBIGUOUS_ACRONYM_TAG in token.tags):
282 ambiguities.append(PendingAmbiguity(
283 AmbiguityKind.SUFFIX_OR_NICKNAME,
284 f"delimited {token.text!r} is also a post-nominal; read "
285 f"as a nickname rather than a suffix",
286 (i,)))
287 # #383/#479: narrows the fork's own "initial" tag with the same
288 # inputs the fork used, rather than re-deciding from scratch.
289 # Only the casedness test is inherited from the tag --
290 # `is_initial(token.text)` also tags a bare capital "initial"
291 # in the fork's else branch, so the tag alone does not tell
292 # this apart from that.
293 #
294 # Of the two clauses beside it, one is load-bearing and one is
295 # not. `conjunctions_ambiguous` is IMPLIED by the others for a
296 # fresh parse -- the contrapositive of what it looks like:
297 # `is_initial` matches an ASCII capital only, so ONLY the
298 # fork's branch can tag a bare LOWERCASE letter "initial"
299 # ("jose e maria santos" tags a lowercase e), and it does so
300 # only for a conjunctions_ambiguous member. So for a letter of
301 # EITHER case, "initial" plus len 1 implies membership. It is
302 # kept only because `_tags_for` starts from `set(token.tags)`, so
303 # this clause is what keeps the emitter honest if a runner ever
304 # hands classify tokens it did not build; no such path exists
305 # today. `conjunctions` is the clause doing real work: it keeps
306 # an orphan marker (a conjunctions_ambiguous entry no longer in
307 # conjunctions) inert rather than reported (decisions.md#P3).
308 # `i < clause_at and token.role is None` mirrors the fork's own
309 # "own words" test above, for the same reason (rules.md#P3): a
310 # clause's words were never eligible for the fork, so they must
311 # never be eligible to report either. Emitted at the decision
312 # site (mechanisms.md#AMBIGUITY-AT-THE-DECISION-SITE), per
313 # token -- 'e and e' reports twice.
314 if (one_case and "initial" in token.tags
315 and len(token.text) == 1
316 and folded[i] in state.lexicon.conjunctions_ambiguous
317 and folded[i] in state.lexicon.conjunctions
318 and i < clause_at and token.role is None):
319 ambiguities.append(PendingAmbiguity(
320 AmbiguityKind.CONJUNCTION_OR_INITIAL,
321 f"{token.text!r} is both a connective and an initial; "
322 f"the name is written in one case, so nothing marks "
323 f"which, and it is read as an initial",
324 (i,)))
325 # The write rides the replace this stage already makes, so
326 # recording the fact costs no frame of its own.
327 return dataclasses.replace(state, tokens=tokens,
328 ambiguities=tuple(ambiguities),
329 one_case=one_case)