1"""Shared vocabulary predicates for pipeline stages.
2
3Text-level tests used by more than one stage; piece-level ones live
4in _pieces, the sibling layer over tokens-plus-tags. All take
5normalized-or-raw text explicitly and no state, with the departures
6named below.
7
8is_wholly_suffix departs from that shape twice, deliberately. It is
9RUN-level rather than text-level, because the question it answers is
10genuinely about a run: the Ph./D. merge spans two tokens, so no
11per-token predicate composed with all() can express it. And it takes
12the Policy OBJECT, where delimiter_cores takes a pre-extracted
13frozenset so its caller hands in one field rather than the config --
14is_wholly_suffix needs TWO policy fields (lenient_comma_suffixes and
15extra_suffix_delimiters), and threading both past every caller costs
16more than the config parameter saves. Still no state: Policy is frozen
17config, not pipeline state.
18
19maiden_marker_run is run-level for the first of those reasons and not
20the second: a maiden marker may be a PHRASE ('z domu'), so how far one
21reaches is a question about a run of words that no per-word membership
22test can answer, and the vocabulary reaches it as a plain frozenset
23field like every other predicate here.
24
25tag_marker_runs is the third departure, and the reason this module also
26imports _state (WorkToken, comma_bucket): deciding which tokens open a
27maiden-marker RUN needs each token's role and span, not its text alone,
28because a run must not cross a role change or a comma. It moved here
29from classify (#289/#516) so _pieces.own_words -- which may import
30only _state and _vocab (tests/v2/test_layering.py) -- can call the
31SAME function classify does rather than approximating it with a
32text-only walk. Two callers building a map from the same tokens with
33the same function cannot disagree, which a from-scratch approximation
34could (measured: 'ANNA z Nowak, MD' flips the one-case verdict when
35the approximation and the real run-completion test disagree on
36whether 'z' opens a run that completes).
37
38Layering: imports _lexicon, _policy, and _pipeline._state (WorkToken,
39comma_bucket -- read-only, never a whole ParseState).
40"""
41from __future__ import annotations
42
43import functools
44import re
45import unicodedata
46from collections.abc import Callable, Iterable, Sequence
47from typing import Literal
48
49from nameparser._lexicon import (
50 FULL_STOPS, Lexicon, _VOCAB_FIELDS, _normalize,
51)
52from nameparser._policy import (Policy, Script, _JA_SCRIPTS, _NO_INITIALS,
53 _SCRIPT_RANGES, _script_matcher)
54from nameparser._pipeline._state import WorkToken, comma_bucket
55
56# Ported verbatim from v1 (nameparser/config/regexes.py "initial") minus
57# its empty-string alternative -- WorkToken text is never empty. Kept in
58# sync by hand; layering forbids importing the config package here.
59# "Verbatim" is a promise about the PATTERN, not about the predicate:
60# since #320 is_initial is this SHAPE test ANDed with a repertoire test
61# (in_initialless_script, below), so _INITIAL.fullmatch(text) and
62# is_initial(text) are no longer the same question -- '씨.' answers yes
63# to the first and no to the second. Call is_initial; the bare pattern
64# is not the thing to ask. The narrowing lives in the predicate
65# precisely so this copy can stay exactly as verbatim as it ever was
66# -- REGEXES["initial"] is public v1 API and cannot narrow, and the
67# only difference between the two remains the empty alternative noted
68# above (config's `?`), which test_regex_sync splices back in.
69_INITIAL = re.compile(r"^(\w\.|[A-Z])$")
70
71# Ported verbatim from v1 (nameparser/config/regexes.py
72# "period_not_at_end") -- layering forbids the config import; keep in
73# sync by hand.
74_PERIOD_NOT_AT_END = re.compile(r".*\..+$", re.I)
75
76# The fix_phd credential pair ('Ph.' + 'D.' as adjacent tokens), shared
77# by is_wholly_suffix below and group's merge (v1 extracted the
78# credential pre-parse; the predicate and the stage must agree on the
79# pattern).
80PH = re.compile(r"^ph\.?$", re.IGNORECASE)
81D = re.compile(r"^d\.?$", re.IGNORECASE)
82
83# The codepoint table lives in _policy beside Script -- one copy
84# importable from the pipeline and the locale packs alike; everything
85# here DERIVES from it. (single_script's sweep was first written
86# per-char on the _EMOJI_RANGES precedent in _tokenize.py, on the
87# theory that a range test needs no regex; measured at token scale the
88# compiled regex wins by 3-9x, and by roughly 70x on long tokens.)
89# Derived, never hand-written -- even the class construction goes
90# through the shared factory: one wholly-of predicate per script, in
91# the table's key order -- the FIRST-covering-entry rule _classify
92# documents.
93_SCRIPT_MATCHERS: dict[Script, Callable[[str], bool]] = {
94 script: _script_matcher(script, whole=True)
95 for script in _SCRIPT_RANGES
96}
97
98# The whole-token matcher over _policy's _JA_SCRIPTS union, backing
99# effective_script's kana license.
100_wholly_ja = _script_matcher(*_JA_SCRIPTS, whole=True)
101
102# The repertoire half of is_initial (_policy._NO_INITIALS), kept apart
103# from _INITIAL's SHAPE half so the pattern itself stays v1-verbatim
104# and its three copies stay pinned by tests/v2/test_regex_sync.py.
105# contains-any, not whole=True: the shape half has already admitted the
106# trailing period, so the text reaching here is '씨.' rather than '씨'
107# and a wholly-of match would be False for every case this exists for.
108# The second caller, is_title_shaped below, admits two or more
109# characters, so contains-any there means one CJK character anywhere
110# vetoes the whole word -- 'Kim김.' is refused as a title along with
111# '田中.' -- and that is deliberate: a word carrying a script with no
112# abbreviations is not wearing an abbreviation's period.
113in_initialless_script = _script_matcher(*_NO_INITIALS, whole=False)
114
115
116# H2's own shape, text-level: an unlisted period-marked abbreviation,
117# two or more letters, one trailing period. Moved here from _pieces
118# (#289/#516, quality-review finding) so _vocab.name_word_count can
119# ask the SAME question _pieces.is_leading_title asks -- measured
120# divergence before the move: 'Dr. Smith, Ed' (LISTED title 'Dr.') and
121# 'Xyz. Smith, Ed' (UNLISTED, H2-shaped) both read family 'Dr.
122# Smith'/'Xyz. Smith', given 'Ed' at the peel, because is_leading_title
123# reads H2's shape and a bare `folded in lexicon.titles` lookup does
124# not -- but name_word_count's OWN count disagreed: 'Xyz.' counted as
125# a name word where 'Dr.' did not, so 'Xyz. Smith, Ed' alone flipped
126# the comma structure to a credential run
127# (mechanisms.md#ONE-PREDICATE-PER-QUESTION).
128#
129# Ported verbatim from v1 (nameparser/config/regexes.py
130# "period_abbreviation") -- layering forbids the config import; keep
131# in sync by hand (tests/v2/test_regex_sync.py, which reaches this
132# object through `_pieces._PERIOD_ABBREV`, an IMPORT of this one, not
133# a second definition -- the sync test's target name did not move).
134_PERIOD_ABBREV = re.compile(r'^[^\W\d_]{2,}\.$')
135
136
137# rules.md#H2: "an abbreviation opening the part of the name that
138# carries the given name — the whole name, or the part after a
139# family comma — reads as a title even when unlisted"
140# (history: decisions.md#H2)
141def is_title_shaped(text: str) -> bool:
142 """Whether TEXT wears H2's shape alone -- vocabulary-free, the
143 LISTED half being a separate lookup at each caller's own site
144 (is_title_piece/lexicon.titles). `name_word_count` below is the
145 only CALLER; `_pieces.is_leading_title` asks the same question
146 but keeps this body inline, its path being hot (every leading
147 piece of every parse) where this one is cold (comma names only)
148 -- the measured figure lives at that inline copy and is not
149 restated here, two homes for one measurement being how they come
150 to disagree. Touch one, touch both:
151 `test_pieces.test_is_title_shaped_and_is_leading_title_agree`
152 runs both over the union of the two example tables, so the
153 agreement is checked rather than asserted in prose
154 (mechanisms.md#ONE-PREDICATE-PER-QUESTION).
155
156 The shape reads a Latin convention: a period marks an
157 abbreviation. Scripts with no initials have no period
158 abbreviations either (_policy._NO_INITIALS, the #320 veto
159 is_initial carries), so a CJK word wearing a period is a name
160 word, not a title -- a lone '田中.' is the family name (#323).
161 `_PERIOD_ABBREV` stays ASCII-period only: a word wearing '。' never
162 matches it, and the veto is what makes the ASCII spelling agree.
163 ASCII text can carry no _NO_INITIALS character (every range sits
164 above U+3000), so the C-level test declines before the regex
165 search runs -- four frames per unlisted-abbreviation opener per
166 parse, this running four times per piece.
167 """
168 return (bool(_PERIOD_ABBREV.match(text))
169 and (text.isascii() or not in_initialless_script(text)))
170
171
172def is_initial_shaped(text: str) -> bool:
173 """v1's is_an_initial verbatim: the SHAPE half alone -- one word
174 character plus a period, or a bare ASCII capital.
175
176 Callers asking whether a token is STRUCTURALLY part of an initial
177 run want this; callers asking whether it can really stand in for a
178 name want is_initial (#320). The two answers differ only inside
179 _NO_INITIALS scripts, where '씨.' is initial-SHAPED but is not an
180 initial -- see assign's roman-numeral fork, the shape caller, for
181 what picking the wrong one costs."""
182 return bool(_INITIAL.fullmatch(text))
183
184
185# v1 regexes.py "roman_numeral", pinned by tests/v2/test_regex_sync.py.
186_ROMAN = re.compile(r'^(X|IX|IV|V?I{0,3})$', re.I)
187
188
189# rules.md#S2: "a trailing word of the suffix vocabulary reads as a
190# suffix" -- the roman-numeral half of that rule, which the initial
191# veto would otherwise take: V, I and X are suffix vocabulary AND bare
192# capitals, and a bare capital inside a name is a middle initial.
193def is_trailing_numeral_suffix(text: str, preceding: str) -> bool:
194 """assign's roman-numeral fork, shared with group's bound-given
195 reserve (#401): a FINAL single-token piece that is a roman numeral
196 reads as the suffix when the piece before it does not look like
197 part of an initial run. `preceding` is that piece's first token;
198 the callers establish that `text` is last and that a name piece
199 precedes it.
200
201 is_initial_shaped, not is_initial: this asks whether the preceding
202 piece looks like part of an initial run, which is a question
203 about layout, and #320 narrowed the tag to initials that can
204 really stand in for a name. Reading the tag here made '씨.' stop
205 suppressing the fork and cost 'John 씨. V' its family name."""
206 return (_ROMAN.match(text) is not None
207 and not is_initial_shaped(preceding))
208
209
210def is_initial(text: str) -> bool:
211 """'A.' / 'j.' / bare capital -- v1's is_an_initial, narrowed to
212 scripts that HAVE initials (#320). v1's \\w is Unicode-aware and
213 matched CJK too, which made period-written CJK honorifics ('씨.')
214 fail is_suffix_strict -- the veto in _is_suffix_strict_n, NOT the
215 vocabulary: suffix_as_written has no veto, so classify tagged '씨.'
216 'vocab:suffix' either way, and is_suffix_lenient took it either way
217 too. Downstream of that one strict-test No, the glued honorific in
218 a name carrying such a token went unpeeled ('田中さん 様.')."""
219 return is_initial_shaped(text) and not in_initialless_script(text)
220
221
222def is_one_case(texts: Sequence[str]) -> bool:
223 """Whether a name is written wholly in ONE case -- all upper or all
224 lower alike -- and so carries no case EVIDENCE about any letter in
225 it (rules.md#P3, #383/#479). The caller passes the name's OWN
226 words: a maiden marker's clause and any delimited (nickname)
227 content are not among them, and appending one must not flip the
228 reading of words that did not change.
229
230 Mirrors the SHAPE of the comparison the R5 gate in
231 `_render.capitalized` makes, not its SPAN: R5 joins every token,
232 nickname and maiden content included, while classify's caller hands
233 in only the name's own words (rules.md#P3's own-words doctrine, see
234 above) -- so a clause-bearing name can be one-case to this function
235 and mixed to R5 (measured: `'JUAN GARCIA Y LOPEZ née Jones'` is one
236 case here, mixed there). Not shared by import today -- render is a
237 layer this module does not reach into, and #492 is where the two
238 spans are reconciled if they ever need to be.
239
240 `Sequence`, not `Iterable`: the caller passes a list it already
241 built rather than a fresh generator, so `is_one_case` costs one
242 profiler frame per parse rather than one per token (#475).
243
244 A CASELESS script answers True, harmlessly: `'محمد و علي'.upper()`
245 is the string itself, so the comparison holds, and the only caller
246 also requires a token whose own `upper()` and `lower()` differ --
247 which a caseless letter's never do. So a caseless name never
248 reaches the decision this gates, and "one case" is the honest
249 verdict for text that has only one.
250 """
251 joined = " ".join(texts)
252 return joined in (joined.upper(), joined.lower())
253
254
255#: Which way the WRITING leans for a member of the ambiguous
256#: credential class; None is no lean at all. Named so the two
257#: predicates that answer it -- `ambiguous_lean` here and
258#: `_pieces.listed_lean`, which wraps it -- carry the same three-value
259#: type rather than a bare `str`. Under mypy's `strict_equality` that
260#: makes a misspelled comparison (`== "credentail"`) an error at the
261#: two call sites that branch on the answer, where a `str` return
262#: leaves it silently False forever -- which is what the value is FOR:
263#: `_pieces.peel_trailing` peels or declines on it.
264#: `period_joined_vocab`'s own verdict is spelled inline for the same
265#: reason; it has one caller shape and no wrapper to keep in step.
266Lean = Literal["credential", "name"]
267
268
269# #289's credential lean: rules.md#S2 records the words-to-spare count
270# today; this predicate is the WRITING evidence that decides ahead of
271# it, in a mixed-case name only (history: decisions.md#S2).
272def ambiguous_lean(text: str, one_case: bool) -> Lean | None:
273 """Which way the WRITING leans for a member of the ambiguous
274 credential set: "credential", "name", or None for no lean at all.
275
276 None is the fall-through to rules.md#S2's count, and it is the
277 answer for three inputs. A name written wholly in one case says
278 nothing about any word in it. A member written wholly in lower
279 case says nothing either -- lower is how most of a mixed-case
280 name is written, so it is not a contrast. And a CASELESS token
281 ('씨', '毛') can be written against nothing, so neither test
282 fires and the count decides, which is what keeps this rule out of
283 a caseless script by construction.
284
285 `one_case` is the recorded fact (ParseState.one_case), taken over
286 the name's OWN words -- handed in rather than recomputed, because
287 three sites read this and a fact two of them derived apart is the
288 bug the field exists to prevent
289 (mechanisms.md#ONE-PREDICATE-PER-QUESTION).
290
291 The caller decides MEMBERSHIP: this answers only about the
292 writing. Periods do not disturb either test, '.' having no case,
293 so 'MA.' leans as 'MA' does -- the lean reads CASE, not periods,
294 and the dotted spelling is the vocabulary's own question
295 (suffix_as_written).
296 """
297 if one_case:
298 return None
299 if text.upper() == text.lower(): # caseless: no contrast to read
300 return None
301 if text.isupper():
302 return "credential"
303 if text.islower():
304 return None
305 return "name"
306
307
308_DOTTED = re.compile(r"(?:[^\W\d_]\.)+")
309
310
311def _dotted(text: str) -> bool:
312 """Written with its periods: one after each letter ('M.A.',
313 'J.D.'), the acronym's own spelling. A single trailing period
314 ('Ma.', 'Ed.', 'Ms.') is the abbreviation shape any word can wear
315 -- the honorific's, a name's -- and is not the gate's "written
316 with periods" (rules.md#S2). Until #296's review the gate was
317 "any period", and 'Smith, Ms.' passed it as the degree."""
318 return _DOTTED.fullmatch(text) is not None
319
320
321def suffix_as_written(n: str, text: str, lexicon: Lexicon) -> bool:
322 """Counts as a suffix as written, with NO initial veto (the veto
323 differs by caller): unambiguous suffix vocabulary, or an ambiguous
324 acronym written with periods ('M.A.' yes, 'Ma' no). `n` is
325 _normalize(text), passed in so callers normalize once.
326
327 Single source for classify's "vocab:suffix" tag and the segment/
328 assign predicates. The ambiguous subset is EXCLUDED from the plain
329 membership test: in the real data suffix_acronyms_ambiguous is a
330 subset of suffix_acronyms, and without the exclusion the period
331 gate is dead code (bare 'Ed'/'Jd' would silently become suffixes).
332 """
333 # acronyms may be written with periods ('M.B.A.'): the ACRONYM
334 # membership alone uses the period-free form (v1's is_suffix
335 # removed periods only for the suffix_acronyms test); suffix WORDS
336 # match on the plain normalized form
337 a = n.replace(".", "")
338 if a in lexicon.suffix_acronyms_ambiguous and _dotted(text):
339 return True
340 return (a in lexicon.suffix_acronyms
341 and a not in lexicon.suffix_acronyms_ambiguous) \
342 or n in lexicon.suffix_words
343
344
345def _is_suffix_strict_n(n: str, text: str, lexicon: Lexicon) -> bool:
346 if is_initial(text):
347 # period-written ambiguous acronyms are exempt from the veto
348 return _dotted(text) and \
349 n.replace(".", "") in lexicon.suffix_acronyms_ambiguous
350 return suffix_as_written(n, text, lexicon)
351
352
353def is_suffix_strict(text: str, lexicon: Lexicon) -> bool:
354 """v1's is_suffix: suffix_as_written with the initial veto ('V.' in
355 'John V. Smith' is a middle initial, not roman five)."""
356 return _is_suffix_strict_n(_normalize(text), text, lexicon)
357
358
359def is_suffix_lenient(text: str, lexicon: Lexicon) -> bool:
360 """v1's is_suffix_lenient: suffix_words accepted unconditionally,
361 bypassing the initial veto -- only safe in unambiguous positions
362 (after a comma)."""
363 n = _normalize(text)
364 return n in lexicon.suffix_words \
365 or _is_suffix_strict_n(n, text, lexicon)
366
367
368def delimiter_cores(policy_delimiters: frozenset[str]) -> frozenset[str]:
369 """Configured suffix delimiters with surrounding whitespace
370 stripped: ' - ' -> '-'. Whitespace-padded delimiters surface as
371 standalone tokens; the stripped core is what tokenize produced."""
372 return frozenset(d.strip() for d in policy_delimiters if d.strip())
373
374
375def splits_into_suffixes(text: str, cores: frozenset[str],
376 lexicon: Lexicon) -> bool:
377 """v1 expand_suffix_delimiter parity for delimiters WITHOUT
378 whitespace ('RN/CRNA' with '/'): the token counts as a suffix when
379 some core splits it into >=2 non-empty parts that are all suffixes.
380 The token text is never rewritten (anti-#100): it takes Role.SUFFIX
381 whole, which renders 'RN/CRNA' where v1 rendered 'RN, CRNA' -- the
382 documented divergence, release-log classified."""
383 for core in cores:
384 if core in text:
385 parts = [part for part in text.split(core) if part]
386 if len(parts) >= 2 and all(
387 is_suffix_lenient(part, lexicon) for part in parts):
388 return True
389 return False
390
391
392# rules.md#S3: "a word with interior periods reads as a suffix when
393# any of its period-separated chunks is suffix vocabulary — except
394# where every chunk the vocabulary matches is a single ASCII
395# character, the roman numerals and the lone digit the vocabulary
396# lists, which are about generations rather than credentials"
397def period_joined_vocab(
398 text: str, lexicon: Lexicon,
399) -> Literal["title", "suffix", "shape"] | None:
400 """v1's parse_pieces derivation for interior-period tokens
401 ('Lt.Gov.', 'Msc.Ed.', and by the ANY rule 'Mr.Smith'): ANY title
402 chunk makes the token a title (checked first, v1's continue); else
403 ANY suffix chunk makes it a suffix. Chunk-level suffix membership
404 is v1's is_suffix: bare ambiguous acronyms COUNT ('Msc.Ed.'
405 derives via 'ed') -- the ambiguous period-gate applies to whole
406 tokens only. Returns "title", "suffix", "shape", or None.
407
408 "shape" is the third verdict and it is the absence of a claim
409 worth acting on (#516): two or more chunks, EVERY one of them
410 alphabetic, and nothing the vocabulary matches except -- possibly
411 -- chunks that are a SINGLE ASCII CHARACTER. That exception is the
412 roman-numeral accident retired narrowly: measured 2026-09-15 the
413 one-character suffix vocabulary is {'2', 'i', 'v'} plus the glued
414 CJK honorific tails, so 'John Smith R.A.I.' was reading as a
415 generational suffix off the chunk 'i'. CHARACTER rather than
416 LETTER because '2' is a digit and is in the set; ASCII because
417 '씨' is the single character that must KEEP its claim ('J.씨'). A
418 multi-character match still reads as it always did, so 'Msc.Ed.',
419 'JD.CPA' and 'Lt.Gov.' are untouched -- the WIDE retirement, where
420 any chunk match yields to the shape, was measured and rejected: it
421 costs 'Doe, John Msc.Ed.' a real credential and re-routes the
422 honorific peel of '김민준씨, J.씨' for nothing this design wants
423 (decisions.md#S2).
424
425 The ALPHABETIC requirement is a SEPARATE, narrower gate on the
426 shape verdict alone (#516 review round, decided by the
427 orchestrator): a bare digit chunk is not an acronym letter by any
428 reading, so 'Smith, 1.4' and 'John Smith 1.4' do not join the
429 class by shape -- the only PROTECTED digit control this design
430 had was the delimited 'Bridge (1.4)', and 'Smith, 1.4' and its
431 no-comma twin were the gap that control did not cover, since
432 delimited content is excluded by a different mechanism (extract's
433 escape) and digits reach THIS detector, never that one.
434 '.isalpha()' is asked of EVERY chunk, not just the matched ones,
435 so a mixed token like 'X.Y.2.' is not shape-admitted either -- one
436 non-letter chunk is enough to say the writing is not spelling an
437 acronym. Chunks the vocabulary itself matches are already
438 alphabetic in the shipped lexicon, so this narrows only the
439 previously-unclaimed "shape" answer, never the "suffix" one.
440
441 The INITIALLESS-SCRIPT guard is a second, independent narrowing of
442 the same verdict, on the same reasoning `is_title_shaped` already
443 gives its own period-abbreviation inference (#323): a script with
444 no period abbreviations at all has nothing for interior periods to
445 ABBREVIATE, so a CJK word glued into period-separated single
446 characters is not spelling an acronym either -- 'John Smith
447 田.中.' and '김 민준 이.박.' stayed family at every release before
448 this gate existed, and would otherwise have joined the ambiguous
449 class by shape for the first time (measured regression, #516
450 review round).
451
452 What a "shape" verdict MEANS is the caller's question, not this
453 one's: classify writes the tag, and Policy.unlisted_dotted_suffixes
454 decides whether it is class membership.
455 """
456 if not _PERIOD_NOT_AT_END.match(text):
457 return None
458 chunks = [_normalize(c) for c in text.split(".") if c]
459 if any(c in lexicon.titles for c in chunks):
460 return "title"
461 matched = [c for c in chunks
462 if c in lexicon.suffix_acronyms or c in lexicon.suffix_words]
463 if matched and not all(len(c) == 1 and c.isascii() for c in matched):
464 return "suffix"
465 if (len(chunks) >= 2 and all(c.isalpha() for c in chunks)
466 and (text.isascii() or not in_initialless_script(text))):
467 return "shape"
468 return None
469
470
471def ambiguous_class_member(text: str, lexicon: Lexicon) -> bool:
472 """Whether TEXT is a member of the LISTED ambiguous credential
473 class, ignoring case and policy entirely (#289/#516).
474
475 The case-INDEPENDENT half of the comma form's own candidate test
476 (`ambiguous_class_candidate`, below, which ALSO admits a by-shape
477 member where Policy allows it) and of `is_wholly_suffix`'s
478 credential-lean disjunct: membership by vocabulary never needs the
479 case fact, only the lean does. That split is what lets the comma
480 form's lazy gate ask membership FIRST and pay for `one_case` only
481 where this says yes (measured regression, #289/#516: forcing the
482 fact first ran `own_words` -> `tag_marker_runs` for every
483 single-token-after-the-comma name, none of them able to reach the
484 class at all).
485
486 Membership is the listed set, bare: a whole-token vocabulary match
487 is not in this class at all, being settled ('M.A.', 'Ph.D.',
488 'A.B.C.').
489
490 Cheaper than `suffix_as_written(n, text, lexicon) or ...` would be
491 here, and provably the same answer for UNDOTTED text: the Lexicon
492 invariant that an ambiguous acronym is never also a suffix WORD
493 (`__post_init__`'s gate_bypassed check) and is never counted
494 without periods once it IS one (`suffix_as_written`'s own
495 exclusion) together kill that predicate's two disjuncts once text
496 is known to hold no period. One frame (`_normalize`) on the common
497 path that never reaches the class, where the general predicate
498 cost at least two (mechanisms.md#ONE-PREDICATE-PER-QUESTION's cost
499 clause).
500
501 The '.' gate here is DELIBERATELY STRICTER than S2's own
502 dotted-form test (`_dotted`, a period after EACH letter): '.'
503 anywhere excludes membership, so a single TRAILING period ('MA.',
504 'Ed.') is excluded here even though the LEAN still reads it as
505 the bare acronym's case ('Smith, MA.' -> suffix 'MA.', measured).
506 Not a contradiction: the two questions this answers -- the
507 comma-form CANDIDATE test below and the credential-lean disjunct
508 -- are never asked of those shapes, the TAG path
509 (`vocab:suffix-ambiguous`, read directly by the trailing peel and
510 the post-comma slot) and H2's leading-title shape test already
511 carrying them where they need to go.
512 """
513 if "." in text:
514 return False
515 return _normalize(text) in lexicon.suffix_acronyms_ambiguous
516
517
518# #516's all-caps half, ONE PREDICATE for the shape test and its
519# WHOLE-VOCABULARY exclusion, shared by the three sites that each
520# needed the identical question answered (classify's tag emission,
521# `_segment.py`'s multi-token run test, and this module's own unit
522# tests), where it had been spelled three times over (quality-review
523# finding). The usual objection to sharing -- a call costing every
524# default-policy parse a frame it cannot use -- does not apply: every
525# caller's own first conjunct is `policy.unlisted_caps_suffixes`,
526# False by default, so neither this call nor the loop inside it is
527# ever reached at the default (confirmed against the 412/449 frame
528# band and the default comma harness).
529def caps_shape_candidate(text: str, lexicon: Lexicon, policy: Policy,
530 one_case: bool | None) -> bool:
531 """Whether TEXT is an UNLISTED all-caps credential candidate: two
532 or more alphabetic characters, no period (a period fails
533 `isalpha()` outright, which is what excludes the dotted spelling
534 here for free -- no separate '.' test needed), written in a name
535 `one_case` says is mixed (`one_case is False`; `None`, "not
536 established", declines exactly as if the name were one case).
537
538 UNLISTED means in NO wordlist at all, not merely "no whole-token
539 suffix vocabulary", and the roster is `_lexicon._VOCAB_FIELDS`
540 itself rather than a list written out here -- checked directly
541 against the lexicon rather than through a caller's tags (this
542 function's own callers have none to read, `segment` running before
543 `classify`). A hand-written roster is a second place to remember,
544 and it had already gone wrong: it named eleven of the thirteen
545 fields, leaving `surnames` and `honorific_tails` out, so
546 `Lexicon.default().add(surnames={"dupont"})` still read
547 `Jean Pierre DUPONT` as suffix `DUPONT` -- a caller listing a word
548 as a SURNAME and getting it read as a credential is this switch's
549 own worst failure, arriving through the one wordlist that says
550 "this is a family name" (review round, #289/#516;
551 `honorific_tails` was already excluded transitively, being a
552 subset of `suffix_words` by Lexicon invariant, and joins the
553 roster for completeness rather than for a behavior change).
554 Measured, #516 review rounds: particles, ambiguous particles,
555 conjunctions, bound-given heads, a maiden marker ('NEE'/'GEB') and
556 a title all join the shape by capitalization alone if their field
557 is left unchecked.
558
559 `suffix_acronyms_ambiguous` is in the roster, which also makes
560 this predicate stand in for `ambiguous_class_member` wherever a
561 caller needs "and not already a LISTED member" (undotted text's
562 only path into that function is the identical membership test) --
563 `_segment.py`'s run test relies on exactly that rather than
564 calling both.
565 """
566 if not (policy.unlisted_caps_suffixes and one_case is False
567 and len(text) >= 2 and text.isalpha() and text.isupper()):
568 return False
569 n = _normalize(text)
570 for field in _VOCAB_FIELDS:
571 if n in getattr(lexicon, field):
572 return False
573 return True
574
575
576# The comma form's own candidate test (rules.md#C1, decisions.md#S2).
577def ambiguous_class_candidate(text: str, lexicon: Lexicon,
578 policy: Policy) -> bool:
579 """Whether TEXT is a CANDIDATE for the ambiguous credential class
580 at the comma form's own structure decision (`_segment.py`): the
581 LISTED half (`ambiguous_class_member`, case-free) OR, where Policy
582 admits it, the SHAPE an unlisted dotted token wears
583 (`period_joined_vocab`'s third verdict, #516).
584
585 Case-free throughout, and the CAPS half of the class is
586 deliberately not here: its real site is `_segment.py`'s
587 multi-token run test, which calls `caps_shape_candidate` directly
588 because the run is a property of that shape alone. A second route
589 through here existed briefly, behind an optional `one_case`
590 parameter no production caller ever passed -- `segment` has
591 nothing to hand in at its single-token test -- so the branch
592 answered False for every name the library parsed and only the
593 unit tests reached it (review-round finding, #289/#516). A dead
594 second route is a place for the two to disagree, not a
595 convenience.
596
597 A period anywhere is the gate for even ASKING the shape question,
598 checked before the shape's own two calls: `period_joined_vocab`
599 already declines a period-free text, but a Python-level call is
600 not free and a comma name's post-comma part usually has no period
601 ("Smith, John") -- measured regression, #516 review round, fixed
602 by hoisting the same cheap substring test `ambiguous_class_member`
603 already makes for its own reason. Where a period IS present,
604 `suffix_as_written` runs only after `period_joined_vocab` says
605 "shape": a dotted whole-token match with no single-chunk
606 vocabulary hit of its own ('A.B.C.', via 'abc') would otherwise
607 read "shape" from this function's chunk-level view alone,
608 oblivious to the WHOLE-token match `suffix_as_written` already
609 settled -- the same precedence classify's own tag order gives it
610 (`vocab:suffix` is set before `period_joined_vocab` is even
611 consulted).
612
613 A LISTED member spelled with its periods is excluded from the
614 shape branch by the same test classify's own shape branch makes:
615 a caller who puts a dotted entry in `suffix_acronyms_ambiguous`
616 has said the word is a listed member of this class, and reading
617 it by shape instead loses the case lean the listing asks for
618 (review-round finding, #289/#516 -- `Jack A.B.` with 'a.b' listed).
619 It cannot change the answer for the shipped vocabulary, whose
620 ambiguous entries carry no period at all.
621
622 This function and classify's tag emission (`_tags_for`'s
623 `derived == "shape"` branch) still ask the SAME question twice, of
624 necessity: `segment` runs before `classify` and has no tags to
625 read yet, so the two stages cannot share the call. Kept from
626 drifting by
627 `test_classify.test_ambiguous_class_candidate_agrees_with_the_tag`,
628 which asks both of the same texts, rather than by a sentence
629 alone.
630 """
631 if ambiguous_class_member(text, lexicon):
632 return True
633 if "." in text:
634 # `_normalize` stays behind the shape verdict, where it always
635 # was: it is a call, and a dotted post-comma token that is not
636 # acronym-shaped at all ('Jr.') must not pay for it.
637 if not (policy.unlisted_dotted_suffixes
638 and period_joined_vocab(text, lexicon) == "shape"):
639 return False
640 n = _normalize(text)
641 return (n not in lexicon.suffix_acronyms_ambiguous
642 and not suffix_as_written(n, text, lexicon))
643 return False
644
645
646def name_word_count(texts: Sequence[str], lexicon: Lexicon,
647 policy: Policy) -> int:
648 """How many of these texts are NAME words -- not suffix
649 vocabulary, not title vocabulary.
650
651 rules.md#C1's count for the ambiguous class, and it is of names
652 rather than of words because 'Smith Jr., MA' is two tokens and one
653 name: counting tokens there flips the structure and hands the
654 family to `given`, which no reading of that string wants
655 (decisions.md#S2). The suffix half asks the POLICY-selected
656 predicate, the same one is_wholly_suffix asks, so the two agree
657 about what a suffix word is; the title half asks BOTH the listed
658 lookup and H2's shape test (`is_title_shaped`) -- a bare
659 `lexicon.titles` lookup here read 'Xyz.' as a name word where the
660 leading peel reads it as a title, and that divergence is recorded
661 once, at `is_title_shaped` itself
662 (mechanisms.md#ONE-PREDICATE-PER-QUESTION).
663 """
664 predicate = (is_suffix_lenient if policy.lenient_comma_suffixes
665 else is_suffix_strict)
666 n = 0
667 for text in texts:
668 if (predicate(text, lexicon) or _normalize(text) in lexicon.titles
669 or is_title_shaped(text)):
670 continue
671 n += 1
672 return n
673
674
675def is_wholly_suffix(texts: Sequence[str], lexicon: Lexicon,
676 policy: Policy, one_case: bool | None = None) -> bool:
677 """Every token in a RUN counts as a suffix -- segment's
678 suffix-comma test, lifted out of it so the peel can ask the same
679 question (#319).
680
681 NOT the plural of _script_segment._is_post_nominal, which asks
682 is_suffix_strict per token. This asks the POLICY-selected predicate
683 (lenient by default), plus period_joined_vocab, delimiter
684 transparency and the Ph./D. merge. 'V.' is the input that tells
685 them apart: it satisfies this predicate but is not a post-nominal
686 -- and reading one for the other IS the #319 bug.
687
688 An EMPTY run is False, not vacuously True: v1's suffix-comma
689 detection fails on an empty parts[1] ('John Smith,, MD' is a
690 family-comma parse). The 'wholly' idiom agrees -- _script_matcher's
691 whole=True requires non-empty too -- which is why the name is that
692 one rather than all_suffixes, where Python's all([]) would promise
693 the opposite.
694
695 An adjacent Ph./D. pair counts as ONE unit (v1's fix_phd extracted
696 the credential pre-parse, so 'Smith, Ph. D.' read as suffix-comma);
697 keep in sync with group's _PH/_D merge.
698
699 `one_case` is ParseState.one_case, and it admits the LEAN: a bare
700 ambiguous acronym written in capitals inside a mixed-case name is
701 a credential here, so 'Steven Hardman, MD, DO, DDS' reads its
702 third segment as the credential run it is (#289). None -- the
703 default, and what every caller with no state to ask has -- reads
704 as no lean and is this predicate's behavior in every release
705 before 2.4.
706
707 Neither Policy.unlisted_dotted_suffixes NOR
708 Policy.unlisted_caps_suffixes reaches this predicate: admitting a
709 by-shape member here unconditionally (an earlier version of this
710 docstring described exactly that, for the dotted half alone)
711 bypassed both the lean AND the NAME-word count, and combined with
712 C1's own legacy TOKEN-count disjunct in `_segment.py`
713 (`suffixy(groups[1]) and len(groups[0]) > 1`) it flipped
714 'Smith Jr., A.B.' to given 'Smith', suffix 'Jr., A.B.' with a
715 self-contradicting report ("holds 1 name words, so it is read as
716 a credential run") -- proved by mutation testing to be otherwise
717 unreached: nothing but this predicate's own two unit tests
718 depended on it, and 'John Smith, A.B.' still flips correctly
719 through `pre_comma_names >= 2` alone (#516 review round). Neither
720 by-shape class, dotted or caps, reaches the comma form through
721 this predicate at all -- only through `_vocab.
722 ambiguous_class_candidate`, which segment's structure decision and
723 report both already consult (the caps half as a RUN test over it,
724 #516's second review round).
725 """
726 if not texts:
727 return False
728 predicate = (is_suffix_lenient if policy.lenient_comma_suffixes
729 else is_suffix_strict)
730 # v1 expand_suffix_delimiter parity (#206): a configured delimiter
731 # is TRANSPARENT in the all-suffix tests -- v1 split the part string
732 # on the delimiter before checking, so the delimiter never counted
733 cores = delimiter_cores(policy.extra_suffix_delimiters)
734
735 def counts_as_suffix(text: str) -> bool:
736 if text in cores:
737 return True
738 if (one_case is not None
739 and ambiguous_class_member(text, lexicon)
740 and ambiguous_lean(text, one_case) == "credential"):
741 return True
742 return (predicate(text, lexicon)
743 or period_joined_vocab(text, lexicon) == "suffix"
744 or (bool(cores) and splits_into_suffixes(text, cores, lexicon)))
745
746 merged = list(texts)
747 k = 0
748 while k < len(merged) - 1:
749 if PH.fullmatch(merged[k]) and D.fullmatch(merged[k + 1]):
750 merged[k:k + 2] = ["phd"]
751 else:
752 k += 1
753 return all(counts_as_suffix(t) for t in merged)
754
755
756# The two derived views of a marker vocabulary, cached per-vocabulary
757# on _script_segment._longest_entry's precedent -- same function shape
758# (a scalar derived from a vocabulary frozenset), same key space, and
759# its reasoning for maxsize=16 carries over verbatim: a process holds
760# the default vocabulary plus one per constructed pack parser, so 16
761# bounds many-lexicon churn without ever evicting in normal use.
762# (NOT _extract._delimiter_chars' precedent, which these once cited:
763# that one is consulted once per parse, so it says nothing about a
764# lookup on the per-token path.) Keying on the frozenset costs a
765# cached hash, not a sweep of its contents.
766@functools.lru_cache(maxsize=16)
767def _longest_marker(markers: frozenset[str]) -> int:
768 """How many words the longest entry of `markers` spans -- the
769 lookahead bound, computed from the vocabulary rather than fixed at a
770 literal so a caller's four-word entry works and an all-single-word
771 set leaves the common path at one lookup. Entries are stored
772 space-joined with single separators, so the space count IS the word
773 count."""
774 return max((entry.count(" ") + 1 for entry in markers), default=0)
775
776
777@functools.lru_cache(maxsize=16)
778def _marker_heads(markers: frozenset[str]) -> frozenset[str]:
779 """The first word of every entry -- the set a run can possibly open
780 with. Entries are already stored per-word folded, so an entry's
781 first word is the same fold the lookup builds."""
782 return frozenset(entry.split(" ", 1)[0] for entry in markers)
783
784
785def maiden_marker_head(n: str, markers: frozenset[str]) -> bool:
786 """Could a maiden marker run START here? `n` is _normalize(word),
787 passed in so callers fold once -- suffix_as_written's shape exactly
788 ("`n` is `_normalize(text)`, passed in so callers normalize once"),
789 and with no raw-text sibling for the same reason it has none: every
790 caller is on the per-token path and has the fold in hand already.
791
792 A SUPERSET test. True means only that some entry opens with this
793 word, never that a run matches -- maiden_marker_run is the answer,
794 and it calls this function, so the two cannot drift. Exported
795 because a caller scanning a whole token stream needs to know
796 whether assembling a candidate sequence is worth doing at all, and
797 for almost every token it is not: _classify's pass would otherwise
798 walk structural boundaries once per token to build a lookahead the
799 predicate discards on this very test.
800 """
801 return n in _marker_heads(markers)
802
803
804def maiden_marker_run(words: Sequence[str], markers: frozenset[str]) -> int:
805 """How many of `words` a maiden marker claims, longest first; 0 for
806 none.
807
808 Phrases are stored space-joined and per-word normalized (_title_key's
809 storage rule), so the key is rebuilt the same way here -- normalizing
810 the joined phrase instead would leave interior periods.
811
812 `words` must be words that stand TOGETHER -- one clause's, or one
813 segment's. This answers only what the vocabulary says about the
814 sequence it is handed; whether a sequence is a sequence is the
815 caller's, and classify's contiguity rule is where that is decided
816 for the token stream.
817
818 Longest first, so a caller configuring both 'geb' and 'geb von' gets
819 the phrase where it matches and the bare word everywhere else. The
820 one answer to "does a marker start here, and where does it end": the
821 stages that can call it do (classify over token texts, extract over
822 a clause's whitespace words), and the stage that runs after classify
823 reads the tags classify recorded instead
824 (mechanisms.md#ONE-PREDICATE-PER-QUESTION).
825 """
826 # Fast path first: no entry opens with this word, so no length can
827 # match. It cannot hide a match -- every key the loop builds opens
828 # with _normalize(words[0]) unless that word folds away, and a
829 # folded-away first word fails the n-word test below for every
830 # n > 1 and is not a stored entry for n == 1.
831 if not words:
832 return 0
833 head = _normalize(words[0])
834 if not maiden_marker_head(head, markers):
835 return 0
836 cap = min(len(words), _longest_marker(markers))
837 # Fold each word ONCE, then key from a prefix. _title_key(words[:n])
838 # per candidate length re-folds the whole prefix every time, which
839 # is quadratic in cap and makes a one-word hit cost more than a
840 # two-word one -- the longest key is always built and discarded
841 # first, and all but one shipped entry is a single word. The join
842 # below IS _title_key's body over pre-folded words (per-word fold,
843 # empties dropped, space-joined); test_vocab pins that the two
844 # agree, since this is a copy of a fold whose definition lives in
845 # _lexicon.
846 folded = [head] + [_normalize(w) for w in words[1:cap]]
847 for n in range(cap, 0, -1):
848 key = " ".join(filter(None, folded[:n]))
849 # a word that folds away is DROPPED from the key, so 'née'
850 # followed by a lone '.' would key as 'née' and a one-word
851 # marker would claim the period as part of its run. n words in,
852 # n words out: anything else is not this key's n-word phrase.
853 if key.count(" ") == n - 1 and key in markers:
854 return n
855 return 0
856
857
858def tag_marker_runs(tokens: Sequence[WorkToken], comma_offsets: Sequence[int],
859 markers: frozenset[str],
860 folded: Sequence[str] | None = None) -> dict[int, str]:
861 """Which tokens are maiden marker runs: index -> "vocab:maiden-marker"
862 for a run's head, "vocab:maiden-marker-cont" for the rest.
863
864 Returns the decision rather than rewriting the tokens: a caller
865 that writes tags into the one pass that builds them (classify)
866 consults this map rather than deciding a run twice, so a marker
867 token is never replaced twice; a caller that never writes tags at
868 all (own_words) reads the same map for its span instead.
869
870 Moved here from classify (#289/#516) so a caller outside classify
871 -- _pieces.own_words, when it has no marker map yet -- can build
872 the SAME map classify would, rather than approximating it with a
873 text-only head test. `_pieces.py` may import only _state and
874 _vocab (tests/v2/test_layering.py), so the one shared function
875 both sites call has to live here, in the layer both can reach.
876
877 `folded` is _normalize per token; classify has already built it
878 for the vocabulary-tag pass and hands it over so this pass costs
879 no second fold, and a caller with only tokens (own_words' self-
880 built path) omits it and pays the fold here instead -- paid only
881 on that path, never on classify's, so the reference name's frame
882 count is unchanged (#289/#516).
883
884 The one sequence pass in this stage, and it has to be one: a marker
885 entry may be a PHRASE whose words are not markers individually
886 ('z', 'domu'), so no per-token membership test can find it.
887 Left to right, longest first at each position, then skip past what
888 the run claimed -- a second marker cannot start inside the first.
889
890 This is where the tag is DECIDED for the stages that read it
891 afterwards. group runs later and asks its questions of these tags
892 rather than re-deriving the run (the recorded-answer half of
893 mechanisms.md#ONE-PREDICATE-PER-QUESTION); extract runs EARLIER,
894 before tokens exist, so it calls the predicate itself over the
895 clause's whitespace words.
896
897 A tagged run is structurally contiguous, and the test is
898 one-directional: a role change IS a clause edge, so no run spans
899 one, but not every clause edge is a role change -- two ADJACENT
900 clauses of the same role are indistinguishable here, and
901 'Jane (z) (domu) Jones' does tag a run across them. Both consumers
902 refuse that run for reasons of their own (the piece walk never sees
903 role-bearing tokens at all; the clause drop is scoped to one
904 clause's span), so no reading depends on it today, and the claim
905 this pass can honestly make is the weaker one. What it does
906 guarantee is what _group._marker_run_pieces needs: a run inside the
907 MAIN stream stays inside one segment. Without it this pass walked
908 the whole span-sorted stream while group walked one segment --
909 _segment keeps only role-less tokens and buckets them by the commas
910 before them -- so a run half inside a bracketed clause was tagged
911 whole and consumed as a proper PREFIX of itself, and
912 'Anna z (domu) Nowak' read family 'Anna', maiden 'Nowak': the bare
913 preposition eating the name, which is the exact damage the phrase
914 entry exists to prevent. Refusing to tag such a run is the fix;
915 truncating it instead would hand M2 the same wrong prefix one word
916 shorter.
917 """
918 # the lookahead the vocabulary actually needs; 0 for an empty set,
919 # which skips the pass entirely
920 cap = _longest_marker(markers)
921 if not cap:
922 return {}
923 if folded is None:
924 folded = [_normalize(t.text) for t in tokens]
925 n_tokens = len(tokens)
926 # Deferred, not computed up front: only the contiguity walk reads
927 # it, only a phrase vocabulary runs that walk, and only at a token
928 # that opens an entry -- so a single-word vocabulary, and a
929 # phrase vocabulary over a name holding no marker, never pay the
930 # sweep at all.
931 buckets: list[int] | None = None
932 tags: dict[int, str] = {}
933 i = 0
934 while i < n_tokens:
935 # The predicate's own head test first, over the fold the caller
936 # already has: almost no token opens any entry, and for those
937 # there is nothing to assemble. Same function maiden_marker_run
938 # consults, so a token skipped here is one it would refuse.
939 if not maiden_marker_head(folded[i], markers):
940 i += 1
941 continue
942 # Bound the lookahead at the first structural boundary, so the
943 # predicate is asked over the words that could form one run and
944 # answers longest-first WITHIN them -- a two-word entry refused
945 # at a clause edge still leaves a one-word entry starting there
946 # free to match.
947 limit = 1
948 if cap > 1:
949 if buckets is None:
950 buckets = [comma_bucket(t.span.start, comma_offsets)
951 for t in tokens]
952 role, bucket = tokens[i].role, buckets[i]
953 while (limit < cap and i + limit < n_tokens
954 and tokens[i + limit].role is role
955 and buckets[i + limit] == bucket):
956 limit += 1
957 run = maiden_marker_run(
958 [tokens[k].text for k in range(i, i + limit)], markers)
959 if not run:
960 i += 1
961 continue
962 tags[i] = "vocab:maiden-marker"
963 for k in range(i + 1, i + run):
964 tags[k] = "vocab:maiden-marker-cont"
965 i += run
966 return tags
967
968
969def _normalized_for_script(text: str) -> str | None:
970 """The guard AND the two normalizations single_script and
971 effective_script's license path both need, single-sourced so they
972 cannot drift: trailing full stops are dropped (FULL_STOPS, #323),
973 then None for the two shapes neither ever classifies (nothing
974 left, and the common all-ASCII Latin token -- skipped before
975 normalizing, since ASCII is already NFC and every _SCRIPT_RANGES
976 entry is non-ASCII regardless), else an NFC-normalized copy.
977
978 Trailing stops, not raw: a period glued to a script-written token
979 ('양.', '太郎.') is not a character of any script, so classifying
980 raw text handed the token no script at all, and three readers
981 spent that None -- the surname site stepped past the family name
982 onto the given name ('양. 지훈' cut 지훈 in half), the order rule
983 fell back to positional ('양 지훈.' lost family-first), and the
984 segmenter's neighbour precondition missed a writer-drawn boundary
985 ('山田太郎 田中.' consulted the segmenter on 山田太郎 as if it stood
986 alone). The scripts this classifies -- every _SCRIPT_RANGES entry,
987 which today coincide with _policy._NO_INITIALS (#320), a
988 coincidence _policy says a new member must not inherit -- have no
989 initials and no period abbreviations, so a stop on such a token
990 carries no information about the word; ASCII text is stripped too,
991 but the guard below returns None for it regardless, so 'Smith.'
992 never classifies. TRAILING only, matching the surname site's own
993 rstrip in _script_segment (the same arithmetic, not a shared
994 gate -- this fold decides only whether the surname site, the order
995 rule and the segmenter ever see the token): a leading stop is not
996 a shape any script writes before a name word, and HIDING such a
997 token from those three readers -- no script, so no surname site,
998 which is what the tree before #323 did -- is safer than admitting
999 it. Admitted, '.김민준' classifies as hangul, becomes a surname
1000 site, is declined by the head match (which rstrips) and falls
1001 through to a configured segmenter, which answering offset 1
1002 divides it into the stop and the name. The peel is not gated by
1003 this fold; its own rstrip carries a leading-stop token, see
1004 _script_segment. The vocabulary fold alone reads BOTH edges
1005 (_lexicon._normalize): '.씨' is still the honorific, and a lookup
1006 divides nothing.
1007
1008 NFC, not raw: NFD input decomposes precomposed katakana onto a
1009 base character plus a COMBINING mark (U+3099/U+309A, which sit in
1010 the HIRAGANA block, not katakana's), so classifying raw NFD text
1011 can hand a pure-katakana token the kana license by accident; NFD
1012 also decomposes Hangul syllables onto bare jamo (U+1100-U+11FF),
1013 entirely outside the HANGUL range, so raw NFD Korean input misses
1014 the shipped family-first order rule rather than merely misfiring.
1015 Normalizing first fixes both. Classification-only and read-only:
1016 the returned copy is never what gets tokenized, so token text and
1017 spans stay exactly what the caller wrote.
1018
1019 Vocabulary MATCHING composes NFC too, since #322
1020 (_lexicon._normalize folds every lookup and every stored entry the
1021 same way), so an NFD suffix word reaches its NFC entry. What stays
1022 raw is SEGMENTATION -- the surname site's direct membership test
1023 and the peel's tail slice index the token's own text -- where NFD
1024 degrades to no-split, never to a wrong split (decisions.md#W1,
1025 the 2026-07-29 ja amendment).
1026 """
1027 text = text.rstrip(FULL_STOPS)
1028 if not text or text.isascii():
1029 return None
1030 return unicodedata.normalize("NFC", text)
1031
1032
1033def _classify(normalized: str) -> Script | None:
1034 """The FIRST _SCRIPT_MATCHERS entry covering all of `normalized`
1035 (already NFC, via _normalized_for_script), else None. Shared by
1036 both public classifiers so each of them normalizes exactly once."""
1037 for script, matcher in _SCRIPT_MATCHERS.items():
1038 if matcher(normalized):
1039 return script
1040 return None
1041
1042
1043def single_script(text: str) -> Script | None:
1044 """The one Script whose ranges cover EVERY char of `text`, else
1045 None (mixed-script text has no well-defined convention to apply;
1046 the caller falls back to the positional default). Classifies an
1047 NFC-normalized copy of `text` -- see _normalized_for_script.
1048 Callers wanting the kana-mixed license (a kanji+kana composite
1049 resolving to HIRAGANA) want effective_script, not this function."""
1050 normalized = _normalized_for_script(text)
1051 if normalized is None:
1052 return None
1053 return _classify(normalized)
1054
1055
1056def effective_script(text: str) -> Script | None:
1057 """single_script, extended by the kana license (#272 amendment):
1058 a MIXED token wholly within Han∪hiragana∪katakana is Japanese --
1059 it necessarily contains kana (pure Han is not mixed), cannot be
1060 Chinese, and is not a foreign transcription (those are
1061 katakana-only: マイケル has no kanji, but さくらエミ -- hiragana
1062 plus katakana -- is kana-only AND licensed) -- and resolves to the
1063 HIRAGANA carrier entry. Pure-katakana stays KATAKANA
1064 (single_script's answer): a lone katakana token is predominantly a
1065 transcribed foreign name, so nothing defaults on it."""
1066 # None for both shapes _wholly_ja could never match anyway (empty
1067 # text, or all-ASCII text): real work, not a leftover "if text"
1068 # guard, since the ASCII case is one a bare emptiness check would
1069 # let through. The single normalized copy then serves both the
1070 # single-script answer and the license below.
1071 normalized = _normalized_for_script(text)
1072 if normalized is None:
1073 return None
1074 script = _classify(normalized)
1075 if script is not None:
1076 return script
1077 if _wholly_ja(normalized):
1078 return Script.HIRAGANA
1079 return None
1080
1081
1082def resolve_script_set(scripts: Iterable[Script]) -> Script | None:
1083 """Generalizes effective_script's kana license from one token's
1084 CHARACTERS to a whole name's PIECES (#272): `scripts` is the
1085 effective_script of every name token, already resolved
1086 individually -- '高橋' (Han) and 'みなみ' (Hiragana) are two
1087 separately single-script pieces (split by a space, not mixed
1088 within one token), but together are exactly the repertoire
1089 effective_script licenses inside a single token (高橋みなみ). A
1090 single distinct script is returned as-is (the ordinary case,
1091 including a lone wholly-katakana name, which callers key with no
1092 table entry); more than one collapses to the HIRAGANA carrier
1093 when confined to Han/Hiragana/Katakana, the same set
1094 effective_script's license tests; any other mix (Han+Hangul, or
1095 no scripts at all -- an empty `scripts`) returns None -- the
1096 caller's cue to fall back to the positional default, exactly like
1097 effective_script's own None. A non-None result reports what was
1098 FOUND, not that a license fired: callers wanting to know whether
1099 the kana license specifically was the reason must compare the
1100 result against a specific Script (e.g. `is Script.HIRAGANA`), not
1101 just its truthiness -- a lone wholly-Han name also returns
1102 non-None here, licensing nothing."""
1103 found = frozenset(scripts)
1104 if len(found) <= 1:
1105 return next(iter(found), None)
1106 if found.issubset(_JA_SCRIPTS):
1107 return Script.HIRAGANA
1108 return None