Coverage for /pythoncovmergedfiles/medio/medio/usr/local/lib/python3.11/site-packages/nameparser/_pipeline/_state.py: 100%

Shortcuts on this page

r m x   toggle line displays

j k   next/prev highlighted chunk

0   (zero) top of page

1   (one) first highlighted chunk

51 statements  

1"""Internal pipeline state: WorkToken and ParseState. 

2 

3WorkTokens are pipeline-internal (no validation -- the tokenizer is the 

4only producer) and are addressed BY INDEX in every stage: pieces and 

5segments are runs of token indices, never joined strings, so value-based 

6lookup (v1's #100 family) is structurally impossible. 

7 

8Layering: imports _types, _lexicon, _policy only (enforced by 

9tests/v2/test_layering.py). 

10""" 

11from __future__ import annotations 

12 

13import bisect 

14from collections.abc import Sequence 

15from dataclasses import dataclass 

16from enum import Enum, auto 

17 

18from nameparser._lexicon import Lexicon 

19from nameparser._policy import Policy 

20from nameparser._types import AmbiguityKind, Role, Segmenter, Span 

21 

22 

23# The comma characters (ASCII/Arabic/fullwidth, #265). Shared here so 

24# tokenize (separators/segmentation) and extract (close-quote 

25# boundaries) cannot drift apart. 

26COMMA_CHARS = frozenset({",", "\u060c", "\uff0c"}) 

27 

28 

29def comma_bucket(start: int, comma_offsets: Sequence[int]) -> int: 

30 """Which comma-delimited part of the name a token starting at 

31 `start` falls in: the number of commas before it. 

32 

33 Shared here for the reason COMMA_CHARS is, and the sharing is 

34 load-bearing in the same way. segment BUILDS the segments with this 

35 (comma_offsets is sorted and no offset ever equals a token start, 

36 so bisect_left counts the commas before the token); classify asks 

37 it to decide whether two tokens could be in one segment, which is 

38 half of what stops a maiden marker run from spanning a boundary no 

39 segment holds. Two tokens agree here iff segment would put them in 

40 one segment, and that is an identity rather than a resemblance 

41 only while both sides ask this function. 

42 """ 

43 return bisect.bisect_left(comma_offsets, start) 

44 

45@dataclass(frozen=True, slots=True) 

46class WorkToken: 

47 """One tokenized word. role stays None until assign; extracted 

48 nickname/maiden tokens arrive with their role pre-set. text is 

49 always the exact original slice (tokenize is the sole producer; 

50 the anti-#100 invariant depends on it).""" 

51 

52 text: str 

53 span: Span 

54 tags: frozenset[str] = frozenset() 

55 role: Role | None = None 

56 

57 

58#: M4's two carve-outs, as the tags classify recorded them: a bound 

59#: given-name word is vocabulary claiming the word as a given name, 

60#: and `initial` is the initial reading. Neither is a predicate M4 owns. 

61#: Shared here beside WorkToken.tags for the reason COMMA_CHARS is: 

62#: assign's `_WORD_ALREADY_CLAIMED` is built from this pair, and the 

63#: two stages must not drift (post_rules imports _assign, so _assign 

64#: cannot reach the other way). 

65_NEVER_FLIPPED = frozenset({"vocab:bound-given", "initial"}) 

66 

67#: The by-shape half of #289/#516's ambiguous credential class: a 

68#: token classify admits to `vocab:suffix-ambiguous`'s READING by 

69#: SHAPE rather than by the listed vocabulary 

70#: (`Policy.unlisted_dotted_suffixes` is the first emitter; 

71#: `Policy.unlisted_caps_suffixes` is the second, and classify writes 

72#: the tag from both branches). One constant, not a string literal at 

73#: each site, because the readers that must tell a by-shape member 

74#: apart from a listed one -- `_pieces.peel_trailing` and 

75#: `_pieces.listed_lean`, which `segment_suffix_reading` asks through, 

76#: so the reading it decides is second-hand -- cannot afford to spell 

77#: it several ways and have one of them typo silently past the 

78#: others. The two sites that want EITHER half read 

79#: `_AMBIGUOUS_CREDENTIAL_TAGS` below rather than this constant. 

80SHAPE_ACRONYM_TAG = "shape:acronym" 

81 

82#: The MEMBERSHIP half of the same class: classify's tag for a token 

83#: the ambiguous credential vocabulary claims, by listing 

84#: (`Lexicon.suffix_acronyms_ambiguous`) or -- where a Policy switch 

85#: admits the by-shape half -- beside `SHAPE_ACRONYM_TAG`. Beside that 

86#: constant and for its reason: the string was spelled out at eight 

87#: sites across four modules, each of them a place for a typo to pass 

88#: silently, since a tag that is never written is simply a tag no 

89#: reader ever finds. 

90AMBIGUOUS_ACRONYM_TAG = "vocab:suffix-ambiguous" 

91 

92#: EITHER way a token joins the ambiguous credential class -- the 

93#: vocabulary's claim and the writing's. The two emitters that report 

94#: a fork the peel called and declined ask exactly this: `_group`'s 

95#: prefix chain (a listed member the case lean read as a name, 'John 

96#: van der Berg Ma'; a by-shape member the count left standing, 

97#: 'Freiherr von Berg X.Y.I.') and `_assign`'s family-comma slot. One 

98#: constant beside the two above and for their reason -- the pair was 

99#: spelled two ways, a frozenset here and an `or` of two `in` tests 

100#: there -- and a frozenset so each test is one `isdisjoint`, a C call 

101#: with no Python frame, on a branch every chained name reaches. 

102_AMBIGUOUS_CREDENTIAL_TAGS = frozenset( 

103 {AMBIGUOUS_ACRONYM_TAG, SHAPE_ACRONYM_TAG}) 

104 

105 

106class Structure(Enum): 

107 """segment's comma-structure decision.""" 

108 

109 NO_COMMA = auto() 

110 FAMILY_COMMA = auto() # "Family, Given ..." (v1 lastname-comma) 

111 SUFFIX_COMMA = auto() # "Given Family, Suffix ..." 

112 

113 

114@dataclass(frozen=True, slots=True) 

115# rules.md#A1: "parsing never fails on any input: where the text's 

116# structure or a word's reading is genuinely uncertain, the parse 

117# completes on the best reading and carries an ambiguity report 

118# naming the doubt" 

119class PendingAmbiguity: 

120 """An ambiguity recorded mid-pipeline by token INDEX; assemble 

121 materializes real Ambiguity objects over the final tokens. 

122 

123 ``origin`` is for the one stage that runs BEFORE tokens exist: 

124 extract_delimited knows only a character offset, so it records that 

125 and tokenize resolves it to the containing token's index. Stages 

126 after tokenize set ``indices`` directly and leave ``origin`` None. 

127 """ 

128 

129 kind: AmbiguityKind 

130 detail: str 

131 indices: tuple[int, ...] = () 

132 origin: int | None = None 

133 

134 

135@dataclass(frozen=True, slots=True) 

136class ParseState: 

137 """Carried through the stage fold. Frozen; stages return copies via 

138 dataclasses.replace. Fields are filled progressively: 

139 extract_delimited -> extracted/masked; tokenize -> tokens (span- 

140 sorted)/comma_offsets/interpunct_offsets (the 间隔号 offsets the 

141 order and segmentation decisions consult, #298; the nakaguro 

142 separators record NOTHING); segment -> segments/structure/one_case 

143 (lazily, only where a comma form could turn it on -- #289/#516); 

144 script_segment -> tokens and segments again (the one stage that 

145 changes the token COUNT: an unspaced CJK token splits into n+1 

146 pieces, still as sub-slices of the original, and every later index 

147 in the segment runs shifts by n); classify -> token tags AND 

148 one_case; group -> pieces/piece_tags/dropped AND maiden token 

149 roles; 

150 assign -> the remaining token roles AND `order`, the effective 

151 order it read them under; post_rules -> roles again, and the 

152 ambiguity P6's attachment reports. 

153 Ambiguities are recorded by every stage that DECIDES one -- 

154 extract (resolved to a token index by tokenize), segment, 

155 script_segment, classify, group, assign, and post_rules -- since a 

156 fork whose branches are taken in different stages needs an emitter 

157 in each. Post-group, segments may retain indices of dropped tokens 

158 -- assign iterates pieces, never segments. This ownership map is 

159 pinned by tests/v2/pipeline/test_state.py. 

160 

161 segmenter belongs to no stage: like original/lexicon/policy it is 

162 passed in at construction by Parser.parse and only ever READ (by 

163 script_segment, for a token the vocabulary declined).""" 

164 

165 original: str 

166 lexicon: Lexicon 

167 policy: Policy 

168 #: The optional Parser(segmenter=...) hook; None = not configured. 

169 segmenter: Segmenter | None = None 

170 extracted: tuple[tuple[Role, Span], ...] = () 

171 masked: tuple[Span, ...] = () 

172 tokens: tuple[WorkToken, ...] = () 

173 comma_offsets: tuple[int, ...] = () 

174 interpunct_offsets: tuple[int, ...] = () 

175 segments: tuple[tuple[int, ...], ...] = () 

176 structure: Structure = Structure.NO_COMMA 

177 # pieces[s][p] = run of token indices: piece p of segment s. 

178 # piece_tags[s][p] = derived flags for that piece ("title", "prefix", 

179 # "suffix", "conjunction") set by group's joins. 

180 pieces: tuple[tuple[tuple[int, ...], ...], ...] = () 

181 piece_tags: tuple[tuple[frozenset[str], ...], ...] = () 

182 dropped: tuple[int, ...] = () # structural tokens (maiden markers) 

183 #: The order `assign` actually READ the name under -- name_order, 

184 #: or the script_orders entry that overrode it. None wherever no 

185 #: positional read happened: after a family comma (which fixes the 

186 #: family, so assign consults no order at all), and on every early 

187 #: return in `_assign_main`, where a segment holds no name piece 

188 #: to position. Recorded rather than recomputed downstream, 

189 #: because the two can differ and a post_rules rule keyed on 

190 #: `policy.name_order` would then disagree with the roles assign 

191 #: already wrote (#395). Reaching that divergence needs a custom 

192 #: lexicon -- every shipped particle is Latin, and Latin has no 

193 #: script_orders entry -- which is why the test for it builds its 

194 #: own (test_post_rules.py). 

195 order: tuple[Role, Role, Role] | None = None 

196 #: Whether the name's OWN words are written wholly in one case -- 

197 #: all upper or all lower alike -- and so carry no case EVIDENCE 

198 #: about any word in them (rules.md#P3's own-words span, 

199 #: _pieces.own_words). None means NOT ASKED YET: the fact is 

200 #: computed by whichever of segment and classify needs it first, 

201 #: segment only when a comma form could turn on it, so a reader 

202 #: between the two stages sees None and must not guess. 

203 #: Recorded rather than recomputed, the way `order` above is: the 

204 #: trailing suffix slot, the post-comma slot, the tail-segment 

205 #: reading, the prefix chain's own tail measure (_group) and the 

206 #: glued-honorific peel's decline of a post-comma run 

207 #: (_script_segment) all consult it and must not disagree 

208 #: (#289/#516, decisions.md#S2). Five stages read it in all -- 

209 #: segment, script_segment, classify, group and assign -- plus the 

210 #: two predicate layers they read it through (_pieces, _vocab). 

211 #: A fact segment records SURVIVES script_segment, the one stage 

212 #: that changes the token count, and survives it unrecomputed 

213 #: because splitting a token cannot change the answer: the verdict 

214 #: is `is_one_case` over the name's own words JOINED, and a split 

215 #: only moves a space into a string whose upper/lower comparison 

216 #: ignores spaces entirely -- '김민준씨' and '김민준 씨' fold alike. 

217 #: So the field is carried through rather than invalidated, and 

218 #: script_segment reads it (its suffix-run predicate takes the 

219 #: lean) rather than asking again. 

220 one_case: bool | None = None 

221 ambiguities: tuple[PendingAmbiguity, ...] = ()