Coverage for /pythoncovmergedfiles/medio/medio/usr/local/lib/python3.11/site-packages/wcwidth/_wcswidth.py: 5%
Shortcuts on this page
r m x toggle line displays
j k next/prev highlighted chunk
0 (zero) top of page
1 (one) first highlighted chunk
Shortcuts on this page
r m x toggle line displays
j k next/prev highlighted chunk
0 (zero) top of page
1 (one) first highlighted chunk
1"""This is a python implementation of wcswidth()."""
3from __future__ import annotations
5from typing import Optional
7__lazy_modules__ = [
8 "wcwidth._constants",
9 "wcwidth._wcwidth",
10 "wcwidth.bisearch",
11 "wcwidth.table_grapheme",
12 "wcwidth.table_vs16",
13]
14# local
15from . import table_grapheme_overrides
16from ._wcwidth import wcwidth
17from .bisearch import bisearch
18from ._constants import (_EMOJI_ZWJ_SET,
19 _ISC_VIRAMA_SET,
20 _CATEGORY_MC_TABLE,
21 _FITZPATRICK_RANGE,
22 _REGIONAL_INDICATOR_SET,
23 resolve_terminal,
24 get_term_overrides,
25 _clamp_ambiguous_width)
26from .table_vs15 import VS15_WIDE_TO_NARROW
27from .table_vs16 import VS16_NARROW_TO_WIDE
28from .table_grapheme import GRAPHEME_EXTEND
31def _scan_zwj_cluster_end(text: str, start: int, end: int) -> int:
32 """
33 Scan forward from *start* (base character) to end of a ZWJ grapheme cluster.
35 Follows the UAX #29 GB11 pattern (ExtPict Extend* ZWJ x ExtPict) chained repeatedly until no
36 more ZWJ joins are found.
37 """
38 idx = start + 1
39 # Skip Extend characters (Fitzpatrick modifiers, etc.) before first ZWJ
40 while idx < end and bisearch(ord(text[idx]), GRAPHEME_EXTEND):
41 idx += 1
42 # Follow ZWJ chains
43 while idx < end:
44 if ord(text[idx]) != 0x200D:
45 break
46 idx += 1
47 # GB11: \p{ExtPict} Extend* ZWJ × \p{ExtPict}
48 # Extend modifiers (VS16, Fitzpatrick skin tones, etc.) attach to
49 # the ExtPict *before* the ZWJ. After ZWJ the next codepoint is
50 # always an ExtPict directly, no Extend skip needed.
51 if idx < end and ord(text[idx]) in _EMOJI_ZWJ_SET:
52 idx += 1
53 # Skip trailing Extend (VS16, etc.) after ExtPict before next ZWJ
54 while idx < end and bisearch(ord(text[idx]), GRAPHEME_EXTEND):
55 idx += 1
56 continue
57 break
58 return idx
61def wcswidth(
62 pwcs: str,
63 n: Optional[int] = None,
64 unicode_version: str = 'auto',
65 ambiguous_width: int = 1,
66) -> int:
67 """
68 Given a unicode string, return its printable length on a terminal.
70 See :ref:`Specification` for details of cell measurement.
72 This implementation differs from Markus Khun's original POSIX C implementation, in that this
73 ``wcswidth()`` processes graphemes strings yielded by :func:`wcwidth.iter_graphemes` defined by
74 `Unicode Standard Annex #29`_. POSIX wcswidth(3) is not grapheme-aware and does not measure many
75 kinds of Emojis or complex scripts correctly.
77 :param pwcs: Measure width of given unicode string.
78 :param n: When ``n`` is None (default), return the length of the entire
79 string, otherwise only the first ``n`` characters are measured.
80 :param unicode_version: Ignored. Retained for backwards compatibility.
82 .. deprecated:: 0.3.0
83 Only the latest Unicode version is now shipped.
85 :param ambiguous_width: Width to use for East Asian Ambiguous (A)
86 characters. Default is ``1`` (narrow). Set to ``2`` for CJK contexts.
87 :returns: The width, in cells, needed to display the first ``n`` characters
88 of the unicode string ``pwcs``. Returns ``-1`` for C0 and C1 control
89 characters!
91 .. _`Unicode Standard Annex #29`: https://www.unicode.org/reports/tr29/
92 """
93 # pylint: disable=unused-argument,too-many-locals,too-many-statements,redefined-variable-type
94 # pylint: disable=too-complex,too-many-branches,duplicate-code,too-many-nested-blocks
96 # Fast path: pure ASCII printable strings are always width == length
97 if n is None and pwcs.isascii() and pwcs.isprintable():
98 return len(pwcs)
100 ambiguous_width = _clamp_ambiguous_width(ambiguous_width)
102 _wcwidth = wcwidth if ambiguous_width == 1 else lambda c: wcwidth(c, 'auto', ambiguous_width)
104 end = len(pwcs) if n is None else min(n, len(pwcs))
105 total_width = 0
106 idx = 0
108 last_measured_idx = -2 # -2 sentinel blocks VS16/VS15 (no base available)
109 last_measured_ucs = -1
110 last_measured_w = 0
111 prev_was_virama = False
112 cluster_width = 0
113 vs16_nw_table = VS16_NARROW_TO_WIDE['9.0.0']
114 vs15_wn_table = VS15_WIDE_TO_NARROW['9.0.0']
115 _bisearch = bisearch
117 while idx < end:
118 char = pwcs[idx]
119 ucs = ord(char)
121 # 5. ZWJ (U+200D): consumed without contributing width.
122 # Virama codepoints are treated as zero-width combining marks (Mn). When a
123 # virama+consonant sequence forms a conjunct, its width is capped at 2 cells.
125 # ZWJ (U+200D)
126 if ucs == 0x200D:
127 if prev_was_virama:
128 idx += 1
129 elif idx + 1 < end:
130 last_measured_w = 0
131 prev_was_virama = False
132 idx += 2
133 else:
134 prev_was_virama = False
135 idx += 1
136 continue
138 # 6. VS16 (U+FE0F): converts preceding narrow character to wide.
139 if ucs == 0xFE0F and last_measured_idx >= 0:
140 if _bisearch(last_measured_ucs, vs16_nw_table):
141 cluster_width = 2
142 last_measured_idx = -2
143 idx += 1
144 continue
146 # VS15 (U+FE0E): text variation selector, requests narrow presentation.
147 if ucs == 0xFE0E and last_measured_idx >= 0:
148 if bisearch(last_measured_ucs, vs15_wn_table) and last_measured_w == 2:
149 total_width -= 1
150 last_measured_idx = -2
151 idx += 1
152 continue
154 # 7. Regional Indicator & Fitzpatrick (both above BMP)
155 if ucs > 0xFFFF:
156 if ucs in _REGIONAL_INDICATOR_SET:
157 ri_before = 0
158 j = idx - 1
159 while j >= 0 and ord(pwcs[j]) in _REGIONAL_INDICATOR_SET:
160 ri_before += 1
161 j -= 1
162 if ri_before % 2 == 1:
163 last_measured_ucs = ucs
164 idx += 1
165 continue
166 elif (_FITZPATRICK_RANGE[0] <= ucs <= _FITZPATRICK_RANGE[1]
167 and last_measured_ucs in _EMOJI_ZWJ_SET):
168 idx += 1
169 continue
171 # 8. Normal character: measure with wcwidth
172 w = _wcwidth(char)
173 if w < 0:
174 return -1
175 if w > 0:
176 if prev_was_virama:
177 cluster_width = 2
178 elif cluster_width:
179 total_width += cluster_width
180 cluster_width = w
181 else:
182 cluster_width = w
184 last_measured_idx = idx
185 last_measured_ucs = ucs
186 last_measured_w = w
187 prev_was_virama = False
188 elif ucs in _ISC_VIRAMA_SET:
189 prev_was_virama = True
190 elif last_measured_idx >= 0 and _bisearch(ucs, _CATEGORY_MC_TABLE):
191 cluster_width = 2
192 last_measured_idx = -2
193 prev_was_virama = False
194 else:
195 prev_was_virama = False
196 idx += 1
198 if cluster_width:
199 total_width += cluster_width
200 return total_width
203def wcstwidth(
204 pwcs: str,
205 n: Optional[int] = None,
206 unicode_version: str = 'auto',
207 ambiguous_width: int = 1,
208 term_program: bool | str = True,
209) -> int:
210 """
211 Given a unicode string, return its printable length on a terminal given by ``term_program``.
213 See :ref:`Specification` for details of cell measurement.
215 Unlike :func:`wcswidth`, this function applies per-terminal correction tables for
216 emoji presentation and grapheme clusters.
218 :param pwcs: Measure width of given unicode string.
219 :param n: When ``n`` is None (default), return the length of the entire
220 string, otherwise only the first ``n`` characters are measured.
221 :param unicode_version: Ignored. Retained for backwards compatibility.
222 :param ambiguous_width: Width to use for East Asian Ambiguous (A)
223 characters. Default is ``1`` (narrow). Set to ``2`` for CJK contexts.
224 :param term_program: Terminal software identifier for table correction.
225 ``True`` (default) reads the ``TERM_PROGRAM`` or ``TERM`` environment
226 variable for auto-detection. ``False`` disables override lookup.
227 Accepts a canonical terminal name matching :func:`list_term_programs`,
228 such as from XTVERSION_, ENQ_, or ``TERM_PROGRAM``.
230 .. versionadded:: 0.8.0
231 :returns: The width, in cells, needed to display the first ``n`` characters
232 of the unicode string ``pwcs``. Returns ``-1`` for C0 and C1 control
233 characters!
234 """
235 # pylint: disable=unused-argument,too-many-locals,too-many-statements,redefined-variable-type
236 # pylint: disable=too-complex,too-many-branches,duplicate-code,too-many-nested-blocks
237 # This function intentionally keeps all logic inline for performance.
239 # Fast path: pure ASCII printable strings are always width == length
240 if n is None and pwcs.isascii() and pwcs.isprintable():
241 return len(pwcs)
243 ambiguous_width = _clamp_ambiguous_width(ambiguous_width)
245 # Resolve terminal software for override lookup
246 term_canonical = resolve_terminal(term_program)
248 # Skip override lookup when no terminal detected (avoids lru_cache call overhead).
249 # Extract locals for hot-loop performance (NamedTuple attribute access is slow).
250 if term_canonical:
251 overrides = get_term_overrides(term_canonical)
252 _narrower = overrides.narrower
253 _vs16_narrower = overrides.vs16_narrower
254 _vs15_wider = overrides.vs15_wider
255 _zeroer = overrides.zeroer
256 _narrow_wider = overrides.narrow_wider
257 _narrow_zeroer = overrides.narrow_zeroer
258 _grapheme_overrides = table_grapheme_overrides.get(term_canonical)
259 else:
260 _narrower = ()
261 _vs16_narrower = ()
262 _vs15_wider = ()
263 _zeroer = ()
264 _narrow_wider = ()
265 _narrow_zeroer = ()
266 _grapheme_overrides = {}
268 # Select wcwidth call pattern for best lru_cache performance
269 _wcwidth = wcwidth if ambiguous_width == 1 else lambda c: wcwidth(c, 'auto', ambiguous_width)
271 end = len(pwcs) if n is None else min(n, len(pwcs))
272 total_width = 0
273 idx = 0
275 # grapheme-clustering state and local re-binding for performance.
276 # Widths accumulate in cluster_width and flush at boundaries. A cluster is a base character
277 # plus combining marks, deferring the flush lets grapheme overrides replace the measured width
278 # retrospectively.
279 last_measured_idx = -2 # -2 sentinel blocks VS16/VS15 (no base available)
280 last_measured_ucs = -1
281 last_measured_w = 0
282 prev_was_virama = False
283 cluster_start = -1
284 total_before_cluster = 0
285 cluster_width = 0
286 vs16_nw_table = VS16_NARROW_TO_WIDE['9.0.0']
287 vs15_wn_table = VS15_WIDE_TO_NARROW['9.0.0']
288 _bisearch = bisearch
290 while idx < end:
291 char = pwcs[idx]
292 ucs = ord(char)
294 #
295 # Much of the logic below matches width(); it is repeated here for performance, with
296 # matching index reference numbers (starting at #5).
297 #
298 # 5. ZWJ (U+200D): consumed without contributing width.
299 # Virama codepoints are treated as zero-width combining marks (Mn). When a
300 # virama+consonant sequence forms a conjunct, its width is capped at 2 cells
301 # matching behavior of popular terminals (PR #224)
303 # ZWJ (U+200D)
304 if ucs == 0x200D:
305 if prev_was_virama:
306 idx += 1
307 elif idx + 1 < end:
308 # Check for terminal grapheme override when base char is ExtPict/RI
309 if (_grapheme_overrides
310 and last_measured_idx >= 0
311 and last_measured_ucs in _EMOJI_ZWJ_SET):
312 cluster_end = _scan_zwj_cluster_end(pwcs, last_measured_idx, end)
313 cluster = pwcs[last_measured_idx:cluster_end]
314 override_w = _grapheme_overrides.get(cluster)
315 if override_w is not None:
316 total_width += (override_w - last_measured_w)
317 last_measured_idx = -2
318 last_measured_ucs = -1
319 last_measured_w = 0
320 prev_was_virama = False
321 cluster_start = -1
322 idx = cluster_end
323 continue
324 # No override; ZWJ breaks VS adjacency.
325 # VS16 already set last_measured_idx = -2, blocking further VS16.
326 last_measured_w = 0
327 prev_was_virama = False
328 idx += 2
329 else:
330 prev_was_virama = False
331 idx += 1
332 continue
334 # 6. VS16 (U+FE0F): converts preceding narrow character to wide.
335 if ucs == 0xFE0F and last_measured_idx >= 0:
336 if _vs16_narrower and _bisearch(last_measured_ucs, _vs16_narrower):
337 pass
338 elif _bisearch(last_measured_ucs, vs16_nw_table):
339 cluster_width = 2
340 last_measured_idx = -2
341 idx += 1
342 continue
344 # VS15 (U+FE0E): text variation selector, requests narrow presentation.
345 if ucs == 0xFE0E and last_measured_idx >= 0:
346 base_ucs = last_measured_ucs
347 vs15_narrow = bisearch(base_ucs, vs15_wn_table)
348 if _vs15_wider and bisearch(base_ucs, _vs15_wider):
349 vs15_narrow = False
350 if vs15_narrow and last_measured_w == 2:
351 total_width -= 1
352 last_measured_idx = -2
353 idx += 1
354 continue
356 # 7. Regional Indicator & Fitzpatrick (both above BMP)
357 if ucs > 0xFFFF:
358 if ucs in _REGIONAL_INDICATOR_SET:
359 ri_before = 0
360 j = idx - 1
361 while j >= 0 and ord(pwcs[j]) in _REGIONAL_INDICATOR_SET:
362 ri_before += 1
363 j -= 1
364 if ri_before % 2 == 1 and not (_narrower and _bisearch(ucs, _narrower)):
365 last_measured_ucs = ucs
366 idx += 1
367 continue
368 elif (_FITZPATRICK_RANGE[0] <= ucs <= _FITZPATRICK_RANGE[1]
369 and last_measured_ucs in _EMOJI_ZWJ_SET):
370 idx += 1
371 continue
373 # 8. Normal character: measure with wcwidth
374 w = _wcwidth(char)
375 if w < 0:
376 # C0/C1 control character
377 return -1
378 # Apply single-codepoint terminal overrides (pre-merged tuples)
379 if w == 2 and _narrower and bisearch(ucs, _narrower):
380 w = 1
381 elif w == 2 and _zeroer and bisearch(ucs, _zeroer):
382 w = 0
383 if w == 1 and _narrow_wider and bisearch(ucs, _narrow_wider):
384 w = 2
385 elif w == 1 and _narrow_zeroer and bisearch(ucs, _narrow_zeroer):
386 w = 0
387 if w > 0:
388 # virama+consonant extends current cluster; otherwise start new
389 if prev_was_virama:
390 cluster_width = 2
391 elif cluster_width:
392 # flush previous cluster, check for grapheme overrides
393 flushed = False
394 if _grapheme_overrides and cluster_start >= 0:
395 # Two-phase override lookup: candidate (cluster+current) catches Lo+Lo pairs
396 # where both chars bear width (Thai KO KAI + SARA AM). cluster_text (cluster
397 # alone) catches C+Mc clusters where the override key is shorter.
398 candidate = pwcs[cluster_start:idx + 1]
399 override_w = _grapheme_overrides.get(candidate)
400 if override_w is not None:
401 total_width = total_before_cluster + override_w
402 flushed = True
403 cluster_width = 0
404 else:
405 cluster_text = pwcs[cluster_start:idx]
406 override_w = _grapheme_overrides.get(cluster_text)
407 if override_w is not None:
408 total_width = total_before_cluster + override_w
409 else:
410 total_width += cluster_width
411 else:
412 total_width += cluster_width
413 if not flushed:
414 cluster_width = w
415 cluster_start = idx
416 total_before_cluster = total_width
417 else:
418 cluster_width = w
419 cluster_start = idx
420 total_before_cluster = total_width
421 last_measured_idx = idx
422 last_measured_ucs = ucs
423 last_measured_w = w
424 prev_was_virama = False
425 elif ucs in _ISC_VIRAMA_SET:
426 prev_was_virama = True
427 elif last_measured_idx >= 0 and _bisearch(ucs, _CATEGORY_MC_TABLE):
428 # Spacing Combining Mark (Mc) following a base character
429 cluster_width = 2
430 last_measured_idx = -2
431 prev_was_virama = False
432 else:
433 prev_was_virama = False
434 idx += 1
436 if cluster_width:
437 if _grapheme_overrides and cluster_start >= 0:
438 cluster_text = pwcs[cluster_start:end]
439 override_w = _grapheme_overrides.get(cluster_text)
440 if override_w is not None:
441 total_width = total_before_cluster + override_w
442 else:
443 total_width += cluster_width
444 else:
445 total_width += cluster_width
446 return total_width