Coverage for /pythoncovmergedfiles/medio/medio/usr/local/lib/python3.11/site-packages/wcwidth/_width.py: 6%
Shortcuts on this page
r m x toggle line displays
j k next/prev highlighted chunk
0 (zero) top of page
1 (one) first highlighted chunk
Shortcuts on this page
r m x toggle line displays
j k next/prev highlighted chunk
0 (zero) top of page
1 (one) first highlighted chunk
1"""This is a high-level width() supporting terminal output."""
3from __future__ import annotations
5from typing import Literal
7__lazy_modules__ = [
8 "wcwidth._constants",
9 "wcwidth._wcswidth",
10 "wcwidth._wcwidth",
11 "wcwidth.bisearch",
12 "wcwidth.control_codes",
13 "wcwidth.escape_sequences",
14 "wcwidth.table_grapheme",
15 "wcwidth.table_vs16",
16 "wcwidth.text_sizing",
17]
18# local
19from . import table_grapheme_overrides
20from ._wcwidth import wcwidth
21from .bisearch import bisearch
22from ._wcswidth import wcswidth, wcstwidth, _scan_zwj_cluster_end
23from ._constants import (_EMOJI_ZWJ_SET,
24 _ISC_VIRAMA_SET,
25 _CATEGORY_MC_TABLE,
26 _FITZPATRICK_RANGE,
27 _REGIONAL_INDICATOR_SET,
28 resolve_terminal,
29 get_term_overrides,
30 _clamp_ambiguous_width)
31from .table_vs15 import VS15_WIDE_TO_NARROW
32from .table_vs16 import VS16_NARROW_TO_WIDE
33from .text_sizing import TextSizing, TextSizingParams
34from .control_codes import ILLEGAL_CTRL, VERTICAL_CTRL, HORIZONTAL_CTRL, ZERO_WIDTH_CTRL
35from .escape_sequences import (_SEQUENCE_CLASSIFY,
36 TEXT_SIZING_PATTERN,
37 CURSOR_MOVEMENT_SEQUENCE,
38 INDETERMINATE_EFFECT_SEQUENCE,
39 strip_sequences)
41# In 'parse' mode, strings longer than this are checked for cursor-movement
42# controls (BS, TAB, CR, cursor sequences); when absent, mode downgrades to
43# 'ignore' to skip character-by-character parsing. The detection scan cost is
44# negligible for long strings but wasted on short ones like labels or headings.
45_WIDTH_FAST_PATH_MIN_LEN = 20
47# Translation table to strip C0/C1 control characters for fast 'ignore' mode.
48_CONTROL_CHAR_TABLE = str.maketrans('', '', (
49 ''.join(chr(c) for c in range(0x00, 0x20)) + # C0: NUL through US (including tab)
50 '\x7f' + # DEL
51 ''.join(chr(c) for c in range(0x80, 0xa0)) # C1: U+0080-U+009F
52))
55def _width_ignored_codes(text: str, ambiguous_width: int = 1,
56 term_program: bool | str = False) -> int:
57 """
58 Fast path for width() with control_codes='ignore'.
60 Strips escape sequences and control characters, then measures remaining text.
61 """
62 if term_program is False:
63 return wcswidth(
64 strip_sequences(text).translate(_CONTROL_CHAR_TABLE),
65 ambiguous_width=ambiguous_width,
66 )
67 return wcstwidth(
68 strip_sequences(text).translate(_CONTROL_CHAR_TABLE),
69 ambiguous_width=ambiguous_width,
70 term_program=term_program,
71 )
74def width(
75 text: str,
76 *,
77 control_codes: Literal['parse', 'strict', 'ignore'] = 'parse',
78 tabsize: int = 8,
79 ambiguous_width: int = 1,
80 term_program: bool | str = False,
81) -> int:
82 r"""
83 Return printable width of text containing many kinds of control codes and sequences.
85 Unlike :func:`wcswidth`, this function handles most control characters and many popular terminal
86 output sequences. Never returns -1.
88 :param text: String to measure.
89 :param control_codes: How to handle control characters and sequences:
91 - ``'parse'`` (default): Track horizontal cursor movement like BS ``\b``, CR ``\r``, TAB
92 ``\t``, cursor left and right movement sequences. Vertical movement (LF, VT, FF) and
93 indeterminate terminal sequences are zero-width. OSC 66 Kitty Text Sizing protocol, OSC 8
94 Hyperlink, and many other kinds of output sequences are parsed for displayed measurements.
95 - ``'strict'``: Like parse, but raises :exc:`ValueError` on control characters with
96 indeterminate results of the screen or cursor, like clear or vertical movement. Generally,
97 these should be handled with a virtual terminal emulator (like 'pyte').
98 - ``'ignore'``: All C0 and C1 control characters and escape sequences are measured as
99 width 0. This is the fastest measurement for text already filtered or known not to contain
100 any kinds of control codes or sequences. TAB ``\t`` is zero-width; to ensure
101 tab expansion, pre-process text using :func:`str.expandtabs`.
103 :param tabsize: Tab stop width for ``'parse'`` and ``'strict'`` modes. Default is 8. Must be
104 positive. Has no effect when ``control_codes='ignore'``.
105 :param ambiguous_width: Width to use for East Asian Ambiguous (A) characters. Default is ``1``
106 (narrow). Set to ``2`` for CJK contexts.
107 :param term_program: Terminal software identifier for table correction.
108 ``False`` (default) disables override lookup. ``True`` reads the
109 ``TERM_PROGRAM`` or ``TERM`` environment variable for auto-detection.
110 Accepts a canonical terminal name matching :func:`list_term_programs`,
111 such as from XTVERSION_, ENQ_, or ``TERM_PROGRAM``.
113 .. versionadded:: 0.8.0
114 :returns: Maximum cursor position reached, "extent", accounting for cursor movement sequences
115 present in ``text`` according to given parameters. This represents the rightmost column the
116 cursor reaches. Always a non-negative integer.
117 :raises ValueError: If ``control_codes='strict'`` and control characters with indeterminate
118 effects, such as vertical movement or clear sequences are encountered, or on unexpected
119 C0 or C1 control code. Also raised when ``control_codes`` is not one of the valid values.
121 .. versionadded:: 0.3.0
123 .. versionchanged:: 0.7.0
124 Expanded strict-mode to raise :exc:`ValueError` when cursor-left movement
125 (CSI D) would move beyond the beginning of the string. Previously, cursor-left
126 was silently clamped to column 0 in all modes.
128 Support horizontal cursor sequences (``cub``, ``cuf``, ``hpa``). Cursor-left (``cub``) or
129 backspace (``\b``) now overwrites text. ``column_address`` (``hpa``) and carriage return
130 (``\r``) are now parsed, and some values conditionally raise ``ValueError`` when
131 ``control_codes='parse'``.
133 Examples::
135 >>> width('hello')
136 5
137 >>> width('コンニチハ')
138 10
139 >>> width('\x1b[31mred\x1b[0m')
140 3
141 >>> width('\x1b[31mred\x1b[0m', control_codes='ignore') # same result (ignored)
142 3
143 >>> width('123\b4') # backspace overwrites previous cell (outputs '124')
144 3
145 >>> width('abc\t') # tab caused cursor to move to column 8
146 8
147 >>> width('1\x1b[10C') # '1' + cursor right 10, cursor ends on column 11
148 11
149 >>> width('1\x1b[10C', control_codes='ignore') # faster but wrong in this case
150 1
151 """
152 # pylint: disable=too-complex,too-many-branches,too-many-statements,too-many-locals,redefined-variable-type,too-many-nested-blocks
153 # This could be split into sub-functions (#1, #3 and #6 especially), but this function is a
154 # hot path, so the steps stay inline and the pylint complexity rules are disabled.
156 # Fast path for ASCII printable (no tabs, escapes, or control chars)
157 if text.isascii() and text.isprintable():
158 return len(text)
160 ambiguous_width = _clamp_ambiguous_width(ambiguous_width)
162 # Fast parse: if no horizontal cursor movements are possible, switch to 'ignore' mode.
163 # Only check longer strings - the detection overhead hurts short string performance.
164 if control_codes == 'parse' and len(text) > _WIDTH_FAST_PATH_MIN_LEN:
165 # Check for cursor-affecting control characters
166 if '\b' not in text and '\t' not in text and '\r' not in text:
167 # Check for escape sequences, if none contain cursor movement or
168 # text sizing, downgrade to 'ignore'
169 if '\x1b' not in text or (
170 not CURSOR_MOVEMENT_SEQUENCE.search(text)
171 and not TEXT_SIZING_PATTERN.search(text)
172 ):
173 control_codes = 'ignore'
175 # Fast path for ignore mode, useful if you know the text is already free of control codes
176 if control_codes == 'ignore':
177 return _width_ignored_codes(text, ambiguous_width, term_program=term_program)
179 # Resolve terminal software for override lookup
180 term_canonical = resolve_terminal(term_program)
182 # Skip override lookup when no terminal detected (avoids lru_cache call overhead).
183 # Extract locals for hot-loop performance (NamedTuple attribute access is slow).
184 if term_canonical:
185 overrides = get_term_overrides(term_canonical)
186 _narrower = overrides.narrower
187 _vs16_narrower = overrides.vs16_narrower
188 _vs15_wider = overrides.vs15_wider
189 _zeroer = overrides.zeroer
190 _narrow_wider = overrides.narrow_wider
191 _narrow_zeroer = overrides.narrow_zeroer
192 _grapheme_overrides = table_grapheme_overrides.get(term_canonical)
193 else:
194 _narrower = ()
195 _vs16_narrower = ()
196 _vs15_wider = ()
197 _zeroer = ()
198 _narrow_wider = ()
199 _narrow_zeroer = ()
200 _grapheme_overrides = {}
202 strict = control_codes == 'strict'
203 # Track absolute positions: tab stops need modulo on absolute column, CR resets to 0.
204 # Initialize max_extent to 0 so backward movement (CR, BS) won't yield negative width.
205 current_col = 0
206 max_extent = 0
207 idx = 0
208 text_len = len(text)
210 # Select wcwidth call pattern for best lru_cache performance:
211 # - ambiguous_width=1 (default): single-arg calls share cache with direct wcwidth() calls
212 # - ambiguous_width=2: full positional args needed (results differ, separate cache is correct)
213 _wcwidth = wcwidth if ambiguous_width == 1 else lambda c: wcwidth(c, 'auto', ambiguous_width)
215 # grapheme-clustering state and local re-binding for performance.
216 # Widths accumulate in cluster_width and flush at boundaries (see _wcswidth.py)
217 last_measured_idx = -2 # -2 sentinel blocks VS16/VS15 (no base available)
218 last_measured_ucs = -1
219 last_measured_w = 0
220 prev_was_virama = False
221 _max_extent_before = 0
222 cluster_start = -1
223 col_before_cluster = 0
224 max_extent_before_cluster = 0
225 cluster_width = 0
226 vs16_nw_table = VS16_NARROW_TO_WIDE['9.0.0']
227 vs15_wn_table = VS15_WIDE_TO_NARROW['9.0.0']
228 _bisearch = bisearch
230 while idx < text_len:
231 char = text[idx]
233 # 1. ESC sequences
234 if char == '\x1b':
235 # Flush pending cluster before processing escape sequence
236 if cluster_width:
237 current_col += cluster_width
238 if current_col > max_extent:
239 max_extent = current_col
240 cluster_width = 0
241 m = _SEQUENCE_CLASSIFY.match(text, idx)
242 if not m:
243 # 1a. Errant ESC or unknown sequence: only the first character is zero-width
244 idx += 1
245 else:
246 seq = m.group()
247 if strict and INDETERMINATE_EFFECT_SEQUENCE.match(seq):
248 raise ValueError(f"Indeterminate cursor sequence at position {idx}, {seq!r}")
250 # 2b. horizontal position absolute (before forward/backward to
251 # avoid other_seq match in _SEQUENCE_CLASSIFY)
252 if (hpa_n := m.group('hpa_n')) is not None:
253 target_col = int(hpa_n) if hpa_n else 1
254 if strict:
255 raise ValueError(
256 f"Indeterminate horizontal position at position {idx}, "
257 f"{seq!r} (absolute column unknown)"
258 )
259 current_col = target_col - 1 # HPA is 1-indexed, convert to 0-indexed
260 # 2c. cursor forward, backward
261 elif (cforward_n := m.group('cforward_n')) is not None:
262 current_col += int(cforward_n) if cforward_n else 1
263 elif (cbackward_n := m.group('cbackward_n')) is not None:
264 n_backward = int(cbackward_n) if cbackward_n else 1
265 if strict and n_backward > current_col:
266 raise ValueError(
267 f"Cursor left movement at position {idx} would move "
268 f"{n_backward} cells left from column {current_col}, "
269 f"exceeding string start"
270 )
271 current_col -= n_backward
272 if current_col < 0:
273 current_col = 0
274 # 2d. OSC 66 Text Sizing: positive display width
275 elif (ts_meta := m.group('ts_meta')) is not None:
276 ts_text = m.group('ts_text') or ''
277 ts_term = m.group('ts_term')
278 assert ts_term is not None
279 text_size = TextSizing(
280 TextSizingParams.from_params(ts_meta, control_codes=control_codes),
281 ts_text, ts_term)
282 current_col += text_size.display_width(ambiguous_width)
283 # 2e. SGR and other zero-width sequences: no column advance
284 idx = m.end()
285 # Escape sequences break VS16 adjacency: reset last-measured state
286 last_measured_idx = -2
287 last_measured_ucs = -1
288 cluster_start = -1
289 if current_col > max_extent:
290 max_extent = current_col
291 continue
293 # 2. Vertical or Illegal control characters zero width or error when 'strict'
294 if char in ILLEGAL_CTRL:
295 if strict:
296 raise ValueError(f"Illegal control character {ord(char):#x} at position {idx}")
297 if cluster_width:
298 current_col += cluster_width
299 if current_col > max_extent:
300 max_extent = current_col
301 cluster_width = 0
302 idx += 1
303 last_measured_idx = -2
304 last_measured_ucs = -1
305 cluster_start = -1
306 continue
308 if char in VERTICAL_CTRL:
309 if strict:
310 raise ValueError(f"Vertical movement character {ord(char):#x} at position {idx}")
311 if cluster_width:
312 current_col += cluster_width
313 if current_col > max_extent:
314 max_extent = current_col
315 cluster_width = 0
316 idx += 1
317 last_measured_idx = -2
318 last_measured_ucs = -1
319 cluster_start = -1
320 continue
322 # 3. Horizontal movement characters
323 if char in HORIZONTAL_CTRL:
324 if cluster_width:
325 current_col += cluster_width
326 if current_col > max_extent:
327 max_extent = current_col
328 cluster_width = 0
329 if char == '\t' and tabsize > 0:
330 current_col += tabsize - (current_col % tabsize)
331 elif char == '\b':
332 if current_col > 0:
333 current_col -= 1
334 elif char == '\r':
335 if strict:
336 raise ValueError(
337 f"Horizontal movement character \\r at position {idx}: "
338 "indeterminate starting column"
339 )
340 current_col = 0
341 if current_col > max_extent:
342 max_extent = current_col
343 idx += 1
344 last_measured_idx = -2
345 last_measured_ucs = -1
346 cluster_start = -1
347 continue
349 # 4. Zero-width control characters
350 if char in ZERO_WIDTH_CTRL:
351 if cluster_width:
352 current_col += cluster_width
353 if current_col > max_extent:
354 max_extent = current_col
355 cluster_width = 0
356 idx += 1
357 last_measured_idx = -2
358 last_measured_ucs = -1
359 cluster_start = -1
360 continue
362 # 5. Inline grapheme-clustering: ZWJ, Virama, VS16, Regional Indicators,
363 # Fitzpatrick, Mc, wcwidth
364 ucs = ord(char)
366 # ZWJ (U+200D)
367 if ucs == 0x200D:
368 if prev_was_virama:
369 idx += 1
370 elif idx + 1 < text_len:
371 # Check for terminal grapheme override when base char is ExtPict/RI
372 if (_grapheme_overrides
373 and last_measured_idx >= 0
374 and last_measured_ucs in _EMOJI_ZWJ_SET):
375 cluster_end = _scan_zwj_cluster_end(text, last_measured_idx, text_len)
376 cluster = text[last_measured_idx:cluster_end]
377 override_w = _grapheme_overrides.get(cluster)
378 if override_w is not None:
379 current_col += (override_w - last_measured_w)
380 max_extent = max(max_extent, current_col)
381 last_measured_idx = -2
382 last_measured_ucs = -1
383 last_measured_w = 0
384 prev_was_virama = False
385 cluster_start = -1
386 idx = cluster_end
387 continue
388 # No override; ZWJ breaks VS adjacency.
389 # VS16 already set last_measured_idx = -2, blocking further VS16.
390 last_measured_w = 0
391 prev_was_virama = False
392 idx += 2
393 else:
394 prev_was_virama = False
395 idx += 1
396 continue
398 # 6. VS16 (U+FE0F): converts preceding narrow character to wide.
399 if ucs == 0xFE0F and last_measured_idx >= 0:
400 if _vs16_narrower and _bisearch(last_measured_ucs, _vs16_narrower):
401 pass
402 elif _bisearch(last_measured_ucs, vs16_nw_table):
403 cluster_width = 2
404 last_measured_idx = -2
405 idx += 1
406 continue
408 # VS15 (U+FE0E): text variation selector, requests narrow presentation.
409 if ucs == 0xFE0E and last_measured_idx >= 0:
410 base_ucs = last_measured_ucs
411 vs15_narrow = bisearch(base_ucs, vs15_wn_table)
412 if _vs15_wider and bisearch(base_ucs, _vs15_wider):
413 vs15_narrow = False
414 if vs15_narrow and last_measured_w == 2:
415 current_col -= 1
416 max_extent = max(_max_extent_before, current_col)
417 last_measured_idx = -2
418 idx += 1
419 continue
421 # 7. Regional Indicator & Fitzpatrick (both above BMP)
422 if ucs > 0xFFFF:
423 if ucs in _REGIONAL_INDICATOR_SET:
424 ri_before = 0
425 j = idx - 1
426 while j >= 0 and ord(text[j]) in _REGIONAL_INDICATOR_SET:
427 ri_before += 1
428 j -= 1
429 if ri_before % 2 == 1 and not (_narrower and _bisearch(ucs, _narrower)):
430 last_measured_ucs = ucs
431 idx += 1
432 continue
433 elif (_FITZPATRICK_RANGE[0] <= ucs <= _FITZPATRICK_RANGE[1]
434 and last_measured_ucs in _EMOJI_ZWJ_SET):
435 idx += 1
436 continue
438 # 8. Normal character: measure with wcwidth
439 w = _wcwidth(char)
440 # Apply single-codepoint terminal overrides (pre-merged tuples)
441 if w == 2 and _narrower and bisearch(ucs, _narrower):
442 w = 1
443 elif w == 2 and _zeroer and bisearch(ucs, _zeroer):
444 w = 0
445 if w == 1 and _narrow_wider and bisearch(ucs, _narrow_wider):
446 w = 2
447 elif w == 1 and _narrow_zeroer and bisearch(ucs, _narrow_zeroer):
448 w = 0
449 if w > 0:
450 # virama+consonant extends current cluster; otherwise start new
451 if prev_was_virama:
452 cluster_width = 2
453 elif cluster_width:
454 # flush previous cluster, check for grapheme overrides
455 flushed = False
456 if _grapheme_overrides and cluster_start >= 0:
457 # Two-phase override lookup (see _wcswidth.py)
458 candidate = text[cluster_start:idx + 1]
459 override_w = _grapheme_overrides.get(candidate)
460 if override_w is not None:
461 current_col = col_before_cluster + override_w
462 max_extent = max(max_extent_before_cluster, current_col)
463 flushed = True
464 cluster_width = 0
465 else:
466 cluster_text = text[cluster_start:idx]
467 override_w = _grapheme_overrides.get(cluster_text)
468 if override_w is not None:
469 current_col = col_before_cluster + override_w
470 max_extent = max(max_extent_before_cluster, current_col)
471 else:
472 current_col += cluster_width
473 else:
474 current_col += cluster_width
475 if current_col > max_extent:
476 max_extent = current_col
477 if not flushed:
478 cluster_width = w
479 cluster_start = idx
480 col_before_cluster = current_col
481 max_extent_before_cluster = max_extent
482 else:
483 cluster_width = w
484 cluster_start = idx
485 col_before_cluster = current_col
486 max_extent_before_cluster = max_extent
487 last_measured_idx = idx
488 last_measured_ucs = ucs
489 last_measured_w = w
490 _max_extent_before = max_extent
491 prev_was_virama = False
492 elif ucs in _ISC_VIRAMA_SET:
493 prev_was_virama = True
494 elif last_measured_idx >= 0 and _bisearch(ucs, _CATEGORY_MC_TABLE):
495 # Spacing Combining Mark (Mc) following a base character
496 cluster_width = 2
497 last_measured_idx = -2
498 prev_was_virama = False
499 else:
500 prev_was_virama = False
501 idx += 1
503 if cluster_width:
504 if _grapheme_overrides and cluster_start >= 0:
505 cluster_text = text[cluster_start:text_len]
506 override_w = _grapheme_overrides.get(cluster_text)
507 if override_w is not None:
508 current_col = col_before_cluster + override_w
509 max_extent = max(max_extent_before_cluster, current_col)
510 else:
511 current_col += cluster_width
512 if current_col > max_extent:
513 max_extent = current_col
514 else:
515 current_col += cluster_width
516 if current_col > max_extent:
517 max_extent = current_col
518 return max_extent