1r"""
2Terminal escape sequence patterns.
3
4This module provides regex patterns for matching terminal escape sequences. All patterns match
5sequences that begin with ESC (``\x1b``). Before calling re.match with these patterns, callers
6should first check that the character at the current position is ESC for optimal performance.
7"""
8
9# std imports
10import re
11
12import typing
13
14# local
15from .sgr_state import _SGR_PATTERN
16
17# Text Sizing Protocol (OSC 66), https://sw.kovidgoyal.net/kitty/text-sizing-protocol/
18TEXT_SIZING_PATTERN = re.compile(
19 r'\x1b\]66;([^;\x07\x1b]*)(?:;([^\x07\x1b]*))?(\x07|\x1b\\)'
20)
21
22# Zero-width escape sequences (SGR, OSC, CSI, etc.). This table, like INDETERMINATE_EFFECT_SEQUENCE,
23# originated from the 'blessed' library.
24ZERO_WIDTH_PATTERN = re.compile(
25 # CSI sequences
26 r'\x1b\[[\x30-\x3f]*[\x20-\x2f]*[\x40-\x7e]|'
27 # OSC sequences; OSC 66 text sizing gets special handling in width() and clip() and has positive
28 # width, despite the ZERO_WIDTH_PATTERN name.
29 r'\x1b\][^\x07\x1b]*(?:\x07|\x1b\\)|'
30 # APC sequences
31 r'\x1b_[^\x1b\x07]*(?:\x07|\x1b\\)|'
32 # DCS sequences
33 r'\x1bP[^\x1b\x07]*(?:\x07|\x1b\\)|'
34 # PM sequences
35 r'\x1b\^[^\x1b\x07]*(?:\x07|\x1b\\)|'
36 # Character set designation (subset of nF, handled separately for clarity).
37 # The final byte is (?s:.) so a newline terminates it like any other byte.
38 r'\x1b[()](?s:.)|'
39 # nF sequences: ESC + one or more intermediate bytes (0x20-0x2F) + final byte (0x30-0x7E)
40 r'\x1b[\x20-\x2f]+[\x30-\x7e]|'
41 # Fe sequences (C1 controls)
42 r'\x1b[\x40-\x5f]|'
43 # Fp sequences (private use)
44 r'\x1b[\x30-\x3f]|'
45 # Fs sequences (independent functions)
46 r'\x1b[\x60-\x7e]'
47)
48
49# TEXT_SIZING_PATTERN with named groups, shared by the two patterns below.
50_TEXT_SIZING_NAMED = (r'\x1b\]66;(?P<ts_meta>[^;\x07\x1b]*)'
51 r'(?:;(?P<ts_text>[^\x07\x1b]*))?(?P<ts_term>\x07|\x1b\\)')
52
53# One alternation: a separate OSC 66 pass would splice the text around each removal.
54_STRIP_WITH_TEXT_SIZING = re.compile(_TEXT_SIZING_NAMED + '|' + ZERO_WIDTH_PATTERN.pattern)
55
56
57def _strip_repl(match: "re.Match[str]") -> str:
58 """Keep the inner text of an OSC 66 match, drop every other sequence."""
59 return match.group('ts_text') or ''
60
61
62# Cursor right movement: CSI [n] C, parameter may be parsed by width()
63CURSOR_RIGHT_SEQUENCE = re.compile(r'\x1b\[(\d*)C')
64
65# Cursor left movement: CSI [n] D, parameter may be parsed by width()
66CURSOR_LEFT_SEQUENCE = re.compile(r'\x1b\[(\d*)D')
67
68# Horizontal position absolute: CSI [n] G, parameter may be parsed by width()
69CURSOR_HPA_SEQUENCE = re.compile(r'\x1b\[(\d*)G')
70
71# Combined cursor movement: single regex for fast-path detection of any
72# horizontal cursor movement (left, right, hpa). Avoids two separate search()
73# calls in hot-path width() and clip() pre-checks.
74CURSOR_MOVEMENT_SEQUENCE = re.compile(r'\x1b\[(\d*)[CDG]')
75
76# Combined horizontal cursor movement: matches BS, CR, and CSI C/D/G cursor sequences
77# in a single regex pass. Used by clip() to decide between the simple append path
78# and the painter's algorithm.
79_HORIZONTAL_CURSOR_MOVEMENT = re.compile(r'[\x08\r]|\x1b\[(\d*)[CDG]')
80
81# Combined pattern: a single regex that matches any zero-width escape sequence
82# and classifies it via named groups, aprox 2x faster than redundant re.matches
83# in clip() and width().
84_SEQUENCE_CLASSIFY = re.compile(
85 _SGR_PATTERN.pattern.replace('(', '(?P<sgr_params>', 1)
86 + '|' + CURSOR_HPA_SEQUENCE.pattern.replace('(', '(?P<hpa_n>', 1)
87 + '|' + CURSOR_RIGHT_SEQUENCE.pattern.replace('(', '(?P<cforward_n>', 1)
88 + '|' + CURSOR_LEFT_SEQUENCE.pattern.replace('(', '(?P<cbackward_n>', 1)
89 + '|' + _TEXT_SIZING_NAMED
90 + '|' + r'(?P<other_seq>(?:' + ZERO_WIDTH_PATTERN.pattern + '))'
91)
92
93# Indeterminate effect sequences - raise ValueError in 'strict' mode. The effects of these sequences
94# are likely to be undesirable, moving the cursor vertically or to any unknown position, and
95# otherwise not managed by the 'width' method of this library.
96#
97# This table was created initially with code generation by extraction of termcap library with
98# techniques used at 'blessed' library runtime for 'xterm', 'alacritty', 'kitty', ghostty',
99# 'screen', 'tmux', and others. Then, these common capabilities were merged into the list below.
100INDETERMINATE_EFFECT_SEQUENCE = re.compile(
101 '|'.join(f'(?:{_pattern})' for _pattern in (
102 r'\x1b\[\d+;\d+r', # change_scroll_region
103 r'\x1b\[\d*K', # erase_in_line (clr_eol, clr_bol)
104 r'\x1b\[\d*J', # erase_in_display (clr_eos, erase_display)
105 r'\x1b\[\d+;\d+H', # cursor_address
106 r'\x1b\[\d*H', # cursor_home
107 r'\x1b\[\d*A', # cursor_up
108 r'\x1b\[\d*B', # cursor_down
109 r'\x1b\[\d*P', # delete_character
110 r'\x1b\[\d*M', # delete_line
111 r'\x1b\[\d*L', # insert_line
112 r'\x1b\[\d*@', # insert_character
113 r'\x1b\[\d+X', # erase_chars
114 r'\x1b\[\d*S', # scroll_up (parm_index)
115 r'\x1b\[\d*T', # scroll_down (parm_rindex)
116 r'\x1b\[\d*d', # row_address
117 r'\x1b\[\?1049[hl]', # alternate screen buffer
118 r'\x1b\[\?47[hl]', # alternate screen (legacy)
119 r'\x1b8', # restore_cursor
120 r'\x1bD', # scroll_forward (index)
121 r'\x1bM', # scroll_reverse (reverse index)
122 r'\x1bc', # full_reset (RIS)
123 ))
124)
125
126
127def iter_sequences(text: str) -> typing.Iterator[typing.Tuple[str, bool]]:
128 r"""
129 Iterate through text, yielding segments with sequence identification.
130
131 This generator yields tuples of ``(segment, is_sequence)`` for each part
132 of the input text, where ``is_sequence`` is ``True`` if the segment is
133 a recognized terminal escape sequence.
134
135 :param text: String to iterate through.
136 :returns: Iterator of (segment, is_sequence) tuples.
137
138 .. versionadded:: 0.3.0
139
140 Example::
141
142 >>> list(iter_sequences('hello'))
143 [('hello', False)]
144 >>> list(iter_sequences('\x1b[31mred'))
145 [('\x1b[31m', True), ('red', False)]
146 >>> list(iter_sequences('\x1b[1m\x1b[31m'))
147 [('\x1b[1m', True), ('\x1b[31m', True)]
148 """
149 idx = 0
150 text_len = len(text)
151 segment_start = 0
152
153 while idx < text_len:
154 char = text[idx]
155
156 if char == '\x1b':
157 # Yield any accumulated non-sequence text
158 if idx > segment_start:
159 yield (text[segment_start:idx], False)
160
161 # Try to match an escape sequence
162 match = ZERO_WIDTH_PATTERN.match(text, idx)
163 if match:
164 yield (match.group(), True)
165 idx = match.end()
166 else:
167 # Lone ESC or unrecognized - yield as sequence anyway
168 yield (char, True)
169 idx += 1
170 segment_start = idx
171 else:
172 idx += 1
173
174 # Yield any remaining text
175 if segment_start < text_len:
176 yield (text[segment_start:], False)
177
178
179def strip_sequences(text: str) -> str:
180 r"""
181 Return text with all terminal escape sequences removed.
182
183 Unknown or incomplete ESC sequences are preserved.
184
185 :param text: String that may contain terminal escape sequences.
186 :returns: The input text with all escape sequences stripped.
187
188 .. versionadded:: 0.3.0
189
190 .. versionchanged:: 0.7.0
191 Inner text of OSC 66 (Text sizing protocol) is preserved.
192
193 Example::
194
195 >>> strip_sequences('\x1b[31mred\x1b[0m')
196 'red'
197 >>> strip_sequences('hello')
198 'hello'
199 >>> strip_sequences('\x1b[1m\x1b[31mbold red\x1b[0m text')
200 'bold red text'
201 >>> strip_sequences('\x1b]66;s=2;hello\x07')
202 'hello'
203 >>> strip_sequences('\x1b]8;id=34;https://example.com\x1b\\[view]\x1b]8;;\x1b\\')
204 '[view]'
205 """
206 if '\x1b]66;' in text:
207 return _STRIP_WITH_TEXT_SIZING.sub(_strip_repl, text)
208 return ZERO_WIDTH_PATTERN.sub('', text)