1"""Functions for manipulating strings and bytes."""
2
3import re
4
5from icalendar.compatibility import deprecate_for_version_8
6from icalendar.parser_tools import DEFAULT_ENCODING, to_unicode
7
8
9def _escape_char(text: str | bytes) -> str:
10 r"""Format value according to iCalendar TEXT escaping rules.
11
12 Escapes special characters in text values according to :rfc:`5545#section-3.3.11`
13 rules.
14 The order of replacements matters to avoid double-escaping.
15
16 Parameters:
17 text: The text to escape.
18
19 Returns:
20 The escaped text with special characters escaped.
21
22 Raises:
23 TypeError: If ``text`` is neither ``str`` nor ``bytes``.
24
25 Note:
26 The replacement order is critical:
27
28 1. ``\N`` -> ``\n`` (normalize newlines to lowercase)
29 2. ``\`` -> ``\\`` (escape backslashes)
30 3. ``;`` -> ``\;`` (escape semicolons)
31 4. ``,`` -> ``\,`` (escape commas)
32 5. ``\r\n`` -> ``\n`` (normalize line endings)
33 6. ``"\n"`` -> ``r"\n"`` (transform a newline character to a literal, or raw,
34 newline character)
35 7. ``"\r"`` -> ``r"\n"`` (transform a lone carriage return to a literal
36 newline character)
37
38 Steps 5 to 7 normalize ``\r\n``, ``\n``, or a lone ``\r`` to ``\n``.
39 The line-ending normalization is an implementation convenience,
40 not part of :rfc:`5545`, which only defines ``\n`` or ``\N`` for an
41 intentional line break, and doesn't give an escape form for a lone ``\r``.
42 """
43 if not isinstance(text, (str, bytes)):
44 raise TypeError(f"Expected str or bytes, got {type(text).__name__}")
45 text = to_unicode(text)
46 # NOTE: ORDER MATTERS!
47 return (
48 text.replace(r"\N", "\n")
49 .replace("\\", "\\\\")
50 .replace(";", r"\;")
51 .replace(",", r"\,")
52 .replace("\r\n", r"\n")
53 .replace("\n", r"\n")
54 .replace("\r", r"\n")
55 )
56
57
58escape_char = deprecate_for_version_8(_escape_char)
59"""Format value according to iCalendar TEXT escaping rules.
60
61.. deprecated:: 7.0.0
62 Use the private :func:`_escape_char` internally. For external use,
63 this function is deprecated. Please use alternative escaping methods
64 or contact the maintainers.
65"""
66
67
68def _unescape_char(text: str | bytes) -> str | bytes:
69 r"""Unescape iCalendar TEXT values.
70
71 Reverses the escaping applied by :func:`_escape_char` according to
72 :rfc:`5545#section-3.3.11` TEXT escaping rules.
73
74 Parameters:
75 text: The escaped text.
76
77 Returns:
78 The unescaped text.
79
80 Raises:
81 TypeError: If ``text`` is neither ``str`` nor ``bytes``.
82
83 Note:
84 The replacement order is critical to avoid double-unescaping:
85
86 1. ``\N`` -> ``\n`` (intermediate step)
87 2. ``\r\n`` -> ``\n`` (normalize line endings)
88 3. ``\n`` -> newline (unescape newlines)
89 4. ``\,`` -> ``,`` (unescape commas)
90 5. ``\;`` -> ``;`` (unescape semicolons)
91 6. ``\\`` -> ``\`` (unescape backslashes last)
92 """
93 if not isinstance(text, (str, bytes)):
94 raise TypeError(f"Expected str or bytes, got {type(text).__name__}")
95 # NOTE: ORDER MATTERS!
96 if isinstance(text, str):
97 return (
98 text.replace("\\N", "\\n")
99 .replace("\r\n", "\n")
100 .replace("\\n", "\n")
101 .replace("\\,", ",")
102 .replace("\\;", ";")
103 .replace("\\\\", "\\")
104 )
105 return (
106 text.replace(b"\\N", b"\\n")
107 .replace(b"\r\n", b"\n")
108 .replace(b"\\n", b"\n")
109 .replace(b"\\,", b",")
110 .replace(b"\\;", b";")
111 .replace(b"\\\\", b"\\")
112 )
113
114
115unescape_char = deprecate_for_version_8(_unescape_char)
116"""Unescape iCalendar TEXT values.
117
118.. deprecated:: 7.0.0
119 Use the private :func:`_unescape_char` internally. For external use,
120 this function is deprecated. Please use alternative unescaping methods
121 or contact the maintainers.
122"""
123
124
125def _foldline(line: str, limit: int = 75, fold_sep: str = "\r\n ") -> str:
126 r"""Make a string folded as defined in :rfc:`5545#section-3.1`.
127
128 Lines of text SHOULD NOT be longer than 75 octets, excluding the line
129 break. Long content lines SHOULD be split into a multiple line
130 representations using a line "folding" technique. That is, a long
131 line can be split between any two characters by inserting a CRLF
132 immediately followed by a single linear white-space character (i.e.,
133 SPACE or HTAB).
134
135 Raises:
136 TypeError: If ``line`` is not a ``str``.
137 ValueError: If ``line`` contains ``\n``.
138
139 """
140 if not isinstance(line, str):
141 raise TypeError(f"Expected str, got {type(line).__name__}")
142 if "\n" in line:
143 raise ValueError("Line must not contain unescaped new line characters.")
144
145 folded_lines: list[str] = []
146 current_chars: list[str] = []
147 byte_count = 0
148 for char in line:
149 char_byte_len = len(char.encode(DEFAULT_ENCODING))
150 if current_chars and byte_count + char_byte_len >= limit:
151 # For compatibility with existing clients, avoid splitting escaped
152 # values such as TEXT backslash escapes or RFC 6868 parameter
153 # escapes across a folded line boundary. See issue #1501.
154 if len(current_chars) > 1 and current_chars[-1] in r"\^":
155 escaped_prefix = current_chars.pop()
156 folded_lines.append("".join(current_chars))
157 current_chars = [escaped_prefix]
158 byte_count = len(escaped_prefix.encode(DEFAULT_ENCODING))
159 else:
160 folded_lines.append("".join(current_chars))
161 current_chars = []
162 byte_count = 0
163 current_chars.append(char)
164 byte_count += char_byte_len
165
166 if current_chars:
167 folded_lines.append("".join(current_chars))
168
169 return fold_sep.join(folded_lines)
170
171
172foldline = deprecate_for_version_8(_foldline)
173"""Make a string folded as defined in RFC5545.
174
175.. deprecated:: 7.0.0
176 Use the private :func:`_foldline` internally.
177"""
178
179
180def _escape_string(val: str) -> str:
181 r"""Escape backslash sequences to URL-encoded hex values.
182
183 Converts backslash-escaped characters to their percent-encoded hex
184 equivalents. This is used for parameter parsing to preserve escaped
185 characters during processing.
186
187 Parameters:
188 val: The string with backslash escapes.
189
190 Returns:
191 The string with backslash escapes converted to percent encoding.
192
193 Note:
194 Conversions:
195
196 - ``%`` -> ``%25``
197 - ``\,`` -> ``%2C``
198 - ``\:`` -> ``%3A``
199 - ``\;`` -> ``%3B``
200 - ``\\`` -> ``%5C``
201
202 A literal ``%`` is escaped first so that percent sequences already in
203 the value (e.g. ``%2C`` in a URI) are not confused with the markers
204 introduced here. :func:`_unescape_string` reverses it.
205 """
206 # f'{i:02X}'
207 return (
208 val.replace("%", "%25")
209 .replace(r"\,", "%2C")
210 .replace(r"\:", "%3A")
211 .replace(r"\;", "%3B")
212 .replace(r"\\", "%5C")
213 )
214
215
216escape_string = deprecate_for_version_8(_escape_string)
217"""Escape backslash sequences to URL-encoded hex values.
218
219.. deprecated:: 7.0.0
220 Use the private :func:`_escape_string` internally. For external use,
221 this function is deprecated.
222"""
223
224
225def _unescape_string(val: str) -> str:
226 r"""Unescape URL-encoded hex values to their original characters.
227
228 Reverses :func:`_escape_string` by converting percent-encoded hex values
229 back to their original characters. This is used for parameter parsing.
230
231 Parameters:
232 val: The string with percent-encoded values.
233
234 Returns:
235 The string with percent encoding converted to characters.
236
237 Note:
238 Conversions:
239
240 - ``%2C`` -> ``,``
241 - ``%3A`` -> ``:``
242 - ``%3B`` -> ``;``
243 - ``%5C`` -> ``\``
244 - ``%25`` -> ``%``
245
246 ``%25`` is restored last so a literal ``%`` that :func:`_escape_string`
247 protected does not re-trigger the marker replacements above.
248 """
249 return (
250 val.replace("%2C", ",")
251 .replace("%3A", ":")
252 .replace("%3B", ";")
253 .replace("%5C", "\\")
254 .replace("%25", "%")
255 )
256
257
258unescape_string = deprecate_for_version_8(_unescape_string)
259"""Unescape URL-encoded hex values to their original characters.
260
261.. deprecated:: 7.0.0
262 Use the private :func:`_unescape_string` internally. For external use,
263 this function is deprecated.
264"""
265
266
267# [\w-] because of the iCalendar RFC
268# . because of the vCard RFC
269NAME = re.compile(r"[\w.-]+")
270
271
272def validate_token(name: str) -> None:
273 r"""Validate that a name is a valid iCalendar token.
274
275 Checks if the name matches the :rfc:`5545` token syntax using the NAME
276 regex pattern (``[\w.-]+``).
277
278 Parameters:
279 name: The token name to validate.
280
281 Raises:
282 ValueError: If the name is not a valid token.
283 """
284 match = NAME.findall(name)
285 if len(match) == 1 and name == match[0]:
286 return
287 raise ValueError(name)
288
289
290__all__ = [
291 "_escape_char",
292 "_escape_string",
293 "_foldline",
294 "_unescape_char",
295 "_unescape_string",
296 "escape_char",
297 "escape_string",
298 "foldline",
299 "unescape_char",
300 "unescape_string",
301 "validate_token",
302]