1"""parsing and generation of content lines"""
2
3import re
4
5from icalendar.parser.parameter import Parameters
6from icalendar.parser.property import unescape_backslash, unescape_list_or_string
7from icalendar.parser.string import (
8 _escape_string,
9 _foldline,
10 _unescape_string,
11 validate_token,
12)
13from icalendar.parser_tools import DEFAULT_ENCODING, ICAL_TYPE, to_unicode
14
15# Equivalent to the natural ``(\r?\n)+[ \t]`` but without its quadratic cost:
16# the greedy run is re-tried at every line break of a long unfoldable block, so
17# a megabyte of bare newlines takes seconds. The leading lookbehinds pin a match
18# to the first line break of a run, making a failed attempt O(1) instead of a
19# full rescan, while matching exactly the same strings.
20UFOLD = re.compile(r"(?:(?<!\n)\r\n|(?<![\r\n])\n)(?:\r?\n)*[ \t]")
21NEWLINE = re.compile(r"\r?\n")
22
23OWS = " \t"
24# ``[ \t]*([;=])[ \t]*`` in one pass rescans a long whitespace run at every
25# position. Splitting it into two anchored passes (leading then trailing) keeps
26# the result identical but removes the quadratic blow-up.
27OWS_BEFORE_DELIMITER_RE = re.compile(r"(?<![ \t])[ \t]+([;=])")
28OWS_AFTER_DELIMITER_RE = re.compile(r"([;=])[ \t]+")
29
30
31def _strip_ows_around_delimiters(st: str, delimiters: str = ";=") -> str:
32 """Strip optional whitespace around delimiters outside of quoted sections,
33 respecting backslash escapes so that escaped delimiters are not treated as
34 separators.
35
36 This is a lenient parsing helper (used when strict=False) to support
37 iCalendar content lines that contain extra whitespace around tokens.
38 """
39 if not st:
40 return st
41
42 # Fast path for the common case in non-strict mode:
43 # no whitespace in the parameter section means there is nothing to normalize.
44 if " " not in st and "\t" not in st:
45 return st
46
47 # Fast regex-based path for simple parameter sections without quoting/escaping.
48 if delimiters == ";=" and '"' not in st and "\\" not in st:
49 st = OWS_BEFORE_DELIMITER_RE.sub(r"\1", st)
50 return OWS_AFTER_DELIMITER_RE.sub(r"\1", st).strip()
51
52 out: list[str] = []
53 pending_ws: list[str] = []
54 in_quotes = False
55 escaped = False
56 # True only if the last appended char was a raw delimiter.
57 last_was_delimiter = False
58
59 def flush_pending() -> None:
60 nonlocal pending_ws
61 if not pending_ws:
62 return
63 if not last_was_delimiter:
64 out.extend(pending_ws)
65 pending_ws.clear()
66
67 for ch in st:
68 # Handle escaped character (the backslash set escaped in previous iteration)
69 if escaped:
70 flush_pending()
71 out.append(ch)
72 escaped = False
73 last_was_delimiter = False
74 continue
75
76 # Handle backslash to escape next character
77 if ch == "\\" and not in_quotes:
78 flush_pending()
79 out.append(ch)
80 escaped = True
81 last_was_delimiter = False
82 continue
83
84 # Handle quote toggling
85 if ch == '"' and not escaped:
86 in_quotes = not in_quotes
87 flush_pending()
88 out.append(ch)
89 last_was_delimiter = False
90 continue
91
92 # Whitespace outside quotes is buffered
93 if not in_quotes and not escaped and ch in OWS:
94 pending_ws.append(ch)
95 continue
96
97 # Raw delimiter (unescaped and outside quotes)
98 if not in_quotes and not escaped and ch in delimiters:
99 pending_ws.clear()
100 while out and out[-1] in OWS:
101 out.pop()
102 out.append(ch)
103 last_was_delimiter = True
104 continue
105
106 # Regular character
107 flush_pending()
108 out.append(ch)
109 last_was_delimiter = False
110
111 if pending_ws and not last_was_delimiter:
112 out.extend(pending_ws)
113
114 return "".join(out).strip()
115
116
117class Contentline(str):
118 r"""A content line is a string that can be folded and parsed into parts.
119
120 Raises:
121 ValueError: If the value contains ``\n``.
122
123 """
124
125 __slots__ = ("strict",)
126
127 def __new__(cls, value, strict=False, encoding=DEFAULT_ENCODING):
128 value = to_unicode(value, encoding=encoding)
129 if "\n" in value:
130 raise ValueError(
131 "Content line can not contain unescaped new line characters."
132 )
133 self = super().__new__(cls, value)
134 self.strict = strict
135 return self
136
137 @classmethod
138 def from_parts(
139 cls,
140 name: ICAL_TYPE,
141 params: Parameters,
142 values,
143 sorted: bool = True, # noqa: A002
144 ):
145 r"""Turn a parts into a content line.
146
147 Raises:
148 TypeError: If ``params`` is not type ``Parameters``.
149
150 """
151 if not isinstance(params, Parameters):
152 raise TypeError(f"Expected Parameters, got {type(params).__name__}")
153 if hasattr(values, "to_ical"):
154 values = values.to_ical()
155 else:
156 from icalendar.prop import vText
157
158 values = vText(values).to_ical()
159 # elif isinstance(values, basestring):
160 # values = escape_char(values)
161
162 # TODO: after unicode only, remove this
163 # Convert back to unicode, after to_ical encoded it.
164 name = to_unicode(name)
165 values = to_unicode(values)
166 if params:
167 params = to_unicode(params.to_ical(sorted=sorted))
168 if params:
169 # some parameter values can be skipped during serialization
170 return cls(f"{name};{params}:{values}")
171 return cls(f"{name}:{values}")
172
173 def raw_parts(self) -> tuple[str, Parameters, str]:
174 """Split the line into ``name``, ``parameters``, and raw ``values`` parts.
175
176 This is :meth:`parts` without the unescaping: the values are returned
177 verbatim, preserving both backslash sequences and URL encoding. It is
178 used for :rfc:`7265` ``UNKNOWN`` values, whose real value type—and
179 therefore whose escaping rules—are not known.
180
181 See :meth:`parts` for the parts themselves and for examples.
182 """
183 try:
184 name_split: int | None = None
185 value_split: int | None = None
186 in_quotes: bool = False
187 escaped: bool = False
188
189 for i, ch in enumerate(self):
190 if ch == '"' and not escaped:
191 in_quotes = not in_quotes
192 elif ch == "\\" and not in_quotes:
193 escaped = True
194 continue
195 elif not in_quotes and not escaped:
196 # Find first delimiter for name
197 if ch in ":;" and name_split is None:
198 name_split = i
199 # Find value delimiter (first colon)
200 if ch == ":" and value_split is None:
201 value_split = i
202
203 escaped = False
204
205 # Validate parsing results
206 if not value_split:
207 # No colon found - value is empty, use end of string
208 value_split = len(self)
209
210 # Extract name - if no delimiter,
211 # take whole string for validate_token to reject
212 name = self[:name_split] if name_split else self
213 if not self.strict:
214 name = re.sub(r"[ \t]+", "", name.strip())
215 validate_token(name)
216
217 if not name_split or name_split + 1 == value_split:
218 # No delimiter or empty parameter section
219 raise ValueError("Invalid content line") # noqa: TRY301
220 # Parse parameters - they still need to be escaped/unescaped
221 # for proper handling of commas, semicolons, etc. in parameter values
222 raw_param_str = self[name_split + 1 : value_split]
223 if not self.strict:
224 raw_param_str = _strip_ows_around_delimiters(raw_param_str)
225 param_str = _escape_string(raw_param_str)
226 params = Parameters.from_ical(param_str, strict=self.strict)
227 params = Parameters(
228 (_unescape_string(key), unescape_list_or_string(value))
229 for key, value in iter(params.items())
230 )
231 values = self[value_split + 1 :]
232 except ValueError as exc:
233 raise ValueError(
234 f"Content line could not be parsed into parts: '{self}': {exc}"
235 ) from exc
236 return (name, params, values)
237
238 def parts(self) -> tuple[str, Parameters, str]:
239 """Split the line into ``name``, ``parameters``, and unescaped ``values`` parts.
240
241 Properly handles escaping with backslashes and double-quote sections
242 to avoid corrupting URL-encoded characters in values.
243
244 The backslash sequences in the values are unescaped, as for the values
245 of ``TEXT`` properties, while URL encoding is preserved. Use
246 :meth:`raw_parts` to get the values verbatim instead.
247
248 Examples:
249
250 With parameter:
251
252 .. code-block:: ics
253
254 DESCRIPTION;ALTREP="cid:part1.0001@example.org":The Fall'98 Wild
255
256 Without parameters:
257
258 .. code-block:: ics
259
260 DESCRIPTION:The Fall'98 Wild
261 """
262 name, params, values = self.raw_parts()
263 return (name, params, unescape_backslash(values))
264
265 def value_separator_index(self) -> int:
266 r"""Return the index of the colon that separates the value.
267
268 This is the first colon that is not inside a quoted parameter section.
269 A colon inside a quoted parameter value (for example
270 ``ALTREP="http://x"``) is skipped, and a colon that belongs to the
271 value (``TEXT`` does not escape ``:``) is not mistaken for the
272 separator. Backslash has no special meaning in the parameter grammar
273 (:rfc:`5545#section-3.1`), so it is treated as an ordinary character.
274
275 Returns:
276 An integer representing the index position of the separator,
277 or ``-1`` if there is none.
278 """
279 in_quotes = False
280 for i, ch in enumerate(self):
281 if ch == '"':
282 in_quotes = not in_quotes
283 elif ch == ":" and not in_quotes:
284 return i
285 return -1
286
287 @classmethod
288 def from_ical(cls, ical, strict=False):
289 """Unfold the content lines in an iCalendar into long content lines."""
290 ical = to_unicode(ical)
291 # a fold is carriage return followed by either a space or a tab
292 return cls(UFOLD.sub("", ical), strict=strict)
293
294 def to_ical(self):
295 """Long content lines are folded so they are less than 75 characters
296 wide.
297 """
298 return _foldline(self).encode(DEFAULT_ENCODING)
299
300
301class Contentlines(list[Contentline]):
302 """I assume that iCalendar files generally are a few kilobytes in size.
303 Then this should be efficient. for Huge files, an iterator should probably
304 be used instead.
305 """
306
307 def to_ical(self):
308 """Simply join self."""
309 return b"\r\n".join(line.to_ical() for line in self if line) + b"\r\n"
310
311 @classmethod
312 def from_ical(cls, st):
313 """Parses a string into content lines."""
314 st = to_unicode(st)
315 try:
316 # a fold is carriage return followed by either a space or a tab
317 unfolded = UFOLD.sub("", st)
318 lines = cls(Contentline(line) for line in NEWLINE.split(unfolded) if line)
319 lines.append("") # '\r\n' at the end of every content line
320 except Exception as e:
321 raise ValueError("Expected StringType with content lines") from e
322 return lines
323
324
325__all__ = ["Contentline", "Contentlines"]