Coverage for /pythoncovmergedfiles/medio/medio/usr/local/lib/python3.11/site-packages/icalendar/parser/content_line.py: 97%

Shortcuts on this page

r m x   toggle line displays

j k   next/prev highlighted chunk

0   (zero) top of page

1   (one) first highlighted chunk

156 statements  

1"""parsing and generation of content lines""" 

2 

3import re 

4 

5from icalendar.parser.parameter import Parameters 

6from icalendar.parser.property import unescape_backslash, unescape_list_or_string 

7from icalendar.parser.string import ( 

8 _escape_string, 

9 _foldline, 

10 _unescape_string, 

11 validate_token, 

12) 

13from icalendar.parser_tools import DEFAULT_ENCODING, ICAL_TYPE, to_unicode 

14 

15# Equivalent to the natural ``(\r?\n)+[ \t]`` but without its quadratic cost: 

16# the greedy run is re-tried at every line break of a long unfoldable block, so 

17# a megabyte of bare newlines takes seconds. The leading lookbehinds pin a match 

18# to the first line break of a run, making a failed attempt O(1) instead of a 

19# full rescan, while matching exactly the same strings. 

20UFOLD = re.compile(r"(?:(?<!\n)\r\n|(?<![\r\n])\n)(?:\r?\n)*[ \t]") 

21NEWLINE = re.compile(r"\r?\n") 

22 

23OWS = " \t" 

24# ``[ \t]*([;=])[ \t]*`` in one pass rescans a long whitespace run at every 

25# position. Splitting it into two anchored passes (leading then trailing) keeps 

26# the result identical but removes the quadratic blow-up. 

27OWS_BEFORE_DELIMITER_RE = re.compile(r"(?<![ \t])[ \t]+([;=])") 

28OWS_AFTER_DELIMITER_RE = re.compile(r"([;=])[ \t]+") 

29 

30 

31def _strip_ows_around_delimiters(st: str, delimiters: str = ";=") -> str: 

32 """Strip optional whitespace around delimiters outside of quoted sections, 

33 respecting backslash escapes so that escaped delimiters are not treated as 

34 separators. 

35 

36 This is a lenient parsing helper (used when strict=False) to support 

37 iCalendar content lines that contain extra whitespace around tokens. 

38 """ 

39 if not st: 

40 return st 

41 

42 # Fast path for the common case in non-strict mode: 

43 # no whitespace in the parameter section means there is nothing to normalize. 

44 if " " not in st and "\t" not in st: 

45 return st 

46 

47 # Fast regex-based path for simple parameter sections without quoting/escaping. 

48 if delimiters == ";=" and '"' not in st and "\\" not in st: 

49 st = OWS_BEFORE_DELIMITER_RE.sub(r"\1", st) 

50 return OWS_AFTER_DELIMITER_RE.sub(r"\1", st).strip() 

51 

52 out: list[str] = [] 

53 pending_ws: list[str] = [] 

54 in_quotes = False 

55 escaped = False 

56 # True only if the last appended char was a raw delimiter. 

57 last_was_delimiter = False 

58 

59 def flush_pending() -> None: 

60 nonlocal pending_ws 

61 if not pending_ws: 

62 return 

63 if not last_was_delimiter: 

64 out.extend(pending_ws) 

65 pending_ws.clear() 

66 

67 for ch in st: 

68 # Handle escaped character (the backslash set escaped in previous iteration) 

69 if escaped: 

70 flush_pending() 

71 out.append(ch) 

72 escaped = False 

73 last_was_delimiter = False 

74 continue 

75 

76 # Handle backslash to escape next character 

77 if ch == "\\" and not in_quotes: 

78 flush_pending() 

79 out.append(ch) 

80 escaped = True 

81 last_was_delimiter = False 

82 continue 

83 

84 # Handle quote toggling 

85 if ch == '"' and not escaped: 

86 in_quotes = not in_quotes 

87 flush_pending() 

88 out.append(ch) 

89 last_was_delimiter = False 

90 continue 

91 

92 # Whitespace outside quotes is buffered 

93 if not in_quotes and not escaped and ch in OWS: 

94 pending_ws.append(ch) 

95 continue 

96 

97 # Raw delimiter (unescaped and outside quotes) 

98 if not in_quotes and not escaped and ch in delimiters: 

99 pending_ws.clear() 

100 while out and out[-1] in OWS: 

101 out.pop() 

102 out.append(ch) 

103 last_was_delimiter = True 

104 continue 

105 

106 # Regular character 

107 flush_pending() 

108 out.append(ch) 

109 last_was_delimiter = False 

110 

111 if pending_ws and not last_was_delimiter: 

112 out.extend(pending_ws) 

113 

114 return "".join(out).strip() 

115 

116 

117class Contentline(str): 

118 r"""A content line is a string that can be folded and parsed into parts. 

119 

120 Raises: 

121 ValueError: If the value contains ``\n``. 

122 

123 """ 

124 

125 __slots__ = ("strict",) 

126 

127 def __new__(cls, value, strict=False, encoding=DEFAULT_ENCODING): 

128 value = to_unicode(value, encoding=encoding) 

129 if "\n" in value: 

130 raise ValueError( 

131 "Content line can not contain unescaped new line characters." 

132 ) 

133 self = super().__new__(cls, value) 

134 self.strict = strict 

135 return self 

136 

137 @classmethod 

138 def from_parts( 

139 cls, 

140 name: ICAL_TYPE, 

141 params: Parameters, 

142 values, 

143 sorted: bool = True, # noqa: A002 

144 ): 

145 r"""Turn a parts into a content line. 

146 

147 Raises: 

148 TypeError: If ``params`` is not type ``Parameters``. 

149 

150 """ 

151 if not isinstance(params, Parameters): 

152 raise TypeError(f"Expected Parameters, got {type(params).__name__}") 

153 if hasattr(values, "to_ical"): 

154 values = values.to_ical() 

155 else: 

156 from icalendar.prop import vText 

157 

158 values = vText(values).to_ical() 

159 # elif isinstance(values, basestring): 

160 # values = escape_char(values) 

161 

162 # TODO: after unicode only, remove this 

163 # Convert back to unicode, after to_ical encoded it. 

164 name = to_unicode(name) 

165 values = to_unicode(values) 

166 if params: 

167 params = to_unicode(params.to_ical(sorted=sorted)) 

168 if params: 

169 # some parameter values can be skipped during serialization 

170 return cls(f"{name};{params}:{values}") 

171 return cls(f"{name}:{values}") 

172 

173 def raw_parts(self) -> tuple[str, Parameters, str]: 

174 """Split the line into ``name``, ``parameters``, and raw ``values`` parts. 

175 

176 This is :meth:`parts` without the unescaping: the values are returned 

177 verbatim, preserving both backslash sequences and URL encoding. It is 

178 used for :rfc:`7265` ``UNKNOWN`` values, whose real value type—and 

179 therefore whose escaping rules—are not known. 

180 

181 See :meth:`parts` for the parts themselves and for examples. 

182 """ 

183 try: 

184 name_split: int | None = None 

185 value_split: int | None = None 

186 in_quotes: bool = False 

187 escaped: bool = False 

188 

189 for i, ch in enumerate(self): 

190 if ch == '"' and not escaped: 

191 in_quotes = not in_quotes 

192 elif ch == "\\" and not in_quotes: 

193 escaped = True 

194 continue 

195 elif not in_quotes and not escaped: 

196 # Find first delimiter for name 

197 if ch in ":;" and name_split is None: 

198 name_split = i 

199 # Find value delimiter (first colon) 

200 if ch == ":" and value_split is None: 

201 value_split = i 

202 

203 escaped = False 

204 

205 # Validate parsing results 

206 if not value_split: 

207 # No colon found - value is empty, use end of string 

208 value_split = len(self) 

209 

210 # Extract name - if no delimiter, 

211 # take whole string for validate_token to reject 

212 name = self[:name_split] if name_split else self 

213 if not self.strict: 

214 name = re.sub(r"[ \t]+", "", name.strip()) 

215 validate_token(name) 

216 

217 if not name_split or name_split + 1 == value_split: 

218 # No delimiter or empty parameter section 

219 raise ValueError("Invalid content line") # noqa: TRY301 

220 # Parse parameters - they still need to be escaped/unescaped 

221 # for proper handling of commas, semicolons, etc. in parameter values 

222 raw_param_str = self[name_split + 1 : value_split] 

223 if not self.strict: 

224 raw_param_str = _strip_ows_around_delimiters(raw_param_str) 

225 param_str = _escape_string(raw_param_str) 

226 params = Parameters.from_ical(param_str, strict=self.strict) 

227 params = Parameters( 

228 (_unescape_string(key), unescape_list_or_string(value)) 

229 for key, value in iter(params.items()) 

230 ) 

231 values = self[value_split + 1 :] 

232 except ValueError as exc: 

233 raise ValueError( 

234 f"Content line could not be parsed into parts: '{self}': {exc}" 

235 ) from exc 

236 return (name, params, values) 

237 

238 def parts(self) -> tuple[str, Parameters, str]: 

239 """Split the line into ``name``, ``parameters``, and unescaped ``values`` parts. 

240 

241 Properly handles escaping with backslashes and double-quote sections 

242 to avoid corrupting URL-encoded characters in values. 

243 

244 The backslash sequences in the values are unescaped, as for the values 

245 of ``TEXT`` properties, while URL encoding is preserved. Use 

246 :meth:`raw_parts` to get the values verbatim instead. 

247 

248 Examples: 

249 

250 With parameter: 

251 

252 .. code-block:: ics 

253 

254 DESCRIPTION;ALTREP="cid:part1.0001@example.org":The Fall'98 Wild 

255 

256 Without parameters: 

257 

258 .. code-block:: ics 

259 

260 DESCRIPTION:The Fall'98 Wild 

261 """ 

262 name, params, values = self.raw_parts() 

263 return (name, params, unescape_backslash(values)) 

264 

265 def value_separator_index(self) -> int: 

266 r"""Return the index of the colon that separates the value. 

267 

268 This is the first colon that is not inside a quoted parameter section. 

269 A colon inside a quoted parameter value (for example 

270 ``ALTREP="http://x"``) is skipped, and a colon that belongs to the 

271 value (``TEXT`` does not escape ``:``) is not mistaken for the 

272 separator. Backslash has no special meaning in the parameter grammar 

273 (:rfc:`5545#section-3.1`), so it is treated as an ordinary character. 

274 

275 Returns: 

276 An integer representing the index position of the separator, 

277 or ``-1`` if there is none. 

278 """ 

279 in_quotes = False 

280 for i, ch in enumerate(self): 

281 if ch == '"': 

282 in_quotes = not in_quotes 

283 elif ch == ":" and not in_quotes: 

284 return i 

285 return -1 

286 

287 @classmethod 

288 def from_ical(cls, ical, strict=False): 

289 """Unfold the content lines in an iCalendar into long content lines.""" 

290 ical = to_unicode(ical) 

291 # a fold is carriage return followed by either a space or a tab 

292 return cls(UFOLD.sub("", ical), strict=strict) 

293 

294 def to_ical(self): 

295 """Long content lines are folded so they are less than 75 characters 

296 wide. 

297 """ 

298 return _foldline(self).encode(DEFAULT_ENCODING) 

299 

300 

301class Contentlines(list[Contentline]): 

302 """I assume that iCalendar files generally are a few kilobytes in size. 

303 Then this should be efficient. for Huge files, an iterator should probably 

304 be used instead. 

305 """ 

306 

307 def to_ical(self): 

308 """Simply join self.""" 

309 return b"\r\n".join(line.to_ical() for line in self if line) + b"\r\n" 

310 

311 @classmethod 

312 def from_ical(cls, st): 

313 """Parses a string into content lines.""" 

314 st = to_unicode(st) 

315 try: 

316 # a fold is carriage return followed by either a space or a tab 

317 unfolded = UFOLD.sub("", st) 

318 lines = cls(Contentline(line) for line in NEWLINE.split(unfolded) if line) 

319 lines.append("") # '\r\n' at the end of every content line 

320 except Exception as e: 

321 raise ValueError("Expected StringType with content lines") from e 

322 return lines 

323 

324 

325__all__ = ["Contentline", "Contentlines"]