Coverage for /pythoncovmergedfiles/medio/medio/usr/local/lib/python3.11/site-packages/icalendar/parser/string.py: 88%

Shortcuts on this page

r m x   toggle line displays

j k   next/prev highlighted chunk

0   (zero) top of page

1   (one) first highlighted chunk

60 statements  

1"""Functions for manipulating strings and bytes.""" 

2 

3import re 

4 

5from icalendar.compatibility import deprecate_for_version_8 

6from icalendar.parser_tools import DEFAULT_ENCODING, to_unicode 

7 

8 

9def _escape_char(text: str | bytes) -> str: 

10 r"""Format value according to iCalendar TEXT escaping rules. 

11 

12 Escapes special characters in text values according to :rfc:`5545#section-3.3.11` 

13 rules. 

14 The order of replacements matters to avoid double-escaping. 

15 

16 Parameters: 

17 text: The text to escape. 

18 

19 Returns: 

20 The escaped text with special characters escaped. 

21 

22 Raises: 

23 TypeError: If ``text`` is neither ``str`` nor ``bytes``. 

24 

25 Note: 

26 The replacement order is critical: 

27 

28 1. ``\N`` -> ``\n`` (normalize newlines to lowercase) 

29 2. ``\`` -> ``\\`` (escape backslashes) 

30 3. ``;`` -> ``\;`` (escape semicolons) 

31 4. ``,`` -> ``\,`` (escape commas) 

32 5. ``\r\n`` -> ``\n`` (normalize line endings) 

33 6. ``"\n"`` -> ``r"\n"`` (transform a newline character to a literal, or raw, 

34 newline character) 

35 7. ``"\r"`` -> ``r"\n"`` (transform a lone carriage return to a literal 

36 newline character) 

37 

38 Steps 5 to 7 normalize ``\r\n``, ``\n``, or a lone ``\r`` to ``\n``. 

39 The line-ending normalization is an implementation convenience, 

40 not part of :rfc:`5545`, which only defines ``\n`` or ``\N`` for an 

41 intentional line break, and doesn't give an escape form for a lone ``\r``. 

42 """ 

43 if not isinstance(text, (str, bytes)): 

44 raise TypeError(f"Expected str or bytes, got {type(text).__name__}") 

45 text = to_unicode(text) 

46 # NOTE: ORDER MATTERS! 

47 return ( 

48 text.replace(r"\N", "\n") 

49 .replace("\\", "\\\\") 

50 .replace(";", r"\;") 

51 .replace(",", r"\,") 

52 .replace("\r\n", r"\n") 

53 .replace("\n", r"\n") 

54 .replace("\r", r"\n") 

55 ) 

56 

57 

58escape_char = deprecate_for_version_8(_escape_char) 

59"""Format value according to iCalendar TEXT escaping rules. 

60 

61.. deprecated:: 7.0.0 

62 Use the private :func:`_escape_char` internally. For external use, 

63 this function is deprecated. Please use alternative escaping methods 

64 or contact the maintainers. 

65""" 

66 

67 

68def _unescape_char(text: str | bytes) -> str | bytes: 

69 r"""Unescape iCalendar TEXT values. 

70 

71 Reverses the escaping applied by :func:`_escape_char` according to 

72 :rfc:`5545#section-3.3.11` TEXT escaping rules. 

73 

74 Parameters: 

75 text: The escaped text. 

76 

77 Returns: 

78 The unescaped text. 

79 

80 Raises: 

81 TypeError: If ``text`` is neither ``str`` nor ``bytes``. 

82 

83 Note: 

84 The replacement order is critical to avoid double-unescaping: 

85 

86 1. ``\N`` -> ``\n`` (intermediate step) 

87 2. ``\r\n`` -> ``\n`` (normalize line endings) 

88 3. ``\n`` -> newline (unescape newlines) 

89 4. ``\,`` -> ``,`` (unescape commas) 

90 5. ``\;`` -> ``;`` (unescape semicolons) 

91 6. ``\\`` -> ``\`` (unescape backslashes last) 

92 """ 

93 if not isinstance(text, (str, bytes)): 

94 raise TypeError(f"Expected str or bytes, got {type(text).__name__}") 

95 # NOTE: ORDER MATTERS! 

96 if isinstance(text, str): 

97 return ( 

98 text.replace("\\N", "\\n") 

99 .replace("\r\n", "\n") 

100 .replace("\\n", "\n") 

101 .replace("\\,", ",") 

102 .replace("\\;", ";") 

103 .replace("\\\\", "\\") 

104 ) 

105 return ( 

106 text.replace(b"\\N", b"\\n") 

107 .replace(b"\r\n", b"\n") 

108 .replace(b"\\n", b"\n") 

109 .replace(b"\\,", b",") 

110 .replace(b"\\;", b";") 

111 .replace(b"\\\\", b"\\") 

112 ) 

113 

114 

115unescape_char = deprecate_for_version_8(_unescape_char) 

116"""Unescape iCalendar TEXT values. 

117 

118.. deprecated:: 7.0.0 

119 Use the private :func:`_unescape_char` internally. For external use, 

120 this function is deprecated. Please use alternative unescaping methods 

121 or contact the maintainers. 

122""" 

123 

124 

125def _foldline(line: str, limit: int = 75, fold_sep: str = "\r\n ") -> str: 

126 r"""Make a string folded as defined in :rfc:`5545#section-3.1`. 

127 

128 Lines of text SHOULD NOT be longer than 75 octets, excluding the line 

129 break. Long content lines SHOULD be split into a multiple line 

130 representations using a line "folding" technique. That is, a long 

131 line can be split between any two characters by inserting a CRLF 

132 immediately followed by a single linear white-space character (i.e., 

133 SPACE or HTAB). 

134 

135 Raises: 

136 TypeError: If ``line`` is not a ``str``. 

137 ValueError: If ``line`` contains ``\n``. 

138 

139 """ 

140 if not isinstance(line, str): 

141 raise TypeError(f"Expected str, got {type(line).__name__}") 

142 if "\n" in line: 

143 raise ValueError("Line must not contain unescaped new line characters.") 

144 

145 folded_lines: list[str] = [] 

146 current_chars: list[str] = [] 

147 byte_count = 0 

148 for char in line: 

149 char_byte_len = len(char.encode(DEFAULT_ENCODING)) 

150 if current_chars and byte_count + char_byte_len >= limit: 

151 # For compatibility with existing clients, avoid splitting escaped 

152 # values such as TEXT backslash escapes or RFC 6868 parameter 

153 # escapes across a folded line boundary. See issue #1501. 

154 if len(current_chars) > 1 and current_chars[-1] in r"\^": 

155 escaped_prefix = current_chars.pop() 

156 folded_lines.append("".join(current_chars)) 

157 current_chars = [escaped_prefix] 

158 byte_count = len(escaped_prefix.encode(DEFAULT_ENCODING)) 

159 else: 

160 folded_lines.append("".join(current_chars)) 

161 current_chars = [] 

162 byte_count = 0 

163 current_chars.append(char) 

164 byte_count += char_byte_len 

165 

166 if current_chars: 

167 folded_lines.append("".join(current_chars)) 

168 

169 return fold_sep.join(folded_lines) 

170 

171 

172foldline = deprecate_for_version_8(_foldline) 

173"""Make a string folded as defined in RFC5545. 

174 

175.. deprecated:: 7.0.0 

176 Use the private :func:`_foldline` internally. 

177""" 

178 

179 

180def _escape_string(val: str) -> str: 

181 r"""Escape backslash sequences to URL-encoded hex values. 

182 

183 Converts backslash-escaped characters to their percent-encoded hex 

184 equivalents. This is used for parameter parsing to preserve escaped 

185 characters during processing. 

186 

187 Parameters: 

188 val: The string with backslash escapes. 

189 

190 Returns: 

191 The string with backslash escapes converted to percent encoding. 

192 

193 Note: 

194 Conversions: 

195 

196 - ``%`` -> ``%25`` 

197 - ``\,`` -> ``%2C`` 

198 - ``\:`` -> ``%3A`` 

199 - ``\;`` -> ``%3B`` 

200 - ``\\`` -> ``%5C`` 

201 

202 A literal ``%`` is escaped first so that percent sequences already in 

203 the value (e.g. ``%2C`` in a URI) are not confused with the markers 

204 introduced here. :func:`_unescape_string` reverses it. 

205 """ 

206 # f'{i:02X}' 

207 return ( 

208 val.replace("%", "%25") 

209 .replace(r"\,", "%2C") 

210 .replace(r"\:", "%3A") 

211 .replace(r"\;", "%3B") 

212 .replace(r"\\", "%5C") 

213 ) 

214 

215 

216escape_string = deprecate_for_version_8(_escape_string) 

217"""Escape backslash sequences to URL-encoded hex values. 

218 

219.. deprecated:: 7.0.0 

220 Use the private :func:`_escape_string` internally. For external use, 

221 this function is deprecated. 

222""" 

223 

224 

225def _unescape_string(val: str) -> str: 

226 r"""Unescape URL-encoded hex values to their original characters. 

227 

228 Reverses :func:`_escape_string` by converting percent-encoded hex values 

229 back to their original characters. This is used for parameter parsing. 

230 

231 Parameters: 

232 val: The string with percent-encoded values. 

233 

234 Returns: 

235 The string with percent encoding converted to characters. 

236 

237 Note: 

238 Conversions: 

239 

240 - ``%2C`` -> ``,`` 

241 - ``%3A`` -> ``:`` 

242 - ``%3B`` -> ``;`` 

243 - ``%5C`` -> ``\`` 

244 - ``%25`` -> ``%`` 

245 

246 ``%25`` is restored last so a literal ``%`` that :func:`_escape_string` 

247 protected does not re-trigger the marker replacements above. 

248 """ 

249 return ( 

250 val.replace("%2C", ",") 

251 .replace("%3A", ":") 

252 .replace("%3B", ";") 

253 .replace("%5C", "\\") 

254 .replace("%25", "%") 

255 ) 

256 

257 

258unescape_string = deprecate_for_version_8(_unescape_string) 

259"""Unescape URL-encoded hex values to their original characters. 

260 

261.. deprecated:: 7.0.0 

262 Use the private :func:`_unescape_string` internally. For external use, 

263 this function is deprecated. 

264""" 

265 

266 

267# [\w-] because of the iCalendar RFC 

268# . because of the vCard RFC 

269NAME = re.compile(r"[\w.-]+") 

270 

271 

272def validate_token(name: str) -> None: 

273 r"""Validate that a name is a valid iCalendar token. 

274 

275 Checks if the name matches the :rfc:`5545` token syntax using the NAME 

276 regex pattern (``[\w.-]+``). 

277 

278 Parameters: 

279 name: The token name to validate. 

280 

281 Raises: 

282 ValueError: If the name is not a valid token. 

283 """ 

284 match = NAME.findall(name) 

285 if len(match) == 1 and name == match[0]: 

286 return 

287 raise ValueError(name) 

288 

289 

290__all__ = [ 

291 "_escape_char", 

292 "_escape_string", 

293 "_foldline", 

294 "_unescape_char", 

295 "_unescape_string", 

296 "escape_char", 

297 "escape_string", 

298 "foldline", 

299 "unescape_char", 

300 "unescape_string", 

301 "validate_token", 

302]