Coverage for /pythoncovmergedfiles/medio/medio/usr/local/lib/python3.11/site-packages/werkzeug/urls.py: 33%

Shortcuts on this page

r m x   toggle line displays

j k   next/prev highlighted chunk

0   (zero) top of page

1   (one) first highlighted chunk

91 statements  

1from __future__ import annotations 

2 

3import codecs 

4import re 

5import typing as t 

6import urllib.parse 

7from urllib.parse import quote 

8from urllib.parse import unquote 

9from urllib.parse import urlencode 

10from urllib.parse import urlsplit 

11from urllib.parse import urlunsplit 

12 

13from .datastructures import iter_multi_items 

14 

15 

16def _codec_error_url_quote(e: UnicodeError) -> tuple[str, int]: 

17 """Used in :func:`uri_to_iri` after unquoting to re-quote any 

18 invalid bytes. 

19 """ 

20 # the docs state that UnicodeError does have these attributes, 

21 # but mypy isn't picking them up 

22 out = quote(e.object[e.start : e.end], safe="") # type: ignore 

23 return out, e.end # type: ignore 

24 

25 

26codecs.register_error("werkzeug.url_quote", _codec_error_url_quote) 

27 

28 

29def _make_unquote_part(name: str, chars: str) -> t.Callable[[str], str]: 

30 """Create a function that unquotes all percent encoded characters except those 

31 given. This allows working with unquoted characters if possible while not changing 

32 the meaning of a given part of a URL. 

33 """ 

34 choices = "|".join(f"{ord(c):02X}" for c in sorted(chars)) 

35 pattern = re.compile(f"((?:%(?:{choices}))+)", re.I) 

36 

37 def _unquote_partial(value: str) -> str: 

38 parts = iter(pattern.split(value)) 

39 out = [] 

40 

41 for part in parts: 

42 out.append(unquote(part, "utf-8", "werkzeug.url_quote")) 

43 out.append(next(parts, "")) 

44 

45 return "".join(out) 

46 

47 _unquote_partial.__name__ = f"_unquote_{name}" 

48 return _unquote_partial 

49 

50 

51# characters that should remain quoted in URL parts 

52# based on https://url.spec.whatwg.org/#percent-encoded-bytes 

53# always keep all controls, space, and % quoted 

54_always_unsafe = bytes((*range(0x21), 0x25, 0x7F)).decode() 

55_unquote_fragment = _make_unquote_part("fragment", _always_unsafe) 

56_unquote_query = _make_unquote_part("query", _always_unsafe + "&=+#") 

57_unquote_path = _make_unquote_part("path", _always_unsafe + "/?#") 

58_unquote_user = _make_unquote_part("user", _always_unsafe + ":@/?#") 

59 

60 

61def uri_to_iri(uri: str) -> str: 

62 """Convert a URI to an IRI. All valid UTF-8 characters are unquoted, 

63 leaving all reserved and invalid characters quoted. If the URL has 

64 a domain, it is decoded from Punycode. 

65 

66 >>> uri_to_iri("http://xn--n3h.net/p%C3%A5th?q=%C3%A8ry%DF") 

67 'http://\\u2603.net/p\\xe5th?q=\\xe8ry%DF' 

68 

69 :param uri: The URI to convert. 

70 

71 .. versionchanged:: 3.1.9 

72 Empty username, password, and port 0 are preserved. 

73 

74 .. versionchanged:: 3.0 

75 Passing a tuple or bytes, and the ``charset`` and ``errors`` parameters, 

76 are removed. 

77 

78 .. versionchanged:: 2.3 

79 Which characters remain quoted is specific to each part of the URL. 

80 

81 .. versionchanged:: 0.15 

82 All reserved and invalid characters remain quoted. Previously, 

83 only some reserved characters were preserved, and invalid bytes 

84 were replaced instead of left quoted. 

85 

86 .. versionadded:: 0.6 

87 """ 

88 parts = urlsplit(uri) 

89 path = _unquote_path(parts.path) 

90 query = _unquote_query(parts.query) 

91 fragment = _unquote_fragment(parts.fragment) 

92 

93 if parts.hostname: 

94 netloc = _decode_idna(parts.hostname) 

95 else: 

96 netloc = "" 

97 

98 if ":" in netloc: 

99 netloc = f"[{netloc}]" 

100 

101 if parts.port is not None: 

102 netloc = f"{netloc}:{parts.port}" 

103 

104 if parts.username is not None: 

105 auth = _unquote_user(parts.username) 

106 

107 if parts.password is not None: 

108 password = _unquote_user(parts.password) 

109 auth = f"{auth}:{password}" 

110 

111 netloc = f"{auth}@{netloc}" 

112 

113 return urlunsplit((parts.scheme, netloc, path, query, fragment)) 

114 

115 

116def iri_to_uri(iri: str) -> str: 

117 """Convert an IRI to a URI. All non-ASCII and unsafe characters are 

118 quoted. If the URL has a domain, it is encoded to Punycode. 

119 

120 >>> iri_to_uri('http://\\u2603.net/p\\xe5th?q=\\xe8ry%DF') 

121 'http://xn--n3h.net/p%C3%A5th?q=%C3%A8ry%DF' 

122 

123 :param iri: The IRI to convert. 

124 

125 .. versionchanged:: 3.1.9 

126 Empty username, password, and port 0 are preserved. 

127 

128 .. versionchanged:: 3.0 

129 Passing a tuple or bytes, the ``charset`` and ``errors`` parameters, 

130 and the ``safe_conversion`` parameter, are removed. 

131 

132 .. versionchanged:: 2.3 

133 Which characters remain unquoted is specific to each part of the URL. 

134 

135 .. versionchanged:: 0.15 

136 All reserved characters remain unquoted. Previously, only some reserved 

137 characters were left unquoted. 

138 

139 .. versionchanged:: 0.9.6 

140 The ``safe_conversion`` parameter was added. 

141 

142 .. versionadded:: 0.6 

143 """ 

144 parts = urlsplit(iri) 

145 # safe = https://url.spec.whatwg.org/#url-path-segment-string 

146 # as well as percent for things that are already quoted 

147 path = quote(parts.path, safe="%!$&'()*+,/:;=@") 

148 query = quote(parts.query, safe="%!$&'()*+,/:;=?@") 

149 fragment = quote(parts.fragment, safe="%!#$&'()*+,/:;=?@") 

150 

151 if parts.hostname: 

152 netloc = parts.hostname.encode("idna").decode("ascii") 

153 else: 

154 netloc = "" 

155 

156 if ":" in netloc: 

157 netloc = f"[{netloc}]" 

158 

159 if parts.port is not None: 

160 netloc = f"{netloc}:{parts.port}" 

161 

162 if parts.username is not None: 

163 auth = quote(parts.username, safe="%!$&'()*+,;=") 

164 

165 if parts.password is not None: 

166 password = quote(parts.password, safe="%!$&'()*+,;=") 

167 auth = f"{auth}:{password}" 

168 

169 netloc = f"{auth}@{netloc}" 

170 

171 return urlunsplit((parts.scheme, netloc, path, query, fragment)) 

172 

173 

174# Python < 3.12 

175# itms-services was worked around in previous iri_to_uri implementations, but 

176# we can tell Python directly that it needs to preserve the //. 

177if "itms-services" not in urllib.parse.uses_netloc: 

178 urllib.parse.uses_netloc.append("itms-services") 

179 

180 

181def _decode_idna(domain: str) -> str: 

182 try: 

183 data = domain.encode("ascii") 

184 except UnicodeEncodeError: 

185 # If the domain is not ASCII, it's decoded already. 

186 return domain 

187 

188 try: 

189 # Try decoding in one shot. 

190 return data.decode("idna") 

191 except UnicodeDecodeError: 

192 pass 

193 

194 # Decode each part separately, leaving invalid parts as punycode. 

195 parts = [] 

196 

197 for part in data.split(b"."): 

198 try: 

199 parts.append(part.decode("idna")) 

200 except UnicodeDecodeError: 

201 parts.append(part.decode("ascii")) 

202 

203 return ".".join(parts) 

204 

205 

206def _urlencode(query: t.Mapping[str, str] | t.Iterable[tuple[str, str]]) -> str: 

207 items = [x for x in iter_multi_items(query) if x[1] is not None] 

208 # safe = https://url.spec.whatwg.org/#percent-encoded-bytes 

209 return urlencode(items, safe="!$'()*,/:;?@")