Coverage for /pythoncovmergedfiles/medio/medio/usr/local/lib/python3.11/site-packages/werkzeug/urls.py: 33%
Shortcuts on this page
r m x toggle line displays
j k next/prev highlighted chunk
0 (zero) top of page
1 (one) first highlighted chunk
Shortcuts on this page
r m x toggle line displays
j k next/prev highlighted chunk
0 (zero) top of page
1 (one) first highlighted chunk
1from __future__ import annotations
3import codecs
4import re
5import typing as t
6import urllib.parse
7from urllib.parse import quote
8from urllib.parse import unquote
9from urllib.parse import urlencode
10from urllib.parse import urlsplit
11from urllib.parse import urlunsplit
13from .datastructures import iter_multi_items
16def _codec_error_url_quote(e: UnicodeError) -> tuple[str, int]:
17 """Used in :func:`uri_to_iri` after unquoting to re-quote any
18 invalid bytes.
19 """
20 # the docs state that UnicodeError does have these attributes,
21 # but mypy isn't picking them up
22 out = quote(e.object[e.start : e.end], safe="") # type: ignore
23 return out, e.end # type: ignore
26codecs.register_error("werkzeug.url_quote", _codec_error_url_quote)
29def _make_unquote_part(name: str, chars: str) -> t.Callable[[str], str]:
30 """Create a function that unquotes all percent encoded characters except those
31 given. This allows working with unquoted characters if possible while not changing
32 the meaning of a given part of a URL.
33 """
34 choices = "|".join(f"{ord(c):02X}" for c in sorted(chars))
35 pattern = re.compile(f"((?:%(?:{choices}))+)", re.I)
37 def _unquote_partial(value: str) -> str:
38 parts = iter(pattern.split(value))
39 out = []
41 for part in parts:
42 out.append(unquote(part, "utf-8", "werkzeug.url_quote"))
43 out.append(next(parts, ""))
45 return "".join(out)
47 _unquote_partial.__name__ = f"_unquote_{name}"
48 return _unquote_partial
51# characters that should remain quoted in URL parts
52# based on https://url.spec.whatwg.org/#percent-encoded-bytes
53# always keep all controls, space, and % quoted
54_always_unsafe = bytes((*range(0x21), 0x25, 0x7F)).decode()
55_unquote_fragment = _make_unquote_part("fragment", _always_unsafe)
56_unquote_query = _make_unquote_part("query", _always_unsafe + "&=+#")
57_unquote_path = _make_unquote_part("path", _always_unsafe + "/?#")
58_unquote_user = _make_unquote_part("user", _always_unsafe + ":@/?#")
61def uri_to_iri(uri: str) -> str:
62 """Convert a URI to an IRI. All valid UTF-8 characters are unquoted,
63 leaving all reserved and invalid characters quoted. If the URL has
64 a domain, it is decoded from Punycode.
66 >>> uri_to_iri("http://xn--n3h.net/p%C3%A5th?q=%C3%A8ry%DF")
67 'http://\\u2603.net/p\\xe5th?q=\\xe8ry%DF'
69 :param uri: The URI to convert.
71 .. versionchanged:: 3.1.9
72 Empty username, password, and port 0 are preserved.
74 .. versionchanged:: 3.0
75 Passing a tuple or bytes, and the ``charset`` and ``errors`` parameters,
76 are removed.
78 .. versionchanged:: 2.3
79 Which characters remain quoted is specific to each part of the URL.
81 .. versionchanged:: 0.15
82 All reserved and invalid characters remain quoted. Previously,
83 only some reserved characters were preserved, and invalid bytes
84 were replaced instead of left quoted.
86 .. versionadded:: 0.6
87 """
88 parts = urlsplit(uri)
89 path = _unquote_path(parts.path)
90 query = _unquote_query(parts.query)
91 fragment = _unquote_fragment(parts.fragment)
93 if parts.hostname:
94 netloc = _decode_idna(parts.hostname)
95 else:
96 netloc = ""
98 if ":" in netloc:
99 netloc = f"[{netloc}]"
101 if parts.port is not None:
102 netloc = f"{netloc}:{parts.port}"
104 if parts.username is not None:
105 auth = _unquote_user(parts.username)
107 if parts.password is not None:
108 password = _unquote_user(parts.password)
109 auth = f"{auth}:{password}"
111 netloc = f"{auth}@{netloc}"
113 return urlunsplit((parts.scheme, netloc, path, query, fragment))
116def iri_to_uri(iri: str) -> str:
117 """Convert an IRI to a URI. All non-ASCII and unsafe characters are
118 quoted. If the URL has a domain, it is encoded to Punycode.
120 >>> iri_to_uri('http://\\u2603.net/p\\xe5th?q=\\xe8ry%DF')
121 'http://xn--n3h.net/p%C3%A5th?q=%C3%A8ry%DF'
123 :param iri: The IRI to convert.
125 .. versionchanged:: 3.1.9
126 Empty username, password, and port 0 are preserved.
128 .. versionchanged:: 3.0
129 Passing a tuple or bytes, the ``charset`` and ``errors`` parameters,
130 and the ``safe_conversion`` parameter, are removed.
132 .. versionchanged:: 2.3
133 Which characters remain unquoted is specific to each part of the URL.
135 .. versionchanged:: 0.15
136 All reserved characters remain unquoted. Previously, only some reserved
137 characters were left unquoted.
139 .. versionchanged:: 0.9.6
140 The ``safe_conversion`` parameter was added.
142 .. versionadded:: 0.6
143 """
144 parts = urlsplit(iri)
145 # safe = https://url.spec.whatwg.org/#url-path-segment-string
146 # as well as percent for things that are already quoted
147 path = quote(parts.path, safe="%!$&'()*+,/:;=@")
148 query = quote(parts.query, safe="%!$&'()*+,/:;=?@")
149 fragment = quote(parts.fragment, safe="%!#$&'()*+,/:;=?@")
151 if parts.hostname:
152 netloc = parts.hostname.encode("idna").decode("ascii")
153 else:
154 netloc = ""
156 if ":" in netloc:
157 netloc = f"[{netloc}]"
159 if parts.port is not None:
160 netloc = f"{netloc}:{parts.port}"
162 if parts.username is not None:
163 auth = quote(parts.username, safe="%!$&'()*+,;=")
165 if parts.password is not None:
166 password = quote(parts.password, safe="%!$&'()*+,;=")
167 auth = f"{auth}:{password}"
169 netloc = f"{auth}@{netloc}"
171 return urlunsplit((parts.scheme, netloc, path, query, fragment))
174# Python < 3.12
175# itms-services was worked around in previous iri_to_uri implementations, but
176# we can tell Python directly that it needs to preserve the //.
177if "itms-services" not in urllib.parse.uses_netloc:
178 urllib.parse.uses_netloc.append("itms-services")
181def _decode_idna(domain: str) -> str:
182 try:
183 data = domain.encode("ascii")
184 except UnicodeEncodeError:
185 # If the domain is not ASCII, it's decoded already.
186 return domain
188 try:
189 # Try decoding in one shot.
190 return data.decode("idna")
191 except UnicodeDecodeError:
192 pass
194 # Decode each part separately, leaving invalid parts as punycode.
195 parts = []
197 for part in data.split(b"."):
198 try:
199 parts.append(part.decode("idna"))
200 except UnicodeDecodeError:
201 parts.append(part.decode("ascii"))
203 return ".".join(parts)
206def _urlencode(query: t.Mapping[str, str] | t.Iterable[tuple[str, str]]) -> str:
207 items = [x for x in iter_multi_items(query) if x[1] is not None]
208 # safe = https://url.spec.whatwg.org/#percent-encoded-bytes
209 return urlencode(items, safe="!$'()*,/:;?@")