Coverage for /pythoncovmergedfiles/medio/medio/usr/local/lib/python3.11/site-packages/yarl/_parse.py: 77%
Shortcuts on this page
r m x toggle line displays
j k next/prev highlighted chunk
0 (zero) top of page
1 (one) first highlighted chunk
Shortcuts on this page
r m x toggle line displays
j k next/prev highlighted chunk
0 (zero) top of page
1 (one) first highlighted chunk
1"""URL parsing utilities."""
3import codecs
4import re
5import unicodedata
6from functools import lru_cache
7from urllib.parse import parse_qsl, scheme_chars, uses_netloc
9from ._quoters import QUOTER, UNQUOTER_PLUS
11# Leading and trailing C0 control and space to be stripped per WHATWG spec.
12# == "".join([chr(i) for i in range(0, 0x20 + 1)])
13WHATWG_C0_CONTROL_OR_SPACE = (
14 "\x00\x01\x02\x03\x04\x05\x06\x07\x08\t\n\x0b\x0c\r\x0e\x0f\x10"
15 "\x11\x12\x13\x14\x15\x16\x17\x18\x19\x1a\x1b\x1c\x1d\x1e\x1f "
16)
18# Unsafe bytes to be removed per WHATWG spec
19UNSAFE_URL_BYTES_TO_REMOVE = ["\t", "\r", "\n"]
20USES_AUTHORITY = frozenset(uses_netloc)
22SplitURLType = tuple[str, str, str, str, str]
25def split_url(url: str) -> SplitURLType:
26 """Split URL into parts."""
27 # Adapted from urllib.parse.urlsplit
28 # Only lstrip url as some applications rely on preserving trailing space.
29 # (https://url.spec.whatwg.org/#concept-basic-url-parser would strip both)
30 url = url.lstrip(WHATWG_C0_CONTROL_OR_SPACE)
31 for b in UNSAFE_URL_BYTES_TO_REMOVE:
32 if b in url:
33 url = url.replace(b, "")
35 scheme = netloc = query = fragment = ""
36 i = url.find(":")
37 if i > 0 and url[0] in scheme_chars:
38 for c in url[1:i]:
39 if c not in scheme_chars:
40 break
41 else:
42 scheme, url = url[:i].lower(), url[i + 1 :]
43 has_hash = "#" in url
44 has_question_mark = "?" in url
45 if url[:2] == "//":
46 delim = len(url) # position of end of domain part of url, default is end
47 if has_hash and has_question_mark:
48 delim_chars = "/?#"
49 elif has_question_mark:
50 delim_chars = "/?"
51 elif has_hash:
52 delim_chars = "/#"
53 else:
54 delim_chars = "/"
55 for c in delim_chars: # look for delimiters; the order is NOT important
56 wdelim = url.find(c, 2) # find first of this delim
57 if wdelim >= 0 and wdelim < delim: # if found
58 delim = wdelim # use earliest delim position
59 netloc = url[2:delim]
60 url = url[delim:]
61 # Backslash is not valid in the authority component per RFC 3986.
62 # WHATWG parsers treat \ as a path separator for special schemes, so
63 # accepting it in the authority can cause host parsing ambiguity.
64 if "\\" in netloc:
65 raise ValueError(
66 "Invalid URL: backslash ('\\') is not allowed in the authority "
67 "component per RFC 3986."
68 )
69 has_left_bracket = "[" in netloc
70 has_right_bracket = "]" in netloc
71 if (has_left_bracket and not has_right_bracket) or (
72 has_right_bracket and not has_left_bracket
73 ):
74 raise ValueError("Invalid IPv6 URL")
75 if has_left_bracket:
76 # Per RFC 3986, brackets are only valid at the START of the host
77 # for IP-literal addresses. Text before '[' (e.g. '127.0.0.1[::1]')
78 # is invalid and must be rejected to prevent SSRF bypasses. The
79 # count checks reject URLs with more than one bracket pair in the
80 # host subcomponent (e.g. 'http://[:localhost[]].google:80'),
81 # which would otherwise resolve to an unintended host.
82 hostinfo = netloc.rpartition("@")[2]
83 if hostinfo[0] != "[" or hostinfo.count("[") > 1 or hostinfo.count("]") > 1:
84 raise ValueError("Invalid IPv6 URL")
85 bracketed_host, _, after_bracket = hostinfo[1:].partition("]")
86 # Per RFC 3986 §3.2.2, after the closing ']' of an IP-literal
87 # only ":" <port> or end-of-authority is valid. Any other text
88 # (e.g. '[::1]allowed.example:1') must be rejected to prevent
89 # host-confusion where the suffix is silently dropped.
90 if after_bracket and after_bracket[0] != ":":
91 raise ValueError("Invalid IPv6 URL")
92 # Valid bracketed hosts are defined in
93 # https://www.rfc-editor.org/rfc/rfc3986#page-49
94 # https://url.spec.whatwg.org/
95 if bracketed_host and bracketed_host[0] == "v":
96 if not re.match(r"\Av[a-fA-F0-9]+\..+\Z", bracketed_host):
97 raise ValueError("IPvFuture address is invalid")
98 elif ":" not in bracketed_host:
99 raise ValueError("The IPv6 content between brackets is not valid")
100 if has_hash:
101 url, _, fragment = url.partition("#")
102 if has_question_mark:
103 url, _, query = url.partition("?")
104 if netloc and not netloc.isascii():
105 _check_netloc(netloc)
106 return scheme, netloc, url, query, fragment
109def _check_netloc(netloc: str) -> None:
110 # Adapted from urllib.parse._checknetloc
111 # looking for characters like \u2100 that expand to 'a/c'
112 # IDNA uses NFKC equivalence, so normalize for this check
114 # ignore characters already included
115 # but not the surrounding text
116 n = netloc.replace("@", "").replace(":", "").replace("#", "").replace("?", "")
117 normalized_netloc = unicodedata.normalize("NFKC", n)
118 if n == normalized_netloc:
119 return
120 # Note that there are no unicode decompositions for the character '@' so
121 # its currently impossible to have test coverage for this branch, however if the
122 # one should be added in the future we want to make sure its still checked.
123 for c in "/?#@:%": # pragma: no branch
124 if c in normalized_netloc:
125 raise ValueError(
126 f"netloc '{netloc}' contains invalid "
127 "characters under NFKC normalization"
128 )
131@lru_cache # match the same size as urlsplit
132def split_netloc(
133 netloc: str,
134) -> tuple[str | None, str | None, str | None, int | None]:
135 """Split netloc into username, password, host and port."""
136 if "@" not in netloc:
137 username: str | None = None
138 password: str | None = None
139 hostinfo = netloc
140 else:
141 userinfo, _, hostinfo = netloc.rpartition("@")
142 username, have_password, password = userinfo.partition(":")
143 if not have_password:
144 password = None
146 if "[" in hostinfo:
147 if hostinfo[0] != "[" or hostinfo.count("[") > 1 or hostinfo.count("]") > 1:
148 raise ValueError("Invalid IPv6 URL")
149 _, _, bracketed = hostinfo.partition("[")
150 hostname, _, port_str = bracketed.partition("]")
151 # Defense-in-depth: after ']' only ':port' or empty is valid.
152 # split_url() should have already rejected invalid suffixes,
153 # but guard here too for callers that use split_netloc() directly.
154 if port_str and port_str[0] != ":":
155 raise ValueError("Invalid IPv6 URL")
156 _, _, port_str = port_str.partition(":")
157 else:
158 hostname, _, port_str = hostinfo.partition(":")
160 if not port_str:
161 return username or None, password, hostname or None, None
163 try:
164 port = int(port_str)
165 except ValueError:
166 raise ValueError("Invalid URL: port can't be converted to integer")
167 if not (0 <= port <= 65535):
168 raise ValueError("Port out of range 0-65535")
169 return username or None, password, hostname or None, port
172def unsplit_result(
173 scheme: str, netloc: str, url: str, query: str, fragment: str
174) -> str:
175 """Unsplit a URL without any normalization."""
176 if netloc or (scheme and scheme in USES_AUTHORITY) or url[:2] == "//":
177 if url and url[:1] != "/":
178 url = f"{scheme}://{netloc}/{url}" if scheme else f"{scheme}:{url}"
179 else:
180 url = f"{scheme}://{netloc}{url}" if scheme else f"//{netloc}{url}"
181 elif scheme:
182 url = f"{scheme}:{url}"
183 if query:
184 url = f"{url}?{query}"
185 return f"{url}#{fragment}" if fragment else url
188@lru_cache # match the same size as urlsplit
189def make_netloc(
190 user: str | None,
191 password: str | None,
192 host: str | None,
193 port: int | None,
194 encode: bool = False,
195) -> str:
196 """Make netloc from parts.
198 The user and password are encoded if encode is True.
200 The host must already be encoded with _encode_host.
201 """
202 if host is None:
203 return ""
204 ret = host
205 if port is not None:
206 ret = f"{ret}:{port}"
207 if user is None and password is None:
208 return ret
209 if password is not None:
210 if not user:
211 user = ""
212 elif encode:
213 user = QUOTER(user)
214 if encode:
215 password = QUOTER(password)
216 user = f"{user}:{password}"
217 elif user and encode:
218 user = QUOTER(user)
219 return f"{user}@{ret}" if user else ret
222def query_to_pairs(
223 query_string: str, *, max_fields: int | None = None, encoding: str = "utf-8"
224) -> list[tuple[str, str]]:
225 """Parse a query string into a list of decoded name, value pairs.
227 The result is the same as
228 ``urllib.parse.parse_qsl(query_string, keep_blank_values=True,
229 encoding=encoding, max_num_fields=max_fields)``.
231 Raises :exc:`ValueError` if *max_fields* is not ``None`` and the
232 query string has more than *max_fields* fields. An empty query string
233 returns an empty list on every Python version, even when *max_fields*
234 is ``0``, where ``parse_qsl`` on Python 3.10 raises instead.
235 """
236 if not query_string:
237 return []
238 if max_fields is not None and query_string.count("&") >= max_fields:
239 raise ValueError("Max number of fields exceeded")
240 pairs: list[tuple[str, str]] = []
241 if "%" not in query_string:
242 # Nothing to decode except '+', which is the same in every encoding
243 for name_value in query_string.replace("+", " ").split("&"):
244 if name_value:
245 name, _, value = name_value.partition("=")
246 pairs.append((name, value))
247 return pairs
248 if encoding != "utf-8" and codecs.lookup(encoding).name != "utf-8":
249 return parse_qsl(query_string, keep_blank_values=True, encoding=encoding)
250 for name_value in query_string.split("&"):
251 if name_value:
252 name, _, value = name_value.partition("=")
253 pairs.append((UNQUOTER_PLUS(name), UNQUOTER_PLUS(value)))
254 return pairs