Coverage for /pythoncovmergedfiles/medio/medio/usr/local/lib/python3.11/site-packages/yarl/_parse.py: 77%

Shortcuts on this page

r m x   toggle line displays

j k   next/prev highlighted chunk

0   (zero) top of page

1   (one) first highlighted chunk

150 statements  

1"""URL parsing utilities.""" 

2 

3import codecs 

4import re 

5import unicodedata 

6from functools import lru_cache 

7from urllib.parse import parse_qsl, scheme_chars, uses_netloc 

8 

9from ._quoters import QUOTER, UNQUOTER_PLUS 

10 

11# Leading and trailing C0 control and space to be stripped per WHATWG spec. 

12# == "".join([chr(i) for i in range(0, 0x20 + 1)]) 

13WHATWG_C0_CONTROL_OR_SPACE = ( 

14 "\x00\x01\x02\x03\x04\x05\x06\x07\x08\t\n\x0b\x0c\r\x0e\x0f\x10" 

15 "\x11\x12\x13\x14\x15\x16\x17\x18\x19\x1a\x1b\x1c\x1d\x1e\x1f " 

16) 

17 

18# Unsafe bytes to be removed per WHATWG spec 

19UNSAFE_URL_BYTES_TO_REMOVE = ["\t", "\r", "\n"] 

20USES_AUTHORITY = frozenset(uses_netloc) 

21 

22SplitURLType = tuple[str, str, str, str, str] 

23 

24 

25def split_url(url: str) -> SplitURLType: 

26 """Split URL into parts.""" 

27 # Adapted from urllib.parse.urlsplit 

28 # Only lstrip url as some applications rely on preserving trailing space. 

29 # (https://url.spec.whatwg.org/#concept-basic-url-parser would strip both) 

30 url = url.lstrip(WHATWG_C0_CONTROL_OR_SPACE) 

31 for b in UNSAFE_URL_BYTES_TO_REMOVE: 

32 if b in url: 

33 url = url.replace(b, "") 

34 

35 scheme = netloc = query = fragment = "" 

36 i = url.find(":") 

37 if i > 0 and url[0] in scheme_chars: 

38 for c in url[1:i]: 

39 if c not in scheme_chars: 

40 break 

41 else: 

42 scheme, url = url[:i].lower(), url[i + 1 :] 

43 has_hash = "#" in url 

44 has_question_mark = "?" in url 

45 if url[:2] == "//": 

46 delim = len(url) # position of end of domain part of url, default is end 

47 if has_hash and has_question_mark: 

48 delim_chars = "/?#" 

49 elif has_question_mark: 

50 delim_chars = "/?" 

51 elif has_hash: 

52 delim_chars = "/#" 

53 else: 

54 delim_chars = "/" 

55 for c in delim_chars: # look for delimiters; the order is NOT important 

56 wdelim = url.find(c, 2) # find first of this delim 

57 if wdelim >= 0 and wdelim < delim: # if found 

58 delim = wdelim # use earliest delim position 

59 netloc = url[2:delim] 

60 url = url[delim:] 

61 # Backslash is not valid in the authority component per RFC 3986. 

62 # WHATWG parsers treat \ as a path separator for special schemes, so 

63 # accepting it in the authority can cause host parsing ambiguity. 

64 if "\\" in netloc: 

65 raise ValueError( 

66 "Invalid URL: backslash ('\\') is not allowed in the authority " 

67 "component per RFC 3986." 

68 ) 

69 has_left_bracket = "[" in netloc 

70 has_right_bracket = "]" in netloc 

71 if (has_left_bracket and not has_right_bracket) or ( 

72 has_right_bracket and not has_left_bracket 

73 ): 

74 raise ValueError("Invalid IPv6 URL") 

75 if has_left_bracket: 

76 # Per RFC 3986, brackets are only valid at the START of the host 

77 # for IP-literal addresses. Text before '[' (e.g. '127.0.0.1[::1]') 

78 # is invalid and must be rejected to prevent SSRF bypasses. The 

79 # count checks reject URLs with more than one bracket pair in the 

80 # host subcomponent (e.g. 'http://[:localhost[]].google:80'), 

81 # which would otherwise resolve to an unintended host. 

82 hostinfo = netloc.rpartition("@")[2] 

83 if hostinfo[0] != "[" or hostinfo.count("[") > 1 or hostinfo.count("]") > 1: 

84 raise ValueError("Invalid IPv6 URL") 

85 bracketed_host, _, after_bracket = hostinfo[1:].partition("]") 

86 # Per RFC 3986 §3.2.2, after the closing ']' of an IP-literal 

87 # only ":" <port> or end-of-authority is valid. Any other text 

88 # (e.g. '[::1]allowed.example:1') must be rejected to prevent 

89 # host-confusion where the suffix is silently dropped. 

90 if after_bracket and after_bracket[0] != ":": 

91 raise ValueError("Invalid IPv6 URL") 

92 # Valid bracketed hosts are defined in 

93 # https://www.rfc-editor.org/rfc/rfc3986#page-49 

94 # https://url.spec.whatwg.org/ 

95 if bracketed_host and bracketed_host[0] == "v": 

96 if not re.match(r"\Av[a-fA-F0-9]+\..+\Z", bracketed_host): 

97 raise ValueError("IPvFuture address is invalid") 

98 elif ":" not in bracketed_host: 

99 raise ValueError("The IPv6 content between brackets is not valid") 

100 if has_hash: 

101 url, _, fragment = url.partition("#") 

102 if has_question_mark: 

103 url, _, query = url.partition("?") 

104 if netloc and not netloc.isascii(): 

105 _check_netloc(netloc) 

106 return scheme, netloc, url, query, fragment 

107 

108 

109def _check_netloc(netloc: str) -> None: 

110 # Adapted from urllib.parse._checknetloc 

111 # looking for characters like \u2100 that expand to 'a/c' 

112 # IDNA uses NFKC equivalence, so normalize for this check 

113 

114 # ignore characters already included 

115 # but not the surrounding text 

116 n = netloc.replace("@", "").replace(":", "").replace("#", "").replace("?", "") 

117 normalized_netloc = unicodedata.normalize("NFKC", n) 

118 if n == normalized_netloc: 

119 return 

120 # Note that there are no unicode decompositions for the character '@' so 

121 # its currently impossible to have test coverage for this branch, however if the 

122 # one should be added in the future we want to make sure its still checked. 

123 for c in "/?#@:%": # pragma: no branch 

124 if c in normalized_netloc: 

125 raise ValueError( 

126 f"netloc '{netloc}' contains invalid " 

127 "characters under NFKC normalization" 

128 ) 

129 

130 

131@lru_cache # match the same size as urlsplit 

132def split_netloc( 

133 netloc: str, 

134) -> tuple[str | None, str | None, str | None, int | None]: 

135 """Split netloc into username, password, host and port.""" 

136 if "@" not in netloc: 

137 username: str | None = None 

138 password: str | None = None 

139 hostinfo = netloc 

140 else: 

141 userinfo, _, hostinfo = netloc.rpartition("@") 

142 username, have_password, password = userinfo.partition(":") 

143 if not have_password: 

144 password = None 

145 

146 if "[" in hostinfo: 

147 if hostinfo[0] != "[" or hostinfo.count("[") > 1 or hostinfo.count("]") > 1: 

148 raise ValueError("Invalid IPv6 URL") 

149 _, _, bracketed = hostinfo.partition("[") 

150 hostname, _, port_str = bracketed.partition("]") 

151 # Defense-in-depth: after ']' only ':port' or empty is valid. 

152 # split_url() should have already rejected invalid suffixes, 

153 # but guard here too for callers that use split_netloc() directly. 

154 if port_str and port_str[0] != ":": 

155 raise ValueError("Invalid IPv6 URL") 

156 _, _, port_str = port_str.partition(":") 

157 else: 

158 hostname, _, port_str = hostinfo.partition(":") 

159 

160 if not port_str: 

161 return username or None, password, hostname or None, None 

162 

163 try: 

164 port = int(port_str) 

165 except ValueError: 

166 raise ValueError("Invalid URL: port can't be converted to integer") 

167 if not (0 <= port <= 65535): 

168 raise ValueError("Port out of range 0-65535") 

169 return username or None, password, hostname or None, port 

170 

171 

172def unsplit_result( 

173 scheme: str, netloc: str, url: str, query: str, fragment: str 

174) -> str: 

175 """Unsplit a URL without any normalization.""" 

176 if netloc or (scheme and scheme in USES_AUTHORITY) or url[:2] == "//": 

177 if url and url[:1] != "/": 

178 url = f"{scheme}://{netloc}/{url}" if scheme else f"{scheme}:{url}" 

179 else: 

180 url = f"{scheme}://{netloc}{url}" if scheme else f"//{netloc}{url}" 

181 elif scheme: 

182 url = f"{scheme}:{url}" 

183 if query: 

184 url = f"{url}?{query}" 

185 return f"{url}#{fragment}" if fragment else url 

186 

187 

188@lru_cache # match the same size as urlsplit 

189def make_netloc( 

190 user: str | None, 

191 password: str | None, 

192 host: str | None, 

193 port: int | None, 

194 encode: bool = False, 

195) -> str: 

196 """Make netloc from parts. 

197 

198 The user and password are encoded if encode is True. 

199 

200 The host must already be encoded with _encode_host. 

201 """ 

202 if host is None: 

203 return "" 

204 ret = host 

205 if port is not None: 

206 ret = f"{ret}:{port}" 

207 if user is None and password is None: 

208 return ret 

209 if password is not None: 

210 if not user: 

211 user = "" 

212 elif encode: 

213 user = QUOTER(user) 

214 if encode: 

215 password = QUOTER(password) 

216 user = f"{user}:{password}" 

217 elif user and encode: 

218 user = QUOTER(user) 

219 return f"{user}@{ret}" if user else ret 

220 

221 

222def query_to_pairs( 

223 query_string: str, *, max_fields: int | None = None, encoding: str = "utf-8" 

224) -> list[tuple[str, str]]: 

225 """Parse a query string into a list of decoded name, value pairs. 

226 

227 The result is the same as 

228 ``urllib.parse.parse_qsl(query_string, keep_blank_values=True, 

229 encoding=encoding, max_num_fields=max_fields)``. 

230 

231 Raises :exc:`ValueError` if *max_fields* is not ``None`` and the 

232 query string has more than *max_fields* fields. An empty query string 

233 returns an empty list on every Python version, even when *max_fields* 

234 is ``0``, where ``parse_qsl`` on Python 3.10 raises instead. 

235 """ 

236 if not query_string: 

237 return [] 

238 if max_fields is not None and query_string.count("&") >= max_fields: 

239 raise ValueError("Max number of fields exceeded") 

240 pairs: list[tuple[str, str]] = [] 

241 if "%" not in query_string: 

242 # Nothing to decode except '+', which is the same in every encoding 

243 for name_value in query_string.replace("+", " ").split("&"): 

244 if name_value: 

245 name, _, value = name_value.partition("=") 

246 pairs.append((name, value)) 

247 return pairs 

248 if encoding != "utf-8" and codecs.lookup(encoding).name != "utf-8": 

249 return parse_qsl(query_string, keep_blank_values=True, encoding=encoding) 

250 for name_value in query_string.split("&"): 

251 if name_value: 

252 name, _, value = name_value.partition("=") 

253 pairs.append((UNQUOTER_PLUS(name), UNQUOTER_PLUS(value))) 

254 return pairs