Coverage for /pythoncovmergedfiles/medio/medio/usr/local/lib/python3.11/site-packages/yarl/_url.py: 39%
Shortcuts on this page
r m x toggle line displays
j k next/prev highlighted chunk
0 (zero) top of page
1 (one) first highlighted chunk
Shortcuts on this page
r m x toggle line displays
j k next/prev highlighted chunk
0 (zero) top of page
1 (one) first highlighted chunk
1import re
2import sys
3import warnings
4from collections.abc import Mapping, Sequence
5from enum import Enum
6from functools import _CacheInfo, lru_cache
7from importlib.util import find_spec
8from ipaddress import ip_address
9from typing import (
10 TYPE_CHECKING,
11 Any,
12 NoReturn,
13 TypedDict,
14 TypeVar,
15 Union,
16 cast,
17 overload,
18)
19from urllib.parse import SplitResult, scheme_chars, uses_relative
21import idna
22from multidict import MultiDict, MultiDictProxy, istr
23from propcache.api import under_cached_property as cached_property
25from ._parse import (
26 USES_AUTHORITY,
27 SplitURLType,
28 make_netloc,
29 query_to_pairs,
30 split_netloc,
31 split_url,
32 unsplit_result,
33)
34from ._path import normalize_path, normalize_path_segments
35from ._query import (
36 Query,
37 QueryVariable,
38 SimpleQuery,
39 get_str_query,
40 get_str_query_from_iterable,
41 get_str_query_from_sequence_iterable,
42)
43from ._quoters import (
44 FRAGMENT_QUOTER,
45 FRAGMENT_REQUOTER,
46 PATH_QUOTER,
47 PATH_REQUOTER,
48 PATH_SAFE_UNQUOTER,
49 QS_UNQUOTER,
50 QUERY_QUOTER,
51 QUERY_REQUOTER,
52 QUOTER,
53 REQUOTER,
54 UNQUOTER,
55 human_quote,
56)
58# Avoid Pydantic import if not used (increases yarl's import time by 3-7x).
59HAS_PYDANTIC = find_spec("pydantic_core") is not None
60if TYPE_CHECKING:
61 from pydantic import GetCoreSchemaHandler, GetJsonSchemaHandler
62 from pydantic.json_schema import JsonSchemaValue
63 from pydantic_core import CoreSchema
66DEFAULT_PORTS = {"http": 80, "https": 443, "ws": 80, "wss": 443, "ftp": 21}
67USES_RELATIVE = frozenset(uses_relative)
68_SCHEME_CHARS = frozenset(scheme_chars)
70# Special schemes https://url.spec.whatwg.org/#special-scheme
71# are not allowed to have an empty host https://url.spec.whatwg.org/#url-representation
72SCHEME_REQUIRES_HOST = frozenset(("http", "https", "ws", "wss", "ftp"))
75# reg-name: unreserved / pct-encoded / sub-delims
76# this pattern matches anything that is *not* in those classes. and is only used
77# on lower-cased ASCII values.
78NOT_REG_NAME = re.compile(
79 r"""
80 # any character not in the unreserved or sub-delims sets, plus %
81 # (validated with the additional check for pct-encoded sequences below)
82 [^a-z0-9\-._~!$&'()*+,;=%]
83 |
84 # % only allowed if it is part of a pct-encoded
85 # sequence of 2 hex digits.
86 %(?![0-9a-f]{2})
87 """,
88 re.VERBOSE,
89)
91# Invisible default-ignorable / format code points that must not appear in a
92# host (soft hyphen, zero-width space, word joiner, bidi controls, variation
93# selectors, ...). Depending on the code point IDNA either silently deletes it
94# (so ``e<ZWSP>vil.com`` encodes to ``evil.com``) or folds it into a different
95# punycode host; either way the parsed host differs from the string an
96# application validated. The set is the union of two authoritative sources,
97# matching the two encoders _idna_encode dispatches to:
98#
99# 1. Unicode Default_Ignorable_Code_Point (uts46=True path via the ``idna``
100# package). Ranges taken from the DerivedCoreProperties data file:
101# https://www.unicode.org/Public/UCD/latest/ucd/DerivedCoreProperties.txt
102# (the E0000..E0FFF block is contiguous under this property).
103# 2. RFC 3454 (Stringprep) Table B.1 "commonly mapped to nothing", used by the
104# stdlib ``str.encode("idna")`` / IDNA2003 nameprep fallback:
105# https://www.rfc-editor.org/rfc/rfc3454#appendix-B.1
106# This is the source of U+1806, which is not Default_Ignorable.
107#
108# Coverage is pinned to the installed ``idna``/Unicode data by a sweep test
109# (test_default_ignorable_covers_idna_stripped in tests/test_url.py) that
110# brute-forces every code point through _idna_encode and fails if any point it
111# silently deletes is not matched here.
112_DEFAULT_IGNORABLE_RE = re.compile(
113 "["
114 "\u00ad" # SOFT HYPHEN
115 "\u034f" # COMBINING GRAPHEME JOINER
116 "\u061c" # ARABIC LETTER MARK
117 "\u115f-\u1160" # HANGUL CHOSEONG/JUNGSEONG FILLER
118 "\u17b4-\u17b5" # KHMER VOWEL INHERENT AQ/AA
119 "\u1806" # MONGOLIAN TODO SOFT HYPHEN (nameprep maps to nothing)
120 "\u180b-\u180f" # MONGOLIAN FVS ONE..FOUR and VOWEL SEPARATOR
121 "\u200b-\u200f" # ZERO WIDTH SPACE..RIGHT-TO-LEFT MARK
122 "\u202a-\u202e" # bidi embedding/override controls
123 "\u2060-\u206f" # WORD JOINER..NOMINAL DIGIT SHAPES
124 "\u3164" # HANGUL FILLER
125 "\ufe00-\ufe0f" # VARIATION SELECTOR-1..16
126 "\ufeff" # ZERO WIDTH NO-BREAK SPACE (BOM)
127 "\uffa0" # HALFWIDTH HANGUL FILLER
128 "\ufff0-\ufff8" # reserved default-ignorables
129 "\U0001bca0-\U0001bca3" # SHORTHAND FORMAT controls
130 "\U0001d173-\U0001d17a" # MUSICAL SYMBOL begin/end controls
131 "\U000e0000-\U000e0fff" # tags and VARIATION SELECTOR SUPPLEMENT
132 "]"
133)
135# Zone IDs are OS-specific text strings with no format defined by the RFCs:
136# https://datatracker.ietf.org/doc/html/rfc4007#section-11.2
137# RFC 9844 §6.3 recommends rejecting characters inappropriate for the
138# environment; for yarl we reject ASCII control characters (CTL):
139# https://datatracker.ietf.org/doc/html/rfc9844#section-6-3
140_ZONE_ID_UNSAFE_RE = re.compile(r"[\x00-\x1f\x7f]")
142_T = TypeVar("_T")
144if sys.version_info >= (3, 11):
145 from typing import Self
146else:
147 Self = Any
150class UndefinedType(Enum):
151 """Singleton type for use with not set sentinel values."""
153 _singleton = 0
156UNDEFINED = UndefinedType._singleton
159class CacheInfo(TypedDict):
160 """Host encoding cache."""
162 idna_encode: _CacheInfo
163 idna_decode: _CacheInfo
164 ip_address: _CacheInfo
165 host_validate: _CacheInfo
166 encode_host: _CacheInfo
169class _InternalURLCache(TypedDict, total=False):
170 _val: SplitURLType
171 _origin: "URL"
172 absolute: bool
173 hash: int
174 scheme: str
175 raw_authority: str
176 authority: str
177 raw_user: str | None
178 user: str | None
179 raw_password: str | None
180 password: str | None
181 raw_host: str | None
182 host: str | None
183 host_subcomponent: str | None
184 host_port_subcomponent: str | None
185 port: int | None
186 explicit_port: int | None
187 raw_path: str
188 path: str
189 _parsed_query: list[tuple[str, str]]
190 query: "MultiDictProxy[str]"
191 raw_query_string: str
192 query_string: str
193 path_qs: str
194 raw_path_qs: str
195 raw_fragment: str
196 fragment: str
197 raw_parts: tuple[str, ...]
198 parts: tuple[str, ...]
199 parent: "URL"
200 raw_name: str
201 name: str
202 raw_suffix: str
203 suffix: str
204 raw_suffixes: tuple[str, ...]
205 suffixes: tuple[str, ...]
208def rewrite_module(obj: _T) -> _T:
209 obj.__module__ = "yarl"
210 return obj
213def _encode_relative_scheme_colon(path: str) -> str:
214 """Re-encode a scheme-shaped leading ``:`` in a relative path to ``%3A``."""
215 colon_pos = path.find(":")
216 if colon_pos <= 0:
217 return path
218 for c in path[:colon_pos]:
219 if c not in _SCHEME_CHARS:
220 return path
221 return path[:colon_pos] + "%3A" + path[colon_pos + 1 :]
224@lru_cache
225def encode_url(url_str: str) -> "URL":
226 """Parse unencoded URL."""
227 cache: _InternalURLCache = {}
228 host: str | None
229 scheme, netloc, path, query, fragment = split_url(url_str)
230 if not netloc: # netloc
231 host = ""
232 else:
233 if ":" in netloc or "@" in netloc or "[" in netloc:
234 # Complex netloc
235 username, password, host, port = split_netloc(netloc)
236 else:
237 username = password = port = None
238 host = netloc
239 if host is None:
240 if scheme in SCHEME_REQUIRES_HOST:
241 msg = (
242 "Invalid URL: host is required for "
243 f"absolute urls with the {scheme} scheme"
244 )
245 raise ValueError(msg)
246 else:
247 host = ""
248 host = _encode_host(host, validate_host=False)
249 # Remove brackets as host encoder adds back brackets for IPv6 addresses
250 cache["raw_host"] = host[1:-1] if "[" in host else host
251 cache["explicit_port"] = port
252 if password is None and username is None:
253 # Fast path for URLs without user, password
254 netloc = host if port is None else f"{host}:{port}"
255 cache["raw_user"] = None
256 cache["raw_password"] = None
257 else:
258 raw_user = REQUOTER(username) if username else username
259 raw_password = REQUOTER(password) if password else password
260 netloc = make_netloc(raw_user, raw_password, host, port)
261 cache["raw_user"] = raw_user
262 cache["raw_password"] = raw_password
264 if path:
265 path = PATH_REQUOTER(path)
266 if netloc and "." in path:
267 path = normalize_path(path)
268 elif not scheme and not netloc:
269 path = _encode_relative_scheme_colon(path)
270 if query:
271 query = QUERY_REQUOTER(query)
272 if fragment:
273 fragment = FRAGMENT_REQUOTER(fragment)
275 cache["scheme"] = scheme
276 cache["raw_path"] = "/" if not path and netloc else path
277 cache["raw_query_string"] = query
278 cache["raw_fragment"] = fragment
280 self = object.__new__(URL)
281 self._scheme = scheme
282 self._netloc = netloc
283 self._path = path
284 self._query = query
285 self._fragment = fragment
286 self._cache = cache
287 return self
290@lru_cache
291def pre_encoded_url(url_str: str) -> "URL":
292 """Parse pre-encoded URL."""
293 self = object.__new__(URL)
294 val = split_url(url_str)
295 self._scheme, self._netloc, self._path, self._query, self._fragment = val
296 self._cache = {}
297 return self
300@lru_cache
301def build_pre_encoded_url(
302 scheme: str,
303 authority: str,
304 user: str | None,
305 password: str | None,
306 host: str,
307 port: int | None,
308 path: str,
309 query_string: str,
310 fragment: str,
311) -> "URL":
312 """Build a pre-encoded URL from parts."""
313 self = object.__new__(URL)
314 self._scheme = scheme
315 if authority:
316 self._netloc = authority
317 elif host:
318 if port is not None:
319 port = None if port == DEFAULT_PORTS.get(scheme) else port
320 if user is None and password is None:
321 self._netloc = host if port is None else f"{host}:{port}"
322 else:
323 self._netloc = make_netloc(user, password, host, port)
324 else:
325 self._netloc = ""
326 if path and not scheme and not self._netloc and ":" in path:
327 path = _encode_relative_scheme_colon(path)
328 self._path = path
329 self._query = query_string
330 self._fragment = fragment
331 self._cache = {}
332 return self
335def from_parts_uncached(
336 scheme: str, netloc: str, path: str, query: str, fragment: str
337) -> "URL":
338 """Create a new URL from parts."""
339 self = object.__new__(URL)
340 self._scheme = scheme
341 self._netloc = netloc
342 if path and not scheme and not netloc and ":" in path:
343 path = _encode_relative_scheme_colon(path)
344 self._path = path
345 self._query = query
346 self._fragment = fragment
347 self._cache = {}
348 return self
351from_parts = lru_cache(from_parts_uncached)
354@rewrite_module
355class URL:
356 # Don't derive from str
357 # follow pathlib.Path design
358 # probably URL will not suffer from pathlib problems:
359 # it's intended for libraries like aiohttp,
360 # not to be passed into standard library functions like os.open etc.
362 # URL grammar (RFC 3986)
363 # pct-encoded = "%" HEXDIG HEXDIG
364 # reserved = gen-delims / sub-delims
365 # gen-delims = ":" / "/" / "?" / "#" / "[" / "]" / "@"
366 # sub-delims = "!" / "$" / "&" / "'" / "(" / ")"
367 # / "*" / "+" / "," / ";" / "="
368 # unreserved = ALPHA / DIGIT / "-" / "." / "_" / "~"
369 # URI = scheme ":" hier-part [ "?" query ] [ "#" fragment ]
370 # hier-part = "//" authority path-abempty
371 # / path-absolute
372 # / path-rootless
373 # / path-empty
374 # scheme = ALPHA *( ALPHA / DIGIT / "+" / "-" / "." )
375 # authority = [ userinfo "@" ] host [ ":" port ]
376 # userinfo = *( unreserved / pct-encoded / sub-delims / ":" )
377 # host = IP-literal / IPv4address / reg-name
378 # IP-literal = "[" ( IPv6address / IPvFuture ) "]"
379 # IPvFuture = "v" 1*HEXDIG "." 1*( unreserved / sub-delims / ":" )
380 # IPv6address = 6( h16 ":" ) ls32
381 # / "::" 5( h16 ":" ) ls32
382 # / [ h16 ] "::" 4( h16 ":" ) ls32
383 # / [ *1( h16 ":" ) h16 ] "::" 3( h16 ":" ) ls32
384 # / [ *2( h16 ":" ) h16 ] "::" 2( h16 ":" ) ls32
385 # / [ *3( h16 ":" ) h16 ] "::" h16 ":" ls32
386 # / [ *4( h16 ":" ) h16 ] "::" ls32
387 # / [ *5( h16 ":" ) h16 ] "::" h16
388 # / [ *6( h16 ":" ) h16 ] "::"
389 # ls32 = ( h16 ":" h16 ) / IPv4address
390 # ; least-significant 32 bits of address
391 # h16 = 1*4HEXDIG
392 # ; 16 bits of address represented in hexadecimal
393 # IPv4address = dec-octet "." dec-octet "." dec-octet "." dec-octet
394 # dec-octet = DIGIT ; 0-9
395 # / %x31-39 DIGIT ; 10-99
396 # / "1" 2DIGIT ; 100-199
397 # / "2" %x30-34 DIGIT ; 200-249
398 # / "25" %x30-35 ; 250-255
399 # reg-name = *( unreserved / pct-encoded / sub-delims )
400 # port = *DIGIT
401 # path = path-abempty ; begins with "/" or is empty
402 # / path-absolute ; begins with "/" but not "//"
403 # / path-noscheme ; begins with a non-colon segment
404 # / path-rootless ; begins with a segment
405 # / path-empty ; zero characters
406 # path-abempty = *( "/" segment )
407 # path-absolute = "/" [ segment-nz *( "/" segment ) ]
408 # path-noscheme = segment-nz-nc *( "/" segment )
409 # path-rootless = segment-nz *( "/" segment )
410 # path-empty = 0<pchar>
411 # segment = *pchar
412 # segment-nz = 1*pchar
413 # segment-nz-nc = 1*( unreserved / pct-encoded / sub-delims / "@" )
414 # ; non-zero-length segment without any colon ":"
415 # pchar = unreserved / pct-encoded / sub-delims / ":" / "@"
416 # query = *( pchar / "/" / "?" )
417 # fragment = *( pchar / "/" / "?" )
418 # URI-reference = URI / relative-ref
419 # relative-ref = relative-part [ "?" query ] [ "#" fragment ]
420 # relative-part = "//" authority path-abempty
421 # / path-absolute
422 # / path-noscheme
423 # / path-empty
424 # absolute-URI = scheme ":" hier-part [ "?" query ]
425 __slots__ = ("_cache", "_scheme", "_netloc", "_path", "_query", "_fragment")
427 _cache: _InternalURLCache
428 _scheme: str
429 _netloc: str
430 _path: str
431 _query: str
432 _fragment: str
434 def __new__(
435 cls,
436 val: Union[str, SplitResult, "URL", UndefinedType] = UNDEFINED,
437 *,
438 encoded: bool = False,
439 strict: bool | None = None,
440 ) -> "URL":
441 if strict is not None: # pragma: no cover
442 warnings.warn("strict parameter is ignored")
443 if type(val) is str:
444 return pre_encoded_url(val) if encoded else encode_url(val)
445 if type(val) is cls:
446 return val
447 if type(val) is SplitResult:
448 if not encoded:
449 raise ValueError("Cannot apply decoding to SplitResult")
450 return from_parts(*val)
451 if isinstance(val, str):
452 return pre_encoded_url(str(val)) if encoded else encode_url(str(val))
453 if val is UNDEFINED:
454 # Special case for UNDEFINED since it might be unpickling and we do
455 # not want to cache as the `__set_state__` call would mutate the URL
456 # object in the `pre_encoded_url` or `encoded_url` caches.
457 self = object.__new__(URL)
458 self._scheme = self._netloc = self._path = self._query = self._fragment = ""
459 self._cache = {}
460 return self
461 raise TypeError("Constructor parameter should be str")
463 @classmethod
464 def build(
465 cls,
466 *,
467 scheme: str = "",
468 authority: str = "",
469 user: str | None = None,
470 password: str | None = None,
471 host: str = "",
472 port: int | None = None,
473 path: str = "",
474 query: Query | None = None,
475 query_string: str = "",
476 fragment: str = "",
477 encoded: bool = False,
478 ) -> "URL":
479 """Creates and returns a new URL"""
481 if authority and (user or password or host or port):
482 raise ValueError(
483 'Can\'t mix "authority" with "user", "password", "host" or "port".'
484 )
485 if port is not None and not isinstance(port, int):
486 raise TypeError(f"The port is required to be int, got {type(port)!r}.")
487 if port and not host:
488 raise ValueError('Can\'t build URL with "port" but without "host".')
489 if query and query_string:
490 raise ValueError('Only one of "query" or "query_string" should be passed')
491 if (
492 scheme is None # type: ignore[redundant-expr]
493 or authority is None # type: ignore[redundant-expr]
494 or host is None # type: ignore[redundant-expr]
495 or path is None # type: ignore[redundant-expr]
496 or query_string is None # type: ignore[redundant-expr]
497 or fragment is None
498 ):
499 raise TypeError(
500 'NoneType is illegal for "scheme", "authority", "host", "path", '
501 '"query_string", and "fragment" args, use empty string instead.'
502 )
504 if query:
505 query_string = get_str_query(query) or ""
507 if encoded:
508 return build_pre_encoded_url(
509 scheme,
510 authority,
511 user,
512 password,
513 host,
514 port,
515 path,
516 query_string,
517 fragment,
518 )
520 self = object.__new__(URL)
521 self._scheme = scheme
522 _host: str | None = None
523 if authority:
524 user, password, _host, port = split_netloc(authority)
525 _host = _encode_host(_host, validate_host=False) if _host else ""
526 elif host:
527 _host = _encode_host(host, validate_host=True)
528 else:
529 self._netloc = ""
531 if _host is not None:
532 if port is not None:
533 port = None if port == DEFAULT_PORTS.get(scheme) else port
534 if user is None and password is None:
535 self._netloc = _host if port is None else f"{_host}:{port}"
536 else:
537 self._netloc = make_netloc(user, password, _host, port, True)
539 path = PATH_QUOTER(path) if path else path
540 if path and self._netloc:
541 if "." in path:
542 path = normalize_path(path)
543 if path[0] != "/":
544 msg = (
545 "Path in a URL with authority should "
546 "start with a slash ('/') if set"
547 )
548 raise ValueError(msg)
550 if path and not self._scheme and not self._netloc and ":" in path:
551 path = _encode_relative_scheme_colon(path)
552 self._path = path
553 if not query and query_string:
554 query_string = QUERY_QUOTER(query_string)
555 self._query = query_string
556 self._fragment = FRAGMENT_QUOTER(fragment) if fragment else fragment
557 self._cache = {}
558 return self
560 def __init_subclass__(cls) -> NoReturn:
561 raise TypeError(f"Inheriting a class {cls!r} from URL is forbidden")
563 def __str__(self) -> str:
564 if not self._path and self._netloc and (self._query or self._fragment):
565 path = "/"
566 else:
567 path = self._path
568 if (port := self.explicit_port) is not None and port == DEFAULT_PORTS.get(
569 self._scheme
570 ):
571 # port normalization - using None for default ports to remove from rendering
572 # https://datatracker.ietf.org/doc/html/rfc3986.html#section-6.2.3
573 host = self.host_subcomponent
574 netloc = make_netloc(self.raw_user, self.raw_password, host, None)
575 else:
576 netloc = self._netloc
577 return unsplit_result(self._scheme, netloc, path, self._query, self._fragment)
579 def __repr__(self) -> str:
580 return f"{self.__class__.__name__}('{str(self)}')"
582 def __bytes__(self) -> bytes:
583 return str(self).encode("ascii")
585 def __eq__(self, other: object) -> bool:
586 if type(other) is not URL:
587 return NotImplemented
589 path1 = "/" if not self._path and self._netloc else self._path
590 path2 = "/" if not other._path and other._netloc else other._path
591 return (
592 self._scheme == other._scheme
593 and self._netloc == other._netloc
594 and path1 == path2
595 and self._query == other._query
596 and self._fragment == other._fragment
597 )
599 def __hash__(self) -> int:
600 if (ret := self._cache.get("hash")) is None:
601 path = "/" if not self._path and self._netloc else self._path
602 ret = self._cache["hash"] = hash(
603 (self._scheme, self._netloc, path, self._query, self._fragment)
604 )
605 return ret
607 def __le__(self, other: object) -> bool:
608 if type(other) is not URL:
609 return NotImplemented
610 return self._val <= other._val
612 def __lt__(self, other: object) -> bool:
613 if type(other) is not URL:
614 return NotImplemented
615 return self._val < other._val
617 def __ge__(self, other: object) -> bool:
618 if type(other) is not URL:
619 return NotImplemented
620 return self._val >= other._val
622 def __gt__(self, other: object) -> bool:
623 if type(other) is not URL:
624 return NotImplemented
625 return self._val > other._val
627 def __truediv__(self, name: str) -> "URL":
628 if not isinstance(name, str):
629 return NotImplemented
630 return self._make_child((str(name),))
632 def __mod__(self, query: Query) -> "URL":
633 return self.update_query(query)
635 def __bool__(self) -> bool:
636 return bool(self._netloc or self._path or self._query or self._fragment)
638 def __getstate__(self) -> tuple[SplitURLType]:
639 # Return a plain tuple rather than a ``SplitResult``. Constructing a
640 # ``SplitResult`` via ``tuple.__new__`` skips its ``__init__`` and on
641 # Python 3.15+ leaves ``_keep_empty`` unset, which breaks pickling: the
642 # new ``SplitResult.__getstate__`` indexes a state that ends up as
643 # ``None`` (gh-1632). ``__setstate__`` already unpacks both shapes, so
644 # pickles produced by older yarl releases (which embed a real
645 # ``SplitResult``) still load correctly.
646 return (self._val,)
648 def __setstate__(
649 self, state: tuple[SplitURLType] | tuple[None, _InternalURLCache]
650 ) -> None:
651 if state[0] is None and isinstance(state[1], dict):
652 # default style pickle
653 val = state[1]["_val"]
654 else:
655 unused: list[object]
656 val, *unused = state
657 self._scheme, self._netloc, self._path, self._query, self._fragment = val
658 self._cache = {}
660 def _cache_netloc(self) -> None:
661 """Cache the netloc parts of the URL."""
662 c = self._cache
663 split_loc = split_netloc(self._netloc)
664 c["raw_user"], c["raw_password"], c["raw_host"], c["explicit_port"] = split_loc
666 def is_absolute(self) -> bool:
667 """A check for absolute URLs.
669 Return True for absolute ones (having scheme or starting
670 with //), False otherwise.
672 Is is preferred to call the .absolute property instead
673 as it is cached.
674 """
675 return self.absolute
677 def is_default_port(self) -> bool:
678 """A check for default port.
680 Return True if port is default for specified scheme,
681 e.g. 'http://python.org' or 'http://python.org:80', False
682 otherwise.
684 Return False for relative URLs.
686 """
687 if (explicit := self.explicit_port) is None:
688 # If the explicit port is None, then the URL must be
689 # using the default port unless its a relative URL
690 # which does not have an implicit port / default port
691 return self._netloc != ""
692 return explicit == DEFAULT_PORTS.get(self._scheme)
694 def origin(self) -> "URL":
695 """Return an URL with scheme, host and port parts only.
697 user, password, path, query and fragment are removed.
699 """
700 # TODO: add a keyword-only option for keeping user/pass maybe?
701 return self._origin
703 @cached_property
704 def _val(self) -> SplitURLType:
705 return (self._scheme, self._netloc, self._path, self._query, self._fragment)
707 @cached_property
708 def _origin(self) -> "URL":
709 """Return an URL with scheme, host and port parts only.
711 user, password, path, query and fragment are removed.
712 """
713 if not (netloc := self._netloc):
714 raise ValueError("URL should be absolute")
715 if not (scheme := self._scheme):
716 raise ValueError("URL should have scheme")
717 if "@" in netloc:
718 encoded_host = self.host_subcomponent
719 netloc = make_netloc(None, None, encoded_host, self.explicit_port)
720 elif not self._path and not self._query and not self._fragment:
721 return self
722 return from_parts(scheme, netloc, "", "", "")
724 def relative(self) -> "URL":
725 """Return a relative part of the URL.
727 scheme, user, password, host and port are removed.
729 """
730 if not self._netloc:
731 raise ValueError("URL should be absolute")
732 return from_parts("", "", self._path, self._query, self._fragment)
734 @cached_property
735 def absolute(self) -> bool:
736 """A check for absolute URLs.
738 Return True for absolute ones (having scheme or starting
739 with //), False otherwise.
741 """
742 # `netloc`` is an empty string for relative URLs
743 # Checking `netloc` is faster than checking `hostname`
744 # because `hostname` is a property that does some extra work
745 # to parse the host from the `netloc`
746 return self._netloc != ""
748 @cached_property
749 def scheme(self) -> str:
750 """Scheme for absolute URLs.
752 Empty string for relative URLs or URLs starting with //
754 """
755 return self._scheme
757 @cached_property
758 def raw_authority(self) -> str:
759 """Encoded authority part of URL.
761 Empty string for relative URLs.
763 """
764 return self._netloc
766 @cached_property
767 def authority(self) -> str:
768 """Decoded authority part of URL.
770 Empty string for relative URLs.
772 """
773 return make_netloc(self.user, self.password, self.host, self.port)
775 @cached_property
776 def raw_user(self) -> str | None:
777 """Encoded user part of URL.
779 None if user is missing.
781 """
782 # not .username
783 self._cache_netloc()
784 return self._cache["raw_user"]
786 @cached_property
787 def user(self) -> str | None:
788 """Decoded user part of URL.
790 None if user is missing.
792 """
793 if (raw_user := self.raw_user) is None:
794 return None
795 return UNQUOTER(raw_user)
797 @cached_property
798 def raw_password(self) -> str | None:
799 """Encoded password part of URL.
801 None if password is missing.
803 """
804 self._cache_netloc()
805 return self._cache["raw_password"]
807 @cached_property
808 def password(self) -> str | None:
809 """Decoded password part of URL.
811 None if password is missing.
813 """
814 if (raw_password := self.raw_password) is None:
815 return None
816 return UNQUOTER(raw_password)
818 @cached_property
819 def raw_host(self) -> str | None:
820 """Encoded host part of URL.
822 None for relative URLs.
824 When working with IPv6 addresses, use the `host_subcomponent` property instead
825 as it will return the host subcomponent with brackets.
826 """
827 # Use host instead of hostname for sake of shortness
828 # May add .hostname prop later
829 self._cache_netloc()
830 return self._cache["raw_host"]
832 @cached_property
833 def host(self) -> str | None:
834 """Decoded host part of URL.
836 None for relative URLs.
838 For IPv6 hosts that carry an RFC 6874 zone identifier, the
839 ``%25`` zone separator is decoded back to ``%``; the encoded
840 form is still available via :attr:`raw_host` and
841 :attr:`host_subcomponent`.
843 """
844 if (raw := self.raw_host) is None:
845 return None
846 if raw and raw[-1].isdigit() or ":" in raw:
847 # IP addresses are never IDNA encoded. The replace decodes
848 # every %25 in the raw host, i.e. the RFC 6874 zone
849 # separator and any %25 that percent-encodes a literal %
850 # inside the zone identifier.
851 if "%25" in raw:
852 return raw.replace("%25", "%")
853 return raw
854 return _idna_decode(raw)
856 @cached_property
857 def host_subcomponent(self) -> str | None:
858 """Return the host subcomponent part of URL.
860 None for relative URLs.
862 https://datatracker.ietf.org/doc/html/rfc3986#section-3.2.2
864 `IP-literal = "[" ( IPv6address / IPvFuture ) "]"`
866 Examples:
867 - `http://example.com:8080` -> `example.com`
868 - `http://example.com:80` -> `example.com`
869 - `https://127.0.0.1:8443` -> `127.0.0.1`
870 - `https://[::1]:8443` -> `[::1]`
871 - `http://[::1]` -> `[::1]`
873 """
874 if (raw := self.raw_host) is None:
875 return None
876 return f"[{raw}]" if ":" in raw else raw
878 @cached_property
879 def host_port_subcomponent(self) -> str | None:
880 """Return the host and port subcomponent part of URL.
882 Trailing dots are removed from the host part.
884 This value is suitable for use in the Host header of an HTTP request.
886 None for relative URLs.
888 https://datatracker.ietf.org/doc/html/rfc3986#section-3.2.2
889 `IP-literal = "[" ( IPv6address / IPvFuture ) "]"`
890 https://datatracker.ietf.org/doc/html/rfc3986#section-3.2.3
891 port = *DIGIT
893 Examples:
894 - `http://example.com:8080` -> `example.com:8080`
895 - `http://example.com:80` -> `example.com`
896 - `http://example.com.:80` -> `example.com`
897 - `https://127.0.0.1:8443` -> `127.0.0.1:8443`
898 - `https://[::1]:8443` -> `[::1]:8443`
899 - `http://[::1]` -> `[::1]`
901 """
902 if (raw := self.raw_host) is None:
903 return None
904 if raw[-1] == ".":
905 # Remove all trailing dots from the netloc as while
906 # they are valid FQDNs in DNS, TLS validation fails.
907 # See https://github.com/aio-libs/aiohttp/issues/3636.
908 # To avoid string manipulation we only call rstrip if
909 # the last character is a dot.
910 raw = raw.rstrip(".")
911 port = self.explicit_port
912 if port is None or port == DEFAULT_PORTS.get(self._scheme):
913 return f"[{raw}]" if ":" in raw else raw
914 return f"[{raw}]:{port}" if ":" in raw else f"{raw}:{port}"
916 @cached_property
917 def port(self) -> int | None:
918 """Port part of URL, with scheme-based fallback.
920 None for relative URLs or URLs without explicit port and
921 scheme without default port substitution.
923 """
924 if (explicit_port := self.explicit_port) is not None:
925 return explicit_port
926 return DEFAULT_PORTS.get(self._scheme)
928 @cached_property
929 def explicit_port(self) -> int | None:
930 """Port part of URL, without scheme-based fallback.
932 None for relative URLs or URLs without explicit port.
934 """
935 self._cache_netloc()
936 return self._cache["explicit_port"]
938 @cached_property
939 def raw_path(self) -> str:
940 """Encoded path of URL.
942 / for absolute URLs without path part.
944 """
945 return self._path if self._path or not self._netloc else "/"
947 @cached_property
948 def path(self) -> str:
949 """Decoded path of URL.
951 / for absolute URLs without path part.
953 """
954 return UNQUOTER(self._path) if self._path else "/" if self._netloc else ""
956 @cached_property
957 def path_safe(self) -> str:
958 """Decoded path of URL.
960 / for absolute URLs without path part.
962 / (%2F) and % (%25) are not decoded
964 """
965 if self._path:
966 return PATH_SAFE_UNQUOTER(self._path)
967 return "/" if self._netloc else ""
969 @cached_property
970 def _parsed_query(self) -> list[tuple[str, str]]:
971 """Parse query part of URL."""
972 return query_to_pairs(self._query)
974 @cached_property
975 def query(self) -> "MultiDictProxy[str]":
976 """A MultiDictProxy representing parsed query parameters in decoded
977 representation.
979 Empty value if URL has no query part.
981 """
982 return MultiDictProxy(MultiDict(self._parsed_query))
984 @cached_property
985 def raw_query_string(self) -> str:
986 """Encoded query part of URL.
988 Empty string if query is missing.
990 """
991 return self._query
993 @cached_property
994 def query_string(self) -> str:
995 """Decoded query part of URL.
997 Empty string if query is missing.
999 """
1000 return QS_UNQUOTER(self._query) if self._query else ""
1002 @cached_property
1003 def path_qs(self) -> str:
1004 """Decoded path of URL with query."""
1005 return self.path if not (q := self.query_string) else f"{self.path}?{q}"
1007 @cached_property
1008 def raw_path_qs(self) -> str:
1009 """Encoded path of URL with query."""
1010 if q := self._query:
1011 return f"{self._path}?{q}" if self._path or not self._netloc else f"/?{q}"
1012 return self._path if self._path or not self._netloc else "/"
1014 @cached_property
1015 def raw_fragment(self) -> str:
1016 """Encoded fragment part of URL.
1018 Empty string if fragment is missing.
1020 """
1021 return self._fragment
1023 @cached_property
1024 def fragment(self) -> str:
1025 """Decoded fragment part of URL.
1027 Empty string if fragment is missing.
1029 """
1030 return UNQUOTER(self._fragment) if self._fragment else ""
1032 @cached_property
1033 def raw_parts(self) -> tuple[str, ...]:
1034 """A tuple containing encoded *path* parts.
1036 ('/',) for absolute URLs if *path* is missing.
1038 """
1039 path = self._path
1040 if self._netloc:
1041 return ("/", *path[1:].split("/")) if path else ("/",)
1042 if path and path[0] == "/":
1043 return ("/", *path[1:].split("/"))
1044 return tuple(path.split("/"))
1046 @cached_property
1047 def parts(self) -> tuple[str, ...]:
1048 """A tuple containing decoded *path* parts.
1050 ('/',) for absolute URLs if *path* is missing.
1052 """
1053 return tuple(UNQUOTER(part) for part in self.raw_parts)
1055 @cached_property
1056 def parent(self) -> "URL":
1057 """A new URL with last part of path removed and cleaned up query and
1058 fragment.
1060 """
1061 path = self._path
1062 if not path or path == "/":
1063 if self._fragment or self._query:
1064 return from_parts(self._scheme, self._netloc, path, "", "")
1065 return self
1066 parts = path.split("/")
1067 return from_parts(self._scheme, self._netloc, "/".join(parts[:-1]), "", "")
1069 @cached_property
1070 def raw_name(self) -> str:
1071 """The last part of raw_parts."""
1072 parts = self.raw_parts
1073 if not self._netloc:
1074 return parts[-1]
1075 parts = parts[1:]
1076 return parts[-1] if parts else ""
1078 @cached_property
1079 def name(self) -> str:
1080 """The last part of parts."""
1081 return UNQUOTER(self.raw_name)
1083 @cached_property
1084 def raw_suffix(self) -> str:
1085 name = self.raw_name
1086 i = name.rfind(".")
1087 return name[i:] if 0 < i < len(name) - 1 else ""
1089 @cached_property
1090 def suffix(self) -> str:
1091 return UNQUOTER(self.raw_suffix)
1093 @cached_property
1094 def raw_suffixes(self) -> tuple[str, ...]:
1095 name = self.raw_name
1096 if name.endswith("."):
1097 return ()
1098 name = name.lstrip(".")
1099 return tuple("." + suffix for suffix in name.split(".")[1:])
1101 @cached_property
1102 def suffixes(self) -> tuple[str, ...]:
1103 return tuple(UNQUOTER(suffix) for suffix in self.raw_suffixes)
1105 def _make_child(self, paths: "Sequence[str]", encoded: bool = False) -> "URL":
1106 """
1107 add paths to self._path, accounting for absolute vs relative paths,
1108 keep existing, but do not create new, empty segments
1109 """
1110 parsed: list[str] = []
1111 needs_normalize: bool = False
1112 for idx, path in enumerate(reversed(paths)):
1113 # empty segment of last is not removed
1114 last = idx == 0
1115 if path and path[0] == "/":
1116 raise ValueError(
1117 f"Appending path {path!r} starting from slash is forbidden"
1118 )
1119 # We need to quote the path if it is not already encoded
1120 # This cannot be done at the end because the existing
1121 # path is already quoted and we do not want to double quote
1122 # the existing path.
1123 path = path if encoded else PATH_QUOTER(path)
1124 needs_normalize |= "." in path
1125 segments = path.split("/")
1126 segments.reverse()
1127 # remove trailing empty segment for all but the last path
1128 parsed += segments[1:] if not last and segments[0] == "" else segments
1130 if (path := self._path) and (old_segments := path.split("/")):
1131 # If the old path ends with a slash, the last segment is an empty string
1132 # and should be removed before adding the new path segments.
1133 old = old_segments[:-1] if old_segments[-1] == "" else old_segments
1134 old.reverse()
1135 parsed += old
1137 # If the netloc is present, inject a leading slash when adding a
1138 # path to an absolute URL where there was none before.
1139 if (netloc := self._netloc) and parsed and parsed[-1] != "":
1140 parsed.append("")
1142 parsed.reverse()
1143 if not netloc or not needs_normalize:
1144 return from_parts(self._scheme, netloc, "/".join(parsed), "", "")
1146 path = "/".join(normalize_path_segments(parsed))
1147 # If normalizing the path segments removed the leading slash, add it back.
1148 if path and path[0] != "/":
1149 path = f"/{path}"
1150 return from_parts(self._scheme, netloc, path, "", "")
1152 def with_scheme(self, scheme: str) -> "URL":
1153 """Return a new URL with scheme replaced."""
1154 # N.B. doesn't cleanup query/fragment
1155 if not isinstance(scheme, str):
1156 raise TypeError("Invalid scheme type")
1157 lower_scheme = scheme.lower()
1158 netloc = self._netloc
1159 if not netloc and lower_scheme in SCHEME_REQUIRES_HOST:
1160 msg = (
1161 "scheme replacement is not allowed for "
1162 f"relative URLs for the {lower_scheme} scheme"
1163 )
1164 raise ValueError(msg)
1165 return from_parts(lower_scheme, netloc, self._path, self._query, self._fragment)
1167 def with_user(self, user: str | None) -> "URL":
1168 """Return a new URL with user replaced.
1170 Autoencode user if needed.
1172 Clear user/password if user is None.
1174 """
1175 # N.B. doesn't cleanup query/fragment
1176 if user is None:
1177 password = None
1178 elif isinstance(user, str):
1179 user = QUOTER(user)
1180 password = self.raw_password
1181 else:
1182 raise TypeError("Invalid user type")
1183 if not (netloc := self._netloc):
1184 raise ValueError("user replacement is not allowed for relative URLs")
1185 encoded_host = self.host_subcomponent or ""
1186 netloc = make_netloc(user, password, encoded_host, self.explicit_port)
1187 return from_parts(self._scheme, netloc, self._path, self._query, self._fragment)
1189 def with_password(self, password: str | None) -> "URL":
1190 """Return a new URL with password replaced.
1192 Autoencode password if needed.
1194 Clear password if argument is None.
1196 """
1197 # N.B. doesn't cleanup query/fragment
1198 if password is None:
1199 pass
1200 elif isinstance(password, str):
1201 password = QUOTER(password)
1202 else:
1203 raise TypeError("Invalid password type")
1204 if not (netloc := self._netloc):
1205 raise ValueError("password replacement is not allowed for relative URLs")
1206 encoded_host = self.host_subcomponent or ""
1207 port = self.explicit_port
1208 netloc = make_netloc(self.raw_user, password, encoded_host, port)
1209 return from_parts(self._scheme, netloc, self._path, self._query, self._fragment)
1211 def with_host(self, host: str) -> "URL":
1212 """Return a new URL with host replaced.
1214 Autoencode host if needed.
1216 Changing host for relative URLs is not allowed, use .join()
1217 instead.
1219 """
1220 # N.B. doesn't cleanup query/fragment
1221 if not isinstance(host, str):
1222 raise TypeError("Invalid host type")
1223 if not (netloc := self._netloc):
1224 raise ValueError("host replacement is not allowed for relative URLs")
1225 if not host:
1226 raise ValueError("host removing is not allowed")
1227 encoded_host = _encode_host(host, validate_host=True) if host else ""
1228 port = self.explicit_port
1229 netloc = make_netloc(self.raw_user, self.raw_password, encoded_host, port)
1230 return from_parts(self._scheme, netloc, self._path, self._query, self._fragment)
1232 def with_port(self, port: int | None) -> "URL":
1233 """Return a new URL with port replaced.
1235 Clear port to default if None is passed.
1237 """
1238 # N.B. doesn't cleanup query/fragment
1239 if port is not None:
1240 if isinstance(port, bool) or not isinstance(port, int):
1241 raise TypeError(f"port should be int or None, got {type(port)}")
1242 if not (0 <= port <= 65535):
1243 raise ValueError(f"port must be between 0 and 65535, got {port}")
1244 if not (netloc := self._netloc):
1245 raise ValueError("port replacement is not allowed for relative URLs")
1246 encoded_host = self.host_subcomponent or ""
1247 netloc = make_netloc(self.raw_user, self.raw_password, encoded_host, port)
1248 return from_parts(self._scheme, netloc, self._path, self._query, self._fragment)
1250 def with_path(
1251 self,
1252 path: str,
1253 *,
1254 encoded: bool = False,
1255 keep_query: bool = False,
1256 keep_fragment: bool = False,
1257 ) -> "URL":
1258 """Return a new URL with path replaced."""
1259 netloc = self._netloc
1260 if not encoded:
1261 path = PATH_QUOTER(path)
1262 if netloc:
1263 path = normalize_path(path) if "." in path else path
1264 if path and path[0] != "/":
1265 path = f"/{path}"
1266 query = self._query if keep_query else ""
1267 fragment = self._fragment if keep_fragment else ""
1268 return from_parts(self._scheme, netloc, path, query, fragment)
1270 @overload
1271 def with_query(self, query: Query) -> "URL": ...
1273 @overload
1274 def with_query(self, **kwargs: QueryVariable) -> "URL": ...
1276 def with_query(self, *args: Any, **kwargs: Any) -> "URL":
1277 """Return a new URL with query part replaced.
1279 Accepts any Mapping (e.g. dict, multidict.MultiDict instances)
1280 or str, autoencode the argument if needed.
1282 A sequence of (key, value) pairs is supported as well.
1284 It also can take an arbitrary number of keyword arguments.
1286 Clear query if None is passed.
1288 """
1289 # N.B. doesn't cleanup query/fragment
1290 query = get_str_query(*args, **kwargs) or ""
1291 return from_parts_uncached(
1292 self._scheme, self._netloc, self._path, query, self._fragment
1293 )
1295 @overload
1296 def extend_query(self, query: Query) -> "URL": ...
1298 @overload
1299 def extend_query(self, **kwargs: QueryVariable) -> "URL": ...
1301 def extend_query(self, *args: Any, **kwargs: Any) -> "URL":
1302 """Return a new URL with query part combined with the existing.
1304 This method will not remove existing query parameters.
1306 Example:
1307 >>> url = URL('http://example.com/?a=1&b=2')
1308 >>> url.extend_query(a=3, c=4)
1309 URL('http://example.com/?a=1&b=2&a=3&c=4')
1310 """
1311 if not (new_query := get_str_query(*args, **kwargs)):
1312 return self
1313 if query := self._query:
1314 # both strings are already encoded so we can use a simple
1315 # string join
1316 query += new_query if query[-1] == "&" else f"&{new_query}"
1317 else:
1318 query = new_query
1319 return from_parts_uncached(
1320 self._scheme, self._netloc, self._path, query, self._fragment
1321 )
1323 @overload
1324 def update_query(self, query: Query) -> "URL": ...
1326 @overload
1327 def update_query(self, **kwargs: QueryVariable) -> "URL": ...
1329 def update_query(self, *args: Any, **kwargs: Any) -> "URL":
1330 """Return a new URL with query part updated.
1332 This method will overwrite existing query parameters.
1334 Example:
1335 >>> url = URL('http://example.com/?a=1&b=2')
1336 >>> url.update_query(a=3, c=4)
1337 URL('http://example.com/?a=3&b=2&c=4')
1338 """
1339 in_query: (
1340 str
1341 | Mapping[str, QueryVariable]
1342 | Sequence[tuple[str | istr, SimpleQuery]]
1343 | None
1344 )
1345 if kwargs:
1346 if args:
1347 msg = "Either kwargs or single query parameter must be present"
1348 raise ValueError(msg)
1349 in_query = kwargs
1350 elif len(args) == 1:
1351 in_query = args[0]
1352 else:
1353 raise ValueError("Either kwargs or single query parameter must be present")
1355 if in_query is None:
1356 query = ""
1357 elif not in_query:
1358 query = self._query
1359 elif isinstance(in_query, Mapping):
1360 qm: MultiDict[QueryVariable] = MultiDict(self._parsed_query)
1361 qm.update(in_query)
1362 query = get_str_query_from_sequence_iterable(qm.items())
1363 elif isinstance(in_query, str):
1364 qstr: MultiDict[str] = MultiDict(self._parsed_query)
1365 qstr.update(query_to_pairs(in_query))
1366 query = get_str_query_from_iterable(qstr.items())
1367 elif isinstance(in_query, (bytes, bytearray, memoryview)):
1368 msg = "Invalid query type: bytes, bytearray and memoryview are forbidden"
1369 raise TypeError(msg)
1370 elif isinstance(in_query, Sequence):
1371 # We don't expect sequence values if we're given a list of pairs
1372 # already; only mappings like builtin `dict` which can't have the
1373 # same key pointing to multiple values are allowed to use
1374 # `_query_seq_pairs`.
1375 if TYPE_CHECKING:
1376 in_query = cast(
1377 Sequence[tuple[Union[str, istr], SimpleQuery]], in_query
1378 )
1379 qs: MultiDict[SimpleQuery] = MultiDict(self._parsed_query)
1380 qs.update(in_query)
1381 query = get_str_query_from_iterable(qs.items())
1382 else:
1383 raise TypeError(
1384 "Invalid query type: only str, mapping or "
1385 "sequence of (key, value) pairs is allowed"
1386 )
1387 return from_parts_uncached(
1388 self._scheme, self._netloc, self._path, query, self._fragment
1389 )
1391 def without_query_params(self, *query_params: str) -> "URL":
1392 """Remove some keys from query part and return new URL."""
1393 params_to_remove = set(query_params) & self.query.keys()
1394 if not params_to_remove:
1395 return self
1396 return self.with_query(
1397 tuple(
1398 (name, value)
1399 for name, value in self.query.items()
1400 if name not in params_to_remove
1401 )
1402 )
1404 def with_fragment(self, fragment: str | None) -> "URL":
1405 """Return a new URL with fragment replaced.
1407 Autoencode fragment if needed.
1409 Clear fragment to default if None is passed.
1411 """
1412 # N.B. doesn't cleanup query/fragment
1413 if fragment is None:
1414 raw_fragment = ""
1415 elif not isinstance(fragment, str):
1416 raise TypeError("Invalid fragment type")
1417 else:
1418 raw_fragment = FRAGMENT_QUOTER(fragment)
1419 if self._fragment == raw_fragment:
1420 return self
1421 return from_parts(
1422 self._scheme, self._netloc, self._path, self._query, raw_fragment
1423 )
1425 def with_name(
1426 self,
1427 name: str,
1428 *,
1429 keep_query: bool = False,
1430 keep_fragment: bool = False,
1431 ) -> "URL":
1432 """Return a new URL with name (last part of path) replaced.
1434 Query and fragment parts are cleaned up.
1436 Name is encoded if needed.
1438 """
1439 # N.B. DOES cleanup query/fragment
1440 if not isinstance(name, str):
1441 raise TypeError("Invalid name type")
1442 if "/" in name:
1443 raise ValueError("Slash in name is not allowed")
1444 name = PATH_QUOTER(name)
1445 if name in (".", ".."):
1446 raise ValueError(". and .. values are forbidden")
1447 parts = list(self.raw_parts)
1448 if netloc := self._netloc:
1449 if len(parts) == 1:
1450 parts.append(name)
1451 else:
1452 parts[-1] = name
1453 parts[0] = "" # replace leading '/'
1454 else:
1455 parts[-1] = name
1456 if parts[0] == "/":
1457 parts[0] = "" # replace leading '/'
1459 query = self._query if keep_query else ""
1460 fragment = self._fragment if keep_fragment else ""
1461 return from_parts(self._scheme, netloc, "/".join(parts), query, fragment)
1463 def with_suffix(
1464 self,
1465 suffix: str,
1466 *,
1467 keep_query: bool = False,
1468 keep_fragment: bool = False,
1469 ) -> "URL":
1470 """Return a new URL with suffix (file extension of name) replaced.
1472 Query and fragment parts are cleaned up.
1474 suffix is encoded if needed.
1475 """
1476 if not isinstance(suffix, str):
1477 raise TypeError("Invalid suffix type")
1478 if suffix and not suffix[0] == "." or suffix == "." or "/" in suffix:
1479 raise ValueError(f"Invalid suffix {suffix!r}")
1480 name = self.raw_name
1481 if not name:
1482 raise ValueError(f"{self!r} has an empty name")
1483 old_suffix = self.raw_suffix
1484 suffix = PATH_QUOTER(suffix)
1485 name = name + suffix if not old_suffix else name[: -len(old_suffix)] + suffix
1486 if name in (".", ".."):
1487 raise ValueError(". and .. values are forbidden")
1488 parts = list(self.raw_parts)
1489 if netloc := self._netloc:
1490 if len(parts) == 1:
1491 parts.append(name)
1492 else:
1493 parts[-1] = name
1494 parts[0] = "" # replace leading '/'
1495 else:
1496 parts[-1] = name
1497 if parts[0] == "/":
1498 parts[0] = "" # replace leading '/'
1500 query = self._query if keep_query else ""
1501 fragment = self._fragment if keep_fragment else ""
1502 return from_parts(self._scheme, netloc, "/".join(parts), query, fragment)
1504 def join(self, url: "URL") -> "URL":
1505 """Join URLs
1507 Construct a full (“absolute”) URL by combining a “base URL”
1508 (self) with another URL (url).
1510 Informally, this uses components of the base URL, in
1511 particular the addressing scheme, the network location and
1512 (part of) the path, to provide missing components in the
1513 relative URL.
1515 """
1516 if type(url) is not URL:
1517 raise TypeError("url should be URL")
1519 scheme = url._scheme or self._scheme
1520 if scheme != self._scheme or scheme not in USES_RELATIVE:
1521 return url
1523 # scheme is in uses_authority as uses_authority is a superset of uses_relative
1524 if (join_netloc := url._netloc) and scheme in USES_AUTHORITY:
1525 return from_parts(scheme, join_netloc, url._path, url._query, url._fragment)
1527 orig_path = self._path
1528 if join_path := url._path:
1529 if join_path[0] == "/":
1530 path = join_path
1531 elif not orig_path:
1532 path = f"/{join_path}"
1533 elif orig_path[-1] == "/":
1534 path = f"{orig_path}{join_path}"
1535 else:
1536 # …
1537 # and relativizing ".."
1538 # parts[0] is / for absolute urls,
1539 # this join will add a double slash there
1540 path = "/".join([*self.parts[:-1], ""]) + join_path
1541 # which has to be removed
1542 if orig_path[0] == "/":
1543 path = path[1:]
1544 path = normalize_path(path) if "." in path else path
1545 else:
1546 path = orig_path
1548 return from_parts(
1549 scheme,
1550 self._netloc,
1551 path,
1552 url._query if join_path or url._query else self._query,
1553 url._fragment if join_path or url._fragment else self._fragment,
1554 )
1556 def joinpath(self, *other: str, encoded: bool = False) -> "URL":
1557 """Return a new URL with the elements in other appended to the path."""
1558 return self._make_child(other, encoded=encoded)
1560 def human_repr(self) -> str:
1561 """Return decoded human readable string for URL representation."""
1562 user = human_quote(self.user, "#/:?@[]\\")
1563 password = human_quote(self.password, "#/:?@[]\\")
1564 if (host := self.host) and ":" in host:
1565 host = f"[{host}]"
1566 path = human_quote(self.path, "#?")
1567 if TYPE_CHECKING:
1568 assert path is not None
1569 if not self._scheme and not self._netloc:
1570 path = _encode_relative_scheme_colon(path)
1571 query_string = "&".join(
1572 "{}={}".format(human_quote(k, "#&+;="), human_quote(v, "#&+;="))
1573 for k, v in self.query.items()
1574 )
1575 fragment = human_quote(self.fragment, "")
1576 if TYPE_CHECKING:
1577 assert fragment is not None
1578 netloc = make_netloc(user, password, host, self.explicit_port)
1579 return unsplit_result(self._scheme, netloc, path, query_string, fragment)
1581 if HAS_PYDANTIC:
1582 # Borrowed from https://docs.pydantic.dev/latest/concepts/types/#handling-third-party-types
1583 @classmethod
1584 def __get_pydantic_json_schema__(
1585 cls,
1586 core_schema: "CoreSchema",
1587 handler: "GetJsonSchemaHandler",
1588 ) -> "JsonSchemaValue":
1589 field_schema: dict[str, Any] = {}
1590 field_schema.update(type="string", format="uri")
1591 return field_schema
1593 @classmethod
1594 def __get_pydantic_core_schema__(
1595 cls,
1596 source_type: type[Self] | type[str],
1597 handler: "GetCoreSchemaHandler",
1598 ) -> "CoreSchema":
1599 # Lazy import: pulling in pydantic_core at module load time
1600 # increases yarl's import cost 3-7x for users who don't use
1601 # pydantic. Keep this import function-scoped.
1602 from pydantic_core import core_schema # noqa: PLC0415
1604 from_str_schema = core_schema.chain_schema(
1605 [
1606 core_schema.str_schema(),
1607 core_schema.no_info_plain_validator_function(URL),
1608 ]
1609 )
1611 return core_schema.json_or_python_schema(
1612 json_schema=from_str_schema,
1613 python_schema=core_schema.union_schema(
1614 [
1615 # check if it's an instance first before doing any further work
1616 core_schema.is_instance_schema(URL),
1617 from_str_schema,
1618 ]
1619 ),
1620 serialization=core_schema.plain_serializer_function_ser_schema(str),
1621 )
1624_DEFAULT_IDNA_SIZE = 256
1625_DEFAULT_ENCODE_SIZE = 512
1628@lru_cache(_DEFAULT_IDNA_SIZE)
1629def _idna_decode(raw: str) -> str:
1630 try:
1631 return idna.decode(raw.encode("ascii"))
1632 except UnicodeError: # e.g. '::1'
1633 return raw.encode("ascii").decode("idna")
1636@lru_cache(_DEFAULT_IDNA_SIZE)
1637def _idna_encode(host: str) -> str:
1638 try:
1639 return idna.encode(host, uts46=True).decode("ascii")
1640 except UnicodeError:
1641 return host.encode("idna").decode("ascii")
1644@lru_cache(_DEFAULT_ENCODE_SIZE)
1645def _encode_host(host: str, validate_host: bool) -> str:
1646 """Encode host part of URL."""
1647 # If the host ends with a digit or contains a colon, its likely
1648 # an IP address.
1649 if host and (host[-1].isdigit() or ":" in host):
1650 # RFC 6874 spells the IPv6 zone separator as the percent-encoded
1651 # ``%25``; bare ``%`` is still accepted so that hosts constructed
1652 # programmatically (e.g. ``with_host("fe80::1%1")``) keep working.
1653 part = "%25" if "%25" in host else "%"
1654 raw_ip, sep, zone = host.partition(part)
1655 # If it looks like an IP, we check with _ip_compressed_version
1656 # and fall-through if its not an IP address. This is a performance
1657 # optimization to avoid parsing IP addresses as much as possible
1658 # because it is orders of magnitude slower than almost any other
1659 # operation this library does.
1660 # Might be an IP address, check it
1661 #
1662 # IP Addresses can look like:
1663 # https://datatracker.ietf.org/doc/html/rfc3986#section-3.2.2
1664 # - 127.0.0.1 (last character is a digit)
1665 # - 2001:db8::ff00:42:8329 (contains a colon)
1666 # - 2001:db8::ff00:42:8329%eth0 (contains a colon)
1667 # - [2001:db8::ff00:42:8329] (contains a colon -- brackets should
1668 # have been removed before it gets here)
1669 # Rare IP Address formats are not supported per:
1670 # https://datatracker.ietf.org/doc/html/rfc3986#section-7.4
1671 #
1672 # IP parsing is slow, so its wrapped in an LRU
1673 try:
1674 ip = ip_address(raw_ip)
1675 except ValueError:
1676 pass
1677 else:
1678 if sep and validate_host and (not zone or _ZONE_ID_UNSAFE_RE.search(zone)):
1679 raise ValueError("Invalid characters in zone identifier")
1680 # These checks should not happen in the
1681 # LRU to keep the cache size small
1682 host = ip.compressed
1683 if ip.version == 6:
1684 return f"[{host}{sep}{zone}]" if sep else f"[{host}]"
1685 return f"{host}{sep}{zone}" if sep else host
1687 # IDNA encoding is slow, skip it for ASCII-only strings
1688 if host.isascii():
1689 # Check for invalid characters explicitly; _idna_encode() does this
1690 # for non-ascii host names.
1691 host = host.lower()
1692 if validate_host and (invalid := NOT_REG_NAME.search(host)):
1693 value, pos, extra = invalid.group(), invalid.start(), ""
1694 if value == "@" or (value == ":" and "@" in host[pos:]):
1695 # this looks like an authority string
1696 extra = (
1697 ", if the value includes a username or password, "
1698 "use 'authority' instead of 'host'"
1699 )
1700 raise ValueError(
1701 f"Host {host!r} cannot contain {value!r} (at position {pos}){extra}"
1702 ) from None
1703 return host
1705 # IDNA/UTS-46 mapping silently deletes default-ignorable code points, which
1706 # would turn e.g. ``e<ZWSP>vil.com`` into ``evil.com``, a different host
1707 # than the string the caller supplied. Reject them on every path (this runs
1708 # regardless of ``validate_host`` since the plain ``URL(str)`` constructor
1709 # encodes with ``validate_host=False``) so the parsed host cannot diverge
1710 # from the input, matching idna/httpx/urllib3.
1711 if invalid := _DEFAULT_IGNORABLE_RE.search(host):
1712 raise ValueError(
1713 f"Host {host!r} cannot contain {invalid.group()!r} "
1714 f"(at position {invalid.start()})"
1715 ) from None
1716 encoded = _idna_encode(host)
1717 # IDNA uses NFKC equivalence, so normalization can expand a non-ascii
1718 # character into an ASCII delimiter (e.g. the fullwidth solidus U+FF0F
1719 # becomes '/'). The ascii branch above rejects such delimiters directly;
1720 # apply the same check to the IDNA output so the builder APIs agree with
1721 # the parser's _check_netloc.
1722 if validate_host and (invalid := NOT_REG_NAME.search(encoded)):
1723 raise ValueError(
1724 f"Host {host!r} cannot contain {invalid.group()!r} "
1725 f"after IDNA normalization to {encoded!r}"
1726 ) from None
1727 return encoded
1730@rewrite_module
1731def cache_clear() -> None:
1732 """Clear all LRU caches."""
1733 _idna_encode.cache_clear()
1734 _idna_decode.cache_clear()
1735 _encode_host.cache_clear()
1738@rewrite_module
1739def cache_info() -> CacheInfo:
1740 """Report cache statistics."""
1741 return {
1742 "idna_encode": _idna_encode.cache_info(),
1743 "idna_decode": _idna_decode.cache_info(),
1744 "ip_address": _encode_host.cache_info(),
1745 "host_validate": _encode_host.cache_info(),
1746 "encode_host": _encode_host.cache_info(),
1747 }
1750@rewrite_module
1751def cache_configure(
1752 *,
1753 idna_encode_size: int | None = _DEFAULT_IDNA_SIZE,
1754 idna_decode_size: int | None = _DEFAULT_IDNA_SIZE,
1755 ip_address_size: int | None | UndefinedType = UNDEFINED,
1756 host_validate_size: int | None | UndefinedType = UNDEFINED,
1757 encode_host_size: int | None | UndefinedType = UNDEFINED,
1758) -> None:
1759 """Configure LRU cache sizes."""
1760 global _idna_decode, _idna_encode, _encode_host
1761 # ip_address_size, host_validate_size are no longer
1762 # used, but are kept for backwards compatibility.
1763 if ip_address_size is not UNDEFINED or host_validate_size is not UNDEFINED:
1764 warnings.warn(
1765 "cache_configure() no longer accepts the "
1766 "ip_address_size or host_validate_size arguments, "
1767 "they are used to set the encode_host_size instead "
1768 "and will be removed in the future",
1769 DeprecationWarning,
1770 stacklevel=2,
1771 )
1773 if encode_host_size is not None:
1774 for size in (ip_address_size, host_validate_size):
1775 if size is None:
1776 encode_host_size = None
1777 elif encode_host_size is UNDEFINED:
1778 if size is not UNDEFINED:
1779 encode_host_size = size
1780 elif size is not UNDEFINED:
1781 if TYPE_CHECKING:
1782 assert isinstance(size, int)
1783 assert isinstance(encode_host_size, int)
1784 encode_host_size = max(size, encode_host_size)
1785 if encode_host_size is UNDEFINED:
1786 encode_host_size = _DEFAULT_ENCODE_SIZE
1788 _encode_host = lru_cache(encode_host_size)(_encode_host.__wrapped__)
1789 _idna_decode = lru_cache(idna_decode_size)(_idna_decode.__wrapped__)
1790 _idna_encode = lru_cache(idna_encode_size)(_idna_encode.__wrapped__)