Coverage for /pythoncovmergedfiles/medio/medio/usr/local/lib/python3.11/site-packages/werkzeug/http.py: 20%
Shortcuts on this page
r m x toggle line displays
j k next/prev highlighted chunk
0 (zero) top of page
1 (one) first highlighted chunk
Shortcuts on this page
r m x toggle line displays
j k next/prev highlighted chunk
0 (zero) top of page
1 (one) first highlighted chunk
1from __future__ import annotations
3import email.utils
4import re
5import typing as t
6import warnings
7from datetime import date
8from datetime import datetime
9from datetime import time
10from datetime import timedelta
11from datetime import timezone
12from enum import Enum
13from hashlib import sha1
14from time import mktime
15from time import struct_time
16from urllib.parse import quote
17from urllib.parse import unquote
19from ._internal import _dt_as_utc
20from ._internal import _plain_int
22if t.TYPE_CHECKING:
23 from _typeshed.wsgi import WSGIEnvironment
25_token_chars = frozenset(
26 "!#$%&'*+-.0123456789ABCDEFGHIJKLMNOPQRSTUVWXYZ^_`abcdefghijklmnopqrstuvwxyz|~"
27)
28_entity_headers = frozenset(
29 [
30 "allow",
31 "content-encoding",
32 "content-language",
33 "content-length",
34 "content-location",
35 "content-md5",
36 "content-range",
37 "content-type",
38 "expires",
39 "last-modified",
40 ]
41)
42_hop_by_hop_headers = frozenset(
43 [
44 "connection",
45 "keep-alive",
46 "proxy-authenticate",
47 "proxy-authorization",
48 "te",
49 "trailer",
50 "transfer-encoding",
51 "upgrade",
52 ]
53)
54HTTP_STATUS_CODES = {
55 100: "Continue",
56 101: "Switching Protocols",
57 102: "Processing",
58 103: "Early Hints", # see RFC 8297
59 200: "OK",
60 201: "Created",
61 202: "Accepted",
62 203: "Non Authoritative Information",
63 204: "No Content",
64 205: "Reset Content",
65 206: "Partial Content",
66 207: "Multi Status",
67 208: "Already Reported", # see RFC 5842
68 226: "IM Used", # see RFC 3229
69 300: "Multiple Choices",
70 301: "Moved Permanently",
71 302: "Found",
72 303: "See Other",
73 304: "Not Modified",
74 305: "Use Proxy",
75 306: "Switch Proxy", # unused
76 307: "Temporary Redirect",
77 308: "Permanent Redirect",
78 400: "Bad Request",
79 401: "Unauthorized",
80 402: "Payment Required", # unused
81 403: "Forbidden",
82 404: "Not Found",
83 405: "Method Not Allowed",
84 406: "Not Acceptable",
85 407: "Proxy Authentication Required",
86 408: "Request Timeout",
87 409: "Conflict",
88 410: "Gone",
89 411: "Length Required",
90 412: "Precondition Failed",
91 413: "Request Entity Too Large",
92 414: "Request URI Too Long",
93 415: "Unsupported Media Type",
94 416: "Requested Range Not Satisfiable",
95 417: "Expectation Failed",
96 418: "I'm a teapot", # see RFC 2324
97 421: "Misdirected Request", # see RFC 7540
98 422: "Unprocessable Entity",
99 423: "Locked",
100 424: "Failed Dependency",
101 425: "Too Early", # see RFC 8470
102 426: "Upgrade Required",
103 428: "Precondition Required", # see RFC 6585
104 429: "Too Many Requests",
105 431: "Request Header Fields Too Large",
106 449: "Retry With", # proprietary MS extension
107 451: "Unavailable For Legal Reasons",
108 500: "Internal Server Error",
109 501: "Not Implemented",
110 502: "Bad Gateway",
111 503: "Service Unavailable",
112 504: "Gateway Timeout",
113 505: "HTTP Version Not Supported",
114 506: "Variant Also Negotiates", # see RFC 2295
115 507: "Insufficient Storage",
116 508: "Loop Detected", # see RFC 5842
117 510: "Not Extended",
118 511: "Network Authentication Failed",
119}
122class COEP(Enum):
123 """Cross Origin Embedder Policies"""
125 UNSAFE_NONE = "unsafe-none"
126 REQUIRE_CORP = "require-corp"
129class COOP(Enum):
130 """Cross Origin Opener Policies"""
132 UNSAFE_NONE = "unsafe-none"
133 SAME_ORIGIN_ALLOW_POPUPS = "same-origin-allow-popups"
134 SAME_ORIGIN = "same-origin"
137def quote_header_value(value: t.Any, allow_token: bool = True) -> str:
138 """Add double quotes around a header value. If the header contains only ASCII token
139 characters, it will be returned unchanged. If the header contains ``"`` or ``\\``
140 characters, they will be escaped with an additional ``\\`` character.
142 This is the reverse of :func:`unquote_header_value`.
144 :param value: The value to quote. Will be converted to a string.
145 :param allow_token: Disable to quote the value even if it only has token characters.
147 .. versionchanged:: 3.0
148 Passing bytes is not supported.
150 .. versionchanged:: 3.0
151 The ``extra_chars`` parameter is removed.
153 .. versionchanged:: 2.3
154 The value is quoted if it is the empty string.
156 .. versionadded:: 0.5
157 """
158 value_str = str(value)
160 if not value_str:
161 return '""'
163 if allow_token:
164 token_chars = _token_chars
166 if token_chars.issuperset(value_str):
167 return value_str
169 value_str = value_str.replace("\\", "\\\\").replace('"', '\\"')
170 return f'"{value_str}"'
173_unslash_re = re.compile(r"\\(.)", re.A)
176def unquote_header_value(value: str) -> str:
177 """Remove double quotes and backslash escapes from a header value.
179 This is the reverse of :func:`quote_header_value`.
181 :param value: The header value to unquote.
183 .. versionchanged:: 3.2
184 Removes escape preceding any character.
186 .. versionchanged:: 3.0
187 The ``is_filename`` parameter is removed.
188 """
189 if len(value) >= 2 and value.startswith('"') and value.endswith('"'):
190 return _unslash_re.sub(r"\g<1>", value[1:-1])
192 return value
195def dump_options_header(header: str | None, options: t.Mapping[str, t.Any]) -> str:
196 """Produce a header value and ``key=value`` parameters separated by semicolons
197 ``;``. For example, the ``Content-Type`` header.
199 .. code-block:: python
201 dump_options_header("text/html", {"charset": "UTF-8"})
202 'text/html; charset=UTF-8'
204 This is the reverse of :func:`parse_options_header`.
206 If a value contains non-token characters, it will be quoted.
208 If a value is ``None``, the parameter is skipped.
210 In some keys for some headers, a UTF-8 value can be encoded using a special
211 ``key*=UTF-8''value`` form, where ``value`` is percent encoded. This function will
212 not produce that format automatically, but if a given key ends with an asterisk
213 ``*``, the value is assumed to have that form and will not be quoted further.
215 :param header: The primary header value.
216 :param options: Parameters to encode as ``key=value`` pairs.
218 .. versionchanged:: 2.3
219 Keys with ``None`` values are skipped rather than treated as a bare key.
221 .. versionchanged:: 2.2.3
222 If a key ends with ``*``, its value will not be quoted.
223 """
224 segments = []
226 if header is not None:
227 segments.append(header)
229 for key, value in options.items():
230 if value is None:
231 continue
233 if key.endswith("*"):
234 segments.append(f"{key}={value}")
235 else:
236 segments.append(f"{key}={quote_header_value(value)}")
238 return "; ".join(segments)
241def dump_header(iterable: dict[str, t.Any] | t.Iterable[t.Any]) -> str:
242 """Produce a header value from a list of items or ``key=value`` pairs, separated by
243 commas ``,``.
245 This is the reverse of :func:`parse_list_header`, :func:`parse_dict_header`, and
246 :func:`parse_set_header`.
248 If a value contains non-token characters, it will be quoted.
250 If a value is ``None``, the key is output alone.
252 In some keys for some headers, a UTF-8 value can be encoded using a special
253 ``key*=UTF-8''value`` form, where ``value`` is percent encoded. This function will
254 not produce that format automatically, but if a given key ends with an asterisk
255 ``*``, the value is assumed to have that form and will not be quoted further.
257 .. code-block:: python
259 dump_header(["foo", "bar baz"])
260 'foo, "bar baz"'
262 dump_header({"foo": "bar baz"})
263 'foo="bar baz"'
265 :param iterable: The items to create a header from.
267 .. versionchanged:: 3.0
268 The ``allow_token`` parameter is removed.
270 .. versionchanged:: 2.2.3
271 If a key ends with ``*``, its value will not be quoted.
272 """
273 if isinstance(iterable, dict):
274 items = []
276 for key, value in iterable.items():
277 if value is None:
278 items.append(key)
279 elif key.endswith("*"):
280 items.append(f"{key}={value}")
281 else:
282 items.append(f"{key}={quote_header_value(value)}")
283 else:
284 items = [quote_header_value(x) for x in iterable]
286 return ", ".join(items)
289def dump_csp_header(header: ds.ContentSecurityPolicy) -> str:
290 """Dump a Content Security Policy header.
292 These are structured into policies such as "default-src 'self';
293 script-src 'self'".
295 .. versionadded:: 1.0.0
296 Support for Content Security Policy headers was added.
298 """
299 return "; ".join(f"{key} {value}" for key, value in header.items())
302def parse_list_header(value: str) -> list[str]:
303 """Parse a header value that consists of a list of comma separated items according
304 to `RFC 9110 <https://httpwg.org/specs/rfc9110.html#abnf.extension>`__.
306 Surrounding quotes are removed from items, but internal quotes are left for
307 future parsing. Empty values are discarded.
309 .. code-block:: python
311 parse_list_header('token, "quoted value"')
312 ['token', 'quoted value']
314 This is the reverse of :func:`dump_header`.
316 :param value: The header value to parse.
318 .. versionchanged:: 3.2
319 Quotes and escapes are kept if only part of an item is quoted. Empty
320 values are omitted. An empty list is returned if the value contains an
321 unclosed quoted string.
322 """
323 items = []
324 item = ""
325 escape = False
326 quote = False
328 for char in value:
329 if escape:
330 escape = False
331 item += char
332 continue
334 if quote:
335 if char == "\\":
336 escape = True
337 elif char == '"':
338 quote = False
340 item += char
341 continue
343 if char == ",":
344 items.append(item)
345 item = ""
346 continue
348 if char == '"':
349 quote = True
351 item += char
353 if quote:
354 # invalid, unclosed quoted string
355 return []
357 items.append(item)
358 return [
359 unquote_header_value(item)
360 for item in (item.strip(" \t") for item in items)
361 if item
362 ]
365def parse_dict_header(value: str) -> dict[str, str | None]:
366 """Parse a list header using :func:`parse_list_header`, then parse each item as a
367 ``key=value`` pair.
369 .. code-block:: python
371 parse_dict_header('a=b, c="d, e", f')
372 {"a": "b", "c": "d, e", "f": None}
374 This is the reverse of :func:`dump_header`.
376 If a key does not have a value, it is ``None``.
378 This handles charsets for values as described in
379 `RFC 2231 <https://www.rfc-editor.org/rfc/rfc2231#section-3>`__. Only ASCII, UTF-8,
380 and ISO-8859-1 charsets are accepted, otherwise the value remains quoted.
382 :param value: The header value to parse.
384 .. versionchanged:: 3.2
385 An empty dict is returned if the value contains an unclosed quoted
386 string.
388 .. versionchanged:: 3.0
389 Passing bytes is not supported.
391 .. versionchanged:: 3.0
392 The ``cls`` argument is removed.
394 .. versionchanged:: 2.3
395 Added support for ``key*=charset''value`` encoded items.
397 .. versionchanged:: 0.9
398 The ``cls`` argument was added.
399 """
400 result: dict[str, str | None] = {}
402 for item in parse_list_header(value):
403 key, has_value, value = item.partition("=")
404 key = key.strip(" \t")
406 if not key:
407 # =value is not valid
408 continue
410 if not has_value:
411 result[key] = None
412 continue
414 value = value.strip(" \t")
415 encoding: str | None = None
417 if key.endswith("*"):
418 # key*=charset''value becomes key=value, where value is percent encoded
419 # adapted from parse_options_header, without the continuation handling
420 key = key[:-1]
422 if (m_charset := _charset_value_re.match(value)) is not None:
423 # If there is a charset marker in the value, split it off.
424 encoding, value = m_charset.groups()
425 encoding = encoding.lower()
427 # A safe list of encodings. Modern clients should only send ASCII or UTF-8.
428 # This list will not be extended further. An invalid encoding will leave the
429 # value quoted.
430 if encoding in {"ascii", "us-ascii", "utf-8", "iso-8859-1"}:
431 # invalid bytes are replaced during unquoting
432 value = unquote(value, encoding=encoding)
434 result[key] = unquote_header_value(value)
436 return result
439# https://httpwg.org/specs/rfc9110.html#parameter
440_parameter_key_re = re.compile(r"([\w!#$%&'*+\-.^`|~]+)=", flags=re.ASCII)
441# https://www.rfc-editor.org/rfc/rfc2231#section-3
442_parameter_mark_re = re.compile(r"(?:\*(\d+))?(\*)?$", flags=re.ASCII)
443_parameter_token_value_re = re.compile(r"[\w!#$%&'*+\-.^`|~]+", flags=re.ASCII)
444# https://www.rfc-editor.org/rfc/rfc2231#section-4
445_charset_value_re = re.compile(
446 r"""
447 ([\w!#$%&*+\-.^`|~]*)' # charset part, could be empty
448 [\w!#$%&*+\-.^`|~]*' # don't care about language part, usually empty
449 ([\w!#$%&'*+\-.^`|~]+)$ # one or more token chars with percent encoding
450 """,
451 re.ASCII | re.VERBOSE,
452)
453_parameter_delimiter_re = re.compile(r";[ \t]*")
456def parse_options_header(value: str | None) -> tuple[str, dict[str, str]]:
457 """Parse a header that consists of a value with ``key=value`` parameters separated
458 by semicolons ``;``. For example, the ``Content-Type`` header.
460 .. code-block:: python
462 parse_options_header("text/html; charset=UTF-8")
463 ('text/html', {'charset': 'UTF-8'})
465 parse_options_header("")
466 ("", {})
468 This is the reverse of :func:`dump_options_header`.
470 This parses valid parameter parts as described in
471 `RFC 9110 <https://httpwg.org/specs/rfc9110.html#parameter>`__. Invalid parts are
472 skipped.
474 This handles continuations and charsets as described in
475 `RFC 2231 <https://www.rfc-editor.org/rfc/rfc2231#section-3>`__, although not as
476 strictly as the RFC. Only ASCII, UTF-8, and ISO-8859-1 charsets are accepted,
477 otherwise the value remains quoted.
479 Clients may not be consistent in how they handle a quote character within a quoted
480 value. The `HTML Standard <https://html.spec.whatwg.org/#multipart-form-data>`__
481 replaces it with ``%22`` in multipart form data.
482 `RFC 9110 <https://httpwg.org/specs/rfc9110.html#quoted.strings>`__ uses backslash
483 escapes in HTTP headers. Both are decoded to the ``"`` character.
485 Clients may not be consistent in how they handle non-ASCII characters. HTML
486 documents must declare ``<meta charset=UTF-8>``, otherwise browsers may replace with
487 HTML character references, which can be decoded using :func:`html.unescape`.
489 :param value: The header value to parse.
490 :return: ``(value, options)``, where ``options`` is a dict
492 .. versionchanged:: 2.3
493 Invalid parts, such as keys with no value, quoted keys, and incorrectly quoted
494 values, are discarded instead of treating as ``None``.
496 .. versionchanged:: 2.3
497 Only ASCII, UTF-8, and ISO-8859-1 are accepted for charset values.
499 .. versionchanged:: 2.3
500 Escaped quotes in quoted values, like ``%22`` and ``\\"``, are handled.
502 .. versionchanged:: 2.2
503 Option names are always converted to lowercase.
505 .. versionchanged:: 2.2
506 The ``multiple`` parameter was removed.
508 .. versionchanged:: 0.15
509 :rfc:`2231` parameter continuations are handled.
511 .. versionadded:: 0.5
512 """
513 if value is None:
514 return "", {}
516 value, _, rest = value.partition(";")
517 value = value.strip(" \t")
518 rest = rest.strip(" \t")
520 if not value or not rest:
521 # empty (invalid) value, or value without options
522 return value, {}
524 # Collect all valid key=value parts without processing the value.
525 parts: list[tuple[str, str]] = []
526 scan_pos = 0
527 length = len(rest)
529 while scan_pos < length:
530 if (m_key := _parameter_key_re.match(rest, scan_pos)) is not None:
531 pk = m_key.group(1).lower()
532 scan_pos = m_key.end()
534 # Value may be a token.
535 if (m_value := _parameter_token_value_re.match(rest, scan_pos)) is not None:
536 parts.append((pk, m_value.group()))
538 # Value may be a quoted string, find the closing quote.
539 elif rest.startswith('"', scan_pos):
540 quote_pos = scan_pos + 1
542 while quote_pos < length:
543 if rest.startswith(("\\\\", '\\"'), quote_pos):
544 # Consume escaped slashes and quotes.
545 quote_pos += 2
546 elif rest.startswith('"', quote_pos):
547 # Stop at an unescaped quote.
548 quote_pos += 1
549 parts.append((pk, rest[scan_pos:quote_pos]))
550 scan_pos = quote_pos
551 break
552 else:
553 # Consume any other character.
554 quote_pos += 1
555 else:
556 # Scanned all without finding closing quote.
557 break
559 # Continue to the next delimited part, skipping spaces.
560 if m_delim := _parameter_delimiter_re.search(rest, scan_pos):
561 scan_pos = m_delim.end()
562 else:
563 break
565 options: dict[str, str] = {}
566 continuations: dict[str, list[str]] = {}
567 continuation_encodings: dict[str, str] = {}
569 # For each collected part, process optional charset and continuation,
570 # unquote quoted values.
571 for pk, pv in parts:
572 encoding: str | None = None
573 has_continuation = has_charset = False
575 # Check for and remove markers at end of key.
576 if (
577 m_mark := _parameter_mark_re.search(pk)
578 ) is not None and m_mark.end() - m_mark.start() > 0:
579 has_continuation = m_mark.group(1) is not None
580 has_charset = m_mark.group(2) is not None
581 pk = pk[: m_mark.start()]
583 # key*=charset''value becomes key=value, where value is percent encoded
584 if has_charset:
585 if (m_charset := _charset_value_re.match(pv)) is not None:
586 # If there is a valid charset marker in the value, split it off.
587 encoding, pv = m_charset.groups()
588 # This might be the empty string, handled next.
589 encoding = encoding.lower()
591 # Use the prior continuation encoding if not set on this part.
592 if not encoding and has_continuation:
593 encoding = continuation_encodings.get(pk)
595 # A safe list of encodings. Modern clients should only send ASCII or UTF-8.
596 # This list will not be extended further. An invalid encoding will leave the
597 # value quoted.
598 if encoding in {"ascii", "us-ascii", "utf-8", "iso-8859-1"}:
599 # invalid bytes are replaced during unquoting
600 pv = unquote(pv, encoding=encoding)
601 else:
602 # Remove quotes, replace header slash escapes, replace multipart
603 # percent-encoded quotes.
604 pv = unquote_header_value(pv).replace("%22", '"')
606 # key*0=a; key*1=b becomes key=ab
607 if has_continuation:
608 # Remove prior normal option.
609 if pk in options:
610 del options[pk]
612 # For simplicity, this uses the scan order rather than enforcing
613 # sequential numbering.
614 if pk in continuations:
615 continuations[pk].append(pv)
616 else:
617 continuations[pk] = [pv]
619 # Further parts don't require their own charset marker.
620 if encoding:
621 continuation_encodings[pk] = encoding
622 else:
623 # Remove prior continuation option.
624 if pk in continuations:
625 del continuations[pk]
626 continuation_encodings.pop(pk, None)
628 options[pk] = pv
630 options.update({pk: "".join(pv) for pk, pv in continuations.items()})
631 return value, options
634_q_value_re = re.compile(r"-?\d+(\.\d+)?", re.ASCII)
635_TAnyAccept = t.TypeVar("_TAnyAccept", bound="ds.Accept")
638@t.overload
639def parse_accept_header(value: str | None) -> ds.Accept: ...
642@t.overload
643def parse_accept_header(value: str | None, cls: type[_TAnyAccept]) -> _TAnyAccept: ...
646def parse_accept_header(
647 value: str | None, cls: type[_TAnyAccept] | None = None
648) -> _TAnyAccept:
649 """Parse an ``Accept`` header according to
650 `RFC 9110 <https://httpwg.org/specs/rfc9110.html#field.accept>`__.
652 Returns an :class:`.Accept` instance, which can sort and inspect items based on
653 their quality parameter. When parsing ``Accept-Charset``, ``Accept-Encoding``, or
654 ``Accept-Language``, pass the appropriate :class:`.Accept` subclass.
656 :param value: The header value to parse.
657 :param cls: The :class:`.Accept` class to wrap the result in.
658 :return: An instance of ``cls``.
660 .. versionchanged:: 2.3
661 Parse according to RFC 9110. Items with invalid ``q`` values are skipped.
662 """
663 if cls is None:
664 cls = t.cast(type[_TAnyAccept], ds.Accept)
666 if not value:
667 return cls(None)
669 result = []
671 for item in parse_list_header(value):
672 item, options = parse_options_header(item)
674 if "q" in options:
675 # pop q, remaining options are reconstructed
676 q_str = options.pop("q").strip(" \t")
678 if _q_value_re.fullmatch(q_str) is None:
679 # ignore an invalid q
680 continue
682 q = float(q_str)
684 if q < 0 or q > 1:
685 # ignore an invalid q
686 continue
687 else:
688 q = 1
690 if options:
691 # reconstruct the media type with any options
692 item = dump_options_header(item, options)
694 result.append((item, q))
696 return cls(result)
699_TAnyCC = t.TypeVar("_TAnyCC", bound="ds.cache_control._CacheControl")
702@t.overload
703def parse_cache_control_header(
704 value: str | None,
705 on_update: t.Callable[[ds.cache_control._CacheControl], None] | None = None,
706) -> ds.RequestCacheControl: ...
709@t.overload
710def parse_cache_control_header(
711 value: str | None,
712 on_update: t.Callable[[ds.cache_control._CacheControl], None] | None = None,
713 cls: type[_TAnyCC] = ...,
714) -> _TAnyCC: ...
717def parse_cache_control_header(
718 value: str | None,
719 on_update: t.Callable[[ds.cache_control._CacheControl], None] | None = None,
720 cls: type[_TAnyCC] | None = None,
721) -> _TAnyCC:
722 """Parse a cache control header. The RFC differs between response and
723 request cache control, this method does not. It's your responsibility
724 to not use the wrong control statements.
726 .. versionadded:: 0.5
727 The `cls` was added. If not specified an immutable
728 :class:`~werkzeug.datastructures.RequestCacheControl` is returned.
730 :param value: a cache control header to be parsed.
731 :param on_update: an optional callable that is called every time a value
732 on the :class:`~werkzeug.datastructures.CacheControl`
733 object is changed.
734 :param cls: the class for the returned object. By default
735 :class:`~werkzeug.datastructures.RequestCacheControl` is used.
736 :return: a `cls` object.
737 """
738 if cls is None:
739 cls = t.cast("type[_TAnyCC]", ds.RequestCacheControl)
741 if not value:
742 return cls((), on_update)
744 return cls(parse_dict_header(value), on_update)
747_TAnyCSP = t.TypeVar("_TAnyCSP", bound="ds.ContentSecurityPolicy")
750@t.overload
751def parse_csp_header(
752 value: str | None,
753 on_update: t.Callable[[ds.ContentSecurityPolicy], None] | None = None,
754) -> ds.ContentSecurityPolicy: ...
757@t.overload
758def parse_csp_header(
759 value: str | None,
760 on_update: t.Callable[[ds.ContentSecurityPolicy], None] | None = None,
761 cls: type[_TAnyCSP] = ...,
762) -> _TAnyCSP: ...
765def parse_csp_header(
766 value: str | None,
767 on_update: t.Callable[[ds.ContentSecurityPolicy], None] | None = None,
768 cls: type[_TAnyCSP] | None = None,
769) -> _TAnyCSP:
770 """Parse a Content Security Policy header.
772 .. versionadded:: 1.0.0
773 Support for Content Security Policy headers was added.
775 :param value: a csp header to be parsed.
776 :param on_update: an optional callable that is called every time a value
777 on the object is changed.
778 :param cls: the class for the returned object. By default
779 :class:`~werkzeug.datastructures.ContentSecurityPolicy` is used.
780 :return: a `cls` object.
781 """
782 if cls is None:
783 cls = t.cast("type[_TAnyCSP]", ds.ContentSecurityPolicy)
785 if value is None:
786 return cls((), on_update)
788 items = []
790 for policy in value.split(";"):
791 policy = policy.strip(" \t")
793 # Ignore badly formatted policies (no space)
794 if " " in policy:
795 directive, _, value = policy.strip(" \t").partition(" ")
796 items.append((directive.strip(" \t"), value.strip(" \t")))
798 return cls(items, on_update)
801def parse_set_header(
802 value: str | None,
803 on_update: t.Callable[[ds.HeaderSet], None] | None = None,
804) -> ds.HeaderSet:
805 """Parse a set-like header and return a
806 :class:`~werkzeug.datastructures.HeaderSet` object:
808 >>> hs = parse_set_header('token, "quoted value"')
810 The return value is an object that treats the items case-insensitively
811 and keeps the order of the items:
813 >>> 'TOKEN' in hs
814 True
815 >>> hs.index('quoted value')
816 1
817 >>> hs
818 HeaderSet(['token', 'quoted value'])
820 To create a header from the :class:`HeaderSet` again, use the
821 :func:`dump_header` function.
823 :param value: a set header to be parsed.
824 :param on_update: an optional callable that is called every time a
825 value on the :class:`~werkzeug.datastructures.HeaderSet`
826 object is changed.
827 :return: a :class:`~werkzeug.datastructures.HeaderSet`
828 """
829 if not value:
830 return ds.HeaderSet(None, on_update)
831 return ds.HeaderSet(parse_list_header(value), on_update)
834def parse_if_range_header(value: str | None) -> ds.IfRange:
835 """Parses an if-range header which can be an etag or a date. Returns
836 a :class:`~werkzeug.datastructures.IfRange` object.
838 .. versionchanged:: 2.0
839 If the value represents a datetime, it is timezone-aware.
841 .. versionadded:: 0.7
842 """
843 if not value:
844 return ds.IfRange()
845 date = parse_date(value)
846 if date is not None:
847 return ds.IfRange(date=date)
848 # drop weakness information
849 return ds.IfRange(unquote_etag(value)[0])
852def parse_range_header(
853 value: str | None, make_inclusive: bool = True
854) -> ds.Range | None:
855 """Parses a range header into a :class:`~werkzeug.datastructures.Range`
856 object. If the header is missing or malformed `None` is returned.
857 `ranges` is a list of ``(start, stop)`` tuples where the ranges are
858 non-inclusive.
860 .. versionadded:: 0.7
861 """
862 if not value or "=" not in value:
863 return None
865 ranges = []
866 last_end = 0
867 units, _, rng = value.partition("=")
868 units = units.strip(" \t").lower()
870 for item in rng.split(","):
871 item = item.strip(" \t")
872 if "-" not in item:
873 return None
874 if item.startswith("-"):
875 if last_end < 0:
876 return None
877 try:
878 begin = _plain_int(item)
879 except ValueError:
880 return None
882 # -0 will parse to 0, and is an invalid suffix length.
883 if begin == 0:
884 return None
886 end = None
887 last_end = -1
888 elif "-" in item:
889 begin_str, _, end_str = item.partition("-")
890 begin_str = begin_str.strip(" \t")
891 end_str = end_str.strip(" \t")
893 try:
894 begin = _plain_int(begin_str)
895 except ValueError:
896 return None
898 if begin < last_end or last_end < 0:
899 return None
900 if end_str:
901 try:
902 end = _plain_int(end_str) + 1
903 except ValueError:
904 return None
906 if begin >= end:
907 return None
908 else:
909 end = None
910 last_end = end if end is not None else -1
911 ranges.append((begin, end))
913 return ds.Range(units, ranges)
916def parse_content_range_header(
917 value: str | None,
918 on_update: t.Callable[[ds.ContentRange], None] | None = None,
919) -> ds.ContentRange | None:
920 """Parses a range header into a
921 :class:`~werkzeug.datastructures.ContentRange` object or `None` if
922 parsing is not possible.
924 .. versionadded:: 0.7
926 :param value: a content range header to be parsed.
927 :param on_update: an optional callable that is called every time a value
928 on the :class:`~werkzeug.datastructures.ContentRange`
929 object is changed.
930 """
931 if value is None:
932 return None
933 try:
934 units, _, rangedef = (value or "").strip(" \t").partition(" ")
935 except ValueError:
936 return None
938 if "/" not in rangedef:
939 return None
940 rng, _, length_str = rangedef.partition("/")
941 if length_str == "*":
942 length = None
943 else:
944 try:
945 length = _plain_int(length_str)
946 except ValueError:
947 return None
949 if rng == "*":
950 if not is_byte_range_valid(None, None, length):
951 return None
953 return ds.ContentRange(units, None, None, length, on_update=on_update)
954 elif "-" not in rng:
955 return None
957 start_str, _, stop_str = rng.partition("-")
958 try:
959 start = _plain_int(start_str)
960 stop = _plain_int(stop_str) + 1
961 except ValueError:
962 return None
964 if is_byte_range_valid(start, stop, length):
965 return ds.ContentRange(units, start, stop, length, on_update=on_update)
967 return None
970def quote_etag(etag: str, weak: bool = False) -> str:
971 """Quote an etag.
973 :param etag: the etag to quote.
974 :param weak: set to `True` to tag it "weak".
975 """
976 if '"' in etag:
977 raise ValueError("invalid etag")
979 if weak:
980 return f'W/"{etag}"'
982 return f'"{etag}"'
985@t.overload
986def unquote_etag(etag: str) -> tuple[str, bool]: ...
987@t.overload
988def unquote_etag(etag: None) -> tuple[None, None]: ...
989def unquote_etag(
990 etag: str | None,
991) -> tuple[str, bool] | tuple[None, None]:
992 """Unquote a single etag:
994 >>> unquote_etag('W/"bar"')
995 ('bar', True)
996 >>> unquote_etag('"bar"')
997 ('bar', False)
999 :param etag: the etag identifier to unquote.
1000 :return: a ``(etag, weak)`` tuple.
1001 """
1002 if not etag:
1003 return None, None
1005 weak = False
1006 start = 0
1008 if etag.startswith(("W/", "w/")):
1009 weak = True
1010 start = 2
1012 if etag.startswith('"', start) and etag.endswith('"', start):
1013 return etag[start + 1 : -1], weak
1015 # invalid unquoted
1016 return etag[start:], weak
1019_etag_re = re.compile(
1020 r"""
1021 [ \t]* # ignore leading space
1022 ([Ww]/)? # optional weak marker
1023 (?:
1024 "([^"]*)" # quoted value
1025 |
1026 ([^" \t,]+) # invalid unquoted value, exclude syntax characters
1027 )
1028 [ \t]* # ignore trailing space
1029 (?:,|\Z) # only if followed by comma or end
1030 """,
1031 flags=re.ASCII | re.VERBOSE,
1032)
1035def parse_etags(value: str | None) -> ds.ETags:
1036 """Parse an etag header.
1038 :param value: the tag header to parse
1039 :return: an :class:`~werkzeug.datastructures.ETags` object.
1040 """
1041 if not value:
1042 return ds.ETags()
1044 if value == "*":
1045 return ds.ETags(star_tag=True)
1047 strong = []
1048 weak = []
1049 pos = 0
1051 while True:
1052 if (m := _etag_re.match(value, pos)) is None:
1053 # Skip invalid chars until the next comma.
1054 if (pos := value.find(",", pos) + 1) == 0:
1055 break
1057 continue
1059 is_weak, tag, invalid_unquoted = m.groups()
1060 pos = m.end()
1061 tag = tag or invalid_unquoted
1063 if is_weak:
1064 weak.append(tag)
1065 else:
1066 strong.append(tag)
1068 return ds.ETags(strong, weak)
1071def generate_etag(data: bytes) -> str:
1072 """Generate an etag for some data.
1074 .. versionchanged:: 2.0
1075 Use SHA-1. MD5 may not be available in some environments.
1076 """
1077 return sha1(data).hexdigest()
1080def parse_date(value: str | None) -> datetime | None:
1081 """Parse an :rfc:`2822` date into a timezone-aware
1082 :class:`datetime.datetime` object, or ``None`` if parsing fails.
1084 This is a wrapper for :func:`email.utils.parsedate_to_datetime`. It
1085 returns ``None`` if parsing fails instead of raising an exception,
1086 and always returns a timezone-aware datetime object. If the string
1087 doesn't have timezone information, it is assumed to be UTC.
1089 :param value: A string with a supported date format.
1091 .. versionchanged:: 2.0
1092 Return a timezone-aware datetime object. Use
1093 ``email.utils.parsedate_to_datetime``.
1094 """
1095 if value is None:
1096 return None
1098 try:
1099 dt = email.utils.parsedate_to_datetime(value)
1100 except (TypeError, ValueError):
1101 return None
1103 if dt.tzinfo is None:
1104 return dt.replace(tzinfo=timezone.utc)
1106 return dt
1109def http_date(
1110 timestamp: datetime | date | int | float | struct_time | None = None,
1111) -> str:
1112 """Format a datetime object or timestamp into an :rfc:`2822` date
1113 string.
1115 This is a wrapper for :func:`email.utils.format_datetime`. It
1116 assumes naive datetime objects are in UTC instead of raising an
1117 exception.
1119 :param timestamp: The datetime or timestamp to format. Defaults to
1120 the current time.
1122 .. versionchanged:: 2.0
1123 Use ``email.utils.format_datetime``. Accept ``date`` objects.
1124 """
1125 if isinstance(timestamp, date):
1126 if not isinstance(timestamp, datetime):
1127 # Assume plain date is midnight UTC.
1128 timestamp = datetime.combine(timestamp, time(), tzinfo=timezone.utc)
1129 else:
1130 # Ensure datetime is timezone-aware.
1131 timestamp = _dt_as_utc(timestamp)
1133 return email.utils.format_datetime(timestamp, usegmt=True)
1135 if isinstance(timestamp, struct_time):
1136 timestamp = mktime(timestamp)
1138 return email.utils.formatdate(timestamp, usegmt=True)
1141def parse_age(value: str | None = None) -> timedelta | None:
1142 """Parses a base-10 integer count of seconds into a timedelta.
1144 If parsing fails, the return value is `None`.
1146 :param value: a string consisting of an integer represented in base-10
1147 :return: a :class:`datetime.timedelta` object or `None`.
1148 """
1149 if not value:
1150 return None
1151 try:
1152 seconds = int(value)
1153 except ValueError:
1154 return None
1155 if seconds < 0:
1156 return None
1157 try:
1158 return timedelta(seconds=seconds)
1159 except OverflowError:
1160 return None
1163def dump_age(age: timedelta | int | None = None) -> str | None:
1164 """Formats the duration as a base-10 integer.
1166 :param age: should be an integer number of seconds,
1167 a :class:`datetime.timedelta` object, or,
1168 if the age is unknown, `None` (default).
1169 """
1170 if age is None:
1171 return None
1172 if isinstance(age, timedelta):
1173 age = int(age.total_seconds())
1174 else:
1175 age = int(age)
1177 if age < 0:
1178 raise ValueError("age cannot be negative")
1180 return str(age)
1183def is_resource_modified(
1184 environ: WSGIEnvironment,
1185 etag: str | None = None,
1186 data: bytes | None = None,
1187 last_modified: datetime | str | None = None,
1188 ignore_if_range: bool = True,
1189) -> bool:
1190 """Convenience method for conditional requests.
1192 :param environ: the WSGI environment of the request to be checked.
1193 :param etag: the etag for the response for comparison.
1194 :param data: or alternatively the data of the response to automatically
1195 generate an etag using :func:`generate_etag`.
1196 :param last_modified: an optional date of the last modification.
1197 :param ignore_if_range: If `False`, `If-Range` header will be taken into
1198 account.
1199 :return: `True` if the resource was modified, otherwise `False`.
1201 .. versionchanged:: 2.0
1202 SHA-1 is used to generate an etag value for the data. MD5 may
1203 not be available in some environments.
1205 .. versionchanged:: 1.0.0
1206 The check is run for methods other than ``GET`` and ``HEAD``.
1207 """
1208 return _sansio_http.is_resource_modified(
1209 http_range=environ.get("HTTP_RANGE"),
1210 http_if_range=environ.get("HTTP_IF_RANGE"),
1211 http_if_modified_since=environ.get("HTTP_IF_MODIFIED_SINCE"),
1212 http_if_none_match=environ.get("HTTP_IF_NONE_MATCH"),
1213 http_if_match=environ.get("HTTP_IF_MATCH"),
1214 etag=etag,
1215 data=data,
1216 last_modified=last_modified,
1217 ignore_if_range=ignore_if_range,
1218 )
1221def remove_entity_headers(
1222 headers: ds.Headers | list[tuple[str, str]],
1223 allowed: t.Iterable[str] = ("expires", "content-location"),
1224) -> None:
1225 """Remove all entity headers from a list or :class:`Headers` object. This
1226 operation works in-place. `Expires` and `Content-Location` headers are
1227 by default not removed. The reason for this is :rfc:`2616` section
1228 10.3.5 which specifies some entity headers that should be sent.
1230 .. versionchanged:: 0.5
1231 added `allowed` parameter.
1233 :param headers: a list or :class:`Headers` object.
1234 :param allowed: a list of headers that should still be allowed even though
1235 they are entity headers.
1236 """
1237 allowed = {x.lower() for x in allowed}
1238 headers[:] = [
1239 (key, value)
1240 for key, value in headers
1241 if not is_entity_header(key) or key.lower() in allowed
1242 ]
1245def remove_hop_by_hop_headers(headers: ds.Headers | list[tuple[str, str]]) -> None:
1246 """Remove all HTTP/1.1 "Hop-by-Hop" headers from a list or
1247 :class:`Headers` object. This operation works in-place.
1249 .. versionadded:: 0.5
1251 :param headers: a list or :class:`Headers` object.
1252 """
1253 headers[:] = [
1254 (key, value) for key, value in headers if not is_hop_by_hop_header(key)
1255 ]
1258def is_entity_header(header: str) -> bool:
1259 """Check if a header is an entity header.
1261 .. versionadded:: 0.5
1263 :param header: the header to test.
1264 :return: `True` if it's an entity header, `False` otherwise.
1265 """
1266 return header.lower() in _entity_headers
1269def is_hop_by_hop_header(header: str) -> bool:
1270 """Check if a header is an HTTP/1.1 "Hop-by-Hop" header.
1272 .. versionadded:: 0.5
1274 :param header: the header to test.
1275 :return: `True` if it's an HTTP/1.1 "Hop-by-Hop" header, `False` otherwise.
1276 """
1277 return header.lower() in _hop_by_hop_headers
1280def parse_cookie(
1281 header: WSGIEnvironment | str | None,
1282 cls: type[ds.MultiDict[str, str]] | None = None,
1283) -> ds.MultiDict[str, str]:
1284 """Parse a cookie from a string or WSGI environ.
1286 The same key can be provided multiple times, the values are stored
1287 in-order. The default :class:`MultiDict` will have the first value
1288 first, and all values can be retrieved with
1289 :meth:`MultiDict.getlist`.
1291 :param header: The cookie header as a string, or a WSGI environ dict
1292 with a ``HTTP_COOKIE`` key.
1293 :param cls: A dict-like class to store the parsed cookies in.
1294 Defaults to :class:`MultiDict`.
1296 .. versionchanged:: 3.0
1297 Passing bytes, and the ``charset`` and ``errors`` parameters, were removed.
1299 .. versionchanged:: 1.0
1300 Returns a :class:`MultiDict` instead of a ``TypeConversionDict``.
1302 .. versionchanged:: 0.5
1303 Returns a :class:`TypeConversionDict` instead of a regular dict. The ``cls``
1304 parameter was added.
1305 """
1306 if isinstance(header, dict):
1307 cookie = header.get("HTTP_COOKIE")
1308 else:
1309 cookie = header
1311 if cookie:
1312 cookie = cookie.encode("latin1").decode()
1314 return _sansio_http.parse_cookie(cookie=cookie, cls=cls)
1317_cookie_no_quote_re = re.compile(r"[\w!#$%&'()*+\-./:<=>?@\[\]^`{|}~]*", re.A)
1318_cookie_slash_re = re.compile(rb"[\x00-\x19\",;\\\x7f-\xff]", re.A)
1319_cookie_slash_map = {b'"': b'\\"', b"\\": b"\\\\"}
1320_cookie_slash_map.update(
1321 (v.to_bytes(1, "big"), b"\\%03o" % v)
1322 for v in [*range(0x20), *b",;", *range(0x7F, 256)]
1323)
1326def dump_cookie(
1327 key: str,
1328 value: str = "",
1329 max_age: timedelta | int | None = None,
1330 expires: str | datetime | int | float | None = None,
1331 path: str | None = "/",
1332 domain: str | None = None,
1333 secure: bool = False,
1334 httponly: bool = False,
1335 sync_expires: bool = True,
1336 max_size: int = 4093,
1337 samesite: str | None = None,
1338 partitioned: bool = False,
1339) -> str:
1340 """Create a Set-Cookie header without the ``Set-Cookie`` prefix.
1342 The return value is usually restricted to ascii as the vast majority
1343 of values are properly escaped, but that is no guarantee. It's
1344 tunneled through latin1 as required by :pep:`3333`.
1346 The return value is not ASCII safe if the key contains unicode
1347 characters. This is technically against the specification but
1348 happens in the wild. It's strongly recommended to not use
1349 non-ASCII values for the keys.
1351 :param key: The cookie key. This is assumed to be trusted and valid, it
1352 must not come from untrusted user input.
1353 :param value: The cookie value. It will be quoted if it contains characters
1354 not allowed by RFC 6265; ``\123`` octal escapes are used to encode
1355 non-ASCII bytes.
1356 :param max_age: should be a number of seconds, or `None` (default) if
1357 the cookie should last only as long as the client's
1358 browser session. Additionally `timedelta` objects
1359 are accepted, too.
1360 :param expires: should be a `datetime` object or unix timestamp.
1361 :param path: limits the cookie to a given path, per default it will
1362 span the whole domain.
1363 :param domain: Use this if you want to set a cross-domain cookie. For
1364 example, ``domain="example.com"`` will set a cookie
1365 that is readable by the domain ``www.example.com``,
1366 ``foo.example.com`` etc. Otherwise, a cookie will only
1367 be readable by the domain that set it.
1368 :param secure: The cookie will only be available via HTTPS
1369 :param httponly: disallow JavaScript to access the cookie. This is an
1370 extension to the cookie standard and probably not
1371 supported by all browsers.
1372 :param charset: the encoding for string values.
1373 :param sync_expires: automatically set expires if max_age is defined
1374 but expires not.
1375 :param max_size: Warn if the final header value exceeds this size. The
1376 default, 4093, should be safely `supported by most browsers
1377 <cookie_>`_. Set to 0 to disable this check.
1378 :param samesite: Limits the scope of the cookie such that it will
1379 only be attached to requests if those requests are same-site.
1380 :param partitioned: Opts the cookie into partitioned storage. This
1381 will also set secure to True
1383 .. _`cookie`: http://browsercookielimits.squawky.net/
1385 .. versionchanged:: 3.1
1386 The ``partitioned`` parameter was added.
1388 .. versionchanged:: 3.0
1389 Passing bytes, and the ``charset`` parameter, were removed.
1391 .. versionchanged:: 2.3.3
1392 The ``path`` parameter is ``/`` by default.
1394 .. versionchanged:: 2.3.1
1395 The value allows more characters without quoting.
1397 .. versionchanged:: 2.3
1398 ``localhost`` and other names without a dot are allowed for the domain. A
1399 leading dot is ignored.
1401 .. versionchanged:: 2.3
1402 The ``path`` parameter is ``None`` by default.
1404 .. versionchanged:: 1.0.0
1405 The string ``'None'`` is accepted for ``samesite``.
1406 """
1407 if path is not None:
1408 # safe = https://url.spec.whatwg.org/#url-path-segment-string
1409 # as well as percent for things that are already quoted
1410 # excluding semicolon since it's part of the header syntax
1411 path = quote(path, safe="%!$&'()*+,/:=@")
1413 if domain:
1414 domain = domain.partition(":")[0].lstrip(".").encode("idna").decode("ascii")
1416 if isinstance(max_age, timedelta):
1417 max_age = int(max_age.total_seconds())
1419 if expires is not None:
1420 if not isinstance(expires, str):
1421 expires = http_date(expires)
1422 elif max_age is not None and sync_expires:
1423 expires = http_date(datetime.now(tz=timezone.utc).timestamp() + max_age)
1425 if samesite is not None:
1426 samesite = samesite.title()
1428 if samesite not in {"Strict", "Lax", "None"}:
1429 raise ValueError("SameSite must be 'Strict', 'Lax', or 'None'.")
1431 if partitioned:
1432 secure = True
1434 # Quote value if it contains characters not allowed by RFC 6265. Slash-escape with
1435 # three octal digits, which matches http.cookies, although the RFC suggests base64.
1436 if not _cookie_no_quote_re.fullmatch(value):
1437 # Work with bytes here, since a UTF-8 character could be multiple bytes.
1438 value = _cookie_slash_re.sub(
1439 lambda m: _cookie_slash_map[m.group()], value.encode()
1440 ).decode("ascii")
1441 value = f'"{value}"'
1443 # Send a non-ASCII key as mojibake. Everything else should already be ASCII.
1444 # TODO Remove encoding dance, it seems like clients accept UTF-8 keys
1445 buf = [f"{key.encode().decode('latin1')}={value}"]
1447 for k, v in (
1448 ("Domain", domain),
1449 ("Expires", expires),
1450 ("Max-Age", max_age),
1451 ("Secure", secure),
1452 ("HttpOnly", httponly),
1453 ("Path", path),
1454 ("SameSite", samesite),
1455 ("Partitioned", partitioned),
1456 ):
1457 if v is None or v is False:
1458 continue
1460 if v is True:
1461 buf.append(k)
1462 continue
1464 buf.append(f"{k}={v}")
1466 rv = "; ".join(buf)
1468 # Warn if the final value of the cookie is larger than the limit. If the cookie is
1469 # too large, then it may be silently ignored by the browser, which can be quite hard
1470 # to debug.
1471 cookie_size = len(rv)
1473 if max_size and cookie_size > max_size:
1474 value_size = len(value)
1475 warnings.warn(
1476 f"The '{key}' cookie is too large: the value was {value_size} bytes but the"
1477 f" header required {cookie_size - value_size} extra bytes. The final size"
1478 f" was {cookie_size} bytes but the limit is {max_size} bytes. Browsers may"
1479 " silently ignore cookies larger than this.",
1480 stacklevel=2,
1481 )
1483 return rv
1486def is_byte_range_valid(
1487 start: int | None, stop: int | None, length: int | None
1488) -> bool:
1489 """Checks if a given byte content range is valid for the given length.
1491 .. versionadded:: 0.7
1492 """
1493 if (start is None) != (stop is None):
1494 return False
1495 elif start is None:
1496 return length is None or length >= 0
1497 elif length is None:
1498 return 0 <= start < stop # type: ignore
1499 elif start >= stop: # type: ignore
1500 return False
1501 return 0 <= start < length
1504# circular dependencies
1505from . import datastructures as ds # noqa: E402
1506from .sansio import http as _sansio_http # noqa: E402