Coverage for /pythoncovmergedfiles/medio/medio/usr/local/lib/python3.11/site-packages/urllib3/response.py: 20%
Shortcuts on this page
r m x toggle line displays
j k next/prev highlighted chunk
0 (zero) top of page
1 (one) first highlighted chunk
Shortcuts on this page
r m x toggle line displays
j k next/prev highlighted chunk
0 (zero) top of page
1 (one) first highlighted chunk
1from __future__ import annotations
3import collections
4import io
5import json as _json
6import logging
7import socket
8import sys
9import typing
10import warnings
11import zlib
12from contextlib import contextmanager
13from http.client import HTTPMessage as _HttplibHTTPMessage
14from http.client import HTTPResponse as _HttplibHTTPResponse
15from socket import timeout as SocketTimeout
17if typing.TYPE_CHECKING:
18 from ._base_connection import BaseHTTPConnection
20try:
21 try:
22 import brotlicffi as brotli # type: ignore[import-not-found]
23 except ImportError:
24 import brotli # type: ignore[import-not-found]
25except ImportError:
26 brotli = None
28from . import util
29from ._base_connection import _TYPE_BODY
30from ._collections import HTTPHeaderDict
31from .connection import BaseSSLError, HTTPConnection, HTTPException
32from .exceptions import (
33 BodyNotHttplibCompatible,
34 DecodeError,
35 DependencyWarning,
36 HTTPError,
37 IncompleteRead,
38 InvalidChunkLength,
39 InvalidHeader,
40 ProtocolError,
41 ReadTimeoutError,
42 ResponseNotChunked,
43 SSLError,
44)
45from .util.response import is_fp_closed, is_response_to_head
46from .util.retry import Retry
48if typing.TYPE_CHECKING:
49 from .connectionpool import HTTPConnectionPool
51log = logging.getLogger(__name__)
53# Read in 64 KiB chunks
54_READ_CHUNK_SIZE = 2**16
56# Maximum length of a chunk-size line (and the trailing CRLF line) when reading a
57# chunked-transfer-encoded response. Matches ``http.client._MAXLINE`` so urllib3's
58# streaming path bounds these reads exactly like the stdlib ``.read()`` path does,
59# preventing a malicious server from forcing unbounded buffering (memory exhaustion)
60# via an unterminated chunk-size line.
61_MAX_CHUNK_LINE_LENGTH = 2**16
64class ContentDecoder:
65 def decompress(self, data: bytes, max_length: int = -1) -> bytes:
66 raise NotImplementedError()
68 @property
69 def has_unconsumed_tail(self) -> bool:
70 raise NotImplementedError()
72 def flush(self) -> bytes:
73 raise NotImplementedError()
76class DeflateDecoder(ContentDecoder):
77 def __init__(self) -> None:
78 self._first_try = True
79 self._first_try_data = b""
80 self._unfed_data = b""
81 self._obj = zlib.decompressobj()
83 def decompress(self, data: bytes, max_length: int = -1) -> bytes:
84 data = self._unfed_data + data
85 self._unfed_data = b""
86 # Data received after EOF cannot produce more output from this
87 # Deflate stream.
88 if self._obj.eof:
89 return b""
90 if not data and not self._obj.unconsumed_tail:
91 return data
92 original_max_length = max_length
93 if original_max_length < 0:
94 max_length = 0
95 elif original_max_length == 0:
96 # We should not pass 0 to the zlib decompressor because 0 is
97 # the default value that will make zlib decompress without a
98 # length limit.
99 # Data should be stored for subsequent calls.
100 self._unfed_data = data
101 return b""
103 # Subsequent calls always reuse `self._obj`. zlib requires
104 # passing the unconsumed tail if decompression is to continue.
105 if not self._first_try:
106 return self._obj.decompress(
107 self._obj.unconsumed_tail + data, max_length=max_length
108 )
110 # First call tries with RFC 1950 ZLIB format.
111 self._first_try_data += data
112 try:
113 decompressed = self._obj.decompress(data, max_length=max_length)
114 if decompressed:
115 self._first_try = False
116 self._first_try_data = b""
117 return decompressed
118 # On failure, it falls back to RFC 1951 DEFLATE format.
119 except zlib.error:
120 self._first_try = False
121 self._obj = zlib.decompressobj(-zlib.MAX_WBITS)
122 try:
123 return self.decompress(
124 self._first_try_data, max_length=original_max_length
125 )
126 finally:
127 self._first_try_data = b""
129 @property
130 def has_unconsumed_tail(self) -> bool:
131 return bool(self._unfed_data) or (
132 bool(self._obj.unconsumed_tail)
133 and not self._first_try
134 and not self._obj.eof
135 )
137 def flush(self) -> bytes:
138 return self._obj.flush()
141class GzipDecoderState:
142 FIRST_MEMBER = 0
143 OTHER_MEMBERS = 1
144 SWALLOW_DATA = 2
147class GzipDecoder(ContentDecoder):
148 def __init__(self) -> None:
149 self._obj = zlib.decompressobj(16 + zlib.MAX_WBITS)
150 self._state = GzipDecoderState.FIRST_MEMBER
151 self._unconsumed_tail = b""
153 def decompress(self, data: bytes, max_length: int = -1) -> bytes:
154 ret = bytearray()
155 if self._state == GzipDecoderState.SWALLOW_DATA:
156 return bytes(ret)
158 if max_length == 0:
159 # We should not pass 0 to the zlib decompressor because 0 is
160 # the default value that will make zlib decompress without a
161 # length limit.
162 # Data should be stored for subsequent calls.
163 self._unconsumed_tail += data
164 return b""
166 # zlib requires passing the unconsumed tail to the subsequent
167 # call if decompression is to continue.
168 data = self._unconsumed_tail + data
169 if not data and self._obj.eof:
170 return bytes(ret)
172 while True:
173 try:
174 ret += self._obj.decompress(
175 data, max_length=max(max_length - len(ret), 0)
176 )
177 except zlib.error:
178 previous_state = self._state
179 # Ignore data after the first error
180 self._state = GzipDecoderState.SWALLOW_DATA
181 self._unconsumed_tail = b""
182 if previous_state == GzipDecoderState.OTHER_MEMBERS:
183 # Allow trailing garbage acceptable in other gzip clients
184 return bytes(ret)
185 raise
187 self._unconsumed_tail = data = (
188 self._obj.unconsumed_tail or self._obj.unused_data
189 )
190 if max_length > 0 and len(ret) >= max_length:
191 break
193 if not data:
194 return bytes(ret)
195 # When the end of a gzip member is reached, a new decompressor
196 # must be created for unused (possibly future) data.
197 if self._obj.eof:
198 self._state = GzipDecoderState.OTHER_MEMBERS
199 self._obj = zlib.decompressobj(16 + zlib.MAX_WBITS)
201 return bytes(ret)
203 @property
204 def has_unconsumed_tail(self) -> bool:
205 return bool(self._unconsumed_tail)
207 def flush(self) -> bytes:
208 return self._obj.flush()
211if brotli is not None:
213 class BrotliDecoder(ContentDecoder):
214 # Supports both 'brotlipy' and 'Brotli' packages
215 # since they share an import name. The top branches
216 # are for 'brotlipy' and bottom branches for 'Brotli'
217 def __init__(self) -> None:
218 self._obj = brotli.Decompressor()
219 if hasattr(self._obj, "decompress"):
220 setattr(self, "_decompress", self._obj.decompress)
221 else:
222 setattr(self, "_decompress", self._obj.process)
224 # Requires Brotli >= 1.2.0 for `output_buffer_limit`.
225 def _decompress(self, data: bytes, output_buffer_limit: int = -1) -> bytes:
226 raise NotImplementedError()
228 def decompress(self, data: bytes, max_length: int = -1) -> bytes:
229 try:
230 if max_length > 0:
231 return self._decompress(data, output_buffer_limit=max_length)
232 else:
233 return self._decompress(data)
234 except TypeError:
235 # Fallback for Brotli/brotlicffi/brotlipy versions without
236 # the `output_buffer_limit` parameter.
237 warnings.warn(
238 "Brotli >= 1.2.0 is required to prevent decompression bombs.",
239 DependencyWarning,
240 )
241 return self._decompress(data)
243 @property
244 def has_unconsumed_tail(self) -> bool:
245 try:
246 return not self._obj.can_accept_more_data()
247 except AttributeError:
248 return False
250 def flush(self) -> bytes:
251 if hasattr(self._obj, "flush"):
252 return self._obj.flush() # type: ignore[no-any-return]
253 return b""
256try:
257 if sys.version_info >= (3, 14):
258 from compression import zstd
259 else:
260 from backports import zstd
261except ImportError:
262 HAS_ZSTD = False
263else:
264 HAS_ZSTD = True
266 class ZstdDecoder(ContentDecoder):
267 def __init__(self) -> None:
268 self._obj = zstd.ZstdDecompressor()
270 def decompress(self, data: bytes, max_length: int = -1) -> bytes:
271 if not data and not self.has_unconsumed_tail:
272 return b""
273 if self._obj.eof:
274 data = self._obj.unused_data + data
275 self._obj = zstd.ZstdDecompressor()
276 part = self._obj.decompress(data, max_length=max_length)
277 length = len(part)
278 data_parts = [part]
279 # Every loop iteration is supposed to read data from a separate frame.
280 # The loop breaks when:
281 # - enough data is read;
282 # - no more unused data is available;
283 # - end of the last read frame has not been reached (i.e.,
284 # more data has to be fed).
285 while (
286 self._obj.eof
287 and self._obj.unused_data
288 and (max_length < 0 or length < max_length)
289 ):
290 unused_data = self._obj.unused_data
291 if not self._obj.needs_input:
292 self._obj = zstd.ZstdDecompressor()
293 part = self._obj.decompress(
294 unused_data,
295 max_length=(max_length - length) if max_length > 0 else -1,
296 )
297 if part_length := len(part):
298 data_parts.append(part)
299 length += part_length
300 elif self._obj.needs_input:
301 break
302 return b"".join(data_parts)
304 @property
305 def has_unconsumed_tail(self) -> bool:
306 return not (self._obj.needs_input or self._obj.eof) or bool(
307 self._obj.unused_data
308 )
310 def flush(self) -> bytes:
311 if not self._obj.eof:
312 raise DecodeError("Zstandard data is incomplete")
313 return b""
316class MultiDecoder(ContentDecoder):
317 """
318 From RFC7231:
319 If one or more encodings have been applied to a representation, the
320 sender that applied the encodings MUST generate a Content-Encoding
321 header field that lists the content codings in the order in which
322 they were applied.
323 """
325 # Maximum allowed number of chained HTTP encodings in the
326 # Content-Encoding header.
327 max_decode_links = 5
329 def __init__(self, modes: str) -> None:
330 encodings = [m.strip() for m in modes.split(",")]
331 if len(encodings) > self.max_decode_links:
332 raise DecodeError(
333 "Too many content encodings in the chain: "
334 f"{len(encodings)} > {self.max_decode_links}"
335 )
336 self._decoders = [_get_decoder(e) for e in encodings]
338 def flush(self) -> bytes:
339 return self._decoders[0].flush()
341 def decompress(self, data: bytes, max_length: int = -1) -> bytes:
342 if max_length <= 0:
343 for d in reversed(self._decoders):
344 data = d.decompress(data)
345 return data
347 ret = bytearray()
348 # Every while loop iteration goes through all decoders once.
349 # It exits when enough data is read or no more data can be read.
350 # It is possible that the while loop iteration does not produce
351 # any data because we retrieve up to `max_length` from every
352 # decoder, and the amount of bytes may be insufficient for the
353 # next decoder to produce enough/any output.
354 while True:
355 any_data = False
356 for d in reversed(self._decoders):
357 data = d.decompress(data, max_length=max_length - len(ret))
358 if data:
359 any_data = True
360 # We should not break when no data is returned because
361 # next decoders may produce data even with empty input.
362 ret += data
363 if not any_data or len(ret) >= max_length:
364 return bytes(ret)
365 data = b""
367 @property
368 def has_unconsumed_tail(self) -> bool:
369 return any(d.has_unconsumed_tail for d in self._decoders)
372def _get_decoder(mode: str) -> ContentDecoder:
373 if "," in mode:
374 return MultiDecoder(mode)
376 # According to RFC 9110 section 8.4.1.3, recipients should
377 # consider x-gzip equivalent to gzip
378 if mode in ("gzip", "x-gzip"):
379 return GzipDecoder()
381 if brotli is not None and mode == "br":
382 return BrotliDecoder()
384 if HAS_ZSTD and mode == "zstd":
385 return ZstdDecoder()
387 return DeflateDecoder()
390class BytesQueueBuffer:
391 """Memory-efficient bytes buffer
393 To return decoded data in read() and still follow the BufferedIOBase API, we need a
394 buffer to always return the correct amount of bytes.
396 This buffer should be filled using calls to put()
398 Our maximum memory usage is determined by the sum of the size of:
400 * self.buffer, which contains the full data
401 * the largest chunk that we will copy in get()
402 """
404 def __init__(self) -> None:
405 self.buffer: typing.Deque[bytes | memoryview[bytes]] = collections.deque()
406 self._size: int = 0
408 def __len__(self) -> int:
409 return self._size
411 def put(self, data: bytes) -> None:
412 self.buffer.append(data)
413 self._size += len(data)
415 def get(self, n: int) -> bytes:
416 if n == 0:
417 return b""
418 elif not self.buffer:
419 raise RuntimeError("buffer is empty")
420 elif n < 0:
421 raise ValueError("n should be > 0")
423 if len(self.buffer[0]) == n and isinstance(self.buffer[0], bytes):
424 self._size -= n
425 return self.buffer.popleft()
427 fetched = 0
428 ret = io.BytesIO()
429 while fetched < n:
430 remaining = n - fetched
431 chunk = self.buffer.popleft()
432 chunk_length = len(chunk)
433 if remaining < chunk_length:
434 chunk = memoryview(chunk)
435 left_chunk, right_chunk = chunk[:remaining], chunk[remaining:]
436 ret.write(left_chunk)
437 self.buffer.appendleft(right_chunk)
438 self._size -= remaining
439 break
440 else:
441 ret.write(chunk)
442 self._size -= chunk_length
443 fetched += chunk_length
445 if not self.buffer:
446 break
448 return ret.getvalue()
450 def get_all(self) -> bytes:
451 buffer = self.buffer
452 if not buffer:
453 assert self._size == 0
454 return b""
455 if len(buffer) == 1:
456 result = buffer.pop()
457 if isinstance(result, memoryview):
458 result = result.tobytes()
459 else:
460 ret = io.BytesIO()
461 ret.writelines(buffer.popleft() for _ in range(len(buffer)))
462 result = ret.getvalue()
463 self._size = 0
464 return result
467class BaseHTTPResponse(io.IOBase):
468 CONTENT_DECODERS = ["gzip", "x-gzip", "deflate"]
469 if brotli is not None:
470 CONTENT_DECODERS += ["br"]
471 if HAS_ZSTD:
472 CONTENT_DECODERS += ["zstd"]
473 REDIRECT_STATUSES = [301, 302, 303, 307, 308]
475 DECODER_ERROR_CLASSES: tuple[type[Exception], ...] = (IOError, zlib.error)
476 if brotli is not None:
477 DECODER_ERROR_CLASSES += (brotli.error,)
479 if HAS_ZSTD:
480 DECODER_ERROR_CLASSES += (zstd.ZstdError,)
482 def __init__(
483 self,
484 *,
485 headers: typing.Mapping[str, str] | typing.Mapping[bytes, bytes] | None = None,
486 status: int,
487 version: int,
488 version_string: str,
489 reason: str | None,
490 decode_content: bool,
491 request_url: str | None,
492 retries: Retry | None = None,
493 ) -> None:
494 if isinstance(headers, HTTPHeaderDict):
495 self.headers = headers
496 else:
497 self.headers = HTTPHeaderDict(headers) # type: ignore[arg-type]
498 self.status = status
499 self.version = version
500 self.version_string = version_string
501 self.reason = reason
502 self.decode_content = decode_content
503 self._has_decoded_content = False
504 self._request_url: str | None = request_url
505 self.retries = retries
507 self.chunked = False
508 tr_enc = self.headers.get("transfer-encoding", "").lower()
509 # Don't incur the penalty of creating a list and then discarding it
510 encodings = (enc.strip() for enc in tr_enc.split(","))
511 if "chunked" in encodings:
512 self.chunked = True
514 self._decoder: ContentDecoder | None = None
515 # Distinguish an uninitialized decoder from a response needing no decoder.
516 self._decoder_initialized = False
517 self.length_remaining: int | None
519 def get_redirect_location(self) -> str | None | typing.Literal[False]:
520 """
521 Should we redirect and where to?
523 :returns: Truthy redirect location string if we got a redirect status
524 code and valid location. ``None`` if redirect status and no
525 location. ``False`` if not a redirect status code.
526 """
527 if self.status in self.REDIRECT_STATUSES:
528 return self.headers.get("location")
529 return False
531 @property
532 def data(self) -> bytes:
533 raise NotImplementedError()
535 def json(self) -> typing.Any:
536 """
537 Deserializes the body of the HTTP response as a Python object.
539 The body of the HTTP response must be encoded using UTF-8, as per
540 `RFC 8529 Section 8.1 <https://www.rfc-editor.org/rfc/rfc8259#section-8.1>`_.
542 To use a custom JSON decoder pass the result of :attr:`HTTPResponse.data` to
543 your custom decoder instead.
545 If the body of the HTTP response is not decodable to UTF-8, a
546 `UnicodeDecodeError` will be raised. If the body of the HTTP response is not a
547 valid JSON document, a `json.JSONDecodeError` will be raised.
549 Read more :ref:`here <json_content>`.
551 :returns: The body of the HTTP response as a Python object.
552 """
553 data = self.data.decode("utf-8")
554 return _json.loads(data)
556 @property
557 def url(self) -> str | None:
558 raise NotImplementedError()
560 @url.setter
561 def url(self, url: str | None) -> None:
562 raise NotImplementedError()
564 @property
565 def connection(self) -> BaseHTTPConnection | None:
566 raise NotImplementedError()
568 @property
569 def retries(self) -> Retry | None:
570 return self._retries
572 @retries.setter
573 def retries(self, retries: Retry | None) -> None:
574 # Override the request_url if retries has a redirect location.
575 if retries is not None and retries.history:
576 self.url = retries.history[-1].redirect_location
577 self._retries = retries
579 def stream(
580 self, amt: int | None = _READ_CHUNK_SIZE, decode_content: bool | None = None
581 ) -> typing.Iterator[bytes]:
582 raise NotImplementedError()
584 def read(
585 self,
586 amt: int | None = None,
587 decode_content: bool | None = None,
588 cache_content: bool = False,
589 ) -> bytes:
590 raise NotImplementedError()
592 def read1(
593 self,
594 amt: int | None = None,
595 decode_content: bool | None = None,
596 ) -> bytes:
597 raise NotImplementedError()
599 def read_chunked(
600 self,
601 amt: int | None = None,
602 decode_content: bool | None = None,
603 ) -> typing.Iterator[bytes]:
604 raise NotImplementedError()
606 def release_conn(self) -> None:
607 raise NotImplementedError()
609 def drain_conn(self) -> None:
610 raise NotImplementedError()
612 def shutdown(self) -> None:
613 raise NotImplementedError()
615 def close(self) -> None:
616 raise NotImplementedError()
618 def _init_decoder(self) -> None:
619 """
620 Set-up the _decoder attribute if necessary.
621 """
622 if self._decoder_initialized:
623 return
625 if self._decoder is None:
626 # Note: content-encoding value should be case-insensitive, per RFC 7230
627 # Section 3.2
628 content_encoding = self.headers.get("content-encoding", "").lower()
629 if content_encoding in self.CONTENT_DECODERS:
630 self._decoder = _get_decoder(content_encoding)
631 elif "," in content_encoding:
632 encodings = [
633 e.strip()
634 for e in content_encoding.split(",")
635 if e.strip() in self.CONTENT_DECODERS
636 ]
637 if encodings:
638 self._decoder = _get_decoder(content_encoding)
640 self._decoder_initialized = True
642 def _decode(
643 self,
644 data: bytes,
645 decode_content: bool | None,
646 flush_decoder: bool,
647 max_length: int | None = None,
648 ) -> bytes:
649 """
650 Decode the data passed in and potentially flush the decoder.
651 """
652 if not decode_content:
653 if self._has_decoded_content:
654 raise RuntimeError(
655 "Calling read(decode_content=False) is not supported after "
656 "read(decode_content=True) was called."
657 )
658 return data
660 if max_length is None or flush_decoder:
661 max_length = -1
663 try:
664 if self._decoder:
665 data = self._decoder.decompress(data, max_length=max_length)
666 self._has_decoded_content = True
667 except self.DECODER_ERROR_CLASSES as e:
668 content_encoding = self.headers.get("content-encoding", "").lower()
669 raise DecodeError(
670 "Received response with content-encoding: %s, but "
671 "failed to decode it." % content_encoding,
672 e,
673 ) from e
674 if flush_decoder:
675 data += self._flush_decoder()
677 return data
679 def _flush_decoder(self) -> bytes:
680 """
681 Flushes the decoder. Should only be called if the decoder is actually
682 being used.
683 """
684 if self._decoder:
685 return self._decoder.decompress(b"") + self._decoder.flush()
686 return b""
688 # Compatibility methods for `io` module
689 def readinto(self, b: bytearray | memoryview[int]) -> int:
690 temp = self.read(len(b))
691 if len(temp) == 0:
692 return 0
693 else:
694 b[: len(temp)] = temp
695 return len(temp)
697 # Methods used by dependent libraries
698 def getheaders(self) -> HTTPHeaderDict:
699 return self.headers
701 def getheader(self, name: str, default: str | None = None) -> str | None:
702 return self.headers.get(name, default)
704 # Compatibility method for http.cookiejar
705 def info(self) -> HTTPHeaderDict:
706 return self.headers
708 def geturl(self) -> str | None:
709 return self.url
712class HTTPResponse(BaseHTTPResponse):
713 """
714 HTTP Response container.
716 Backwards-compatible with :class:`http.client.HTTPResponse` but the response ``body`` is
717 loaded and decoded on-demand when the ``data`` property is accessed. This
718 class is also compatible with the Python standard library's :mod:`io`
719 module, and can hence be treated as a readable object in the context of that
720 framework.
722 Extra parameters for behaviour not present in :class:`http.client.HTTPResponse`:
724 :param preload_content:
725 If True, the response's body will be preloaded during construction.
727 :param decode_content:
728 If True, will attempt to decode the body based on the
729 'content-encoding' header.
731 :param original_response:
732 When this HTTPResponse wrapper is generated from an :class:`http.client.HTTPResponse`
733 object, it's convenient to include the original for debug purposes. It's
734 otherwise unused.
736 :param retries:
737 The retries contains the last :class:`~urllib3.util.retry.Retry` that
738 was used during the request.
740 :param enforce_content_length:
741 Enforce content length checking. Body returned by server must match
742 value of Content-Length header, if present. Otherwise, raise error.
743 """
745 def __init__(
746 self,
747 body: _TYPE_BODY = "",
748 headers: typing.Mapping[str, str] | typing.Mapping[bytes, bytes] | None = None,
749 status: int = 0,
750 version: int = 0,
751 version_string: str = "HTTP/?",
752 reason: str | None = None,
753 preload_content: bool = True,
754 decode_content: bool = True,
755 original_response: _HttplibHTTPResponse | None = None,
756 pool: HTTPConnectionPool | None = None,
757 connection: HTTPConnection | None = None,
758 msg: _HttplibHTTPMessage | None = None,
759 retries: Retry | None = None,
760 enforce_content_length: bool = True,
761 request_method: str | None = None,
762 request_url: str | None = None,
763 auto_close: bool = True,
764 sock_shutdown: typing.Callable[[int], None] | None = None,
765 ) -> None:
766 super().__init__(
767 headers=headers,
768 status=status,
769 version=version,
770 version_string=version_string,
771 reason=reason,
772 decode_content=decode_content,
773 request_url=request_url,
774 retries=retries,
775 )
777 self.enforce_content_length = enforce_content_length
778 self.auto_close = auto_close
780 self._body = None
781 self._uncached_read_occurred = False
782 self._fp: _HttplibHTTPResponse | None = None
783 self._original_response = original_response
784 self._fp_bytes_read = 0
785 self.msg = msg
787 if body and isinstance(body, (str, bytes)):
788 self._body = body
790 self._pool = pool
791 self._connection = connection
793 if hasattr(body, "read"):
794 self._fp = body # type: ignore[assignment]
795 self._sock_shutdown = sock_shutdown
797 # Are we using the chunked-style of transfer encoding?
798 self.chunk_left: int | None = None
800 # Determine length of response
801 self.length_remaining = self._init_length(request_method)
803 # Used to return the correct amount of bytes for partial read()s
804 self._decoded_buffer = BytesQueueBuffer()
806 # If requested, preload the body.
807 if preload_content and not self._body:
808 self._body = self.read(decode_content=decode_content)
810 def release_conn(self) -> None:
811 if not self._pool or not self._connection:
812 return None
814 self._pool._put_conn(self._connection)
815 self._connection = None
817 def drain_conn(self) -> None:
818 """
819 Read and discard any remaining HTTP response data in the response connection.
821 Unread data in the HTTPResponse connection blocks the connection from being released back to the pool.
822 """
823 try:
824 while self._raw_read(_READ_CHUNK_SIZE):
825 pass
826 except (HTTPError, OSError, BaseSSLError, HTTPException):
827 pass
828 if self._has_decoded_content:
829 # `_raw_read` skips decompression, so we should clean up the
830 # decoder to avoid keeping unnecessary data in memory.
831 self._decoded_buffer = BytesQueueBuffer()
832 self._decoder = None
834 @property
835 def data(self) -> bytes:
836 # For backwards-compat with earlier urllib3 0.4 and earlier.
837 if self._body:
838 return self._body # type: ignore[return-value]
840 if self._fp:
841 return self.read(cache_content=True)
843 return None # type: ignore[return-value]
845 @property
846 def connection(self) -> HTTPConnection | None:
847 return self._connection
849 def isclosed(self) -> bool:
850 return is_fp_closed(self._fp)
852 def tell(self) -> int:
853 """
854 Obtain the number of bytes pulled over the wire so far. May differ from
855 the amount of content returned by :meth:`HTTPResponse.read`
856 if bytes are encoded on the wire (e.g, compressed).
857 """
858 return self._fp_bytes_read
860 def _init_length(self, request_method: str | None) -> int | None:
861 """
862 Set initial length value for Response content if available.
863 """
864 length: int | None
865 content_length: str | None = self.headers.get("content-length")
867 if content_length is not None:
868 if self.chunked:
869 # This Response will fail with an IncompleteRead if it can't be
870 # received as chunked. This method falls back to attempt reading
871 # the response before raising an exception.
872 log.warning(
873 "Received response with both Content-Length and "
874 "Transfer-Encoding set. This is expressly forbidden "
875 "by RFC 7230 sec 3.3.2. Ignoring Content-Length and "
876 "attempting to process response as Transfer-Encoding: "
877 "chunked."
878 )
879 return None
881 try:
882 # RFC 7230 section 3.3.2 specifies multiple content lengths can
883 # be sent in a single Content-Length header
884 # (e.g. Content-Length: 42, 42). This line ensures the values
885 # are all valid ints and that as long as the `set` length is 1,
886 # all values are the same. Otherwise, the header is invalid.
887 lengths = {int(val) for val in content_length.split(",")}
888 if len(lengths) > 1:
889 raise InvalidHeader(
890 "Content-Length contained multiple "
891 "unmatching values (%s)" % content_length
892 )
893 length = lengths.pop()
894 except ValueError:
895 length = None
896 else:
897 if length < 0:
898 length = None
900 else: # if content_length is None
901 length = None
903 # Convert status to int for comparison
904 # In some cases, httplib returns a status of "_UNKNOWN"
905 try:
906 status = int(self.status)
907 except ValueError:
908 status = 0
910 # Check for responses that shouldn't include a body
911 if status in (204, 304) or 100 <= status < 200 or request_method == "HEAD":
912 length = 0
914 return length
916 @contextmanager
917 def _error_catcher(self) -> typing.Generator[None]:
918 """
919 Catch low-level python exceptions, instead re-raising urllib3
920 variants, so that low-level exceptions are not leaked in the
921 high-level api.
923 On exit, release the connection back to the pool.
924 """
925 clean_exit = False
927 try:
928 try:
929 yield
931 except SocketTimeout as e:
932 # FIXME: Ideally we'd like to include the url in the ReadTimeoutError but
933 # there is yet no clean way to get at it from this context.
934 raise ReadTimeoutError(self._pool, None, "Read timed out.") from e # type: ignore[arg-type]
936 except BaseSSLError as e:
937 # SSL errors related to framing/MAC get wrapped and reraised here
938 raise SSLError(e) from e
940 except IncompleteRead as e:
941 if (
942 e.expected is not None
943 and e.partial is not None
944 and e.expected == -e.partial
945 ):
946 arg = "Response may not contain content."
947 else:
948 arg = f"Connection broken: {e!r}"
949 raise ProtocolError(arg, e) from e
951 except (HTTPException, OSError) as e:
952 raise ProtocolError(f"Connection broken: {e!r}", e) from e
954 # If no exception is thrown, we should avoid cleaning up
955 # unnecessarily.
956 clean_exit = True
957 finally:
958 # If we didn't terminate cleanly, we need to throw away our
959 # connection.
960 if not clean_exit:
961 # The response may not be closed but we're not going to use it
962 # anymore so close it now to ensure that the connection is
963 # released back to the pool.
964 if self._original_response:
965 self._original_response.close()
967 # Closing the response may not actually be sufficient to close
968 # everything, so if we have a hold of the connection close that
969 # too.
970 if self._connection:
971 self._connection.close()
973 # If we hold the original response but it's closed now, we should
974 # return the connection back to the pool.
975 if self._original_response and self._original_response.isclosed():
976 self.release_conn()
978 def _fp_read(
979 self,
980 amt: int | None = None,
981 *,
982 read1: bool = False,
983 ) -> bytes:
984 """
985 Read a response with the thought that reading the number of bytes
986 larger than can fit in a 32-bit int at a time via SSL in some
987 known cases leads to an overflow error that has to be prevented
988 if `amt` or `self.length_remaining` indicate that a problem may
989 happen.
991 This happens to urllib3 injected with pyOpenSSL-backed SSL-support.
992 """
993 assert self._fp
994 c_int_max = 2**31 - 1
995 if (
996 (amt and amt > c_int_max)
997 or (
998 amt is None
999 and self.length_remaining
1000 and self.length_remaining > c_int_max
1001 )
1002 ) and util.IS_PYOPENSSL:
1003 if read1:
1004 return self._fp.read1(c_int_max)
1005 buffer = io.BytesIO()
1006 # Besides `max_chunk_amt` being a maximum chunk size, it
1007 # affects memory overhead of reading a response by this
1008 # method in CPython.
1009 # `c_int_max` equal to 2 GiB - 1 byte is the actual maximum
1010 # chunk size that does not lead to an overflow error, but
1011 # 256 MiB is a compromise.
1012 max_chunk_amt = 2**28
1013 while amt is None or amt != 0:
1014 if amt is not None:
1015 chunk_amt = min(amt, max_chunk_amt)
1016 amt -= chunk_amt
1017 else:
1018 chunk_amt = max_chunk_amt
1019 data = self._fp.read(chunk_amt)
1020 if not data:
1021 break
1022 buffer.write(data)
1023 del data # to reduce peak memory usage by `max_chunk_amt`.
1024 return buffer.getvalue()
1025 elif read1:
1026 return self._fp.read1(amt) if amt is not None else self._fp.read1()
1027 else:
1028 # StringIO doesn't like amt=None
1029 return self._fp.read(amt) if amt is not None else self._fp.read()
1031 def _raw_read(
1032 self,
1033 amt: int | None = None,
1034 *,
1035 read1: bool = False,
1036 ) -> bytes:
1037 """
1038 Reads `amt` of bytes from the socket.
1039 """
1040 if self._fp is None:
1041 return None # type: ignore[return-value]
1043 fp_closed = getattr(self._fp, "closed", False)
1045 with self._error_catcher():
1046 data = self._fp_read(amt, read1=read1) if not fp_closed else b""
1047 if amt is not None and amt != 0 and not data:
1048 # Platform-specific: Buggy versions of Python.
1049 # Close the connection when no data is returned
1050 #
1051 # This is redundant to what httplib/http.client _should_
1052 # already do. However, versions of python released before
1053 # December 15, 2012 (http://bugs.python.org/issue16298) do
1054 # not properly close the connection in all cases. There is
1055 # no harm in redundantly calling close.
1056 self._fp.close()
1057 if (
1058 self.enforce_content_length
1059 and self.length_remaining is not None
1060 and self.length_remaining != 0
1061 ):
1062 # This is an edge case that httplib failed to cover due
1063 # to concerns of backward compatibility. We're
1064 # addressing it here to make sure IncompleteRead is
1065 # raised during streaming, so all calls with incorrect
1066 # Content-Length are caught.
1067 raise IncompleteRead(self._fp_bytes_read, self.length_remaining)
1068 elif read1 and (
1069 (amt != 0 and not data) or self.length_remaining == len(data)
1070 ):
1071 # All data has been read, but `self._fp.read1` in
1072 # CPython 3.12 and older doesn't always close
1073 # `http.client.HTTPResponse`, so we close it here.
1074 # See https://github.com/python/cpython/issues/113199
1075 self._fp.close()
1077 if data:
1078 self._fp_bytes_read += len(data)
1079 if self.length_remaining is not None:
1080 self.length_remaining -= len(data)
1081 return data
1083 def read(
1084 self,
1085 amt: int | None = None,
1086 decode_content: bool | None = None,
1087 cache_content: bool = False,
1088 ) -> bytes:
1089 """
1090 Similar to :meth:`http.client.HTTPResponse.read`, but with two additional
1091 parameters: ``decode_content`` and ``cache_content``.
1093 :param amt:
1094 How much of the content to read. If specified, caching is skipped
1095 because it doesn't make sense to cache partial content as the full
1096 response.
1098 :param decode_content:
1099 If True, will attempt to decode the body based on the
1100 'content-encoding' header.
1102 :param cache_content:
1103 If True, will save the returned data such that the same result is
1104 returned despite of the state of the underlying file object. This
1105 is useful if you want the ``.data`` property to continue working
1106 after having ``.read()`` the file object. (Overridden if ``amt`` is
1107 set.)
1108 """
1109 self._init_decoder()
1110 if decode_content is None:
1111 decode_content = self.decode_content
1113 if amt and amt < 0:
1114 # Negative numbers and `None` should be treated the same.
1115 amt = None
1116 elif amt is not None:
1117 cache_content = False
1119 if (
1120 self._decoder
1121 and self._decoder.has_unconsumed_tail
1122 and len(self._decoded_buffer) < amt
1123 ):
1124 decoded_data = self._decode(
1125 b"",
1126 decode_content,
1127 flush_decoder=False,
1128 max_length=amt - len(self._decoded_buffer),
1129 )
1130 self._decoded_buffer.put(decoded_data)
1131 if len(self._decoded_buffer) >= amt:
1132 return self._decoded_buffer.get(amt)
1134 data = self._raw_read(amt)
1135 if not cache_content:
1136 self._uncached_read_occurred = True
1138 flush_decoder = amt is None or (amt != 0 and not data)
1140 if (
1141 not data
1142 and len(self._decoded_buffer) == 0
1143 and not (self._decoder and self._decoder.has_unconsumed_tail)
1144 ):
1145 return data
1147 if amt is None:
1148 data = self._decode(data, decode_content, flush_decoder)
1149 # It's possible that there is buffered decoded data after a
1150 # partial read.
1151 if decode_content and len(self._decoded_buffer) > 0:
1152 self._decoded_buffer.put(data)
1153 data = self._decoded_buffer.get_all()
1155 if cache_content and not self._uncached_read_occurred:
1156 self._body = data
1157 else:
1158 # do not waste memory on buffer when not decoding
1159 if not decode_content:
1160 if self._has_decoded_content:
1161 raise RuntimeError(
1162 "Calling read(decode_content=False) is not supported after "
1163 "read(decode_content=True) was called."
1164 )
1165 return data
1167 decoded_data = self._decode(
1168 data,
1169 decode_content,
1170 flush_decoder,
1171 max_length=amt - len(self._decoded_buffer),
1172 )
1173 self._decoded_buffer.put(decoded_data)
1175 while len(self._decoded_buffer) < amt and data:
1176 # TODO make sure to initially read enough data to get past the headers
1177 # For example, the GZ file header takes 10 bytes, we don't want to read
1178 # it one byte at a time
1179 data = self._raw_read(amt)
1180 decoded_data = self._decode(
1181 data,
1182 decode_content,
1183 flush_decoder,
1184 max_length=amt - len(self._decoded_buffer),
1185 )
1186 self._decoded_buffer.put(decoded_data)
1187 data = self._decoded_buffer.get(amt)
1189 return data
1191 def read1(
1192 self,
1193 amt: int | None = None,
1194 decode_content: bool | None = None,
1195 ) -> bytes:
1196 """
1197 Similar to ``http.client.HTTPResponse.read1`` and documented
1198 in :meth:`io.BufferedReader.read1`, but with an additional parameter:
1199 ``decode_content``.
1201 :param amt:
1202 How much of the content to read.
1204 :param decode_content:
1205 If True, will attempt to decode the body based on the
1206 'content-encoding' header.
1207 """
1208 if decode_content is None:
1209 decode_content = self.decode_content
1210 if amt and amt < 0:
1211 # Negative numbers and `None` should be treated the same.
1212 amt = None
1213 # try and respond without going to the network
1214 if self._has_decoded_content:
1215 if not decode_content:
1216 raise RuntimeError(
1217 "Calling read1(decode_content=False) is not supported after "
1218 "read1(decode_content=True) was called."
1219 )
1220 if (
1221 self._decoder
1222 and self._decoder.has_unconsumed_tail
1223 and (amt is None or len(self._decoded_buffer) < amt)
1224 ):
1225 decoded_data = self._decode(
1226 b"",
1227 decode_content,
1228 flush_decoder=False,
1229 max_length=(
1230 amt - len(self._decoded_buffer) if amt is not None else None
1231 ),
1232 )
1233 self._decoded_buffer.put(decoded_data)
1234 if len(self._decoded_buffer) > 0:
1235 if amt is None:
1236 return self._decoded_buffer.get_all()
1237 return self._decoded_buffer.get(amt)
1238 if amt == 0:
1239 return b""
1241 # FIXME, this method's type doesn't say returning None is possible
1242 data = self._raw_read(amt, read1=True)
1243 self._uncached_read_occurred = True
1244 if not decode_content or data is None:
1245 return data
1247 self._init_decoder()
1248 while True:
1249 flush_decoder = not data
1250 decoded_data = self._decode(
1251 data, decode_content, flush_decoder, max_length=amt
1252 )
1253 self._decoded_buffer.put(decoded_data)
1254 if decoded_data or flush_decoder:
1255 break
1256 data = self._raw_read(8192, read1=True)
1258 if amt is None:
1259 return self._decoded_buffer.get_all()
1260 return self._decoded_buffer.get(amt)
1262 def stream(
1263 self, amt: int | None = _READ_CHUNK_SIZE, decode_content: bool | None = None
1264 ) -> typing.Generator[bytes]:
1265 """
1266 A generator wrapper for the read() method. A call will block until
1267 ``amt`` bytes have been read from the connection or until the
1268 connection is closed.
1270 :param amt:
1271 How much of the content to read. The generator will return up to
1272 much data per iteration, but may return less. This is particularly
1273 likely when using compressed data. However, the empty string will
1274 never be returned.
1276 :param decode_content:
1277 If True, will attempt to decode the body based on the
1278 'content-encoding' header.
1279 """
1280 if amt == 0:
1281 return
1283 if self.chunked and self.supports_chunked_reads():
1284 yield from self.read_chunked(amt, decode_content=decode_content)
1285 else:
1286 while (
1287 not is_fp_closed(self._fp)
1288 or len(self._decoded_buffer) > 0
1289 or (self._decoder and self._decoder.has_unconsumed_tail)
1290 ):
1291 data = self.read(amt=amt, decode_content=decode_content)
1293 if data:
1294 yield data
1296 # Overrides from io.IOBase
1297 def readable(self) -> bool:
1298 return True
1300 def shutdown(self) -> None:
1301 if not self._sock_shutdown:
1302 raise ValueError("Cannot shutdown socket as self._sock_shutdown is not set")
1303 if self._connection is None:
1304 raise RuntimeError(
1305 "Cannot shutdown as connection has already been released to the pool"
1306 )
1307 self._sock_shutdown(socket.SHUT_RD)
1309 def close(self) -> None:
1310 self._sock_shutdown = None
1312 if not self.closed and self._fp:
1313 self._fp.close()
1315 if self._connection:
1316 self._connection.close()
1318 if not self.auto_close:
1319 io.IOBase.close(self)
1321 @property
1322 def closed(self) -> bool:
1323 if not self.auto_close:
1324 return io.IOBase.closed.__get__(self) # type: ignore[no-any-return]
1325 elif self._fp is None:
1326 return True
1327 elif hasattr(self._fp, "isclosed"):
1328 return self._fp.isclosed()
1329 elif hasattr(self._fp, "closed"):
1330 return self._fp.closed
1331 else:
1332 return True
1334 def fileno(self) -> int:
1335 if self._fp is None:
1336 raise OSError("HTTPResponse has no file to get a fileno from")
1337 elif hasattr(self._fp, "fileno"):
1338 return self._fp.fileno()
1339 else:
1340 raise OSError(
1341 "The file-like object this HTTPResponse is wrapped "
1342 "around has no file descriptor"
1343 )
1345 def flush(self) -> None:
1346 if (
1347 self._fp is not None
1348 and hasattr(self._fp, "flush")
1349 and not getattr(self._fp, "closed", False)
1350 ):
1351 return self._fp.flush()
1353 def supports_chunked_reads(self) -> bool:
1354 """
1355 Checks if the underlying file-like object looks like a
1356 :class:`http.client.HTTPResponse` object. We do this by testing for
1357 the fp attribute. If it is present we assume it returns raw chunks as
1358 processed by read_chunked().
1359 """
1360 return hasattr(self._fp, "fp")
1362 def _update_chunk_length(self) -> None:
1363 # First, we'll figure out length of a chunk and then
1364 # we'll try to read it from socket.
1365 if self.chunk_left is not None:
1366 return None
1367 line = self._fp.fp.readline(_MAX_CHUNK_LINE_LENGTH + 1) # type: ignore[union-attr]
1368 if len(line) > _MAX_CHUNK_LINE_LENGTH:
1369 self.close()
1370 raise ProtocolError(
1371 "Response chunk size line exceeded maximum allowed length"
1372 ) from None
1373 line = line.split(b";", 1)[0]
1374 try:
1375 self.chunk_left = int(line, 16)
1376 except ValueError:
1377 self.close()
1378 if line:
1379 # Invalid chunked protocol response, abort.
1380 raise InvalidChunkLength(self, line) from None
1381 else:
1382 # Truncated at start of next chunk
1383 raise ProtocolError("Response ended prematurely") from None
1385 def _handle_chunk(self, amt: int | None) -> bytes:
1386 returned_chunk = None
1387 if amt is None:
1388 chunk = self._fp._safe_read(self.chunk_left) # type: ignore[union-attr]
1389 returned_chunk = chunk
1390 self._fp._safe_read(2) # type: ignore[union-attr] # Toss the CRLF at the end of the chunk.
1391 self.chunk_left = None
1392 elif self.chunk_left is not None and amt < self.chunk_left:
1393 value = self._fp._safe_read(amt) # type: ignore[union-attr]
1394 self.chunk_left = self.chunk_left - amt
1395 returned_chunk = value
1396 elif amt == self.chunk_left:
1397 value = self._fp._safe_read(amt) # type: ignore[union-attr]
1398 self._fp._safe_read(2) # type: ignore[union-attr] # Toss the CRLF at the end of the chunk.
1399 self.chunk_left = None
1400 returned_chunk = value
1401 else: # amt > self.chunk_left
1402 returned_chunk = self._fp._safe_read(self.chunk_left) # type: ignore[union-attr]
1403 self._fp._safe_read(2) # type: ignore[union-attr] # Toss the CRLF at the end of the chunk.
1404 self.chunk_left = None
1405 return returned_chunk # type: ignore[no-any-return]
1407 def read_chunked(
1408 self, amt: int | None = None, decode_content: bool | None = None
1409 ) -> typing.Generator[bytes]:
1410 """
1411 Similar to :meth:`HTTPResponse.read`, but with an additional
1412 parameter: ``decode_content``.
1414 :param amt:
1415 How much of the content to read. If specified, caching is skipped
1416 because it doesn't make sense to cache partial content as the full
1417 response.
1419 :param decode_content:
1420 If True, will attempt to decode the body based on the
1421 'content-encoding' header.
1422 """
1423 self._init_decoder()
1424 # FIXME: Rewrite this method and make it a class with a better structured logic.
1425 if not self.chunked:
1426 raise ResponseNotChunked(
1427 "Response is not chunked. "
1428 "Header 'transfer-encoding: chunked' is missing."
1429 )
1430 if not self.supports_chunked_reads():
1431 raise BodyNotHttplibCompatible(
1432 "Body should be http.client.HTTPResponse like. "
1433 "It should have have an fp attribute which returns raw chunks."
1434 )
1436 with self._error_catcher():
1437 # Don't bother reading the body of a HEAD request.
1438 if self._original_response and is_response_to_head(self._original_response):
1439 self._original_response.close()
1440 return None
1442 # If a response is already read and closed
1443 # then return immediately.
1444 if self._fp.fp is None: # type: ignore[union-attr]
1445 return None
1447 if amt == 0:
1448 return
1449 elif amt and amt < 0:
1450 # Negative numbers and `None` should be treated the same,
1451 # but httplib handles only `None` correctly.
1452 amt = None
1454 while True:
1455 # First, check if any data is left in the decoder's buffer.
1456 if self._decoder and self._decoder.has_unconsumed_tail:
1457 chunk = b""
1458 else:
1459 self._update_chunk_length()
1460 self._uncached_read_occurred = True
1461 if self.chunk_left == 0:
1462 break
1463 chunk = self._handle_chunk(amt)
1464 decoded = self._decode(
1465 chunk,
1466 decode_content=decode_content,
1467 flush_decoder=False,
1468 max_length=amt,
1469 )
1470 if decoded:
1471 yield decoded
1473 if decode_content:
1474 # On CPython and PyPy, we should never need to flush the
1475 # decoder. However, on Jython we *might* need to, so
1476 # lets defensively do it anyway.
1477 decoded = self._flush_decoder()
1478 if decoded: # Platform-specific: Jython.
1479 yield decoded
1481 # Chunk content ends with \r\n: discard it.
1482 while self._fp is not None:
1483 line = self._fp.fp.readline(_MAX_CHUNK_LINE_LENGTH + 1)
1484 if len(line) > _MAX_CHUNK_LINE_LENGTH:
1485 raise ProtocolError(
1486 "Response chunk trailer line exceeded maximum allowed length"
1487 )
1488 if not line:
1489 # Some sites may not end with '\r\n'.
1490 break
1491 if line == b"\r\n":
1492 break
1494 # We read everything; close the "file".
1495 if self._original_response:
1496 self._original_response.close()
1498 @property
1499 def url(self) -> str | None:
1500 """
1501 Returns the URL that was the source of this response.
1502 If the request that generated this response redirected, this method
1503 will return the final redirect location.
1504 """
1505 return self._request_url
1507 @url.setter
1508 def url(self, url: str | None) -> None:
1509 self._request_url = url
1511 def __iter__(self) -> typing.Iterator[bytes]:
1512 buffer: list[bytes] = []
1513 for chunk in self.stream(decode_content=True):
1514 if b"\n" in chunk:
1515 chunks = chunk.split(b"\n")
1516 yield b"".join(buffer) + chunks[0] + b"\n"
1517 for x in chunks[1:-1]:
1518 yield x + b"\n"
1519 if chunks[-1]:
1520 buffer = [chunks[-1]]
1521 else:
1522 buffer = []
1523 else:
1524 buffer.append(chunk)
1525 if buffer:
1526 yield b"".join(buffer)