codekingpro/portable-devtools
114k
1from __future__ import annotations2 3import collections4import io5import json as _json6import logging7import socket8import sys9import typing10import warnings11import zlib12from contextlib import contextmanager13from http.client import HTTPMessage as _HttplibHTTPMessage14from http.client import HTTPResponse as _HttplibHTTPResponse15from socket import timeout as SocketTimeout16 17if typing.TYPE_CHECKING:18 from ._base_connection import BaseHTTPConnection19 20try:21 try:22 import brotlicffi as brotli # type: ignore[import-not-found]23 except ImportError:24 import brotli # type: ignore[import-not-found]25except ImportError:26 brotli = None27 28from . import util29from ._base_connection import _TYPE_BODY30from ._collections import HTTPHeaderDict31from .connection import BaseSSLError, HTTPConnection, HTTPException32from .exceptions import (33 BodyNotHttplibCompatible,34 DecodeError,35 DependencyWarning,36 HTTPError,37 IncompleteRead,38 InvalidChunkLength,39 InvalidHeader,40 ProtocolError,41 ReadTimeoutError,42 ResponseNotChunked,43 SSLError,44)45from .util.response import is_fp_closed, is_response_to_head46from .util.retry import Retry47 48if typing.TYPE_CHECKING:49 from .connectionpool import HTTPConnectionPool50 51log = logging.getLogger(__name__)52 53 54class ContentDecoder:55 def decompress(self, data: bytes, max_length: int = -1) -> bytes:56 raise NotImplementedError()57 58 @property59 def has_unconsumed_tail(self) -> bool:60 raise NotImplementedError()61 62 def flush(self) -> bytes:63 raise NotImplementedError()64 65 66class DeflateDecoder(ContentDecoder):67 def __init__(self) -> None:68 self._first_try = True69 self._first_try_data = b""70 self._unfed_data = b""71 self._obj = zlib.decompressobj()72 73 def decompress(self, data: bytes, max_length: int = -1) -> bytes:74 data = self._unfed_data + data75 self._unfed_data = b""76 if not data and not self._obj.unconsumed_tail:77 return data78 original_max_length = max_length79 if original_max_length < 0:80 max_length = 081 elif original_max_length == 0:82 # We should not pass 0 to the zlib decompressor because 0 is83 # the default value that will make zlib decompress without a84 # length limit.85 # Data should be stored for subsequent calls.86 self._unfed_data = data87 return b""88 89 # Subsequent calls always reuse `self._obj`. zlib requires90 # passing the unconsumed tail if decompression is to continue.91 if not self._first_try:92 return self._obj.decompress(93 self._obj.unconsumed_tail + data, max_length=max_length94 )95 96 # First call tries with RFC 1950 ZLIB format.97 self._first_try_data += data98 try:99 decompressed = self._obj.decompress(data, max_length=max_length)100 if decompressed:101 self._first_try = False102 self._first_try_data = b""103 return decompressed104 # On failure, it falls back to RFC 1951 DEFLATE format.105 except zlib.error:106 self._first_try = False107 self._obj = zlib.decompressobj(-zlib.MAX_WBITS)108 try:109 return self.decompress(110 self._first_try_data, max_length=original_max_length111 )112 finally:113 self._first_try_data = b""114 115 @property116 def has_unconsumed_tail(self) -> bool:117 return bool(self._unfed_data) or (118 bool(self._obj.unconsumed_tail) and not self._first_try119 )120 121 def flush(self) -> bytes:122 return self._obj.flush()123 124 125class GzipDecoderState:126 FIRST_MEMBER = 0127 OTHER_MEMBERS = 1128 SWALLOW_DATA = 2129 130 131class GzipDecoder(ContentDecoder):132 def __init__(self) -> None:133 self._obj = zlib.decompressobj(16 + zlib.MAX_WBITS)134 self._state = GzipDecoderState.FIRST_MEMBER135 self._unconsumed_tail = b""136 137 def decompress(self, data: bytes, max_length: int = -1) -> bytes:138 ret = bytearray()139 if self._state == GzipDecoderState.SWALLOW_DATA:140 return bytes(ret)141 142 if max_length == 0:143 # We should not pass 0 to the zlib decompressor because 0 is144 # the default value that will make zlib decompress without a145 # length limit.146 # Data should be stored for subsequent calls.147 self._unconsumed_tail += data148 return b""149 150 # zlib requires passing the unconsumed tail to the subsequent151 # call if decompression is to continue.152 data = self._unconsumed_tail + data153 if not data and self._obj.eof:154 return bytes(ret)155 156 while True:157 try:158 ret += self._obj.decompress(159 data, max_length=max(max_length - len(ret), 0)160 )161 except zlib.error:162 previous_state = self._state163 # Ignore data after the first error164 self._state = GzipDecoderState.SWALLOW_DATA165 self._unconsumed_tail = b""166 if previous_state == GzipDecoderState.OTHER_MEMBERS:167 # Allow trailing garbage acceptable in other gzip clients168 return bytes(ret)169 raise170 171 self._unconsumed_tail = data = (172 self._obj.unconsumed_tail or self._obj.unused_data173 )174 if max_length > 0 and len(ret) >= max_length:175 break176 177 if not data:178 return bytes(ret)179 # When the end of a gzip member is reached, a new decompressor180 # must be created for unused (possibly future) data.181 if self._obj.eof:182 self._state = GzipDecoderState.OTHER_MEMBERS183 self._obj = zlib.decompressobj(16 + zlib.MAX_WBITS)184 185 return bytes(ret)186 187 @property188 def has_unconsumed_tail(self) -> bool:189 return bool(self._unconsumed_tail)190 191 def flush(self) -> bytes:192 return self._obj.flush()193 194 195if brotli is not None:196 197 class BrotliDecoder(ContentDecoder):198 # Supports both 'brotlipy' and 'Brotli' packages199 # since they share an import name. The top branches200 # are for 'brotlipy' and bottom branches for 'Brotli'201 def __init__(self) -> None:202 self._obj = brotli.Decompressor()203 if hasattr(self._obj, "decompress"):204 setattr(self, "_decompress", self._obj.decompress)205 else:206 setattr(self, "_decompress", self._obj.process)207 208 # Requires Brotli >= 1.2.0 for `output_buffer_limit`.209 def _decompress(self, data: bytes, output_buffer_limit: int = -1) -> bytes:210 raise NotImplementedError()211 212 def decompress(self, data: bytes, max_length: int = -1) -> bytes:213 try:214 if max_length > 0:215 return self._decompress(data, output_buffer_limit=max_length)216 else:217 return self._decompress(data)218 except TypeError:219 # Fallback for Brotli/brotlicffi/brotlipy versions without220 # the `output_buffer_limit` parameter.221 warnings.warn(222 "Brotli >= 1.2.0 is required to prevent decompression bombs.",223 DependencyWarning,224 )225 return self._decompress(data)226 227 @property228 def has_unconsumed_tail(self) -> bool:229 try:230 return not self._obj.can_accept_more_data()231 except AttributeError:232 return False233 234 def flush(self) -> bytes:235 if hasattr(self._obj, "flush"):236 return self._obj.flush() # type: ignore[no-any-return]237 return b""238 239 240try:241 if sys.version_info >= (3, 14):242 from compression import zstd243 else:244 from backports import zstd245except ImportError:246 HAS_ZSTD = False247else:248 HAS_ZSTD = True249 250 class ZstdDecoder(ContentDecoder):251 def __init__(self) -> None:252 self._obj = zstd.ZstdDecompressor()253 254 def decompress(self, data: bytes, max_length: int = -1) -> bytes:255 if not data and not self.has_unconsumed_tail:256 return b""257 if self._obj.eof:258 data = self._obj.unused_data + data259 self._obj = zstd.ZstdDecompressor()260 part = self._obj.decompress(data, max_length=max_length)261 length = len(part)262 data_parts = [part]263 # Every loop iteration is supposed to read data from a separate frame.264 # The loop breaks when:265 # - enough data is read;266 # - no more unused data is available;267 # - end of the last read frame has not been reached (i.e.,268 # more data has to be fed).269 while (270 self._obj.eof271 and self._obj.unused_data272 and (max_length < 0 or length < max_length)273 ):274 unused_data = self._obj.unused_data275 if not self._obj.needs_input:276 self._obj = zstd.ZstdDecompressor()277 part = self._obj.decompress(278 unused_data,279 max_length=(max_length - length) if max_length > 0 else -1,280 )281 if part_length := len(part):282 data_parts.append(part)283 length += part_length284 elif self._obj.needs_input:285 break286 return b"".join(data_parts)287 288 @property289 def has_unconsumed_tail(self) -> bool:290 return not (self._obj.needs_input or self._obj.eof) or bool(291 self._obj.unused_data292 )293 294 def flush(self) -> bytes:295 if not self._obj.eof:296 raise DecodeError("Zstandard data is incomplete")297 return b""298 299 300class MultiDecoder(ContentDecoder):301 """302 From RFC7231:303 If one or more encodings have been applied to a representation, the304 sender that applied the encodings MUST generate a Content-Encoding305 header field that lists the content codings in the order in which306 they were applied.307 """308 309 # Maximum allowed number of chained HTTP encodings in the310 # Content-Encoding header.311 max_decode_links = 5312 313 def __init__(self, modes: str) -> None:314 encodings = [m.strip() for m in modes.split(",")]315 if len(encodings) > self.max_decode_links:316 raise DecodeError(317 "Too many content encodings in the chain: "318 f"{len(encodings)} > {self.max_decode_links}"319 )320 self._decoders = [_get_decoder(e) for e in encodings]321 322 def flush(self) -> bytes:323 return self._decoders[0].flush()324 325 def decompress(self, data: bytes, max_length: int = -1) -> bytes:326 if max_length <= 0:327 for d in reversed(self._decoders):328 data = d.decompress(data)329 return data330 331 ret = bytearray()332 # Every while loop iteration goes through all decoders once.333 # It exits when enough data is read or no more data can be read.334 # It is possible that the while loop iteration does not produce335 # any data because we retrieve up to `max_length` from every336 # decoder, and the amount of bytes may be insufficient for the337 # next decoder to produce enough/any output.338 while True:339 any_data = False340 for d in reversed(self._decoders):341 data = d.decompress(data, max_length=max_length - len(ret))342 if data:343 any_data = True344 # We should not break when no data is returned because345 # next decoders may produce data even with empty input.346 ret += data347 if not any_data or len(ret) >= max_length:348 return bytes(ret)349 data = b""350 351 @property352 def has_unconsumed_tail(self) -> bool:353 return any(d.has_unconsumed_tail for d in self._decoders)354 355 356def _get_decoder(mode: str) -> ContentDecoder:357 if "," in mode:358 return MultiDecoder(mode)359 360 # According to RFC 9110 section 8.4.1.3, recipients should361 # consider x-gzip equivalent to gzip362 if mode in ("gzip", "x-gzip"):363 return GzipDecoder()364 365 if brotli is not None and mode == "br":366 return BrotliDecoder()367 368 if HAS_ZSTD and mode == "zstd":369 return ZstdDecoder()370 371 return DeflateDecoder()372 373 374class BytesQueueBuffer:375 """Memory-efficient bytes buffer376 377 To return decoded data in read() and still follow the BufferedIOBase API, we need a378 buffer to always return the correct amount of bytes.379 380 This buffer should be filled using calls to put()381 382 Our maximum memory usage is determined by the sum of the size of:383 384 * self.buffer, which contains the full data385 * the largest chunk that we will copy in get()386 """387 388 def __init__(self) -> None:389 self.buffer: typing.Deque[bytes | memoryview[bytes]] = collections.deque()390 self._size: int = 0391 392 def __len__(self) -> int:393 return self._size394 395 def put(self, data: bytes) -> None:396 self.buffer.append(data)397 self._size += len(data)398 399 def get(self, n: int) -> bytes:400 if n == 0:401 return b""402 elif not self.buffer:403 raise RuntimeError("buffer is empty")404 elif n < 0:405 raise ValueError("n should be > 0")406 407 if len(self.buffer[0]) == n and isinstance(self.buffer[0], bytes):408 self._size -= n409 return self.buffer.popleft()410 411 fetched = 0412 ret = io.BytesIO()413 while fetched < n:414 remaining = n - fetched415 chunk = self.buffer.popleft()416 chunk_length = len(chunk)417 if remaining < chunk_length:418 chunk = memoryview(chunk)419 left_chunk, right_chunk = chunk[:remaining], chunk[remaining:]420 ret.write(left_chunk)421 self.buffer.appendleft(right_chunk)422 self._size -= remaining423 break424 else:425 ret.write(chunk)426 self._size -= chunk_length427 fetched += chunk_length428 429 if not self.buffer:430 break431 432 return ret.getvalue()433 434 def get_all(self) -> bytes:435 buffer = self.buffer436 if not buffer:437 assert self._size == 0438 return b""439 if len(buffer) == 1:440 result = buffer.pop()441 if isinstance(result, memoryview):442 result = result.tobytes()443 else:444 ret = io.BytesIO()445 ret.writelines(buffer.popleft() for _ in range(len(buffer)))446 result = ret.getvalue()447 self._size = 0448 return result449 450 451class BaseHTTPResponse(io.IOBase):452 CONTENT_DECODERS = ["gzip", "x-gzip", "deflate"]453 if brotli is not None:454 CONTENT_DECODERS += ["br"]455 if HAS_ZSTD:456 CONTENT_DECODERS += ["zstd"]457 REDIRECT_STATUSES = [301, 302, 303, 307, 308]458 459 DECODER_ERROR_CLASSES: tuple[type[Exception], ...] = (IOError, zlib.error)460 if brotli is not None:461 DECODER_ERROR_CLASSES += (brotli.error,)462 463 if HAS_ZSTD:464 DECODER_ERROR_CLASSES += (zstd.ZstdError,)465 466 def __init__(467 self,468 *,469 headers: typing.Mapping[str, str] | typing.Mapping[bytes, bytes] | None = None,470 status: int,471 version: int,472 version_string: str,473 reason: str | None,474 decode_content: bool,475 request_url: str | None,476 retries: Retry | None = None,477 ) -> None:478 if isinstance(headers, HTTPHeaderDict):479 self.headers = headers480 else:481 self.headers = HTTPHeaderDict(headers) # type: ignore[arg-type]482 self.status = status483 self.version = version484 self.version_string = version_string485 self.reason = reason486 self.decode_content = decode_content487 self._has_decoded_content = False488 self._request_url: str | None = request_url489 self.retries = retries490 491 self.chunked = False492 tr_enc = self.headers.get("transfer-encoding", "").lower()493 # Don't incur the penalty of creating a list and then discarding it494 encodings = (enc.strip() for enc in tr_enc.split(","))495 if "chunked" in encodings:496 self.chunked = True497 498 self._decoder: ContentDecoder | None = None499 self.length_remaining: int | None500 501 def get_redirect_location(self) -> str | None | typing.Literal[False]:502 """503 Should we redirect and where to?504 505 :returns: Truthy redirect location string if we got a redirect status506 code and valid location. ``None`` if redirect status and no507 location. ``False`` if not a redirect status code.508 """509 if self.status in self.REDIRECT_STATUSES:510 return self.headers.get("location")511 return False512 513 @property514 def data(self) -> bytes:515 raise NotImplementedError()516 517 def json(self) -> typing.Any:518 """519 Deserializes the body of the HTTP response as a Python object.520 521 The body of the HTTP response must be encoded using UTF-8, as per522 `RFC 8529 Section 8.1 <https://www.rfc-editor.org/rfc/rfc8259#section-8.1>`_.523 524 To use a custom JSON decoder pass the result of :attr:`HTTPResponse.data` to525 your custom decoder instead.526 527 If the body of the HTTP response is not decodable to UTF-8, a528 `UnicodeDecodeError` will be raised. If the body of the HTTP response is not a529 valid JSON document, a `json.JSONDecodeError` will be raised.530 531 Read more :ref:`here <json_content>`.532 533 :returns: The body of the HTTP response as a Python object.534 """535 data = self.data.decode("utf-8")536 return _json.loads(data)537 538 @property539 def url(self) -> str | None:540 raise NotImplementedError()541 542 @url.setter543 def url(self, url: str | None) -> None:544 raise NotImplementedError()545 546 @property547 def connection(self) -> BaseHTTPConnection | None:548 raise NotImplementedError()549 550 @property551 def retries(self) -> Retry | None:552 return self._retries553 554 @retries.setter555 def retries(self, retries: Retry | None) -> None:556 # Override the request_url if retries has a redirect location.557 if retries is not None and retries.history:558 self.url = retries.history[-1].redirect_location559 self._retries = retries560 561 def stream(562 self, amt: int | None = 2**16, decode_content: bool | None = None563 ) -> typing.Iterator[bytes]:564 raise NotImplementedError()565 566 def read(567 self,568 amt: int | None = None,569 decode_content: bool | None = None,570 cache_content: bool = False,571 ) -> bytes:572 raise NotImplementedError()573 574 def read1(575 self,576 amt: int | None = None,577 decode_content: bool | None = None,578 ) -> bytes:579 raise NotImplementedError()580 581 def read_chunked(582 self,583 amt: int | None = None,584 decode_content: bool | None = None,585 ) -> typing.Iterator[bytes]:586 raise NotImplementedError()587 588 def release_conn(self) -> None:589 raise NotImplementedError()590 591 def drain_conn(self) -> None:592 raise NotImplementedError()593 594 def shutdown(self) -> None:595 raise NotImplementedError()596 597 def close(self) -> None:598 raise NotImplementedError()599 600 def _init_decoder(self) -> None:601 """602 Set-up the _decoder attribute if necessary.603 """604 # Note: content-encoding value should be case-insensitive, per RFC 7230605 # Section 3.2606 content_encoding = self.headers.get("content-encoding", "").lower()607 if self._decoder is None:608 if content_encoding in self.CONTENT_DECODERS:609 self._decoder = _get_decoder(content_encoding)610 elif "," in content_encoding:611 encodings = [612 e.strip()613 for e in content_encoding.split(",")614 if e.strip() in self.CONTENT_DECODERS615 ]616 if encodings:617 self._decoder = _get_decoder(content_encoding)618 619 def _decode(620 self,621 data: bytes,622 decode_content: bool | None,623 flush_decoder: bool,624 max_length: int | None = None,625 ) -> bytes:626 """627 Decode the data passed in and potentially flush the decoder.628 """629 if not decode_content:630 if self._has_decoded_content:631 raise RuntimeError(632 "Calling read(decode_content=False) is not supported after "633 "read(decode_content=True) was called."634 )635 return data636 637 if max_length is None or flush_decoder:638 max_length = -1639 640 try:641 if self._decoder:642 data = self._decoder.decompress(data, max_length=max_length)643 self._has_decoded_content = True644 except self.DECODER_ERROR_CLASSES as e:645 content_encoding = self.headers.get("content-encoding", "").lower()646 raise DecodeError(647 "Received response with content-encoding: %s, but "648 "failed to decode it." % content_encoding,649 e,650 ) from e651 if flush_decoder:652 data += self._flush_decoder()653 654 return data655 656 def _flush_decoder(self) -> bytes:657 """658 Flushes the decoder. Should only be called if the decoder is actually659 being used.660 """661 if self._decoder:662 return self._decoder.decompress(b"") + self._decoder.flush()663 return b""664 665 # Compatibility methods for `io` module666 def readinto(self, b: bytearray | memoryview[int]) -> int:667 temp = self.read(len(b))668 if len(temp) == 0:669 return 0670 else:671 b[: len(temp)] = temp672 return len(temp)673 674 # Methods used by dependent libraries675 def getheaders(self) -> HTTPHeaderDict:676 return self.headers677 678 def getheader(self, name: str, default: str | None = None) -> str | None:679 return self.headers.get(name, default)680 681 # Compatibility method for http.cookiejar682 def info(self) -> HTTPHeaderDict:683 return self.headers684 685 def geturl(self) -> str | None:686 return self.url687 688 689class HTTPResponse(BaseHTTPResponse):690 """691 HTTP Response container.692 693 Backwards-compatible with :class:`http.client.HTTPResponse` but the response ``body`` is694 loaded and decoded on-demand when the ``data`` property is accessed. This695 class is also compatible with the Python standard library's :mod:`io`696 module, and can hence be treated as a readable object in the context of that697 framework.698 699 Extra parameters for behaviour not present in :class:`http.client.HTTPResponse`:700 701 :param preload_content:702 If True, the response's body will be preloaded during construction.703 704 :param decode_content:705 If True, will attempt to decode the body based on the706 'content-encoding' header.707 708 :param original_response:709 When this HTTPResponse wrapper is generated from an :class:`http.client.HTTPResponse`710 object, it's convenient to include the original for debug purposes. It's711 otherwise unused.712 713 :param retries:714 The retries contains the last :class:`~urllib3.util.retry.Retry` that715 was used during the request.716 717 :param enforce_content_length:718 Enforce content length checking. Body returned by server must match719 value of Content-Length header, if present. Otherwise, raise error.720 """721 722 def __init__(723 self,724 body: _TYPE_BODY = "",725 headers: typing.Mapping[str, str] | typing.Mapping[bytes, bytes] | None = None,726 status: int = 0,727 version: int = 0,728 version_string: str = "HTTP/?",729 reason: str | None = None,730 preload_content: bool = True,731 decode_content: bool = True,732 original_response: _HttplibHTTPResponse | None = None,733 pool: HTTPConnectionPool | None = None,734 connection: HTTPConnection | None = None,735 msg: _HttplibHTTPMessage | None = None,736 retries: Retry | None = None,737 enforce_content_length: bool = True,738 request_method: str | None = None,739 request_url: str | None = None,740 auto_close: bool = True,741 sock_shutdown: typing.Callable[[int], None] | None = None,742 ) -> None:743 super().__init__(744 headers=headers,745 status=status,746 version=version,747 version_string=version_string,748 reason=reason,749 decode_content=decode_content,750 request_url=request_url,751 retries=retries,752 )753 754 self.enforce_content_length = enforce_content_length755 self.auto_close = auto_close756 757 self._body = None758 self._uncached_read_occurred = False759 self._fp: _HttplibHTTPResponse | None = None760 self._original_response = original_response761 self._fp_bytes_read = 0762 self.msg = msg763 764 if body and isinstance(body, (str, bytes)):765 self._body = body766 767 self._pool = pool768 self._connection = connection769 770 if hasattr(body, "read"):771 self._fp = body # type: ignore[assignment]772 self._sock_shutdown = sock_shutdown773 774 # Are we using the chunked-style of transfer encoding?775 self.chunk_left: int | None = None776 777 # Determine length of response778 self.length_remaining = self._init_length(request_method)779 780 # Used to return the correct amount of bytes for partial read()s781 self._decoded_buffer = BytesQueueBuffer()782 783 # If requested, preload the body.784 if preload_content and not self._body:785 self._body = self.read(decode_content=decode_content)786 787 def release_conn(self) -> None:788 if not self._pool or not self._connection:789 return None790 791 self._pool._put_conn(self._connection)792 self._connection = None793 794 def drain_conn(self) -> None:795 """796 Read and discard any remaining HTTP response data in the response connection.797 798 Unread data in the HTTPResponse connection blocks the connection from being released back to the pool.799 """800 try:801 self._raw_read()802 except (HTTPError, OSError, BaseSSLError, HTTPException):803 pass804 if self._has_decoded_content:805 # `_raw_read` skips decompression, so we should clean up the806 # decoder to avoid keeping unnecessary data in memory.807 self._decoded_buffer = BytesQueueBuffer()808 self._decoder = None809 810 @property811 def data(self) -> bytes:812 # For backwards-compat with earlier urllib3 0.4 and earlier.813 if self._body:814 return self._body # type: ignore[return-value]815 816 if self._fp:817 return self.read(cache_content=True)818 819 return None # type: ignore[return-value]820 821 @property822 def connection(self) -> HTTPConnection | None:823 return self._connection824 825 def isclosed(self) -> bool:826 return is_fp_closed(self._fp)827 828 def tell(self) -> int:829 """830 Obtain the number of bytes pulled over the wire so far. May differ from831 the amount of content returned by :meth:`HTTPResponse.read`832 if bytes are encoded on the wire (e.g, compressed).833 """834 return self._fp_bytes_read835 836 def _init_length(self, request_method: str | None) -> int | None:837 """838 Set initial length value for Response content if available.839 """840 length: int | None841 content_length: str | None = self.headers.get("content-length")842 843 if content_length is not None:844 if self.chunked:845 # This Response will fail with an IncompleteRead if it can't be846 # received as chunked. This method falls back to attempt reading847 # the response before raising an exception.848 log.warning(849 "Received response with both Content-Length and "850 "Transfer-Encoding set. This is expressly forbidden "851 "by RFC 7230 sec 3.3.2. Ignoring Content-Length and "852 "attempting to process response as Transfer-Encoding: "853 "chunked."854 )855 return None856 857 try:858 # RFC 7230 section 3.3.2 specifies multiple content lengths can859 # be sent in a single Content-Length header860 # (e.g. Content-Length: 42, 42). This line ensures the values861 # are all valid ints and that as long as the `set` length is 1,862 # all values are the same. Otherwise, the header is invalid.863 lengths = {int(val) for val in content_length.split(",")}864 if len(lengths) > 1:865 raise InvalidHeader(866 "Content-Length contained multiple "867 "unmatching values (%s)" % content_length868 )869 length = lengths.pop()870 except ValueError:871 length = None872 else:873 if length < 0:874 length = None875 876 else: # if content_length is None877 length = None878 879 # Convert status to int for comparison880 # In some cases, httplib returns a status of "_UNKNOWN"881 try:882 status = int(self.status)883 except ValueError:884 status = 0885 886 # Check for responses that shouldn't include a body887 if status in (204, 304) or 100 <= status < 200 or request_method == "HEAD":888 length = 0889 890 return length891 892 @contextmanager893 def _error_catcher(self) -> typing.Generator[None]:894 """895 Catch low-level python exceptions, instead re-raising urllib3896 variants, so that low-level exceptions are not leaked in the897 high-level api.898 899 On exit, release the connection back to the pool.900 """901 clean_exit = False902 903 try:904 try:905 yield906 907 except SocketTimeout as e:908 # FIXME: Ideally we'd like to include the url in the ReadTimeoutError but909 # there is yet no clean way to get at it from this context.910 raise ReadTimeoutError(self._pool, None, "Read timed out.") from e # type: ignore[arg-type]911 912 except BaseSSLError as e:913 # SSL errors related to framing/MAC get wrapped and reraised here914 raise SSLError(e) from e915 916 except IncompleteRead as e:917 if (918 e.expected is not None919 and e.partial is not None920 and e.expected == -e.partial921 ):922 arg = "Response may not contain content."923 else:924 arg = f"Connection broken: {e!r}"925 raise ProtocolError(arg, e) from e926 927 except (HTTPException, OSError) as e:928 raise ProtocolError(f"Connection broken: {e!r}", e) from e929 930 # If no exception is thrown, we should avoid cleaning up931 # unnecessarily.932 clean_exit = True933 finally:934 # If we didn't terminate cleanly, we need to throw away our935 # connection.936 if not clean_exit:937 # The response may not be closed but we're not going to use it938 # anymore so close it now to ensure that the connection is939 # released back to the pool.940 if self._original_response:941 self._original_response.close()942 943 # Closing the response may not actually be sufficient to close944 # everything, so if we have a hold of the connection close that945 # too.946 if self._connection:947 self._connection.close()948 949 # If we hold the original response but it's closed now, we should950 # return the connection back to the pool.951 if self._original_response and self._original_response.isclosed():952 self.release_conn()953 954 def _fp_read(955 self,956 amt: int | None = None,957 *,958 read1: bool = False,959 ) -> bytes:960 """961 Read a response with the thought that reading the number of bytes962 larger than can fit in a 32-bit int at a time via SSL in some963 known cases leads to an overflow error that has to be prevented964 if `amt` or `self.length_remaining` indicate that a problem may965 happen.966 967 This happens to urllib3 injected with pyOpenSSL-backed SSL-support.968 """969 assert self._fp970 c_int_max = 2**31 - 1971 if (972 (amt and amt > c_int_max)973 or (974 amt is None975 and self.length_remaining976 and self.length_remaining > c_int_max977 )978 ) and util.IS_PYOPENSSL:979 if read1:980 return self._fp.read1(c_int_max)981 buffer = io.BytesIO()982 # Besides `max_chunk_amt` being a maximum chunk size, it983 # affects memory overhead of reading a response by this984 # method in CPython.985 # `c_int_max` equal to 2 GiB - 1 byte is the actual maximum986 # chunk size that does not lead to an overflow error, but987 # 256 MiB is a compromise.988 max_chunk_amt = 2**28989 while amt is None or amt != 0:990 if amt is not None:991 chunk_amt = min(amt, max_chunk_amt)992 amt -= chunk_amt993 else:994 chunk_amt = max_chunk_amt995 data = self._fp.read(chunk_amt)996 if not data:997 break998 buffer.write(data)999 del data # to reduce peak memory usage by `max_chunk_amt`.1000 return buffer.getvalue()1001 elif read1:1002 return self._fp.read1(amt) if amt is not None else self._fp.read1()1003 else:1004 # StringIO doesn't like amt=None1005 return self._fp.read(amt) if amt is not None else self._fp.read()1006 1007 def _raw_read(1008 self,1009 amt: int | None = None,1010 *,1011 read1: bool = False,1012 ) -> bytes:1013 """1014 Reads `amt` of bytes from the socket.1015 """1016 if self._fp is None:1017 return None # type: ignore[return-value]1018 1019 fp_closed = getattr(self._fp, "closed", False)1020 1021 with self._error_catcher():1022 data = self._fp_read(amt, read1=read1) if not fp_closed else b""1023 if amt is not None and amt != 0 and not data:1024 # Platform-specific: Buggy versions of Python.1025 # Close the connection when no data is returned1026 #1027 # This is redundant to what httplib/http.client _should_1028 # already do. However, versions of python released before1029 # December 15, 2012 (http://bugs.python.org/issue16298) do1030 # not properly close the connection in all cases. There is1031 # no harm in redundantly calling close.1032 self._fp.close()1033 if (1034 self.enforce_content_length1035 and self.length_remaining is not None1036 and self.length_remaining != 01037 ):1038 # This is an edge case that httplib failed to cover due1039 # to concerns of backward compatibility. We're1040 # addressing it here to make sure IncompleteRead is1041 # raised during streaming, so all calls with incorrect1042 # Content-Length are caught.1043 raise IncompleteRead(self._fp_bytes_read, self.length_remaining)1044 elif read1 and (1045 (amt != 0 and not data) or self.length_remaining == len(data)1046 ):1047 # All data has been read, but `self._fp.read1` in1048 # CPython 3.12 and older doesn't always close1049 # `http.client.HTTPResponse`, so we close it here.1050 # See https://github.com/python/cpython/issues/1131991051 self._fp.close()1052 1053 if data:1054 self._fp_bytes_read += len(data)1055 if self.length_remaining is not None:1056 self.length_remaining -= len(data)1057 return data1058 1059 def read(1060 self,1061 amt: int | None = None,1062 decode_content: bool | None = None,1063 cache_content: bool = False,1064 ) -> bytes:1065 """1066 Similar to :meth:`http.client.HTTPResponse.read`, but with two additional1067 parameters: ``decode_content`` and ``cache_content``.1068 1069 :param amt:1070 How much of the content to read. If specified, caching is skipped1071 because it doesn't make sense to cache partial content as the full1072 response.1073 1074 :param decode_content:1075 If True, will attempt to decode the body based on the1076 'content-encoding' header.1077 1078 :param cache_content:1079 If True, will save the returned data such that the same result is1080 returned despite of the state of the underlying file object. This1081 is useful if you want the ``.data`` property to continue working1082 after having ``.read()`` the file object. (Overridden if ``amt`` is1083 set.)1084 """1085 self._init_decoder()1086 if decode_content is None:1087 decode_content = self.decode_content1088 1089 if amt and amt < 0:1090 # Negative numbers and `None` should be treated the same.1091 amt = None1092 elif amt is not None:1093 cache_content = False1094 1095 if (1096 self._decoder1097 and self._decoder.has_unconsumed_tail1098 and len(self._decoded_buffer) < amt1099 ):1100 decoded_data = self._decode(1101 b"",1102 decode_content,1103 flush_decoder=False,1104 max_length=amt - len(self._decoded_buffer),1105 )1106 self._decoded_buffer.put(decoded_data)1107 if len(self._decoded_buffer) >= amt:1108 return self._decoded_buffer.get(amt)1109 1110 data = self._raw_read(amt)1111 if not cache_content:1112 self._uncached_read_occurred = True1113 1114 flush_decoder = amt is None or (amt != 0 and not data)1115 1116 if (1117 not data1118 and len(self._decoded_buffer) == 01119 and not (self._decoder and self._decoder.has_unconsumed_tail)1120 ):1121 return data1122 1123 if amt is None:1124 data = self._decode(data, decode_content, flush_decoder)1125 # It's possible that there is buffered decoded data after a1126 # partial read.1127 if decode_content and len(self._decoded_buffer) > 0:1128 self._decoded_buffer.put(data)1129 data = self._decoded_buffer.get_all()1130 1131 if cache_content and not self._uncached_read_occurred:1132 self._body = data1133 else:1134 # do not waste memory on buffer when not decoding1135 if not decode_content:1136 if self._has_decoded_content:1137 raise RuntimeError(1138 "Calling read(decode_content=False) is not supported after "1139 "read(decode_content=True) was called."1140 )1141 return data1142 1143 decoded_data = self._decode(1144 data,1145 decode_content,1146 flush_decoder,1147 max_length=amt - len(self._decoded_buffer),1148 )1149 self._decoded_buffer.put(decoded_data)1150 1151 while len(self._decoded_buffer) < amt and data:1152 # TODO make sure to initially read enough data to get past the headers1153 # For example, the GZ file header takes 10 bytes, we don't want to read1154 # it one byte at a time1155 data = self._raw_read(amt)1156 decoded_data = self._decode(1157 data,1158 decode_content,1159 flush_decoder,1160 max_length=amt - len(self._decoded_buffer),1161 )1162 self._decoded_buffer.put(decoded_data)1163 data = self._decoded_buffer.get(amt)1164 1165 return data1166 1167 def read1(1168 self,1169 amt: int | None = None,1170 decode_content: bool | None = None,1171 ) -> bytes:1172 """1173 Similar to ``http.client.HTTPResponse.read1`` and documented1174 in :meth:`io.BufferedReader.read1`, but with an additional parameter:1175 ``decode_content``.1176 1177 :param amt:1178 How much of the content to read.1179 1180 :param decode_content:1181 If True, will attempt to decode the body based on the1182 'content-encoding' header.1183 """1184 if decode_content is None:1185 decode_content = self.decode_content1186 if amt and amt < 0:1187 # Negative numbers and `None` should be treated the same.1188 amt = None1189 # try and respond without going to the network1190 if self._has_decoded_content:1191 if not decode_content:1192 raise RuntimeError(1193 "Calling read1(decode_content=False) is not supported after "1194 "read1(decode_content=True) was called."1195 )1196 if (1197 self._decoder1198 and self._decoder.has_unconsumed_tail1199 and (amt is None or len(self._decoded_buffer) < amt)1200 ):