codekingpro/portable-devtools
115k
1import binascii2import json3import os4import time5import urllib.parse6import warnings7from collections.abc import Callable8from collections.abc import Iterable9from collections.abc import Iterator10from collections.abc import Mapping11from collections.abc import Sequence12from dataclasses import dataclass13from dataclasses import fields14from email.utils import formatdate15from email.utils import mktime_tz16from email.utils import parsedate_tz17from typing import Any18from typing import cast19 20from mitmproxy import flow21from mitmproxy.coretypes import multidict22from mitmproxy.coretypes import serializable23from mitmproxy.net import encoding24from mitmproxy.net.http import cookies25from mitmproxy.net.http import multipart26from mitmproxy.net.http import status_codes27from mitmproxy.net.http import url28from mitmproxy.net.http.headers import assemble_content_type29from mitmproxy.net.http.headers import infer_content_encoding30from mitmproxy.net.http.headers import parse_content_type31from mitmproxy.utils import human32from mitmproxy.utils import strutils33from mitmproxy.utils import typecheck34from mitmproxy.utils.strutils import always_bytes35from mitmproxy.utils.strutils import always_str36from mitmproxy.websocket import WebSocketData37 38 39# While headers _should_ be ASCII, it's not uncommon for certain headers to be utf-8 encoded.40def _native(x: bytes) -> str:41 return x.decode("utf-8", "surrogateescape")42 43 44def _always_bytes(x: str | bytes) -> bytes:45 return strutils.always_bytes(x, "utf-8", "surrogateescape")46 47 48# This cannot be easily typed with mypy yet, so we just specify MultiDict without concrete types.49class Headers(multidict.MultiDict): # type: ignore50 """51 Header class which allows both convenient access to individual headers as well as52 direct access to the underlying raw data. Provides a full dictionary interface.53 54 Create headers with keyword arguments:55 >>> h = Headers(host="example.com", content_type="application/xml")56 57 Headers mostly behave like a normal dict:58 >>> h["Host"]59 "example.com"60 61 Headers are case insensitive:62 >>> h["host"]63 "example.com"64 65 Headers can also be created from a list of raw (header_name, header_value) byte tuples:66 >>> h = Headers([67 (b"Host",b"example.com"),68 (b"Accept",b"text/html"),69 (b"accept",b"application/xml")70 ])71 72 Multiple headers are folded into a single header as per RFC 7230:73 >>> h["Accept"]74 "text/html, application/xml"75 76 Setting a header removes all existing headers with the same name:77 >>> h["Accept"] = "application/text"78 >>> h["Accept"]79 "application/text"80 81 `bytes(h)` returns an HTTP/1 header block:82 >>> print(bytes(h))83 Host: example.com84 Accept: application/text85 86 For full control, the raw header fields can be accessed:87 >>> h.fields88 89 Caveats:90 - For use with the "Set-Cookie" and "Cookie" headers, either use `Response.cookies` or see `Headers.get_all`.91 """92 93 def __init__(self, fields: Iterable[tuple[bytes, bytes]] = (), **headers):94 """95 *Args:*96 - *fields:* (optional) list of ``(name, value)`` header byte tuples,97 e.g. ``[(b"Host", b"example.com")]``. All names and values must be bytes.98 - *\\*\\*headers:* Additional headers to set. Will overwrite existing values from `fields`.99 For convenience, underscores in header names will be transformed to dashes -100 this behaviour does not extend to other methods.101 102 If ``**headers`` contains multiple keys that have equal ``.lower()`` representations,103 the behavior is undefined.104 """105 super().__init__(fields)106 107 for key, value in self.fields:108 if not isinstance(key, bytes) or not isinstance(value, bytes):109 raise TypeError("Header fields must be bytes.")110 111 # content_type -> content-type112 self.update(113 {114 _always_bytes(name).replace(b"_", b"-"): _always_bytes(value)115 for name, value in headers.items()116 }117 )118 119 fields: tuple[tuple[bytes, bytes], ...]120 121 @staticmethod122 def _reduce_values(values) -> str:123 # Headers can be folded124 return ", ".join(values)125 126 @staticmethod127 def _kconv(key) -> str:128 # Headers are case-insensitive129 return key.lower()130 131 def __bytes__(self) -> bytes:132 if self.fields:133 return b"\r\n".join(b": ".join(field) for field in self.fields) + b"\r\n"134 else:135 return b""136 137 def __delitem__(self, key: str | bytes) -> None:138 key = _always_bytes(key)139 super().__delitem__(key)140 141 def __iter__(self) -> Iterator[str]:142 for x in super().__iter__():143 yield _native(x)144 145 def get_all(self, name: str | bytes) -> list[str]:146 """147 Like `Headers.get`, but does not fold multiple headers into a single one.148 This is useful for Set-Cookie and Cookie headers, which do not support folding.149 150 *See also:*151 - <https://tools.ietf.org/html/rfc7230#section-3.2.2>152 - <https://datatracker.ietf.org/doc/html/rfc6265#section-5.4>153 - <https://datatracker.ietf.org/doc/html/rfc7540#section-8.1.2.5>154 """155 name = _always_bytes(name)156 return [_native(x) for x in super().get_all(name)]157 158 def set_all(self, name: str | bytes, values: Iterable[str | bytes]):159 """160 Explicitly set multiple headers for the given key.161 See `Headers.get_all`.162 """163 name = _always_bytes(name)164 values = [_always_bytes(x) for x in values]165 return super().set_all(name, values)166 167 def insert(self, index: int, key: str | bytes, value: str | bytes):168 key = _always_bytes(key)169 value = _always_bytes(value)170 super().insert(index, key, value)171 172 def items(self, multi=False):173 if multi:174 return ((_native(k), _native(v)) for k, v in self.fields)175 else:176 return super().items()177 178 179@dataclass180class MessageData(serializable.Serializable):181 http_version: bytes182 headers: Headers183 content: bytes | None184 trailers: Headers | None185 timestamp_start: float186 timestamp_end: float | None187 188 # noinspection PyUnreachableCode189 if __debug__:190 191 def __post_init__(self):192 for field in fields(self):193 val = getattr(self, field.name)194 typecheck.check_option_type(field.name, val, field.type)195 196 def set_state(self, state):197 for k, v in state.items():198 if k in ("headers", "trailers") and v is not None:199 v = Headers.from_state(v)200 setattr(self, k, v)201 202 def get_state(self):203 state = vars(self).copy()204 state["headers"] = state["headers"].get_state()205 if state["trailers"] is not None:206 state["trailers"] = state["trailers"].get_state()207 return state208 209 @classmethod210 def from_state(cls, state):211 state["headers"] = Headers.from_state(state["headers"])212 if state["trailers"] is not None:213 state["trailers"] = Headers.from_state(state["trailers"])214 return cls(**state)215 216 217@dataclass218class RequestData(MessageData):219 host: str220 port: int221 method: bytes222 scheme: bytes223 authority: bytes224 path: bytes225 226 227@dataclass228class ResponseData(MessageData):229 status_code: int230 reason: bytes231 232 233class Message(serializable.Serializable):234 """Base class for `Request` and `Response`."""235 236 @classmethod237 def from_state(cls, state):238 return cls(**state)239 240 def get_state(self):241 return self.data.get_state()242 243 def set_state(self, state):244 self.data.set_state(state)245 246 data: MessageData247 stream: Callable[[bytes], Iterable[bytes] | bytes] | bool = False248 """249 This attribute controls if the message body should be streamed.250 251 If `False`, mitmproxy will buffer the entire body before forwarding it to the destination.252 This makes it possible to perform string replacements on the entire body.253 If `True`, the message body will not be buffered on the proxy254 but immediately forwarded instead.255 Alternatively, a transformation function can be specified, which will be called for each chunk of data.256 Please note that packet boundaries generally should not be relied upon.257 258 This attribute must be set in the `requestheaders` or `responseheaders` hook.259 Setting it in `request` or `response` is already too late, mitmproxy has buffered the message body already.260 """261 262 @property263 def http_version(self) -> str:264 """265 HTTP version string, for example `HTTP/1.1`.266 """267 return self.data.http_version.decode("utf-8", "surrogateescape")268 269 @http_version.setter270 def http_version(self, http_version: str | bytes) -> None:271 self.data.http_version = strutils.always_bytes(272 http_version, "utf-8", "surrogateescape"273 )274 275 @property276 def is_http10(self) -> bool:277 return self.data.http_version == b"HTTP/1.0"278 279 @property280 def is_http11(self) -> bool:281 return self.data.http_version == b"HTTP/1.1"282 283 @property284 def is_http2(self) -> bool:285 return self.data.http_version == b"HTTP/2.0"286 287 @property288 def is_http3(self) -> bool:289 return self.data.http_version == b"HTTP/3"290 291 @property292 def headers(self) -> Headers:293 """294 The HTTP headers.295 """296 return self.data.headers297 298 @headers.setter299 def headers(self, h: Headers) -> None:300 self.data.headers = h301 302 @property303 def trailers(self) -> Headers | None:304 """305 The [HTTP trailers](https://developer.mozilla.org/en-US/docs/Web/HTTP/Headers/Trailer).306 """307 return self.data.trailers308 309 @trailers.setter310 def trailers(self, h: Headers | None) -> None:311 self.data.trailers = h312 313 @property314 def raw_content(self) -> bytes | None:315 """316 The raw (potentially compressed) HTTP message body.317 318 In contrast to `Message.content` and `Message.text`, accessing this property never raises.319 `raw_content` may be `None` if the content is missing, for example due to body streaming320 (see `Message.stream`). In contrast, `b""` signals a present but empty message body.321 322 *See also:* `Message.content`, `Message.text`323 """324 return self.data.content325 326 @raw_content.setter327 def raw_content(self, content: bytes | None) -> None:328 self.data.content = content329 330 @property331 def content(self) -> bytes | None:332 """333 The uncompressed HTTP message body as bytes.334 335 Accessing this attribute may raise a `ValueError` when the HTTP content-encoding is invalid.336 337 *See also:* `Message.raw_content`, `Message.text`338 """339 return self.get_content()340 341 @content.setter342 def content(self, value: bytes | None) -> None:343 self.set_content(value)344 345 @property346 def text(self) -> str | None:347 """348 The uncompressed and decoded HTTP message body as text.349 350 Accessing this attribute may raise a `ValueError` when either content-encoding or charset is invalid.351 352 *See also:* `Message.raw_content`, `Message.content`353 """354 return self.get_text()355 356 @text.setter357 def text(self, value: str | None) -> None:358 self.set_text(value)359 360 def set_content(self, value: bytes | None) -> None:361 if value is None:362 self.raw_content = None363 return364 if not isinstance(value, bytes):365 raise TypeError(366 f"Message content must be bytes, not {type(value).__name__}. "367 "Please use .text if you want to assign a str."368 )369 ce = self.headers.get("content-encoding")370 try:371 self.raw_content = encoding.encode(value, ce or "identity")372 except ValueError:373 # So we have an invalid content-encoding?374 # Let's remove it!375 del self.headers["content-encoding"]376 self.raw_content = value377 378 if "transfer-encoding" in self.headers:379 # https://httpwg.org/specs/rfc7230.html#header.content-length380 # don't set content-length if a transfer-encoding is provided381 pass382 else:383 self.headers["content-length"] = str(len(self.raw_content))384 385 def get_content(self, strict: bool = True) -> bytes | None:386 """387 Similar to `Message.content`, but does not raise if `strict` is `False`.388 Instead, the compressed message body is returned as-is.389 """390 if self.raw_content is None:391 return None392 ce = self.headers.get("content-encoding")393 if ce:394 try:395 content = encoding.decode(self.raw_content, ce)396 # A client may illegally specify a byte -> str encoding here (e.g. utf8)397 if isinstance(content, str):398 raise ValueError(f"Invalid Content-Encoding: {ce}")399 return content400 except ValueError:401 if strict:402 raise403 return self.raw_content404 else:405 return self.raw_content406 407 def set_text(self, text: str | None) -> None:408 if text is None:409 self.content = None410 return411 enc = infer_content_encoding(self.headers.get("content-type", ""))412 413 try:414 self.content = cast(bytes, encoding.encode(text, enc))415 except ValueError:416 # Fall back to UTF-8 and update the content-type header.417 ct = parse_content_type(self.headers.get("content-type", "")) or (418 "text",419 "plain",420 {},421 )422 ct[2]["charset"] = "utf-8"423 self.headers["content-type"] = assemble_content_type(*ct)424 enc = "utf8"425 self.content = text.encode(enc, "surrogateescape")426 427 def get_text(self, strict: bool = True) -> str | None:428 """429 Similar to `Message.text`, but does not raise if `strict` is `False`.430 Instead, the message body is returned as surrogate-escaped UTF-8.431 """432 content = self.get_content(strict)433 if content is None:434 return None435 enc = infer_content_encoding(self.headers.get("content-type", ""), content)436 try:437 return cast(str, encoding.decode(content, enc))438 except ValueError:439 if strict:440 raise441 return content.decode("utf8", "surrogateescape")442 443 @property444 def timestamp_start(self) -> float:445 """446 *Timestamp:* Headers received.447 """448 return self.data.timestamp_start449 450 @timestamp_start.setter451 def timestamp_start(self, timestamp_start: float) -> None:452 self.data.timestamp_start = timestamp_start453 454 @property455 def timestamp_end(self) -> float | None:456 """457 *Timestamp:* Last byte received.458 """459 return self.data.timestamp_end460 461 @timestamp_end.setter462 def timestamp_end(self, timestamp_end: float | None):463 self.data.timestamp_end = timestamp_end464 465 def decode(self, strict: bool = True) -> None:466 """467 Decodes body based on the current Content-Encoding header, then468 removes the header.469 470 If the message body is missing or empty, no action is taken.471 472 *Raises:*473 - `ValueError`, when the content-encoding is invalid and strict is True.474 """475 if not self.raw_content:476 # The body is missing (for example, because of body streaming or because it's a response477 # to a HEAD request), so we can't correctly update content-length.478 return479 decoded = self.get_content(strict)480 self.headers.pop("content-encoding", None)481 self.content = decoded482 483 def encode(self, encoding: str) -> None:484 """485 Encodes body with the given encoding, where e is "gzip", "deflate", "identity", "br", or "zstd".486 Any existing content-encodings are overwritten, the content is not decoded beforehand.487 488 *Raises:*489 - `ValueError`, when the specified content-encoding is invalid.490 """491 self.headers["content-encoding"] = encoding492 self.content = self.raw_content493 if "content-encoding" not in self.headers:494 raise ValueError(f"Invalid content encoding {encoding!r}")495 496 def json(self, **kwargs: Any) -> Any:497 """498 Returns the JSON encoded content of the response, if any.499 `**kwargs` are optional arguments that will be500 passed to `json.loads()`.501 502 Will raise if the content can not be decoded and then parsed as JSON.503 504 *Raises:*505 - `json.decoder.JSONDecodeError` if content is not valid JSON.506 - `TypeError` if the content is not available, for example because the response507 has been streamed.508 """509 content = self.get_content(strict=False)510 if content is None:511 raise TypeError("Message content is not available.")512 else:513 return json.loads(content, **kwargs)514 515 516class Request(Message):517 """518 An HTTP request.519 """520 521 data: RequestData522 523 def __init__(524 self,525 host: str,526 port: int,527 method: bytes,528 scheme: bytes,529 authority: bytes,530 path: bytes,531 http_version: bytes,532 headers: Headers | tuple[tuple[bytes, bytes], ...],533 content: bytes | None,534 trailers: Headers | tuple[tuple[bytes, bytes], ...] | None,535 timestamp_start: float,536 timestamp_end: float | None,537 ):538 # auto-convert invalid types to retain compatibility with older code.539 if isinstance(host, bytes):540 host = host.decode("idna", "strict")541 if isinstance(method, str):542 method = method.encode("ascii", "strict")543 if isinstance(scheme, str):544 scheme = scheme.encode("ascii", "strict")545 if isinstance(authority, str):546 authority = authority.encode("ascii", "strict")547 if isinstance(path, str):548 path = path.encode("ascii", "strict")549 if isinstance(http_version, str):550 http_version = http_version.encode("ascii", "strict")551 552 if isinstance(content, str):553 raise ValueError(f"Content must be bytes, not {type(content).__name__}")554 if not isinstance(headers, Headers):555 headers = Headers(headers)556 if trailers is not None and not isinstance(trailers, Headers):557 trailers = Headers(trailers)558 559 self.data = RequestData(560 host=host,561 port=port,562 method=method,563 scheme=scheme,564 authority=authority,565 path=path,566 http_version=http_version,567 headers=headers,568 content=content,569 trailers=trailers,570 timestamp_start=timestamp_start,571 timestamp_end=timestamp_end,572 )573 574 def __repr__(self) -> str:575 if self.host and self.port:576 hostport = f"{self.host}:{self.port}"577 else:578 hostport = ""579 path = self.path or ""580 return f"Request({self.method} {hostport}{path})"581 582 @classmethod583 def make(584 cls,585 method: str,586 url: str,587 content: bytes | str = "",588 headers: (589 Headers | dict[str | bytes, str | bytes] | Iterable[tuple[bytes, bytes]]590 ) = (),591 ) -> "Request":592 """593 Simplified API for creating request objects.594 """595 # Headers can be list or dict, we differentiate here.596 if isinstance(headers, Headers):597 pass598 elif isinstance(headers, dict):599 headers = Headers(600 (601 always_bytes(k, "utf-8", "surrogateescape"),602 always_bytes(v, "utf-8", "surrogateescape"),603 )604 for k, v in headers.items()605 )606 elif isinstance(headers, Iterable):607 headers = Headers(headers) # type: ignore608 else:609 raise TypeError(610 "Expected headers to be an iterable or dict, but is {}.".format(611 type(headers).__name__612 )613 )614 615 req = cls(616 "",617 0,618 method.encode("utf-8", "surrogateescape"),619 b"",620 b"",621 b"",622 b"HTTP/1.1",623 headers,624 b"",625 None,626 time.time(),627 time.time(),628 )629 630 req.url = url631 # Assign this manually to update the content-length header.632 if isinstance(content, bytes):633 req.content = content634 elif isinstance(content, str):635 req.text = content636 else:637 raise TypeError(638 f"Expected content to be str or bytes, but is {type(content).__name__}."639 )640 641 return req642 643 @property644 def first_line_format(self) -> str:645 """646 *Read-only:* HTTP request form as defined in [RFC 7230](https://tools.ietf.org/html/rfc7230#section-5.3).647 648 origin-form and asterisk-form are subsumed as "relative".649 """650 if self.method == "CONNECT":651 return "authority"652 elif self.authority:653 return "absolute"654 else:655 return "relative"656 657 @property658 def method(self) -> str:659 """660 HTTP request method, e.g. "GET".661 """662 return self.data.method.decode("utf-8", "surrogateescape").upper()663 664 @method.setter665 def method(self, val: str | bytes) -> None:666 self.data.method = always_bytes(val, "utf-8", "surrogateescape")667 668 @property669 def scheme(self) -> str:670 """671 HTTP request scheme, which should be "http" or "https".672 """673 return self.data.scheme.decode("utf-8", "surrogateescape")674 675 @scheme.setter676 def scheme(self, val: str | bytes) -> None:677 self.data.scheme = always_bytes(val, "utf-8", "surrogateescape")678 679 @property680 def authority(self) -> str:681 """682 HTTP request authority.683 684 For HTTP/1, this is the authority portion of the request target685 (in either absolute-form or authority-form).686 For origin-form and asterisk-form requests, this property is set to an empty string.687 688 For HTTP/2, this is the :authority pseudo header.689 690 *See also:* `Request.host`, `Request.host_header`, `Request.pretty_host`691 """692 try:693 return self.data.authority.decode("idna")694 except UnicodeError:695 return self.data.authority.decode("utf8", "surrogateescape")696 697 @authority.setter698 def authority(self, val: str | bytes) -> None:699 if isinstance(val, str):700 try:701 val = val.encode("idna", "strict")702 except UnicodeError:703 val = val.encode("utf8", "surrogateescape") # type: ignore704 self.data.authority = val705 706 @property707 def host(self) -> str:708 """709 Target server for this request. This may be parsed from the raw request710 (e.g. from a ``GET http://example.com/ HTTP/1.1`` request line)711 or inferred from the proxy mode (e.g. an IP in transparent mode).712 713 Setting the host attribute also updates the host header and authority information, if present.714 715 *See also:* `Request.authority`, `Request.host_header`, `Request.pretty_host`716 """717 return self.data.host718 719 @host.setter720 def host(self, val: str | bytes) -> None:721 self.data.host = always_str(val, "idna", "strict")722 self._update_host_and_authority()723 724 @property725 def host_header(self) -> str | None:726 """727 The request's host/authority header.728 729 This property maps to either ``request.headers["Host"]`` or730 ``request.authority``, depending on whether it's HTTP/1.x or HTTP/2.0.731 732 *See also:* `Request.authority`,`Request.host`, `Request.pretty_host`733 """734 if self.is_http2 or self.is_http3:735 return self.authority or self.data.headers.get("Host", None)736 else:737 return self.data.headers.get("Host", None)738 739 @host_header.setter740 def host_header(self, val: None | str | bytes) -> None:741 if val is None:742 if self.is_http2 or self.is_http3:743 self.data.authority = b""744 self.headers.pop("Host", None)745 else:746 if self.is_http2 or self.is_http3:747 self.authority = val # type: ignore748 if not (self.is_http2 or self.is_http3) or "Host" in self.headers:749 # For h2, we only overwrite, but not create, as :authority is the h2 host header.750 self.headers["Host"] = val751 752 @property753 def port(self) -> int:754 """755 Target port.756 """757 return self.data.port758 759 @port.setter760 def port(self, port: int) -> None:761 if not isinstance(port, int):762 raise ValueError(f"Port must be an integer, not {port!r}.")763 764 self.data.port = port765 self._update_host_and_authority()766 767 def _update_host_and_authority(self) -> None:768 val = url.hostport(self.scheme, self.host, self.port)769 770 # Update host header771 if "Host" in self.data.headers:772 self.data.headers["Host"] = val773 # Update authority774 if self.data.authority:775 self.authority = val776 777 @property778 def path(self) -> str:779 """780 HTTP request path, e.g. "/index.html" or "/index.html?a=b".781 Usually starts with a slash, except for OPTIONS requests, which may just be "*".782 783 This attribute includes both path and query parts of the target URI784 (see Sections 3.3 and 3.4 of [RFC3986](https://datatracker.ietf.org/doc/html/rfc3986)).785 """786 return self.data.path.decode("utf-8", "surrogateescape")787 788 @path.setter789 def path(self, val: str | bytes) -> None:790 self.data.path = always_bytes(val, "utf-8", "surrogateescape")791 792 @property793 def url(self) -> str:794 """795 The full URL string, constructed from `Request.scheme`, `Request.host`, `Request.port` and `Request.path`.796 797 Settings this property updates these attributes as well.798 """799 if self.first_line_format == "authority":800 return f"{self.host}:{self.port}"801 path = self.path if self.path != "*" else ""802 return url.unparse(self.scheme, self.host, self.port, path)803 804 @url.setter805 def url(self, val: str | bytes) -> None:806 val = always_str(val, "utf-8", "surrogateescape")807 self.scheme, self.host, self.port, self.path = url.parse(val) # type: ignore808 809 @property810 def pretty_host(self) -> str:811 """812 *Read-only:* Like `Request.host`, but using `Request.host_header` header as an additional (preferred) data source.813 This is useful in transparent mode where `Request.host` is only an IP address.814 815 *Warning:* When working in adversarial environments, this may not reflect the actual destination816 as the Host header could be spoofed.817 """818 authority = self.host_header819 if authority:820 return url.parse_authority(authority, check=False)[0]821 else:822 return self.host823 824 @property825 def pretty_url(self) -> str:826 """827 *Read-only:* Like `Request.url`, but using `Request.pretty_host` instead of `Request.host`.828 """829 if self.first_line_format == "authority":830 return self.authority831 832 host_header = self.host_header833 if not host_header:834 return self.url835 836 pretty_host, pretty_port = url.parse_authority(host_header, check=False)837 pretty_port = pretty_port or url.default_port(self.scheme) or 443838 path = self.path if self.path != "*" else ""839 840 return url.unparse(self.scheme, pretty_host, pretty_port, path)841 842 def _get_query(self):843 query = urllib.parse.urlparse(self.url).query844 return tuple(url.decode(query))845 846 def _set_query(self, query_data):847 query = url.encode(query_data)848 _, _, path, params, _, fragment = urllib.parse.urlparse(self.url)849 self.path = urllib.parse.urlunparse(["", "", path, params, query, fragment])850 851 @property852 def query(self) -> multidict.MultiDictView[str, str]:853 """854 The request query as a mutable mapping view on the request's path.855 For the most part, this behaves like a dictionary.856 Modifications to the MultiDictView update `Request.path`, and vice versa.857 """858 return multidict.MultiDictView(self._get_query, self._set_query)859 860 @query.setter861 def query(self, value):862 self._set_query(value)863 864 def _get_cookies(self):865 h = self.headers.get_all("Cookie")866 return tuple(cookies.parse_cookie_headers(h))867 868 def _set_cookies(self, value):869 self.headers["cookie"] = cookies.format_cookie_header(value)870 871 @property872 def cookies(self) -> multidict.MultiDictView[str, str]:873 """874 The request cookies.875 For the most part, this behaves like a dictionary.876 Modifications to the MultiDictView update `Request.headers`, and vice versa.877 """878 return multidict.MultiDictView(self._get_cookies, self._set_cookies)879 880 @cookies.setter881 def cookies(self, value):882 self._set_cookies(value)883 884 @property885 def path_components(self) -> tuple[str, ...]:886 """887 The URL's path components as a tuple of strings.888 Components are unquoted.889 """890 path = urllib.parse.urlparse(self.url).path891 # This needs to be a tuple so that it's immutable.892 # Otherwise, this would fail silently:893 # request.path_components.append("foo")894 return tuple(url.unquote(i) for i in path.split("/") if i)895 896 @path_components.setter897 def path_components(self, components: Iterable[str]):898 components = map(lambda x: url.quote(x, safe=""), components)899 path = "/" + "/".join(components)900 _, _, _, params, query, fragment = urllib.parse.urlparse(self.url)901 self.path = urllib.parse.urlunparse(["", "", path, params, query, fragment])902 903 def anticache(self) -> None:904 """905 Modifies this request to remove headers that might produce a cached response.906 """907 delheaders = (908 "if-modified-since",909 "if-none-match",910 )911 for i in delheaders:912 self.headers.pop(i, None)913 914 def anticomp(self) -> None:915 """916 Modify the Accept-Encoding header to only accept uncompressed responses.917 """918 self.headers["accept-encoding"] = "identity"919 920 def constrain_encoding(self) -> None:921 """922 Limits the permissible Accept-Encoding values, based on what we can decode appropriately.923 """924 accept_encoding = self.headers.get("accept-encoding")925 if accept_encoding:926 self.headers["accept-encoding"] = ", ".join(927 e928 for e in {"gzip", "identity", "deflate", "br", "zstd"}929 if e in accept_encoding930 )931 932 def _get_urlencoded_form(self):933 is_valid_content_type = (934 "application/x-www-form-urlencoded"935 in self.headers.get("content-type", "").lower()936 )937 if is_valid_content_type:938 return tuple(url.decode(self.get_text(strict=False)))939 return ()940 941 def _set_urlencoded_form(self, form_data: Sequence[tuple[str, str]]) -> None:942 """943 Sets the body to the URL-encoded form data, and adds the appropriate content-type header.944 This will overwrite the existing content if there is one.945 """946 self.headers["content-type"] = "application/x-www-form-urlencoded"947 self.content = url.encode(form_data, self.get_text(strict=False)).encode()948 949 @property950 def urlencoded_form(self) -> multidict.MultiDictView[str, str]:951 """952 The URL-encoded form data.953 954 If the content-type indicates non-form data or the form could not be parsed, this is set to955 an empty `MultiDictView`.956 957 Modifications to the MultiDictView update `Request.content`, and vice versa.958 """959 return multidict.MultiDictView(960 self._get_urlencoded_form, self._set_urlencoded_form961 )962 963 @urlencoded_form.setter964 def urlencoded_form(self, value):965 self._set_urlencoded_form(value)966 967 def _get_multipart_form(self) -> list[tuple[bytes, bytes]]:968 is_valid_content_type = (969 "multipart/form-data" in self.headers.get("content-type", "").lower()970 )971 if is_valid_content_type and self.content is not None:972 try:973 return multipart.decode_multipart(974 self.headers.get("content-type"), self.content975 )976 except ValueError:977 pass978 return []979 980 def _set_multipart_form(self, value: list[tuple[bytes, bytes]]) -> None:981 ct = self.headers.get("content-type", "")982 is_valid_content_type = ct.lower().startswith("multipart/form-data")983 if not is_valid_content_type:984 """985 Generate a random boundary here.986 987 See <https://datatracker.ietf.org/doc/html/rfc2046#section-5.1.1> for specifications988 on generating the boundary.989 """990 boundary = "-" * 20 + binascii.hexlify(os.urandom(16)).decode()991 self.headers["content-type"] = ct = f"multipart/form-data; {boundary=!s}"992 self.content = multipart.encode_multipart(ct, value)993 994 @property995 def multipart_form(self) -> multidict.MultiDictView[bytes, bytes]:996 """997 The multipart form data.998 999 If the content-type indicates non-form data or the form could not be parsed, this is set to1000 an empty `MultiDictView`.1001 1002 Modifications to the MultiDictView update `Request.content`, and vice versa.1003 """1004 return multidict.MultiDictView(1005 self._get_multipart_form, self._set_multipart_form1006 )1007 1008 @multipart_form.setter1009 def multipart_form(self, value: list[tuple[bytes, bytes]]) -> None:1010 self._set_multipart_form(value)1011 1012 1013class Response(Message):1014 """1015 An HTTP response.1016 """1017 1018 data: ResponseData1019 1020 def __init__(1021 self,1022 http_version: bytes,1023 status_code: int,1024 reason: bytes,1025 headers: Headers | tuple[tuple[bytes, bytes], ...],1026 content: bytes | None,1027 trailers: None | Headers | tuple[tuple[bytes, bytes], ...],1028 timestamp_start: float,1029 timestamp_end: float | None,1030 ):1031 # auto-convert invalid types to retain compatibility with older code.1032 if isinstance(http_version, str):1033 http_version = http_version.encode("ascii", "strict")1034 if isinstance(reason, str):1035 reason = reason.encode("ascii", "strict")1036 1037 if isinstance(content, str):1038 raise ValueError(f"Content must be bytes, not {type(content).__name__}")1039 if not isinstance(headers, Headers):1040 headers = Headers(headers)1041 if trailers is not None and not isinstance(trailers, Headers):1042 trailers = Headers(trailers)1043 1044 self.data = ResponseData(1045 http_version=http_version,1046 status_code=status_code,1047 reason=reason,1048 headers=headers,1049 content=content,1050 trailers=trailers,1051 timestamp_start=timestamp_start,1052 timestamp_end=timestamp_end,1053 )1054 1055 def __repr__(self) -> str:1056 if self.raw_content:1057 ct = self.headers.get("content-type", "unknown content type")1058 size = human.pretty_size(len(self.raw_content))1059 details = f"{ct}, {size}"1060 else:1061 details = "no content"1062 return f"Response({self.status_code}, {details})"1063 1064 @classmethod1065 def make(1066 cls,1067 status_code: int = 200,1068 content: bytes | str = b"",1069 headers: (1070 Headers | Mapping[str, str | bytes] | Iterable[tuple[bytes, bytes]]1071 ) = (),1072 ) -> "Response":1073 """1074 Simplified API for creating response objects.1075 """1076 if isinstance(headers, Headers):1077 headers = headers1078 elif isinstance(headers, dict):1079 headers = Headers(1080 (1081 always_bytes(k, "utf-8", "surrogateescape"), # type: ignore1082 always_bytes(v, "utf-8", "surrogateescape"),1083 )1084 for k, v in headers.items()1085 )1086 elif isinstance(headers, Iterable):1087 headers = Headers(headers) # type: ignore1088 else:1089 raise TypeError(1090 "Expected headers to be an iterable or dict, but is {}.".format(1091 type(headers).__name__1092 )1093 )1094 1095 resp = cls(1096 b"HTTP/1.1",1097 status_code,1098 status_codes.RESPONSES.get(status_code, "").encode(),1099 headers,1100 None,1101 None,1102 time.time(),1103 time.time(),1104 )1105 1106 # Assign this manually to update the content-length header.1107 if isinstance(content, bytes):1108 resp.content = content1109 elif isinstance(content, str):1110 resp.text = content1111 else:1112 raise TypeError(1113 f"Expected content to be str or bytes, but is {type(content).__name__}."1114 )1115 1116 return resp1117 1118 @property1119 def status_code(self) -> int:1120 """1121 HTTP Status Code, e.g. ``200``.1122 """1123 return self.data.status_code1124 1125 @status_code.setter1126 def status_code(self, status_code: int) -> None:1127 self.data.status_code = status_code1128 1129 @property1130 def reason(self) -> str:1131 """1132 HTTP reason phrase, for example "Not Found".1133 1134 HTTP/2 responses do not contain a reason phrase, an empty string will be returned instead.1135 """1136 # Encoding: http://stackoverflow.com/a/16674906/9347191137 return self.data.reason.decode("ISO-8859-1")1138 1139 @reason.setter1140 def reason(self, reason: str | bytes) -> None:1141 self.data.reason = strutils.always_bytes(reason, "ISO-8859-1")1142 1143 def _get_cookies(self):1144 h = self.headers.get_all("set-cookie")1145 all_cookies = cookies.parse_set_cookie_headers(h)1146 return tuple((name, (value, attrs)) for name, value, attrs in all_cookies)1147 1148 def _set_cookies(self, value):1149 cookie_headers = []1150 for k, v in value:1151 header = cookies.format_set_cookie_header([(k, v[0], v[1])])1152 cookie_headers.append(header)1153 self.headers.set_all("set-cookie", cookie_headers)1154 1155 @property1156 def cookies(1157 self,1158 ) -> multidict.MultiDictView[str, tuple[str, multidict.MultiDict[str, str | None]]]:1159 """1160 The response cookies. A possibly empty `MultiDictView`, where the keys are cookie1161 name strings, and values are `(cookie value, attributes)` tuples. Within1162 attributes, unary attributes (e.g. `HTTPOnly`) are indicated by a `None` value.1163 Modifications to the MultiDictView update `Response.headers`, and vice versa.1164 1165 *Warning:* Changes to `attributes` will not be picked up unless you also reassign1166 the `(cookie value, attributes)` tuple directly in the `MultiDictView`.1167 """1168 return multidict.MultiDictView(self._get_cookies, self._set_cookies)1169 1170 @cookies.setter1171 def cookies(self, value):1172 self._set_cookies(value)1173 1174 def refresh(self, now=None):1175 """1176 This fairly complex and heuristic function refreshes a server1177 response for replay.1178 1179 - It adjusts date, expires, and last-modified headers.1180 - It adjusts cookie expiration.1181 """1182 if not now:1183 now = time.time()1184 delta = now - self.timestamp_start1185 refresh_headers = [1186 "date",1187 "expires",1188 "last-modified",1189 ]1190 for i in refresh_headers:1191 if i in self.headers:1192 d = parsedate_tz(self.headers[i])1193 if d:1194 new = mktime_tz(d) + delta1195 try:1196 self.headers[i] = formatdate(new, usegmt=True)1197 except OSError: # pragma: no cover1198 pass # value out of bounds on Windows only (which is why we exclude it from coverage).1199 c = []1200 for set_cookie_header in self.headers.get_all("set-cookie"):