codekingpro/portable-devtools
114k
1r"""HTTP/1.1 client library2 3<intro stuff goes here>4<other stuff, too>5 6HTTPConnection goes through a number of "states", which define when a client7may legally make another request or fetch the response for a particular8request. This diagram details these state transitions:9 10 (null)11 |12 | HTTPConnection()13 v14 Idle15 |16 | putrequest()17 v18 Request-started19 |20 | ( putheader() )* endheaders()21 v22 Request-sent23 |\_____________________________24 | | getresponse() raises25 | response = getresponse() | ConnectionError26 v v27 Unread-response Idle28 [Response-headers-read]29 |\____________________30 | |31 | response.read() | putrequest()32 v v33 Idle Req-started-unread-response34 ______/|35 / |36 response.read() | | ( putheader() )* endheaders()37 v v38 Request-started Req-sent-unread-response39 |40 | response.read()41 v42 Request-sent43 44This diagram presents the following rules:45 -- a second request may not be started until {response-headers-read}46 -- a response [object] cannot be retrieved until {request-sent}47 -- there is no differentiation between an unread response body and a48 partially read response body49 50Note: this enforcement is applied by the HTTPConnection class. The51 HTTPResponse class does not enforce this state machine, which52 implies sophisticated clients may accelerate the request/response53 pipeline. Caution should be taken, though: accelerating the states54 beyond the above pattern may imply knowledge of the server's55 connection-close behavior for certain requests. For example, it56 is impossible to tell whether the server will close the connection57 UNTIL the response headers have been read; this means that further58 requests cannot be placed into the pipeline until it is known that59 the server will NOT be closing the connection.60 61Logical State __state __response62------------- ------- ----------63Idle _CS_IDLE None64Request-started _CS_REQ_STARTED None65Request-sent _CS_REQ_SENT None66Unread-response _CS_IDLE <response_class>67Req-started-unread-response _CS_REQ_STARTED <response_class>68Req-sent-unread-response _CS_REQ_SENT <response_class>69"""70 71import email.parser72import email.message73import errno74import http75import io76import re77import socket78import sys79import collections.abc80from urllib.parse import urlsplit81 82# HTTPMessage, parse_headers(), and the HTTP status code constants are83# intentionally omitted for simplicity84__all__ = ["HTTPResponse", "HTTPConnection",85 "HTTPException", "NotConnected", "UnknownProtocol",86 "UnknownTransferEncoding", "UnimplementedFileMode",87 "IncompleteRead", "InvalidURL", "ImproperConnectionState",88 "CannotSendRequest", "CannotSendHeader", "ResponseNotReady",89 "BadStatusLine", "LineTooLong", "RemoteDisconnected", "error",90 "responses"]91 92HTTP_PORT = 8093HTTPS_PORT = 44394 95_UNKNOWN = 'UNKNOWN'96 97# connection states98_CS_IDLE = 'Idle'99_CS_REQ_STARTED = 'Request-started'100_CS_REQ_SENT = 'Request-sent'101 102 103# hack to maintain backwards compatibility104globals().update(http.HTTPStatus.__members__)105 106# another hack to maintain backwards compatibility107# Mapping status codes to official W3C names108responses = {v: v.phrase for v in http.HTTPStatus.__members__.values()}109 110# maximal line length when calling readline().111_MAXLINE = 65536112_MAXHEADERS = 100113 114# Data larger than this will be read in chunks, to prevent extreme115# overallocation.116_MIN_READ_BUF_SIZE = 1 << 20117 118 119# Header name/value ABNF (http://tools.ietf.org/html/rfc7230#section-3.2)120#121# VCHAR = %x21-7E122# obs-text = %x80-FF123# header-field = field-name ":" OWS field-value OWS124# field-name = token125# field-value = *( field-content / obs-fold )126# field-content = field-vchar [ 1*( SP / HTAB ) field-vchar ]127# field-vchar = VCHAR / obs-text128#129# obs-fold = CRLF 1*( SP / HTAB )130# ; obsolete line folding131# ; see Section 3.2.4132 133# token = 1*tchar134#135# tchar = "!" / "#" / "$" / "%" / "&" / "'" / "*"136# / "+" / "-" / "." / "^" / "_" / "`" / "|" / "~"137# / DIGIT / ALPHA138# ; any VCHAR, except delimiters139#140# VCHAR defined in http://tools.ietf.org/html/rfc5234#appendix-B.1141 142# the patterns for both name and value are more lenient than RFC143# definitions to allow for backwards compatibility144_is_legal_header_name = re.compile(rb'[^:\s][^:\r\n]*').fullmatch145_is_illegal_header_value = re.compile(rb'\n(?![ \t])|\r(?![ \t\n])').search146 147# These characters are not allowed within HTTP URL paths.148# See https://tools.ietf.org/html/rfc3986#section-3.3 and the149# https://tools.ietf.org/html/rfc3986#appendix-A pchar definition.150# Prevents CVE-2019-9740. Includes control characters such as \r\n.151# We don't restrict chars above \x7f as putrequest() limits us to ASCII.152_contains_disallowed_url_pchar_re = re.compile('[\x00-\x20\x7f]')153# Arguably only these _should_ allowed:154# _is_allowed_url_pchars_re = re.compile(r"^[/!$&'()*+,;=:@%a-zA-Z0-9._~-]+$")155# We are more lenient for assumed real world compatibility purposes.156 157# These characters are not allowed within HTTP method names158# to prevent http header injection.159_contains_disallowed_method_pchar_re = re.compile('[\x00-\x1f]')160 161# We always set the Content-Length header for these methods because some162# servers will otherwise respond with a 411163_METHODS_EXPECTING_BODY = {'PATCH', 'POST', 'PUT'}164 165 166def _encode(data, name='data'):167 """Call data.encode("latin-1") but show a better error message."""168 try:169 return data.encode("latin-1")170 except UnicodeEncodeError as err:171 raise UnicodeEncodeError(172 err.encoding,173 err.object,174 err.start,175 err.end,176 "%s (%.20r) is not valid Latin-1. Use %s.encode('utf-8') "177 "if you want to send it encoded in UTF-8." %178 (name.title(), data[err.start:err.end], name)) from None179 180def _strip_ipv6_iface(enc_name: bytes) -> bytes:181 """Remove interface scope from IPv6 address."""182 enc_name, percent, _ = enc_name.partition(b"%")183 if percent:184 assert enc_name.startswith(b'['), enc_name185 enc_name += b']'186 return enc_name187 188class HTTPMessage(email.message.Message):189 # XXX The only usage of this method is in190 # http.server.CGIHTTPRequestHandler. Maybe move the code there so191 # that it doesn't need to be part of the public API. The API has192 # never been defined so this could cause backwards compatibility193 # issues.194 195 def getallmatchingheaders(self, name):196 """Find all header lines matching a given header name.197 198 Look through the list of headers and find all lines matching a given199 header name (and their continuation lines). A list of the lines is200 returned, without interpretation. If the header does not occur, an201 empty list is returned. If the header occurs multiple times, all202 occurrences are returned. Case is not important in the header name.203 204 """205 name = name.lower() + ':'206 n = len(name)207 lst = []208 hit = 0209 for line in self.keys():210 if line[:n].lower() == name:211 hit = 1212 elif not line[:1].isspace():213 hit = 0214 if hit:215 lst.append(line)216 return lst217 218def _read_headers(fp):219 """Reads potential header lines into a list from a file pointer.220 221 Length of line is limited by _MAXLINE, and number of222 headers is limited by _MAXHEADERS.223 """224 headers = []225 while True:226 line = fp.readline(_MAXLINE + 1)227 if len(line) > _MAXLINE:228 raise LineTooLong("header line")229 headers.append(line)230 if len(headers) > _MAXHEADERS:231 raise HTTPException("got more than %d headers" % _MAXHEADERS)232 if line in (b'\r\n', b'\n', b''):233 break234 return headers235 236def _parse_header_lines(header_lines, _class=HTTPMessage):237 """238 Parses only RFC 5322 headers from header lines.239 240 email Parser wants to see strings rather than bytes.241 But a TextIOWrapper around self.rfile would buffer too many bytes242 from the stream, bytes which we later need to read as bytes.243 So we read the correct bytes here, as bytes, for email Parser244 to parse.245 246 """247 hstring = b''.join(header_lines).decode('iso-8859-1')248 return email.parser.Parser(_class=_class).parsestr(hstring)249 250def parse_headers(fp, _class=HTTPMessage):251 """Parses only RFC 5322 headers from a file pointer."""252 253 headers = _read_headers(fp)254 return _parse_header_lines(headers, _class)255 256 257class HTTPResponse(io.BufferedIOBase):258 259 # See RFC 2616 sec 19.6 and RFC 1945 sec 6 for details.260 261 # The bytes from the socket object are iso-8859-1 strings.262 # See RFC 2616 sec 2.2 which notes an exception for MIME-encoded263 # text following RFC 2047. The basic status line parsing only264 # accepts iso-8859-1.265 266 def __init__(self, sock, debuglevel=0, method=None, url=None):267 # If the response includes a content-length header, we need to268 # make sure that the client doesn't read more than the269 # specified number of bytes. If it does, it will block until270 # the server times out and closes the connection. This will271 # happen if a self.fp.read() is done (without a size) whether272 # self.fp is buffered or not. So, no self.fp.read() by273 # clients unless they know what they are doing.274 self.fp = sock.makefile("rb")275 self.debuglevel = debuglevel276 self._method = method277 278 # The HTTPResponse object is returned via urllib. The clients279 # of http and urllib expect different attributes for the280 # headers. headers is used here and supports urllib. msg is281 # provided as a backwards compatibility layer for http282 # clients.283 284 self.headers = self.msg = None285 286 # from the Status-Line of the response287 self.version = _UNKNOWN # HTTP-Version288 self.status = _UNKNOWN # Status-Code289 self.reason = _UNKNOWN # Reason-Phrase290 291 self.chunked = _UNKNOWN # is "chunked" being used?292 self.chunk_left = _UNKNOWN # bytes left to read in current chunk293 self.length = _UNKNOWN # number of bytes left in response294 self.will_close = _UNKNOWN # conn will close at end of response295 296 def _read_status(self):297 line = str(self.fp.readline(_MAXLINE + 1), "iso-8859-1")298 if len(line) > _MAXLINE:299 raise LineTooLong("status line")300 if self.debuglevel > 0:301 print("reply:", repr(line))302 if not line:303 # Presumably, the server closed the connection before304 # sending a valid response.305 raise RemoteDisconnected("Remote end closed connection without"306 " response")307 try:308 version, status, reason = line.split(None, 2)309 except ValueError:310 try:311 version, status = line.split(None, 1)312 reason = ""313 except ValueError:314 # empty version will cause next test to fail.315 version = ""316 if not version.startswith("HTTP/"):317 self._close_conn()318 raise BadStatusLine(line)319 320 # The status code is a three-digit number321 try:322 status = int(status)323 if status < 100 or status > 999:324 raise BadStatusLine(line)325 except ValueError:326 raise BadStatusLine(line)327 return version, status, reason328 329 def begin(self):330 if self.headers is not None:331 # we've already started reading the response332 return333 334 # read until we get a non-100 response335 while True:336 version, status, reason = self._read_status()337 if status != CONTINUE:338 break339 # skip the header from the 100 response340 skipped_headers = _read_headers(self.fp)341 if self.debuglevel > 0:342 print("headers:", skipped_headers)343 del skipped_headers344 345 self.code = self.status = status346 self.reason = reason.strip()347 if version in ("HTTP/1.0", "HTTP/0.9"):348 # Some servers might still return "0.9", treat it as 1.0 anyway349 self.version = 10350 elif version.startswith("HTTP/1."):351 self.version = 11 # use HTTP/1.1 code for HTTP/1.x where x>=1352 else:353 raise UnknownProtocol(version)354 355 self.headers = self.msg = parse_headers(self.fp)356 357 if self.debuglevel > 0:358 for hdr, val in self.headers.items():359 print("header:", hdr + ":", val)360 361 # are we using the chunked-style of transfer encoding?362 tr_enc = self.headers.get("transfer-encoding")363 if tr_enc and tr_enc.lower() == "chunked":364 self.chunked = True365 self.chunk_left = None366 else:367 self.chunked = False368 369 # will the connection close at the end of the response?370 self.will_close = self._check_close()371 372 # do we have a Content-Length?373 # NOTE: RFC 2616, S4.4, #3 says we ignore this if tr_enc is "chunked"374 self.length = None375 length = self.headers.get("content-length")376 if length and not self.chunked:377 try:378 self.length = int(length)379 except ValueError:380 self.length = None381 else:382 if self.length < 0: # ignore nonsensical negative lengths383 self.length = None384 else:385 self.length = None386 387 # does the body have a fixed length? (of zero)388 if (status == NO_CONTENT or status == NOT_MODIFIED or389 100 <= status < 200 or # 1xx codes390 self._method == "HEAD"):391 self.length = 0392 393 # if the connection remains open, and we aren't using chunked, and394 # a content-length was not provided, then assume that the connection395 # WILL close.396 if (not self.will_close and397 not self.chunked and398 self.length is None):399 self.will_close = True400 401 def _check_close(self):402 conn = self.headers.get("connection")403 if self.version == 11:404 # An HTTP/1.1 proxy is assumed to stay open unless405 # explicitly closed.406 if conn and "close" in conn.lower():407 return True408 return False409 410 # Some HTTP/1.0 implementations have support for persistent411 # connections, using rules different than HTTP/1.1.412 413 # For older HTTP, Keep-Alive indicates persistent connection.414 if self.headers.get("keep-alive"):415 return False416 417 # At least Akamai returns a "Connection: Keep-Alive" header,418 # which was supposed to be sent by the client.419 if conn and "keep-alive" in conn.lower():420 return False421 422 # Proxy-Connection is a netscape hack.423 pconn = self.headers.get("proxy-connection")424 if pconn and "keep-alive" in pconn.lower():425 return False426 427 # otherwise, assume it will close428 return True429 430 def _close_conn(self):431 fp = self.fp432 self.fp = None433 fp.close()434 435 def close(self):436 try:437 super().close() # set "closed" flag438 finally:439 if self.fp:440 self._close_conn()441 442 # These implementations are for the benefit of io.BufferedReader.443 444 # XXX This class should probably be revised to act more like445 # the "raw stream" that BufferedReader expects.446 447 def flush(self):448 super().flush()449 if self.fp:450 self.fp.flush()451 452 def readable(self):453 """Always returns True"""454 return True455 456 # End of "raw stream" methods457 458 def isclosed(self):459 """True if the connection is closed."""460 # NOTE: it is possible that we will not ever call self.close(). This461 # case occurs when will_close is TRUE, length is None, and we462 # read up to the last byte, but NOT past it.463 #464 # IMPLIES: if will_close is FALSE, then self.close() will ALWAYS be465 # called, meaning self.isclosed() is meaningful.466 return self.fp is None467 468 def read(self, amt=None):469 """Read and return the response body, or up to the next amt bytes."""470 if self.fp is None:471 return b""472 473 if self._method == "HEAD":474 self._close_conn()475 return b""476 477 if self.chunked:478 return self._read_chunked(amt)479 480 if amt is not None and amt >= 0:481 if self.length is not None and amt > self.length:482 # clip the read to the "end of response"483 amt = self.length484 s = self.fp.read(amt)485 if not s and amt:486 # Ideally, we would raise IncompleteRead if the content-length487 # wasn't satisfied, but it might break compatibility.488 self._close_conn()489 elif self.length is not None:490 self.length -= len(s)491 if not self.length:492 self._close_conn()493 return s494 else:495 # Amount is not given (unbounded read) so we must check self.length496 if self.length is None:497 s = self.fp.read()498 else:499 try:500 s = self._safe_read(self.length)501 except IncompleteRead:502 self._close_conn()503 raise504 self.length = 0505 self._close_conn() # we read everything506 return s507 508 def readinto(self, b):509 """Read up to len(b) bytes into bytearray b and return the number510 of bytes read.511 """512 513 if self.fp is None:514 return 0515 516 if self._method == "HEAD":517 self._close_conn()518 return 0519 520 if self.chunked:521 return self._readinto_chunked(b)522 523 if self.length is not None:524 if len(b) > self.length:525 # clip the read to the "end of response"526 b = memoryview(b)[0:self.length]527 528 # we do not use _safe_read() here because this may be a .will_close529 # connection, and the user is reading more bytes than will be provided530 # (for example, reading in 1k chunks)531 n = self.fp.readinto(b)532 if not n and b:533 # Ideally, we would raise IncompleteRead if the content-length534 # wasn't satisfied, but it might break compatibility.535 self._close_conn()536 elif self.length is not None:537 self.length -= n538 if not self.length:539 self._close_conn()540 return n541 542 def _read_next_chunk_size(self):543 # Read the next chunk size from the file544 line = self.fp.readline(_MAXLINE + 1)545 if len(line) > _MAXLINE:546 raise LineTooLong("chunk size")547 i = line.find(b";")548 if i >= 0:549 line = line[:i] # strip chunk-extensions550 try:551 return int(line, 16)552 except ValueError:553 # close the connection as protocol synchronisation is554 # probably lost555 self._close_conn()556 raise557 558 def _read_and_discard_trailer(self):559 # read and discard trailer up to the CRLF terminator560 ### note: we shouldn't have any trailers!561 while True:562 line = self.fp.readline(_MAXLINE + 1)563 if len(line) > _MAXLINE:564 raise LineTooLong("trailer line")565 if not line:566 # a vanishingly small number of sites EOF without567 # sending the trailer568 break569 if line in (b'\r\n', b'\n', b''):570 break571 572 def _get_chunk_left(self):573 # return self.chunk_left, reading a new chunk if necessary.574 # chunk_left == 0: at the end of the current chunk, need to close it575 # chunk_left == None: No current chunk, should read next.576 # This function returns non-zero or None if the last chunk has577 # been read.578 chunk_left = self.chunk_left579 if not chunk_left: # Can be 0 or None580 if chunk_left is not None:581 # We are at the end of chunk, discard chunk end582 self._safe_read(2) # toss the CRLF at the end of the chunk583 try:584 chunk_left = self._read_next_chunk_size()585 except ValueError:586 raise IncompleteRead(b'')587 if chunk_left == 0:588 # last chunk: 1*("0") [ chunk-extension ] CRLF589 self._read_and_discard_trailer()590 # we read everything; close the "file"591 self._close_conn()592 chunk_left = None593 self.chunk_left = chunk_left594 return chunk_left595 596 def _read_chunked(self, amt=None):597 assert self.chunked != _UNKNOWN598 if amt is not None and amt < 0:599 amt = None600 value = []601 try:602 while (chunk_left := self._get_chunk_left()) is not None:603 if amt is not None and amt <= chunk_left:604 value.append(self._safe_read(amt))605 self.chunk_left = chunk_left - amt606 break607 608 value.append(self._safe_read(chunk_left))609 if amt is not None:610 amt -= chunk_left611 self.chunk_left = 0612 return b''.join(value)613 except IncompleteRead as exc:614 raise IncompleteRead(b''.join(value)) from exc615 616 def _readinto_chunked(self, b):617 assert self.chunked != _UNKNOWN618 total_bytes = 0619 mvb = memoryview(b)620 try:621 while True:622 chunk_left = self._get_chunk_left()623 if chunk_left is None:624 return total_bytes625 626 if len(mvb) <= chunk_left:627 n = self._safe_readinto(mvb)628 self.chunk_left = chunk_left - n629 return total_bytes + n630 631 temp_mvb = mvb[:chunk_left]632 n = self._safe_readinto(temp_mvb)633 mvb = mvb[n:]634 total_bytes += n635 self.chunk_left = 0636 637 except IncompleteRead:638 raise IncompleteRead(bytes(b[0:total_bytes]))639 640 def _safe_read(self, amt):641 """Read the number of bytes requested.642 643 This function should be used when <amt> bytes "should" be present for644 reading. If the bytes are truly not available (due to EOF), then the645 IncompleteRead exception can be used to detect the problem.646 """647 cursize = min(amt, _MIN_READ_BUF_SIZE)648 data = self.fp.read(cursize)649 if len(data) >= amt:650 return data651 if len(data) < cursize:652 raise IncompleteRead(data, amt - len(data))653 654 data = io.BytesIO(data)655 data.seek(0, 2)656 while True:657 # This is a geometric increase in read size (never more than658 # doubling out the current length of data per loop iteration).659 delta = min(cursize, amt - cursize)660 data.write(self.fp.read(delta))661 if data.tell() >= amt:662 return data.getvalue()663 cursize += delta664 if data.tell() < cursize:665 raise IncompleteRead(data.getvalue(), amt - data.tell())666 667 def _safe_readinto(self, b):668 """Same as _safe_read, but for reading into a buffer."""669 amt = len(b)670 n = self.fp.readinto(b)671 if n < amt:672 raise IncompleteRead(bytes(b[:n]), amt-n)673 return n674 675 def read1(self, n=-1):676 """Read with at most one underlying system call. If at least one677 byte is buffered, return that instead.678 """679 if self.fp is None or self._method == "HEAD":680 return b""681 if self.chunked:682 return self._read1_chunked(n)683 if self.length is not None and (n < 0 or n > self.length):684 n = self.length685 result = self.fp.read1(n)686 if not result and n:687 self._close_conn()688 elif self.length is not None:689 self.length -= len(result)690 if not self.length:691 self._close_conn()692 return result693 694 def peek(self, n=-1):695 # Having this enables IOBase.readline() to read more than one696 # byte at a time697 if self.fp is None or self._method == "HEAD":698 return b""699 if self.chunked:700 return self._peek_chunked(n)701 return self.fp.peek(n)702 703 def readline(self, limit=-1):704 if self.fp is None or self._method == "HEAD":705 return b""706 if self.chunked:707 # Fallback to IOBase readline which uses peek() and read()708 return super().readline(limit)709 if self.length is not None and (limit < 0 or limit > self.length):710 limit = self.length711 result = self.fp.readline(limit)712 if not result and limit:713 self._close_conn()714 elif self.length is not None:715 self.length -= len(result)716 if not self.length:717 self._close_conn()718 return result719 720 def _read1_chunked(self, n):721 # Strictly speaking, _get_chunk_left() may cause more than one read,722 # but that is ok, since that is to satisfy the chunked protocol.723 chunk_left = self._get_chunk_left()724 if chunk_left is None or n == 0:725 return b''726 if not (0 <= n <= chunk_left):727 n = chunk_left # if n is negative or larger than chunk_left728 read = self.fp.read1(n)729 self.chunk_left -= len(read)730 if not read:731 raise IncompleteRead(b"")732 return read733 734 def _peek_chunked(self, n):735 # Strictly speaking, _get_chunk_left() may cause more than one read,736 # but that is ok, since that is to satisfy the chunked protocol.737 try:738 chunk_left = self._get_chunk_left()739 except IncompleteRead:740 return b'' # peek doesn't worry about protocol741 if chunk_left is None:742 return b'' # eof743 # peek is allowed to return more than requested. Just request the744 # entire chunk, and truncate what we get.745 return self.fp.peek(chunk_left)[:chunk_left]746 747 def fileno(self):748 return self.fp.fileno()749 750 def getheader(self, name, default=None):751 '''Returns the value of the header matching *name*.752 753 If there are multiple matching headers, the values are754 combined into a single string separated by commas and spaces.755 756 If no matching header is found, returns *default* or None if757 the *default* is not specified.758 759 If the headers are unknown, raises http.client.ResponseNotReady.760 761 '''762 if self.headers is None:763 raise ResponseNotReady()764 headers = self.headers.get_all(name) or default765 if isinstance(headers, str) or not hasattr(headers, '__iter__'):766 return headers767 else:768 return ', '.join(headers)769 770 def getheaders(self):771 """Return list of (header, value) tuples."""772 if self.headers is None:773 raise ResponseNotReady()774 return list(self.headers.items())775 776 # We override IOBase.__iter__ so that it doesn't check for closed-ness777 778 def __iter__(self):779 return self780 781 # For compatibility with old-style urllib responses.782 783 def info(self):784 '''Returns an instance of the class mimetools.Message containing785 meta-information associated with the URL.786 787 When the method is HTTP, these headers are those returned by788 the server at the head of the retrieved HTML page (including789 Content-Length and Content-Type).790 791 When the method is FTP, a Content-Length header will be792 present if (as is now usual) the server passed back a file793 length in response to the FTP retrieval request. A794 Content-Type header will be present if the MIME type can be795 guessed.796 797 When the method is local-file, returned headers will include798 a Date representing the file's last-modified time, a799 Content-Length giving file size, and a Content-Type800 containing a guess at the file's type. See also the801 description of the mimetools module.802 803 '''804 return self.headers805 806 def geturl(self):807 '''Return the real URL of the page.808 809 In some cases, the HTTP server redirects a client to another810 URL. The urlopen() function handles this transparently, but in811 some cases the caller needs to know which URL the client was812 redirected to. The geturl() method can be used to get at this813 redirected URL.814 815 '''816 return self.url817 818 def getcode(self):819 '''Return the HTTP status code that was sent with the response,820 or None if the URL is not an HTTP URL.821 822 '''823 return self.status824 825 826def _create_https_context(http_version):827 # Function also used by urllib.request to be able to set the check_hostname828 # attribute on a context object.829 context = ssl._create_default_https_context()830 # send ALPN extension to indicate HTTP/1.1 protocol831 if http_version == 11:832 context.set_alpn_protocols(['http/1.1'])833 # enable PHA for TLS 1.3 connections if available834 if context.post_handshake_auth is not None:835 context.post_handshake_auth = True836 return context837 838 839class HTTPConnection:840 841 _http_vsn = 11842 _http_vsn_str = 'HTTP/1.1'843 844 response_class = HTTPResponse845 default_port = HTTP_PORT846 auto_open = 1847 debuglevel = 0848 849 @staticmethod850 def _is_textIO(stream):851 """Test whether a file-like object is a text or a binary stream.852 """853 return isinstance(stream, io.TextIOBase)854 855 @staticmethod856 def _get_content_length(body, method):857 """Get the content-length based on the body.858 859 If the body is None, we set Content-Length: 0 for methods that expect860 a body (RFC 7230, Section 3.3.2). We also set the Content-Length for861 any method if the body is a str or bytes-like object and not a file.862 """863 if body is None:864 # do an explicit check for not None here to distinguish865 # between unset and set but empty866 if method.upper() in _METHODS_EXPECTING_BODY:867 return 0868 else:869 return None870 871 if hasattr(body, 'read'):872 # file-like object.873 return None874 875 try:876 # does it implement the buffer protocol (bytes, bytearray, array)?877 mv = memoryview(body)878 return mv.nbytes879 except TypeError:880 pass881 882 if isinstance(body, str):883 return len(body)884 885 return None886 887 def __init__(self, host, port=None, timeout=socket._GLOBAL_DEFAULT_TIMEOUT,888 source_address=None, blocksize=8192):889 self.timeout = timeout890 self.source_address = source_address891 self.blocksize = blocksize892 self.sock = None893 self._buffer = []894 self.__response = None895 self.__state = _CS_IDLE896 self._method = None897 self._tunnel_host = None898 self._tunnel_port = None899 self._tunnel_headers = {}900 self._raw_proxy_headers = None901 902 (self.host, self.port) = self._get_hostport(host, port)903 904 self._validate_host(self.host)905 906 # This is stored as an instance variable to allow unit907 # tests to replace it with a suitable mockup908 self._create_connection = socket.create_connection909 910 def set_tunnel(self, host, port=None, headers=None):911 """Set up host and port for HTTP CONNECT tunnelling.912 913 In a connection that uses HTTP CONNECT tunnelling, the host passed to914 the constructor is used as a proxy server that relays all communication915 to the endpoint passed to `set_tunnel`. This done by sending an HTTP916 CONNECT request to the proxy server when the connection is established.917 918 This method must be called before the HTTP connection has been919 established.920 921 The headers argument should be a mapping of extra HTTP headers to send922 with the CONNECT request.923 924 As HTTP/1.1 is used for HTTP CONNECT tunnelling request, as per the RFC925 (https://tools.ietf.org/html/rfc7231#section-4.3.6), a HTTP Host:926 header must be provided, matching the authority-form of the request927 target provided as the destination for the CONNECT request. If a928 HTTP Host: header is not provided via the headers argument, one929 is generated and transmitted automatically.930 """931 932 if self.sock:933 raise RuntimeError("Can't set up tunnel for established connection")934 935 self._tunnel_host, self._tunnel_port = self._get_hostport(host, port)936 if headers:937 self._tunnel_headers = headers.copy()938 else:939 self._tunnel_headers.clear()940 941 if not any(header.lower() == "host" for header in self._tunnel_headers):942 encoded_host = self._tunnel_host.encode("idna").decode("ascii")943 self._tunnel_headers["Host"] = "%s:%d" % (944 encoded_host, self._tunnel_port)945 946 def _get_hostport(self, host, port):947 if port is None:948 i = host.rfind(':')949 j = host.rfind(']') # ipv6 addresses have [...]950 if i > j:951 try:952 port = int(host[i+1:])953 except ValueError:954 if host[i+1:] == "": # http://foo.com:/ == http://foo.com/955 port = self.default_port956 else:957 raise InvalidURL("nonnumeric port: '%s'" % host[i+1:])958 host = host[:i]959 else:960 port = self.default_port961 if host and host[0] == '[' and host[-1] == ']':962 host = host[1:-1]963 964 return (host, port)965 966 def set_debuglevel(self, level):967 self.debuglevel = level968 969 def _wrap_ipv6(self, ip):970 if b':' in ip and ip[0] != b'['[0]:971 return b"[" + ip + b"]"972 return ip973 974 def _tunnel(self):975 connect = b"CONNECT %s:%d %s\r\n" % (976 self._wrap_ipv6(self._tunnel_host.encode("idna")),977 self._tunnel_port,978 self._http_vsn_str.encode("ascii"))979 headers = [connect]980 for header, value in self._tunnel_headers.items():981 headers.append(f"{header}: {value}\r\n".encode("latin-1"))982 headers.append(b"\r\n")983 # Making a single send() call instead of one per line encourages984 # the host OS to use a more optimal packet size instead of985 # potentially emitting a series of small packets.986 self.send(b"".join(headers))987 del headers988 989 response = self.response_class(self.sock, method=self._method)990 try:991 (version, code, message) = response._read_status()992 993 self._raw_proxy_headers = _read_headers(response.fp)994 995 if self.debuglevel > 0:996 for header in self._raw_proxy_headers:997 print('header:', header.decode())998 999 if code != http.HTTPStatus.OK:1000 self.close()1001 raise OSError(f"Tunnel connection failed: {code} {message.strip()}")1002 1003 finally:1004 response.close()1005 1006 def get_proxy_response_headers(self):1007 """1008 Returns a dictionary with the headers of the response1009 received from the proxy server to the CONNECT request1010 sent to set the tunnel.1011 1012 If the CONNECT request was not sent, the method returns None.1013 """1014 return (1015 _parse_header_lines(self._raw_proxy_headers)1016 if self._raw_proxy_headers is not None1017 else None1018 )1019 1020 def connect(self):1021 """Connect to the host and port specified in __init__."""1022 sys.audit("http.client.connect", self, self.host, self.port)1023 self.sock = self._create_connection(1024 (self.host,self.port), self.timeout, self.source_address)1025 # Might fail in OSs that don't implement TCP_NODELAY1026 try:1027 self.sock.setsockopt(socket.IPPROTO_TCP, socket.TCP_NODELAY, 1)1028 except OSError as e:1029 if e.errno != errno.ENOPROTOOPT:1030 raise1031 1032 if self._tunnel_host:1033 self._tunnel()1034 1035 def close(self):1036 """Close the connection to the HTTP server."""1037 self.__state = _CS_IDLE1038 try:1039 sock = self.sock1040 if sock:1041 self.sock = None1042 sock.close() # close it manually... there may be other refs1043 finally:1044 response = self.__response1045 if response:1046 self.__response = None1047 response.close()1048 1049 def send(self, data):1050 """Send 'data' to the server.1051 ``data`` can be a string object, a bytes object, an array object, a1052 file-like object that supports a .read() method, or an iterable object.1053 """1054 1055 if self.sock is None:1056 if self.auto_open:1057 self.connect()1058 else:1059 raise NotConnected()1060 1061 if self.debuglevel > 0:1062 print("send:", repr(data))1063 if hasattr(data, "read") :1064 if self.debuglevel > 0:1065 print("sending a readable")1066 encode = self._is_textIO(data)1067 if encode and self.debuglevel > 0:1068 print("encoding file using iso-8859-1")1069 while datablock := data.read(self.blocksize):1070 if encode:1071 datablock = datablock.encode("iso-8859-1")1072 sys.audit("http.client.send", self, datablock)1073 self.sock.sendall(datablock)1074 return1075 sys.audit("http.client.send", self, data)1076 try:1077 self.sock.sendall(data)1078 except TypeError:1079 if isinstance(data, collections.abc.Iterable):1080 for d in data:1081 self.sock.sendall(d)1082 else:1083 raise TypeError("data should be a bytes-like object "1084 "or an iterable, got %r" % type(data))1085 1086 def _output(self, s):1087 """Add a line of output to the current request buffer.1088 1089 Assumes that the line does *not* end with \\r\\n.1090 """1091 self._buffer.append(s)1092 1093 def _read_readable(self, readable):1094 if self.debuglevel > 0:1095 print("reading a readable")1096 encode = self._is_textIO(readable)1097 if encode and self.debuglevel > 0:1098 print("encoding file using iso-8859-1")1099 while datablock := readable.read(self.blocksize):1100 if encode:1101 datablock = datablock.encode("iso-8859-1")1102 yield datablock1103 1104 def _send_output(self, message_body=None, encode_chunked=False):1105 """Send the currently buffered request and clear the buffer.1106 1107 Appends an extra \\r\\n to the buffer.1108 A message_body may be specified, to be appended to the request.1109 """1110 self._buffer.extend((b"", b""))1111 msg = b"\r\n".join(self._buffer)1112 del self._buffer[:]1113 self.send(msg)1114 1115 if message_body is not None:1116 1117 # create a consistent interface to message_body1118 if hasattr(message_body, 'read'):1119 # Let file-like take precedence over byte-like. This1120 # is needed to allow the current position of mmap'ed1121 # files to be taken into account.1122 chunks = self._read_readable(message_body)1123 else:1124 try:1125 # this is solely to check to see if message_body1126 # implements the buffer API. it /would/ be easier1127 # to capture if PyObject_CheckBuffer was exposed1128 # to Python.1129 memoryview(message_body)1130 except TypeError:1131 try:1132 chunks = iter(message_body)1133 except TypeError:1134 raise TypeError("message_body should be a bytes-like "1135 "object or an iterable, got %r"1136 % type(message_body))1137 else:1138 # the object implements the buffer interface and1139 # can be passed directly into socket methods1140 chunks = (message_body,)1141 1142 for chunk in chunks:1143 if not chunk:1144 if self.debuglevel > 0:1145 print('Zero length chunk ignored')1146 continue1147 1148 if encode_chunked and self._http_vsn == 11:1149 # chunked encoding1150 chunk = f'{len(chunk):X}\r\n'.encode('ascii') + chunk \1151 + b'\r\n'1152 self.send(chunk)1153 1154 if encode_chunked and self._http_vsn == 11:1155 # end chunked transfer1156 self.send(b'0\r\n\r\n')1157 1158 def putrequest(self, method, url, skip_host=False,1159 skip_accept_encoding=False):1160 """Send a request to the server.1161 1162 'method' specifies an HTTP request method, e.g. 'GET'.1163 'url' specifies the object being requested, e.g. '/index.html'.1164 'skip_host' if True does not add automatically a 'Host:' header1165 'skip_accept_encoding' if True does not add automatically an1166 'Accept-Encoding:' header1167 """1168 1169 # if a prior response has been completed, then forget about it.1170 if self.__response and self.__response.isclosed():1171 self.__response = None1172 1173 1174 # in certain cases, we cannot issue another request on this connection.1175 # this occurs when:1176 # 1) we are in the process of sending a request. (_CS_REQ_STARTED)1177 # 2) a response to a previous request has signalled that it is going1178 # to close the connection upon completion.1179 # 3) the headers for the previous response have not been read, thus1180 # we cannot determine whether point (2) is true. (_CS_REQ_SENT)1181 #1182 # if there is no prior response, then we can request at will.1183 #1184 # if point (2) is true, then we will have passed the socket to the1185 # response (effectively meaning, "there is no prior response"), and1186 # will open a new one when a new request is made.1187 #1188 # Note: if a prior response exists, then we *can* start a new request.1189 # We are not allowed to begin fetching the response to this new1190 # request, however, until that prior response is complete.1191 #1192 if self.__state == _CS_IDLE:1193 self.__state = _CS_REQ_STARTED1194 else:1195 raise CannotSendRequest(self.__state)1196 1197 self._validate_method(method)1198 1199 # Save the method for use later in the response phase1200 self._method = method