codekingpro/portable-devtools
115k
1import collections2import re3 4 5def parse_content_type(c: str) -> tuple[str, str, dict[str, str]] | None:6 """7 A simple parser for content-type values. Returns a (type, subtype,8 parameters) tuple, where type and subtype are strings, and parameters9 is a dict. If the string could not be parsed, return None.10 11 E.g. the following string:12 13 text/html; charset=UTF-814 15 Returns:16 17 ("text", "html", {"charset": "UTF-8"})18 """19 parts = c.split(";", 1)20 ts = parts[0].split("/", 1)21 if len(ts) != 2:22 return None23 d = collections.OrderedDict()24 if len(parts) == 2:25 for i in parts[1].split(";"):26 clause = i.split("=", 1)27 if len(clause) == 2:28 d[clause[0].strip()] = clause[1].strip()29 return ts[0].lower(), ts[1].lower(), d30 31 32def assemble_content_type(type, subtype, parameters):33 if not parameters:34 return f"{type}/{subtype}"35 params = "; ".join(f"{k}={v}" for k, v in parameters.items())36 return f"{type}/{subtype}; {params}"37 38 39def infer_content_encoding(content_type: str, content: bytes = b"") -> str:40 """41 Infer the encoding of content from the content-type header.42 """43 enc = None44 45 # BOM has the highest priority46 if content.startswith(b"\x00\x00\xfe\xff"):47 enc = "utf-32be"48 elif content.startswith(b"\xff\xfe\x00\x00"):49 enc = "utf-32le"50 elif content.startswith(b"\xfe\xff"):51 enc = "utf-16be"52 elif content.startswith(b"\xff\xfe"):53 enc = "utf-16le"54 elif content.startswith(b"\xef\xbb\xbf"):55 # 'utf-8-sig' will strip the BOM on decode56 enc = "utf-8-sig"57 elif parsed_content_type := parse_content_type(content_type):58 # Use the charset from the header if possible59 enc = parsed_content_type[2].get("charset")60 61 # Otherwise, infer the encoding62 if not enc and "json" in content_type:63 enc = "utf8"64 65 if not enc and "html" in content_type:66 meta_charset = re.search(67 rb"""<meta[^>]+charset=['"]?([^'">]+)""", content, re.IGNORECASE68 )69 if meta_charset:70 enc = meta_charset.group(1).decode("ascii", "ignore")71 else:72 # Fallback to utf8 for html73 # Ref: https://html.spec.whatwg.org/multipage/parsing.html#determining-the-character-encoding74 # > 9. [snip] the comprehensive UTF-8 encoding is suggested.75 enc = "utf8"76 77 if not enc and "xml" in content_type:78 if xml_encoding := re.search(79 rb"""<\?xml[^\?>]+encoding=['"]([^'"\?>]+)""", content, re.IGNORECASE80 ):81 enc = xml_encoding.group(1).decode("ascii", "ignore")82 else:83 # Fallback to utf8 for xml84 # Ref: https://datatracker.ietf.org/doc/html/rfc7303#section-8.585 # > the XML processor [snip] to determine an encoding of UTF-8.86 enc = "utf8"87 88 if not enc and ("javascript" in content_type or "ecmascript" in content_type):89 # Fallback to utf8 for javascript90 # Ref: https://datatracker.ietf.org/doc/html/rfc9239#section-4.291 # > 3. Else, the character encoding scheme is assumed to be UTF-892 enc = "utf8"93 94 if not enc and "text/css" in content_type:95 # @charset rule must be the very first thing.96 css_charset = re.match(rb"""@charset "([^"]+)";""", content, re.IGNORECASE)97 if css_charset:98 enc = css_charset.group(1).decode("ascii", "ignore")99 else:100 # Fallback to utf8 for css101 # Ref: https://drafts.csswg.org/css-syntax/#determine-the-fallback-encoding102 # > 4. Otherwise, return utf-8103 enc = "utf8"104 105 # Fallback to latin-1106 if not enc:107 enc = "latin-1"108 109 # Use GB 18030 as the superset of GB2312 and GBK to fix common encoding problems on Chinese websites.110 if enc.lower() in ("gb2312", "gbk"):111 enc = "gb18030"112 113 return enc114 