codekingpro/portable-devtools
115k
1from typing import Union, Tuple2 3from charset_normalizer import from_bytes4from charset_normalizer.constant import TOO_SMALL_SEQUENCE5 6UTF8 = 'utf-8'7 8ContentBytes = Union[bytearray, bytes]9 10 11def detect_encoding(content: ContentBytes) -> str:12 """13 We default to UTF-8 if text too short, because the detection14 can return a random encoding leading to confusing results15 given the `charset_normalizer` version (< 2.0.5).16 17 >>> too_short = ']"foo"'18 >>> detected = from_bytes(too_short.encode()).best().encoding19 >>> detected20 'ascii'21 >>> too_short.encode().decode(detected)22 ']"foo"'23 """24 encoding = UTF825 if len(content) > TOO_SMALL_SEQUENCE:26 match = from_bytes(bytes(content)).best()27 if match:28 encoding = match.encoding29 return encoding30 31 32def smart_decode(content: ContentBytes, encoding: str) -> Tuple[str, str]:33 """Decode `content` using the given `encoding`.34 If no `encoding` is provided, the best effort is to guess it from `content`.35 36 Unicode errors are replaced.37 38 """39 if not encoding:40 encoding = detect_encoding(content)41 return content.decode(encoding, 'replace'), encoding42 43 44def smart_encode(content: str, encoding: str) -> bytes:45 """Encode `content` using the given `encoding`.46 47 Unicode errors are replaced.48 49 """50 return content.encode(encoding, 'replace')51 