codekingpro/portable-devtools
115k
1import codecs2import io3import re4from collections.abc import Iterable5from typing import overload6 7# https://mypy.readthedocs.io/en/stable/more_types.html#function-overloading8 9 10@overload11def always_bytes(str_or_bytes: None, *encode_args) -> None: ...12 13 14@overload15def always_bytes(str_or_bytes: str | bytes, *encode_args) -> bytes: ...16 17 18def always_bytes(str_or_bytes: None | str | bytes, *encode_args) -> None | bytes:19 if str_or_bytes is None or isinstance(str_or_bytes, bytes):20 return str_or_bytes21 elif isinstance(str_or_bytes, str):22 return str_or_bytes.encode(*encode_args)23 else:24 raise TypeError(25 f"Expected str or bytes, but got {type(str_or_bytes).__name__}."26 )27 28 29@overload30def always_str(str_or_bytes: None, *encode_args) -> None: ...31 32 33@overload34def always_str(str_or_bytes: str | bytes, *encode_args) -> str: ...35 36 37def always_str(str_or_bytes: None | str | bytes, *decode_args) -> None | str:38 """39 Returns,40 str_or_bytes unmodified, if41 """42 if str_or_bytes is None or isinstance(str_or_bytes, str):43 return str_or_bytes44 elif isinstance(str_or_bytes, bytes):45 return str_or_bytes.decode(*decode_args)46 else:47 raise TypeError(48 f"Expected str or bytes, but got {type(str_or_bytes).__name__}."49 )50 51 52# Translate control characters to "safe" characters. This implementation53# initially replaced them with the matching control pictures54# (http://unicode.org/charts/PDF/U2400.pdf), but that turned out to render badly55# with monospace fonts. We are back to "." therefore.56_control_char_trans = {57 x: ord(".")58 for x in range(32) # x + 0x2400 for unicode control group pictures59}60_control_char_trans[127] = ord(".") # 0x242161_control_char_trans_newline = _control_char_trans.copy()62for x in ("\r", "\n", "\t"):63 del _control_char_trans_newline[ord(x)]64 65_control_char_trans = str.maketrans(_control_char_trans)66_control_char_trans_newline = str.maketrans(_control_char_trans_newline)67 68 69def escape_control_characters(text: str, keep_spacing=True) -> str:70 """71 Replace all unicode C1 control characters from the given text with a single "."72 73 Args:74 keep_spacing: If True, tabs and newlines will not be replaced.75 """76 if not isinstance(text, str):77 raise ValueError(f"text type must be unicode but is {type(text).__name__}")78 79 trans = _control_char_trans_newline if keep_spacing else _control_char_trans80 return text.translate(trans)81 82 83def bytes_to_escaped_str(84 data: bytes, keep_spacing: bool = False, escape_single_quotes: bool = False85) -> str:86 """87 Take bytes and return a safe string that can be displayed to the user.88 89 Single quotes are always escaped, double quotes are never escaped:90 "'" + bytes_to_escaped_str(...) + "'"91 gives a valid Python string.92 93 Args:94 keep_spacing: If True, tabs and newlines will not be escaped.95 """96 97 if not isinstance(data, bytes):98 raise ValueError(f"data must be bytes, but is {data.__class__.__name__}")99 # We always insert a double-quote here so that we get a single-quoted string back100 # https://stackoverflow.com/questions/29019340/why-does-python-use-different-quotes-for-representing-strings-depending-on-their101 ret = repr(b'"' + data).lstrip("b")[2:-1]102 if not escape_single_quotes:103 ret = re.sub(r"(?<!\\)(\\\\)*\\'", lambda m: (m.group(1) or "") + "'", ret)104 if keep_spacing:105 ret = re.sub(106 r"(?<!\\)(\\\\)*\\([nrt])",107 lambda m: (m.group(1) or "") + dict(n="\n", r="\r", t="\t")[m.group(2)],108 ret,109 )110 return ret111 112 113def escaped_str_to_bytes(data: str) -> bytes:114 """115 Take an escaped string and return the unescaped bytes equivalent.116 117 Raises:118 ValueError, if the escape sequence is invalid.119 """120 if not isinstance(data, str):121 raise ValueError(f"data must be str, but is {data.__class__.__name__}")122 123 # This one is difficult - we use an undocumented Python API here124 # as per http://stackoverflow.com/a/23151714/934719125 return codecs.escape_decode(data)[0] # type: ignore126 127 128def is_mostly_bin(s: bytes) -> bool:129 if not s:130 return False131 132 # Cut off at ~100 chars, but do it smartly so that if the input is UTF-8, we don't133 # chop a multibyte code point in half.134 if len(s) > 100:135 for cut in range(100, min(104, len(s))):136 is_continuation_byte = (s[cut] >> 6) == 0b10137 if not is_continuation_byte:138 # A new character starts here, so we cut off just before that.139 s = s[:cut]140 break141 else:142 s = s[:100]143 144 low_bytes = sum(i < 9 or 13 < i < 32 for i in s)145 high_bytes = sum(i > 126 for i in s)146 ascii_bytes = len(s) - low_bytes - high_bytes147 148 # Heuristic 1: If it's mostly printable ASCII, it's not bin.149 if ascii_bytes / len(s) > 0.7:150 return False151 152 # Heuristic 2: If it's UTF-8 without too many ASCII control chars, it's not bin.153 # Note that b"\x00\x00\x00" would be valid UTF-8, so we don't want to accept _any_154 # UTF-8 with higher code points.155 if (ascii_bytes + high_bytes) / len(s) > 0.95:156 try:157 s.decode()158 return False159 except ValueError:160 pass161 162 return True163 164 165def is_xml(s: bytes) -> bool:166 for char in s:167 if char in (9, 10, 32): # is space?168 continue169 return char == 60 # is a "<"?170 return False171 172 173def clean_hanging_newline(t):174 """175 Many editors will silently add a newline to the final line of a176 document (I'm looking at you, Vim). This function fixes this common177 problem at the risk of removing a hanging newline in the rare cases178 where the user actually intends it.179 """180 if t and t[-1] == "\n":181 return t[:-1]182 return t183 184 185def hexdump(s):186 """187 Returns:188 A generator of (offset, hex, str) tuples189 """190 for i in range(0, len(s), 16):191 offset = f"{i:0=10x}"192 part = s[i : i + 16]193 x = " ".join(f"{i:0=2x}" for i in part)194 x = x.ljust(47) # 16*2 + 15195 part_repr = always_str(196 escape_control_characters(197 part.decode("ascii", "replace").replace("\ufffd", "."), False198 )199 )200 yield (offset, x, part_repr)201 202 203def _move_to_private_code_plane(matchobj):204 return chr(ord(matchobj.group(0)) + 0xE000)205 206 207def _restore_from_private_code_plane(matchobj):208 return chr(ord(matchobj.group(0)) - 0xE000)209 210 211NO_ESCAPE = r"(?<!\\)(?:\\\\)*"212MULTILINE_CONTENT = r"[\s\S]*?"213SINGLELINE_CONTENT = r".*?"214MULTILINE_CONTENT_LINE_CONTINUATION = r"(?:.|(?<=\\)\n)*?"215 216 217def split_special_areas(218 data: str,219 area_delimiter: Iterable[str],220):221 """222 Split a string of code into a [code, special area, code, special area, ..., code] list.223 224 For example,225 226 >>> split_special_areas(227 >>> "test /* don't modify me */ foo",228 >>> [r"/\\*[\\s\\S]*?\\*/"]) # (regex matching comments)229 ["test ", "/* don't modify me */", " foo"]230 231 "".join(split_special_areas(x, ...)) == x always holds true.232 """233 return re.split("({})".format("|".join(area_delimiter)), data, flags=re.MULTILINE)234 235 236def escape_special_areas(237 data: str,238 area_delimiter: Iterable[str],239 control_characters,240):241 """242 Escape all control characters present in special areas with UTF8 symbols243 in the private use plane (U+E000 t+ ord(char)).244 This is useful so that one can then use regex replacements on the resulting string without245 interfering with special areas.246 247 control_characters must be 0 < ord(x) < 256.248 249 Example:250 251 >>> print(x)252 if (true) { console.log('{}'); }253 >>> x = escape_special_areas(x, "{", ["'" + SINGLELINE_CONTENT + "'"])254 >>> print(x)255 if (true) { console.log('�}'); }256 >>> x = re.sub(r"\\s*{\\s*", " {\n ", x)257 >>> x = unescape_special_areas(x)258 >>> print(x)259 if (true) {260 console.log('{}'); }261 """262 buf = io.StringIO()263 parts = split_special_areas(data, area_delimiter)264 rex = re.compile(rf"[{control_characters}]")265 for i, x in enumerate(parts):266 if i % 2:267 x = rex.sub(_move_to_private_code_plane, x)268 buf.write(x)269 return buf.getvalue()270 271 272def unescape_special_areas(data: str):273 """274 Invert escape_special_areas.275 276 x == unescape_special_areas(escape_special_areas(x)) always holds true.277 """278 return re.sub(r"[\ue000-\ue0ff]", _restore_from_private_code_plane, data)279 280 281def cut_after_n_lines(content: str, n: int) -> str:282 assert n > 0283 pos = content.find("\n")284 while pos >= 0 and n > 1:285 pos = content.find("\n", pos + 1)286 n -= 1287 if pos >= 0:288 content = content[: pos + 1]289 return content290 