Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes15kdownloads
strutils.py290 linesDownload Raw Back to utils
1import codecs2import io3import re4from collections.abc import Iterable5from typing import overload6 7# https://mypy.readthedocs.io/en/stable/more_types.html#function-overloading8 9 10@overload11def always_bytes(str_or_bytes: None, *encode_args) -> None: ...12 13 14@overload15def always_bytes(str_or_bytes: str | bytes, *encode_args) -> bytes: ...16 17 18def always_bytes(str_or_bytes: None | str | bytes, *encode_args) -> None | bytes:19    if str_or_bytes is None or isinstance(str_or_bytes, bytes):20        return str_or_bytes21    elif isinstance(str_or_bytes, str):22        return str_or_bytes.encode(*encode_args)23    else:24        raise TypeError(25            f"Expected str or bytes, but got {type(str_or_bytes).__name__}."26        )27 28 29@overload30def always_str(str_or_bytes: None, *encode_args) -> None: ...31 32 33@overload34def always_str(str_or_bytes: str | bytes, *encode_args) -> str: ...35 36 37def always_str(str_or_bytes: None | str | bytes, *decode_args) -> None | str:38    """39    Returns,40        str_or_bytes unmodified, if41    """42    if str_or_bytes is None or isinstance(str_or_bytes, str):43        return str_or_bytes44    elif isinstance(str_or_bytes, bytes):45        return str_or_bytes.decode(*decode_args)46    else:47        raise TypeError(48            f"Expected str or bytes, but got {type(str_or_bytes).__name__}."49        )50 51 52# Translate control characters to "safe" characters. This implementation53# initially replaced them with the matching control pictures54# (http://unicode.org/charts/PDF/U2400.pdf), but that turned out to render badly55# with monospace fonts. We are back to "." therefore.56_control_char_trans = {57    x: ord(".")58    for x in range(32)  # x + 0x2400 for unicode control group pictures59}60_control_char_trans[127] = ord(".")  # 0x242161_control_char_trans_newline = _control_char_trans.copy()62for x in ("\r", "\n", "\t"):63    del _control_char_trans_newline[ord(x)]64 65_control_char_trans = str.maketrans(_control_char_trans)66_control_char_trans_newline = str.maketrans(_control_char_trans_newline)67 68 69def escape_control_characters(text: str, keep_spacing=True) -> str:70    """71    Replace all unicode C1 control characters from the given text with a single "."72 73    Args:74        keep_spacing: If True, tabs and newlines will not be replaced.75    """76    if not isinstance(text, str):77        raise ValueError(f"text type must be unicode but is {type(text).__name__}")78 79    trans = _control_char_trans_newline if keep_spacing else _control_char_trans80    return text.translate(trans)81 82 83def bytes_to_escaped_str(84    data: bytes, keep_spacing: bool = False, escape_single_quotes: bool = False85) -> str:86    """87    Take bytes and return a safe string that can be displayed to the user.88 89    Single quotes are always escaped, double quotes are never escaped:90        "'" + bytes_to_escaped_str(...) + "'"91    gives a valid Python string.92 93    Args:94        keep_spacing: If True, tabs and newlines will not be escaped.95    """96 97    if not isinstance(data, bytes):98        raise ValueError(f"data must be bytes, but is {data.__class__.__name__}")99    # We always insert a double-quote here so that we get a single-quoted string back100    # https://stackoverflow.com/questions/29019340/why-does-python-use-different-quotes-for-representing-strings-depending-on-their101    ret = repr(b'"' + data).lstrip("b")[2:-1]102    if not escape_single_quotes:103        ret = re.sub(r"(?<!\\)(\\\\)*\\'", lambda m: (m.group(1) or "") + "'", ret)104    if keep_spacing:105        ret = re.sub(106            r"(?<!\\)(\\\\)*\\([nrt])",107            lambda m: (m.group(1) or "") + dict(n="\n", r="\r", t="\t")[m.group(2)],108            ret,109        )110    return ret111 112 113def escaped_str_to_bytes(data: str) -> bytes:114    """115    Take an escaped string and return the unescaped bytes equivalent.116 117    Raises:118        ValueError, if the escape sequence is invalid.119    """120    if not isinstance(data, str):121        raise ValueError(f"data must be str, but is {data.__class__.__name__}")122 123    # This one is difficult - we use an undocumented Python API here124    # as per http://stackoverflow.com/a/23151714/934719125    return codecs.escape_decode(data)[0]  # type: ignore126 127 128def is_mostly_bin(s: bytes) -> bool:129    if not s:130        return False131 132    # Cut off at ~100 chars, but do it smartly so that if the input is UTF-8, we don't133    # chop a multibyte code point in half.134    if len(s) > 100:135        for cut in range(100, min(104, len(s))):136            is_continuation_byte = (s[cut] >> 6) == 0b10137            if not is_continuation_byte:138                # A new character starts here, so we cut off just before that.139                s = s[:cut]140                break141        else:142            s = s[:100]143 144    low_bytes = sum(i < 9 or 13 < i < 32 for i in s)145    high_bytes = sum(i > 126 for i in s)146    ascii_bytes = len(s) - low_bytes - high_bytes147 148    # Heuristic 1: If it's mostly printable ASCII, it's not bin.149    if ascii_bytes / len(s) > 0.7:150        return False151 152    # Heuristic 2: If it's UTF-8 without too many ASCII control chars, it's not bin.153    # Note that b"\x00\x00\x00" would be valid UTF-8, so we don't want to accept _any_154    # UTF-8 with higher code points.155    if (ascii_bytes + high_bytes) / len(s) > 0.95:156        try:157            s.decode()158            return False159        except ValueError:160            pass161 162    return True163 164 165def is_xml(s: bytes) -> bool:166    for char in s:167        if char in (9, 10, 32):  # is space?168            continue169        return char == 60  # is a "<"?170    return False171 172 173def clean_hanging_newline(t):174    """175    Many editors will silently add a newline to the final line of a176    document (I'm looking at you, Vim). This function fixes this common177    problem at the risk of removing a hanging newline in the rare cases178    where the user actually intends it.179    """180    if t and t[-1] == "\n":181        return t[:-1]182    return t183 184 185def hexdump(s):186    """187    Returns:188        A generator of (offset, hex, str) tuples189    """190    for i in range(0, len(s), 16):191        offset = f"{i:0=10x}"192        part = s[i : i + 16]193        x = " ".join(f"{i:0=2x}" for i in part)194        x = x.ljust(47)  # 16*2 + 15195        part_repr = always_str(196            escape_control_characters(197                part.decode("ascii", "replace").replace("\ufffd", "."), False198            )199        )200        yield (offset, x, part_repr)201 202 203def _move_to_private_code_plane(matchobj):204    return chr(ord(matchobj.group(0)) + 0xE000)205 206 207def _restore_from_private_code_plane(matchobj):208    return chr(ord(matchobj.group(0)) - 0xE000)209 210 211NO_ESCAPE = r"(?<!\\)(?:\\\\)*"212MULTILINE_CONTENT = r"[\s\S]*?"213SINGLELINE_CONTENT = r".*?"214MULTILINE_CONTENT_LINE_CONTINUATION = r"(?:.|(?<=\\)\n)*?"215 216 217def split_special_areas(218    data: str,219    area_delimiter: Iterable[str],220):221    """222    Split a string of code into a [code, special area, code, special area, ..., code] list.223 224    For example,225 226    >>> split_special_areas(227    >>>     "test /* don't modify me */ foo",228    >>>     [r"/\\*[\\s\\S]*?\\*/"])  # (regex matching comments)229    ["test ", "/* don't modify me */", " foo"]230 231    "".join(split_special_areas(x, ...)) == x always holds true.232    """233    return re.split("({})".format("|".join(area_delimiter)), data, flags=re.MULTILINE)234 235 236def escape_special_areas(237    data: str,238    area_delimiter: Iterable[str],239    control_characters,240):241    """242    Escape all control characters present in special areas with UTF8 symbols243    in the private use plane (U+E000 t+ ord(char)).244    This is useful so that one can then use regex replacements on the resulting string without245    interfering with special areas.246 247    control_characters must be 0 < ord(x) < 256.248 249    Example:250 251    >>> print(x)252    if (true) { console.log('{}'); }253    >>> x = escape_special_areas(x, "{", ["'" + SINGLELINE_CONTENT + "'"])254    >>> print(x)255    if (true) { console.log('�}'); }256    >>> x = re.sub(r"\\s*{\\s*", " {\n    ", x)257    >>> x = unescape_special_areas(x)258    >>> print(x)259    if (true) {260        console.log('{}'); }261    """262    buf = io.StringIO()263    parts = split_special_areas(data, area_delimiter)264    rex = re.compile(rf"[{control_characters}]")265    for i, x in enumerate(parts):266        if i % 2:267            x = rex.sub(_move_to_private_code_plane, x)268        buf.write(x)269    return buf.getvalue()270 271 272def unescape_special_areas(data: str):273    """274    Invert escape_special_areas.275 276    x == unescape_special_areas(escape_special_areas(x)) always holds true.277    """278    return re.sub(r"[\ue000-\ue0ff]", _restore_from_private_code_plane, data)279 280 281def cut_after_n_lines(content: str, n: int) -> str:282    assert n > 0283    pos = content.find("\n")284    while pos >= 0 and n > 1:285        pos = content.find("\n", pos + 1)286        n -= 1287    if pos >= 0:288        content = content[: pos + 1]289    return content290 
codekingpro/portable-devtools · Team Ai