Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
str_util.py403 linesDownload Raw Back to urwid
1# Urwid unicode character processing tables2#    Copyright (C) 2004-2011  Ian Ward3#4#    This library is free software; you can redistribute it and/or5#    modify it under the terms of the GNU Lesser General Public6#    License as published by the Free Software Foundation; either7#    version 2.1 of the License, or (at your option) any later version.8#9#    This library is distributed in the hope that it will be useful,10#    but WITHOUT ANY WARRANTY; without even the implied warranty of11#    MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the GNU12#    Lesser General Public License for more details.13#14#    You should have received a copy of the GNU Lesser General Public15#    License along with this library; if not, write to the Free Software16#    Foundation, Inc., 59 Temple Place, Suite 330, Boston, MA  02111-1307  USA17#18# Urwid web site: https://urwid.org/19 20 21from __future__ import annotations22 23import re24import typing25import warnings26 27import wcwidth28 29if typing.TYPE_CHECKING:30    from typing_extensions import Literal31 32SAFE_ASCII_RE = re.compile(r"^[ -~]*$")33SAFE_ASCII_BYTES_RE = re.compile(rb"^[ -~]*$")34 35_byte_encoding: Literal["utf8", "narrow", "wide"] = "narrow"36 37 38def get_char_width(char: str) -> Literal[0, 1, 2]:39    """40    Return the screen column width for a single character.41 42    .. deprecated:: 3.0.443    """44    warnings.warn(45        "get_char_width is deprecated in favor of wcwidth.width",46        DeprecationWarning,47        stacklevel=2,48    )49    if (width := wcwidth.wcwidth(char)) >= 0:50        return width51 52    return 053 54 55def get_width(o: int) -> Literal[0, 1, 2]:56    """57    Return the screen column width for unicode ordinal o.58 59    .. deprecated:: 3.0.460    """61    warnings.warn(62        "get_width is deprecated in favor of wcwidth.width",63        DeprecationWarning,64        stacklevel=2,65    )66    if (width := wcwidth.wcwidth(chr(o))) >= 0:67        return width68 69    return 070 71 72def _decode_grapheme_at(text: bytes, start: int, end: int) -> tuple[str, int]:73    """74    Decode bytes starting at `start` to get the first grapheme cluster.75 76    :param text: UTF-8 encoded bytes77    :param start: starting byte position78    :param end: ending byte position79    :returns: (grapheme_string, next_byte_position)80 81    Assumes caller provides valid UTF-8 byte boundaries.82    """83    decoded = text[start:end].decode("utf-8")84    grapheme = next(wcwidth.iter_graphemes(decoded), "")85    grapheme_bytes = grapheme.encode("utf-8")86    return grapheme, start + len(grapheme_bytes)87 88 89def decode_one(text: bytes | str, pos: int) -> tuple[int, int]:90    """91    Return (ordinal at pos, next position) for UTF-8 encoded text.92    """93    lt = len(text) - pos94 95    b2 = 0  # Fallback, not changing anything96    b3 = 0  # Fallback, not changing anything97    b4 = 0  # Fallback, not changing anything98 99    try:100        if isinstance(text, str):101            b1 = ord(text[pos])102            if lt > 1:103                b2 = ord(text[pos + 1])104            if lt > 2:105                b3 = ord(text[pos + 2])106            if lt > 3:107                b4 = ord(text[pos + 3])108        else:109            b1 = text[pos]110            if lt > 1:111                b2 = text[pos + 1]112            if lt > 2:113                b3 = text[pos + 2]114            if lt > 3:115                b4 = text[pos + 3]116    except Exception as e:117        raise ValueError(f"{e}: text={text!r}, pos={pos!r}, lt={lt!r}").with_traceback(e.__traceback__) from e118 119    if not b1 & 0x80:120        return b1, pos + 1121    error = ord("?"), pos + 1122 123    if lt < 2:124        return error125    if b1 & 0xE0 == 0xC0:126        if b2 & 0xC0 != 0x80:127            return error128        if (o := ((b1 & 0x1F) << 6) | (b2 & 0x3F)) >= 0x80:129            return o, pos + 2130        return error131    if lt < 3:132        return error133    if b1 & 0xF0 == 0xE0:134        if b2 & 0xC0 != 0x80:135            return error136        if b3 & 0xC0 != 0x80:137            return error138        if (o := ((b1 & 0x0F) << 12) | ((b2 & 0x3F) << 6) | (b3 & 0x3F)) >= 0x800:139            return o, pos + 3140        return error141    if lt < 4:142        return error143    if b1 & 0xF8 == 0xF0:144        if b2 & 0xC0 != 0x80:145            return error146        if b3 & 0xC0 != 0x80:147            return error148        if b4 & 0xC0 != 0x80:149            return error150        if (o := ((b1 & 0x07) << 18) | ((b2 & 0x3F) << 12) | ((b3 & 0x3F) << 6) | (b4 & 0x3F)) >= 0x10000:151            return o, pos + 4152        return error153    return error154 155 156def decode_one_uni(text: str, i: int) -> tuple[int, int]:157    """158    decode_one implementation for unicode strings159    """160    return ord(text[i]), i + 1161 162 163def decode_one_right(text: bytes, pos: int) -> tuple[int, int] | None:164    """165    Return (ordinal at pos, next position) for UTF-8 encoded text.166    pos is assumed to be on the trailing byte of a utf-8 sequence.167    """168    if not isinstance(text, bytes):169        raise TypeError(text)170    error = ord("?"), pos - 1171    p = pos172    while p >= 0:173        if text[p] & 0xC0 != 0x80:174            o, _next_pos = decode_one(text, p)175            return o, p - 1176        p -= 1177        if p == p - 4:178            return error179    return None180 181 182def set_byte_encoding(enc: Literal["utf8", "narrow", "wide"]) -> None:183    if enc not in {"utf8", "narrow", "wide"}:184        raise ValueError(enc)185    global _byte_encoding  # noqa: PLW0603  # pylint: disable=global-statement186    _byte_encoding = enc187 188 189def get_byte_encoding() -> Literal["utf8", "narrow", "wide"]:190    return _byte_encoding191 192 193def calc_string_text_pos(text: str, start_offs: int, end_offs: int, pref_col: int) -> tuple[int, int]:194    """195    Calculate the closest position to the screen column pref_col in text196    where start_offs is the offset into text assumed to be screen column 0197    and end_offs is the end of the range to search.198 199    Iterates by grapheme clusters for emoji ZWJ sequences, flags,200    combining characters, and other multi-codepoint unicode sequences.201 202    :param text: string203    :param start_offs: starting text position204    :param end_offs: ending text position205    :param pref_col: target column206    :returns: (position, actual_col)207    """208    if start_offs > end_offs:209        raise ValueError((start_offs, end_offs))210 211    cols = 0212    pos = start_offs213    for grapheme in wcwidth.iter_graphemes(text[start_offs:end_offs]):214        grapheme_width = wcwidth.width(grapheme, control_codes="ignore")215        if grapheme_width + cols > pref_col:216            return pos, cols217        cols += grapheme_width218        pos += len(grapheme)219 220    return end_offs, cols221 222 223def calc_text_pos(text: str | bytes, start_offs: int, end_offs: int, pref_col: int) -> tuple[int, int]:224    """225    Calculate the closest position to the screen column pref_col in text226    where start_offs is the offset into text assumed to be screen column 0227    and end_offs is the end of the range to search.228 229    text may be unicode or a byte string in the target _byte_encoding230 231    Returns (position, actual_col).232    """233    if start_offs > end_offs:234        raise ValueError((start_offs, end_offs))235 236    if isinstance(text, str):237        return calc_string_text_pos(text, start_offs, end_offs, pref_col)238 239    if not isinstance(text, bytes):240        raise TypeError(text)241 242    if _byte_encoding == "utf8":243        decoded = text[start_offs:end_offs].decode("utf-8")244        str_pos, cols = calc_string_text_pos(decoded, 0, len(decoded), pref_col)245        byte_offset = len(decoded[:str_pos].encode("utf-8"))246        return start_offs + byte_offset, cols247 248    # "wide" and "narrow"249    i = start_offs + pref_col250    if i >= end_offs:251        return end_offs, end_offs - start_offs252    if _byte_encoding == "wide" and within_double_byte(text, start_offs, i) == 2:253        i -= 1254    return i, i - start_offs255 256 257def calc_width(text: str | bytes, start_offs: int, end_offs: int) -> int:258    """259    Return the screen column width of text between start_offs and end_offs.260 261    text may be unicode or a byte string in the target _byte_encoding262 263    Some characters are wide (take two columns) and others affect the264    previous character (take zero columns), while others are grouped265    in sequence by "grapheme boundaries" (Emoji, Skin tones, flags, etc).266    """267 268    if start_offs > end_offs:269        msg = f"{start_offs=} > {end_offs=}"270        raise ValueError(msg)271 272    if isinstance(text, str):273        return wcwidth.width(text[start_offs:end_offs], control_codes="ignore")274 275    if _byte_encoding == "utf8":276        try:277            return wcwidth.width(text[start_offs:end_offs].decode("utf-8"), control_codes="ignore")278        except UnicodeDecodeError as exc:279            warnings.warn(280                "`calc_width` with text encoded to bytes can produce incorrect results"281                f"due to possible offset in the middle of character: {exc}",282                UnicodeWarning,283                stacklevel=2,284            )285 286        i = start_offs287        sc = 0288        while i < end_offs:289            o, i = decode_one(text, i)290            if (w := wcwidth.wcwidth(chr(o))) > 0:291                sc += w292        return sc293    # "wide", "narrow" or all printable ASCII, just return the character count294    return end_offs - start_offs295 296 297def is_wide_char(text: str | bytes, offs: int) -> bool:298    """299    Test if the grapheme cluster at offs within text is wide (2 columns).300 301    For Unicode strings, extracts the full grapheme cluster starting at offs302    and checks if it renders as wide. This correctly handles multi-codepoint303    graphemes like emoji ZWJ sequences and flags.304 305    text may be unicode or a byte string in the target _byte_encoding306    """307    if isinstance(text, str):308        grapheme = next(wcwidth.iter_graphemes(text[offs:]))309        return wcwidth.width(grapheme, control_codes="ignore") == 2310    if not isinstance(text, bytes):311        raise TypeError(text)312    if _byte_encoding == "utf8":313        grapheme, _ = _decode_grapheme_at(text, offs, len(text))314        return wcwidth.width(grapheme, control_codes="ignore") == 2315    if _byte_encoding == "wide":316        return within_double_byte(text, offs, offs) == 1317    return False318 319 320def move_prev_char(text: str | bytes, start_offs: int, end_offs: int) -> int:321    """322    Return the position of the grapheme cluster before end_offs.323 324    For Unicode strings, handle multi-codepoint, "grapheme clusters",325    to better measure emoji ZWJ, flags, combining characters, skin tones.326    """327    if start_offs >= end_offs:328        raise ValueError((start_offs, end_offs))329    if isinstance(text, str):330        return wcwidth.grapheme_boundary_before(text, end_offs)331    if not isinstance(text, bytes):332        raise TypeError(text)333    if _byte_encoding == "utf8":334        decoded = text[start_offs:end_offs].decode("utf-8")335        str_pos = len(decoded)336        prev_str_pos = wcwidth.grapheme_boundary_before(decoded, str_pos)337        prefix = decoded[:prev_str_pos]338        return start_offs + len(prefix.encode("utf-8"))339    if _byte_encoding == "wide" and within_double_byte(text, start_offs, end_offs - 1) == 2:340        return end_offs - 2341    return end_offs - 1342 343 344def move_next_char(text: str | bytes, start_offs: int, end_offs: int) -> int:345    """346    Return the position of the next grapheme cluster after start_offs.347 348    For Unicode strings, handle multi-codepoint, "grapheme clusters",349    to better measure emoji ZWJ, flags, combining characters, skin tones.350    """351    if start_offs >= end_offs:352        raise ValueError((start_offs, end_offs))353    if isinstance(text, str):354        grapheme = next(wcwidth.iter_graphemes(text[start_offs:end_offs]))355        return start_offs + len(grapheme)356    if not isinstance(text, bytes):357        raise TypeError(text)358    if _byte_encoding == "utf8":359        _, next_pos = _decode_grapheme_at(text, start_offs, end_offs)360        return next_pos361    if _byte_encoding == "wide" and within_double_byte(text, start_offs, start_offs) == 1:362        return start_offs + 2363    return start_offs + 1364 365 366def within_double_byte(text: bytes, line_start: int, pos: int) -> Literal[0, 1, 2]:367    """Return whether pos is within a double-byte encoded character.368 369    text -- byte string in question370    line_start -- offset of beginning of line (< pos)371    pos -- offset in question372 373    Return values:374    0 -- not within dbe char, or double_byte_encoding == False375    1 -- pos is on the 1st half of a dbe char376    2 -- pos is on the 2nd half of a dbe char377    """378    if not isinstance(text, bytes):379        raise TypeError(text)380    v = text[pos]381 382    if 0x40 <= v < 0x7F:383        # might be second half of big5, uhc or gbk encoding384        if pos == line_start:385            return 0386 387        if text[pos - 1] >= 0x81 and within_double_byte(text, line_start, pos - 1) == 1:388            return 2389        return 0390 391    if v < 0x80:392        return 0393 394    i = pos - 1395    while i >= line_start:396        if text[i] < 0x80:397            break398        i -= 1399 400    if (pos - i) & 1:401        return 1402    return 2403 
codekingpro/portable-devtools · Team Ai