codekingpro/portable-devtools
114k
1# Urwid unicode character processing tables2# Copyright (C) 2004-2011 Ian Ward3#4# This library is free software; you can redistribute it and/or5# modify it under the terms of the GNU Lesser General Public6# License as published by the Free Software Foundation; either7# version 2.1 of the License, or (at your option) any later version.8#9# This library is distributed in the hope that it will be useful,10# but WITHOUT ANY WARRANTY; without even the implied warranty of11# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU12# Lesser General Public License for more details.13#14# You should have received a copy of the GNU Lesser General Public15# License along with this library; if not, write to the Free Software16# Foundation, Inc., 59 Temple Place, Suite 330, Boston, MA 02111-1307 USA17#18# Urwid web site: https://urwid.org/19 20 21from __future__ import annotations22 23import re24import typing25import warnings26 27import wcwidth28 29if typing.TYPE_CHECKING:30 from typing_extensions import Literal31 32SAFE_ASCII_RE = re.compile(r"^[ -~]*$")33SAFE_ASCII_BYTES_RE = re.compile(rb"^[ -~]*$")34 35_byte_encoding: Literal["utf8", "narrow", "wide"] = "narrow"36 37 38def get_char_width(char: str) -> Literal[0, 1, 2]:39 """40 Return the screen column width for a single character.41 42 .. deprecated:: 3.0.443 """44 warnings.warn(45 "get_char_width is deprecated in favor of wcwidth.width",46 DeprecationWarning,47 stacklevel=2,48 )49 if (width := wcwidth.wcwidth(char)) >= 0:50 return width51 52 return 053 54 55def get_width(o: int) -> Literal[0, 1, 2]:56 """57 Return the screen column width for unicode ordinal o.58 59 .. deprecated:: 3.0.460 """61 warnings.warn(62 "get_width is deprecated in favor of wcwidth.width",63 DeprecationWarning,64 stacklevel=2,65 )66 if (width := wcwidth.wcwidth(chr(o))) >= 0:67 return width68 69 return 070 71 72def _decode_grapheme_at(text: bytes, start: int, end: int) -> tuple[str, int]:73 """74 Decode bytes starting at `start` to get the first grapheme cluster.75 76 :param text: UTF-8 encoded bytes77 :param start: starting byte position78 :param end: ending byte position79 :returns: (grapheme_string, next_byte_position)80 81 Assumes caller provides valid UTF-8 byte boundaries.82 """83 decoded = text[start:end].decode("utf-8")84 grapheme = next(wcwidth.iter_graphemes(decoded), "")85 grapheme_bytes = grapheme.encode("utf-8")86 return grapheme, start + len(grapheme_bytes)87 88 89def decode_one(text: bytes | str, pos: int) -> tuple[int, int]:90 """91 Return (ordinal at pos, next position) for UTF-8 encoded text.92 """93 lt = len(text) - pos94 95 b2 = 0 # Fallback, not changing anything96 b3 = 0 # Fallback, not changing anything97 b4 = 0 # Fallback, not changing anything98 99 try:100 if isinstance(text, str):101 b1 = ord(text[pos])102 if lt > 1:103 b2 = ord(text[pos + 1])104 if lt > 2:105 b3 = ord(text[pos + 2])106 if lt > 3:107 b4 = ord(text[pos + 3])108 else:109 b1 = text[pos]110 if lt > 1:111 b2 = text[pos + 1]112 if lt > 2:113 b3 = text[pos + 2]114 if lt > 3:115 b4 = text[pos + 3]116 except Exception as e:117 raise ValueError(f"{e}: text={text!r}, pos={pos!r}, lt={lt!r}").with_traceback(e.__traceback__) from e118 119 if not b1 & 0x80:120 return b1, pos + 1121 error = ord("?"), pos + 1122 123 if lt < 2:124 return error125 if b1 & 0xE0 == 0xC0:126 if b2 & 0xC0 != 0x80:127 return error128 if (o := ((b1 & 0x1F) << 6) | (b2 & 0x3F)) >= 0x80:129 return o, pos + 2130 return error131 if lt < 3:132 return error133 if b1 & 0xF0 == 0xE0:134 if b2 & 0xC0 != 0x80:135 return error136 if b3 & 0xC0 != 0x80:137 return error138 if (o := ((b1 & 0x0F) << 12) | ((b2 & 0x3F) << 6) | (b3 & 0x3F)) >= 0x800:139 return o, pos + 3140 return error141 if lt < 4:142 return error143 if b1 & 0xF8 == 0xF0:144 if b2 & 0xC0 != 0x80:145 return error146 if b3 & 0xC0 != 0x80:147 return error148 if b4 & 0xC0 != 0x80:149 return error150 if (o := ((b1 & 0x07) << 18) | ((b2 & 0x3F) << 12) | ((b3 & 0x3F) << 6) | (b4 & 0x3F)) >= 0x10000:151 return o, pos + 4152 return error153 return error154 155 156def decode_one_uni(text: str, i: int) -> tuple[int, int]:157 """158 decode_one implementation for unicode strings159 """160 return ord(text[i]), i + 1161 162 163def decode_one_right(text: bytes, pos: int) -> tuple[int, int] | None:164 """165 Return (ordinal at pos, next position) for UTF-8 encoded text.166 pos is assumed to be on the trailing byte of a utf-8 sequence.167 """168 if not isinstance(text, bytes):169 raise TypeError(text)170 error = ord("?"), pos - 1171 p = pos172 while p >= 0:173 if text[p] & 0xC0 != 0x80:174 o, _next_pos = decode_one(text, p)175 return o, p - 1176 p -= 1177 if p == p - 4:178 return error179 return None180 181 182def set_byte_encoding(enc: Literal["utf8", "narrow", "wide"]) -> None:183 if enc not in {"utf8", "narrow", "wide"}:184 raise ValueError(enc)185 global _byte_encoding # noqa: PLW0603 # pylint: disable=global-statement186 _byte_encoding = enc187 188 189def get_byte_encoding() -> Literal["utf8", "narrow", "wide"]:190 return _byte_encoding191 192 193def calc_string_text_pos(text: str, start_offs: int, end_offs: int, pref_col: int) -> tuple[int, int]:194 """195 Calculate the closest position to the screen column pref_col in text196 where start_offs is the offset into text assumed to be screen column 0197 and end_offs is the end of the range to search.198 199 Iterates by grapheme clusters for emoji ZWJ sequences, flags,200 combining characters, and other multi-codepoint unicode sequences.201 202 :param text: string203 :param start_offs: starting text position204 :param end_offs: ending text position205 :param pref_col: target column206 :returns: (position, actual_col)207 """208 if start_offs > end_offs:209 raise ValueError((start_offs, end_offs))210 211 cols = 0212 pos = start_offs213 for grapheme in wcwidth.iter_graphemes(text[start_offs:end_offs]):214 grapheme_width = wcwidth.width(grapheme, control_codes="ignore")215 if grapheme_width + cols > pref_col:216 return pos, cols217 cols += grapheme_width218 pos += len(grapheme)219 220 return end_offs, cols221 222 223def calc_text_pos(text: str | bytes, start_offs: int, end_offs: int, pref_col: int) -> tuple[int, int]:224 """225 Calculate the closest position to the screen column pref_col in text226 where start_offs is the offset into text assumed to be screen column 0227 and end_offs is the end of the range to search.228 229 text may be unicode or a byte string in the target _byte_encoding230 231 Returns (position, actual_col).232 """233 if start_offs > end_offs:234 raise ValueError((start_offs, end_offs))235 236 if isinstance(text, str):237 return calc_string_text_pos(text, start_offs, end_offs, pref_col)238 239 if not isinstance(text, bytes):240 raise TypeError(text)241 242 if _byte_encoding == "utf8":243 decoded = text[start_offs:end_offs].decode("utf-8")244 str_pos, cols = calc_string_text_pos(decoded, 0, len(decoded), pref_col)245 byte_offset = len(decoded[:str_pos].encode("utf-8"))246 return start_offs + byte_offset, cols247 248 # "wide" and "narrow"249 i = start_offs + pref_col250 if i >= end_offs:251 return end_offs, end_offs - start_offs252 if _byte_encoding == "wide" and within_double_byte(text, start_offs, i) == 2:253 i -= 1254 return i, i - start_offs255 256 257def calc_width(text: str | bytes, start_offs: int, end_offs: int) -> int:258 """259 Return the screen column width of text between start_offs and end_offs.260 261 text may be unicode or a byte string in the target _byte_encoding262 263 Some characters are wide (take two columns) and others affect the264 previous character (take zero columns), while others are grouped265 in sequence by "grapheme boundaries" (Emoji, Skin tones, flags, etc).266 """267 268 if start_offs > end_offs:269 msg = f"{start_offs=} > {end_offs=}"270 raise ValueError(msg)271 272 if isinstance(text, str):273 return wcwidth.width(text[start_offs:end_offs], control_codes="ignore")274 275 if _byte_encoding == "utf8":276 try:277 return wcwidth.width(text[start_offs:end_offs].decode("utf-8"), control_codes="ignore")278 except UnicodeDecodeError as exc:279 warnings.warn(280 "`calc_width` with text encoded to bytes can produce incorrect results"281 f"due to possible offset in the middle of character: {exc}",282 UnicodeWarning,283 stacklevel=2,284 )285 286 i = start_offs287 sc = 0288 while i < end_offs:289 o, i = decode_one(text, i)290 if (w := wcwidth.wcwidth(chr(o))) > 0:291 sc += w292 return sc293 # "wide", "narrow" or all printable ASCII, just return the character count294 return end_offs - start_offs295 296 297def is_wide_char(text: str | bytes, offs: int) -> bool:298 """299 Test if the grapheme cluster at offs within text is wide (2 columns).300 301 For Unicode strings, extracts the full grapheme cluster starting at offs302 and checks if it renders as wide. This correctly handles multi-codepoint303 graphemes like emoji ZWJ sequences and flags.304 305 text may be unicode or a byte string in the target _byte_encoding306 """307 if isinstance(text, str):308 grapheme = next(wcwidth.iter_graphemes(text[offs:]))309 return wcwidth.width(grapheme, control_codes="ignore") == 2310 if not isinstance(text, bytes):311 raise TypeError(text)312 if _byte_encoding == "utf8":313 grapheme, _ = _decode_grapheme_at(text, offs, len(text))314 return wcwidth.width(grapheme, control_codes="ignore") == 2315 if _byte_encoding == "wide":316 return within_double_byte(text, offs, offs) == 1317 return False318 319 320def move_prev_char(text: str | bytes, start_offs: int, end_offs: int) -> int:321 """322 Return the position of the grapheme cluster before end_offs.323 324 For Unicode strings, handle multi-codepoint, "grapheme clusters",325 to better measure emoji ZWJ, flags, combining characters, skin tones.326 """327 if start_offs >= end_offs:328 raise ValueError((start_offs, end_offs))329 if isinstance(text, str):330 return wcwidth.grapheme_boundary_before(text, end_offs)331 if not isinstance(text, bytes):332 raise TypeError(text)333 if _byte_encoding == "utf8":334 decoded = text[start_offs:end_offs].decode("utf-8")335 str_pos = len(decoded)336 prev_str_pos = wcwidth.grapheme_boundary_before(decoded, str_pos)337 prefix = decoded[:prev_str_pos]338 return start_offs + len(prefix.encode("utf-8"))339 if _byte_encoding == "wide" and within_double_byte(text, start_offs, end_offs - 1) == 2:340 return end_offs - 2341 return end_offs - 1342 343 344def move_next_char(text: str | bytes, start_offs: int, end_offs: int) -> int:345 """346 Return the position of the next grapheme cluster after start_offs.347 348 For Unicode strings, handle multi-codepoint, "grapheme clusters",349 to better measure emoji ZWJ, flags, combining characters, skin tones.350 """351 if start_offs >= end_offs:352 raise ValueError((start_offs, end_offs))353 if isinstance(text, str):354 grapheme = next(wcwidth.iter_graphemes(text[start_offs:end_offs]))355 return start_offs + len(grapheme)356 if not isinstance(text, bytes):357 raise TypeError(text)358 if _byte_encoding == "utf8":359 _, next_pos = _decode_grapheme_at(text, start_offs, end_offs)360 return next_pos361 if _byte_encoding == "wide" and within_double_byte(text, start_offs, start_offs) == 1:362 return start_offs + 2363 return start_offs + 1364 365 366def within_double_byte(text: bytes, line_start: int, pos: int) -> Literal[0, 1, 2]:367 """Return whether pos is within a double-byte encoded character.368 369 text -- byte string in question370 line_start -- offset of beginning of line (< pos)371 pos -- offset in question372 373 Return values:374 0 -- not within dbe char, or double_byte_encoding == False375 1 -- pos is on the 1st half of a dbe char376 2 -- pos is on the 2nd half of a dbe char377 """378 if not isinstance(text, bytes):379 raise TypeError(text)380 v = text[pos]381 382 if 0x40 <= v < 0x7F:383 # might be second half of big5, uhc or gbk encoding384 if pos == line_start:385 return 0386 387 if text[pos - 1] >= 0x81 and within_double_byte(text, line_start, pos - 1) == 1:388 return 2389 return 0390 391 if v < 0x80:392 return 0393 394 i = pos - 1395 while i >= line_start:396 if text[i] < 0x80:397 break398 i -= 1399 400 if (pos - i) & 1:401 return 1402 return 2403 