Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
_encodings.py171 linesDownload Raw Back to psycopg
1"""2Mappings between PostgreSQL and Python encodings.3"""4 5# Copyright (C) 2020 The Psycopg Team6 7import re8import string9import codecs10from typing import Any, Dict, Optional, TYPE_CHECKING11 12from .pq._enums import ConnStatus13from .errors import NotSupportedError14from ._compat import cache15 16if TYPE_CHECKING:17    from .pq.abc import PGconn18    from .connection import BaseConnection19 20OK = ConnStatus.OK21 22 23_py_codecs = {24    "BIG5": "big5",25    "EUC_CN": "gb2312",26    "EUC_JIS_2004": "euc_jis_2004",27    "EUC_JP": "euc_jp",28    "EUC_KR": "euc_kr",29    # "EUC_TW": not available in Python30    "GB18030": "gb18030",31    "GBK": "gbk",32    "ISO_8859_5": "iso8859-5",33    "ISO_8859_6": "iso8859-6",34    "ISO_8859_7": "iso8859-7",35    "ISO_8859_8": "iso8859-8",36    "JOHAB": "johab",37    "KOI8R": "koi8-r",38    "KOI8U": "koi8-u",39    "LATIN1": "iso8859-1",40    "LATIN10": "iso8859-16",41    "LATIN2": "iso8859-2",42    "LATIN3": "iso8859-3",43    "LATIN4": "iso8859-4",44    "LATIN5": "iso8859-9",45    "LATIN6": "iso8859-10",46    "LATIN7": "iso8859-13",47    "LATIN8": "iso8859-14",48    "LATIN9": "iso8859-15",49    # "MULE_INTERNAL": not available in Python50    "SHIFT_JIS_2004": "shift_jis_2004",51    "SJIS": "shift_jis",52    # this actually means no encoding, see PostgreSQL docs53    # it is special-cased by the text loader.54    "SQL_ASCII": "ascii",55    "UHC": "cp949",56    "UTF8": "utf-8",57    "WIN1250": "cp1250",58    "WIN1251": "cp1251",59    "WIN1252": "cp1252",60    "WIN1253": "cp1253",61    "WIN1254": "cp1254",62    "WIN1255": "cp1255",63    "WIN1256": "cp1256",64    "WIN1257": "cp1257",65    "WIN1258": "cp1258",66    "WIN866": "cp866",67    "WIN874": "cp874",68}69 70py_codecs: Dict[bytes, str] = {}71py_codecs.update((k.encode(), v) for k, v in _py_codecs.items())72 73# Add an alias without underscore, for lenient lookups74py_codecs.update(75    (k.replace("_", "").encode(), v) for k, v in _py_codecs.items() if "_" in k76)77 78pg_codecs = {v: k.encode() for k, v in _py_codecs.items()}79 80 81def conn_encoding(conn: "Optional[BaseConnection[Any]]") -> str:82    """83    Return the Python encoding name of a psycopg connection.84 85    Default to utf8 if the connection has no encoding info.86    """87    if not conn or conn.closed:88        return "utf-8"89 90    pgenc = conn.pgconn.parameter_status(b"client_encoding") or b"UTF8"91    return pg2pyenc(pgenc)92 93 94def pgconn_encoding(pgconn: "PGconn") -> str:95    """96    Return the Python encoding name of a libpq connection.97 98    Default to utf8 if the connection has no encoding info.99    """100    if pgconn.status != OK:101        return "utf-8"102 103    pgenc = pgconn.parameter_status(b"client_encoding") or b"UTF8"104    return pg2pyenc(pgenc)105 106 107def conninfo_encoding(conninfo: str) -> str:108    """109    Return the Python encoding name passed in a conninfo string. Default to utf8.110 111    Because the input is likely to come from the user and not normalised by the112    server, be somewhat lenient (non-case-sensitive lookup, ignore noise chars).113    """114    from .conninfo import conninfo_to_dict115 116    params = conninfo_to_dict(conninfo)117    pgenc = params.get("client_encoding")118    if pgenc:119        try:120            return pg2pyenc(pgenc.encode())121        except NotSupportedError:122            pass123 124    return "utf-8"125 126 127@cache128def py2pgenc(name: str) -> bytes:129    """Convert a Python encoding name to PostgreSQL encoding name.130 131    Raise LookupError if the Python encoding is unknown.132    """133    return pg_codecs[codecs.lookup(name).name]134 135 136@cache137def pg2pyenc(name: bytes) -> str:138    """Convert a PostgreSQL encoding name to Python encoding name.139 140    Raise NotSupportedError if the PostgreSQL encoding is not supported by141    Python.142    """143    try:144        return py_codecs[name.replace(b"-", b"").replace(b"_", b"").upper()]145    except KeyError:146        sname = name.decode("utf8", "replace")147        raise NotSupportedError(f"codec not available in Python: {sname!r}")148 149 150def _as_python_identifier(s: str, prefix: str = "f") -> str:151    """152    Reduce a string to a valid Python identifier.153 154    Replace all non-valid chars with '_' and prefix the value with `!prefix` if155    the first letter is an '_'.156    """157    if not s.isidentifier():158        if s[0] in "1234567890":159            s = prefix + s160        if not s.isidentifier():161            s = _re_clean.sub("_", s)162    # namedtuple fields cannot start with underscore. So...163    if s[0] == "_":164        s = prefix + s165    return s166 167 168_re_clean = re.compile(169    f"[^{string.ascii_lowercase}{string.ascii_uppercase}{string.digits}_]"170)171 
codekingpro/portable-devtools · Team Ai