codekingpro/portable-devtools
114k
1"""2Mappings between PostgreSQL and Python encodings.3"""4 5# Copyright (C) 2020 The Psycopg Team6 7import re8import string9import codecs10from typing import Any, Dict, Optional, TYPE_CHECKING11 12from .pq._enums import ConnStatus13from .errors import NotSupportedError14from ._compat import cache15 16if TYPE_CHECKING:17 from .pq.abc import PGconn18 from .connection import BaseConnection19 20OK = ConnStatus.OK21 22 23_py_codecs = {24 "BIG5": "big5",25 "EUC_CN": "gb2312",26 "EUC_JIS_2004": "euc_jis_2004",27 "EUC_JP": "euc_jp",28 "EUC_KR": "euc_kr",29 # "EUC_TW": not available in Python30 "GB18030": "gb18030",31 "GBK": "gbk",32 "ISO_8859_5": "iso8859-5",33 "ISO_8859_6": "iso8859-6",34 "ISO_8859_7": "iso8859-7",35 "ISO_8859_8": "iso8859-8",36 "JOHAB": "johab",37 "KOI8R": "koi8-r",38 "KOI8U": "koi8-u",39 "LATIN1": "iso8859-1",40 "LATIN10": "iso8859-16",41 "LATIN2": "iso8859-2",42 "LATIN3": "iso8859-3",43 "LATIN4": "iso8859-4",44 "LATIN5": "iso8859-9",45 "LATIN6": "iso8859-10",46 "LATIN7": "iso8859-13",47 "LATIN8": "iso8859-14",48 "LATIN9": "iso8859-15",49 # "MULE_INTERNAL": not available in Python50 "SHIFT_JIS_2004": "shift_jis_2004",51 "SJIS": "shift_jis",52 # this actually means no encoding, see PostgreSQL docs53 # it is special-cased by the text loader.54 "SQL_ASCII": "ascii",55 "UHC": "cp949",56 "UTF8": "utf-8",57 "WIN1250": "cp1250",58 "WIN1251": "cp1251",59 "WIN1252": "cp1252",60 "WIN1253": "cp1253",61 "WIN1254": "cp1254",62 "WIN1255": "cp1255",63 "WIN1256": "cp1256",64 "WIN1257": "cp1257",65 "WIN1258": "cp1258",66 "WIN866": "cp866",67 "WIN874": "cp874",68}69 70py_codecs: Dict[bytes, str] = {}71py_codecs.update((k.encode(), v) for k, v in _py_codecs.items())72 73# Add an alias without underscore, for lenient lookups74py_codecs.update(75 (k.replace("_", "").encode(), v) for k, v in _py_codecs.items() if "_" in k76)77 78pg_codecs = {v: k.encode() for k, v in _py_codecs.items()}79 80 81def conn_encoding(conn: "Optional[BaseConnection[Any]]") -> str:82 """83 Return the Python encoding name of a psycopg connection.84 85 Default to utf8 if the connection has no encoding info.86 """87 if not conn or conn.closed:88 return "utf-8"89 90 pgenc = conn.pgconn.parameter_status(b"client_encoding") or b"UTF8"91 return pg2pyenc(pgenc)92 93 94def pgconn_encoding(pgconn: "PGconn") -> str:95 """96 Return the Python encoding name of a libpq connection.97 98 Default to utf8 if the connection has no encoding info.99 """100 if pgconn.status != OK:101 return "utf-8"102 103 pgenc = pgconn.parameter_status(b"client_encoding") or b"UTF8"104 return pg2pyenc(pgenc)105 106 107def conninfo_encoding(conninfo: str) -> str:108 """109 Return the Python encoding name passed in a conninfo string. Default to utf8.110 111 Because the input is likely to come from the user and not normalised by the112 server, be somewhat lenient (non-case-sensitive lookup, ignore noise chars).113 """114 from .conninfo import conninfo_to_dict115 116 params = conninfo_to_dict(conninfo)117 pgenc = params.get("client_encoding")118 if pgenc:119 try:120 return pg2pyenc(pgenc.encode())121 except NotSupportedError:122 pass123 124 return "utf-8"125 126 127@cache128def py2pgenc(name: str) -> bytes:129 """Convert a Python encoding name to PostgreSQL encoding name.130 131 Raise LookupError if the Python encoding is unknown.132 """133 return pg_codecs[codecs.lookup(name).name]134 135 136@cache137def pg2pyenc(name: bytes) -> str:138 """Convert a PostgreSQL encoding name to Python encoding name.139 140 Raise NotSupportedError if the PostgreSQL encoding is not supported by141 Python.142 """143 try:144 return py_codecs[name.replace(b"-", b"").replace(b"_", b"").upper()]145 except KeyError:146 sname = name.decode("utf8", "replace")147 raise NotSupportedError(f"codec not available in Python: {sname!r}")148 149 150def _as_python_identifier(s: str, prefix: str = "f") -> str:151 """152 Reduce a string to a valid Python identifier.153 154 Replace all non-valid chars with '_' and prefix the value with `!prefix` if155 the first letter is an '_'.156 """157 if not s.isidentifier():158 if s[0] in "1234567890":159 s = prefix + s160 if not s.isidentifier():161 s = _re_clean.sub("_", s)162 # namedtuple fields cannot start with underscore. So...163 if s[0] == "_":164 s = prefix + s165 return s166 167 168_re_clean = re.compile(169 f"[^{string.ascii_lowercase}{string.ascii_uppercase}{string.digits}_]"170)171 