codekingpro/portable-devtools
114k
1""" codecs -- Python Codec Registry, API and helpers.2 3 4Written by Marc-Andre Lemburg (mal@lemburg.com).5 6(c) Copyright CNRI, All Rights Reserved. NO WARRANTY.7 8"""9 10import builtins11import sys12 13### Registry and builtin stateless codec functions14 15try:16 from _codecs import *17except ImportError as why:18 raise SystemError('Failed to load the builtin codecs: %s' % why)19 20__all__ = ["register", "lookup", "open", "EncodedFile", "BOM", "BOM_BE",21 "BOM_LE", "BOM32_BE", "BOM32_LE", "BOM64_BE", "BOM64_LE",22 "BOM_UTF8", "BOM_UTF16", "BOM_UTF16_LE", "BOM_UTF16_BE",23 "BOM_UTF32", "BOM_UTF32_LE", "BOM_UTF32_BE",24 "CodecInfo", "Codec", "IncrementalEncoder", "IncrementalDecoder",25 "StreamReader", "StreamWriter",26 "StreamReaderWriter", "StreamRecoder",27 "getencoder", "getdecoder", "getincrementalencoder",28 "getincrementaldecoder", "getreader", "getwriter",29 "encode", "decode", "iterencode", "iterdecode",30 "strict_errors", "ignore_errors", "replace_errors",31 "xmlcharrefreplace_errors",32 "backslashreplace_errors", "namereplace_errors",33 "register_error", "lookup_error"]34 35### Constants36 37#38# Byte Order Mark (BOM = ZERO WIDTH NO-BREAK SPACE = U+FEFF)39# and its possible byte string values40# for UTF8/UTF16/UTF32 output and little/big endian machines41#42 43# UTF-844BOM_UTF8 = b'\xef\xbb\xbf'45 46# UTF-16, little endian47BOM_LE = BOM_UTF16_LE = b'\xff\xfe'48 49# UTF-16, big endian50BOM_BE = BOM_UTF16_BE = b'\xfe\xff'51 52# UTF-32, little endian53BOM_UTF32_LE = b'\xff\xfe\x00\x00'54 55# UTF-32, big endian56BOM_UTF32_BE = b'\x00\x00\xfe\xff'57 58if sys.byteorder == 'little':59 60 # UTF-16, native endianness61 BOM = BOM_UTF16 = BOM_UTF16_LE62 63 # UTF-32, native endianness64 BOM_UTF32 = BOM_UTF32_LE65 66else:67 68 # UTF-16, native endianness69 BOM = BOM_UTF16 = BOM_UTF16_BE70 71 # UTF-32, native endianness72 BOM_UTF32 = BOM_UTF32_BE73 74# Old broken names (don't use in new code)75BOM32_LE = BOM_UTF16_LE76BOM32_BE = BOM_UTF16_BE77BOM64_LE = BOM_UTF32_LE78BOM64_BE = BOM_UTF32_BE79 80 81### Codec base classes (defining the API)82 83class CodecInfo(tuple):84 """Codec details when looking up the codec registry"""85 86 # Private API to allow Python 3.4 to denylist the known non-Unicode87 # codecs in the standard library. A more general mechanism to88 # reliably distinguish test encodings from other codecs will hopefully89 # be defined for Python 3.590 #91 # See http://bugs.python.org/issue1961992 _is_text_encoding = True # Assume codecs are text encodings by default93 94 def __new__(cls, encode, decode, streamreader=None, streamwriter=None,95 incrementalencoder=None, incrementaldecoder=None, name=None,96 *, _is_text_encoding=None):97 self = tuple.__new__(cls, (encode, decode, streamreader, streamwriter))98 self.name = name99 self.encode = encode100 self.decode = decode101 self.incrementalencoder = incrementalencoder102 self.incrementaldecoder = incrementaldecoder103 self.streamwriter = streamwriter104 self.streamreader = streamreader105 if _is_text_encoding is not None:106 self._is_text_encoding = _is_text_encoding107 return self108 109 def __repr__(self):110 return "<%s.%s object for encoding %s at %#x>" % \111 (self.__class__.__module__, self.__class__.__qualname__,112 self.name, id(self))113 114 def __getnewargs__(self):115 return tuple(self)116 117class Codec:118 119 """ Defines the interface for stateless encoders/decoders.120 121 The .encode()/.decode() methods may use different error122 handling schemes by providing the errors argument. These123 string values are predefined:124 125 'strict' - raise a ValueError error (or a subclass)126 'ignore' - ignore the character and continue with the next127 'replace' - replace with a suitable replacement character;128 Python will use the official U+FFFD REPLACEMENT129 CHARACTER for the builtin Unicode codecs on130 decoding and '?' on encoding.131 'surrogateescape' - replace with private code points U+DCnn.132 'xmlcharrefreplace' - Replace with the appropriate XML133 character reference (only for encoding).134 'backslashreplace' - Replace with backslashed escape sequences.135 'namereplace' - Replace with \\N{...} escape sequences136 (only for encoding).137 138 The set of allowed values can be extended via register_error.139 140 """141 def encode(self, input, errors='strict'):142 143 """ Encodes the object input and returns a tuple (output144 object, length consumed).145 146 errors defines the error handling to apply. It defaults to147 'strict' handling.148 149 The method may not store state in the Codec instance. Use150 StreamWriter for codecs which have to keep state in order to151 make encoding efficient.152 153 The encoder must be able to handle zero length input and154 return an empty object of the output object type in this155 situation.156 157 """158 raise NotImplementedError159 160 def decode(self, input, errors='strict'):161 162 """ Decodes the object input and returns a tuple (output163 object, length consumed).164 165 input must be an object which provides the bf_getreadbuf166 buffer slot. Python strings, buffer objects and memory167 mapped files are examples of objects providing this slot.168 169 errors defines the error handling to apply. It defaults to170 'strict' handling.171 172 The method may not store state in the Codec instance. Use173 StreamReader for codecs which have to keep state in order to174 make decoding efficient.175 176 The decoder must be able to handle zero length input and177 return an empty object of the output object type in this178 situation.179 180 """181 raise NotImplementedError182 183class IncrementalEncoder(object):184 """185 An IncrementalEncoder encodes an input in multiple steps. The input can186 be passed piece by piece to the encode() method. The IncrementalEncoder187 remembers the state of the encoding process between calls to encode().188 """189 def __init__(self, errors='strict'):190 """191 Creates an IncrementalEncoder instance.192 193 The IncrementalEncoder may use different error handling schemes by194 providing the errors keyword argument. See the module docstring195 for a list of possible values.196 """197 self.errors = errors198 self.buffer = ""199 200 def encode(self, input, final=False):201 """202 Encodes input and returns the resulting object.203 """204 raise NotImplementedError205 206 def reset(self):207 """208 Resets the encoder to the initial state.209 """210 211 def getstate(self):212 """213 Return the current state of the encoder.214 """215 return 0216 217 def setstate(self, state):218 """219 Set the current state of the encoder. state must have been220 returned by getstate().221 """222 223class BufferedIncrementalEncoder(IncrementalEncoder):224 """225 This subclass of IncrementalEncoder can be used as the baseclass for an226 incremental encoder if the encoder must keep some of the output in a227 buffer between calls to encode().228 """229 def __init__(self, errors='strict'):230 IncrementalEncoder.__init__(self, errors)231 # unencoded input that is kept between calls to encode()232 self.buffer = ""233 234 def _buffer_encode(self, input, errors, final):235 # Overwrite this method in subclasses: It must encode input236 # and return an (output, length consumed) tuple237 raise NotImplementedError238 239 def encode(self, input, final=False):240 # encode input (taking the buffer into account)241 data = self.buffer + input242 (result, consumed) = self._buffer_encode(data, self.errors, final)243 # keep unencoded input until the next call244 self.buffer = data[consumed:]245 return result246 247 def reset(self):248 IncrementalEncoder.reset(self)249 self.buffer = ""250 251 def getstate(self):252 return self.buffer or 0253 254 def setstate(self, state):255 self.buffer = state or ""256 257class IncrementalDecoder(object):258 """259 An IncrementalDecoder decodes an input in multiple steps. The input can260 be passed piece by piece to the decode() method. The IncrementalDecoder261 remembers the state of the decoding process between calls to decode().262 """263 def __init__(self, errors='strict'):264 """265 Create an IncrementalDecoder instance.266 267 The IncrementalDecoder may use different error handling schemes by268 providing the errors keyword argument. See the module docstring269 for a list of possible values.270 """271 self.errors = errors272 273 def decode(self, input, final=False):274 """275 Decode input and returns the resulting object.276 """277 raise NotImplementedError278 279 def reset(self):280 """281 Reset the decoder to the initial state.282 """283 284 def getstate(self):285 """286 Return the current state of the decoder.287 288 This must be a (buffered_input, additional_state_info) tuple.289 buffered_input must be a bytes object containing bytes that290 were passed to decode() that have not yet been converted.291 additional_state_info must be a non-negative integer292 representing the state of the decoder WITHOUT yet having293 processed the contents of buffered_input. In the initial state294 and after reset(), getstate() must return (b"", 0).295 """296 return (b"", 0)297 298 def setstate(self, state):299 """300 Set the current state of the decoder.301 302 state must have been returned by getstate(). The effect of303 setstate((b"", 0)) must be equivalent to reset().304 """305 306class BufferedIncrementalDecoder(IncrementalDecoder):307 """308 This subclass of IncrementalDecoder can be used as the baseclass for an309 incremental decoder if the decoder must be able to handle incomplete310 byte sequences.311 """312 def __init__(self, errors='strict'):313 IncrementalDecoder.__init__(self, errors)314 # undecoded input that is kept between calls to decode()315 self.buffer = b""316 317 def _buffer_decode(self, input, errors, final):318 # Overwrite this method in subclasses: It must decode input319 # and return an (output, length consumed) tuple320 raise NotImplementedError321 322 def decode(self, input, final=False):323 # decode input (taking the buffer into account)324 data = self.buffer + input325 (result, consumed) = self._buffer_decode(data, self.errors, final)326 # keep undecoded input until the next call327 self.buffer = data[consumed:]328 return result329 330 def reset(self):331 IncrementalDecoder.reset(self)332 self.buffer = b""333 334 def getstate(self):335 # additional state info is always 0336 return (self.buffer, 0)337 338 def setstate(self, state):339 # ignore additional state info340 self.buffer = state[0]341 342#343# The StreamWriter and StreamReader class provide generic working344# interfaces which can be used to implement new encoding submodules345# very easily. See encodings/utf_8.py for an example on how this is346# done.347#348 349class StreamWriter(Codec):350 351 def __init__(self, stream, errors='strict'):352 353 """ Creates a StreamWriter instance.354 355 stream must be a file-like object open for writing.356 357 The StreamWriter may use different error handling358 schemes by providing the errors keyword argument. These359 parameters are predefined:360 361 'strict' - raise a ValueError (or a subclass)362 'ignore' - ignore the character and continue with the next363 'replace'- replace with a suitable replacement character364 'xmlcharrefreplace' - Replace with the appropriate XML365 character reference.366 'backslashreplace' - Replace with backslashed escape367 sequences.368 'namereplace' - Replace with \\N{...} escape sequences.369 370 The set of allowed parameter values can be extended via371 register_error.372 """373 self.stream = stream374 self.errors = errors375 376 def write(self, object):377 378 """ Writes the object's contents encoded to self.stream.379 """380 data, consumed = self.encode(object, self.errors)381 self.stream.write(data)382 383 def writelines(self, list):384 385 """ Writes the concatenated list of strings to the stream386 using .write().387 """388 self.write(''.join(list))389 390 def reset(self):391 392 """ Resets the codec buffers used for keeping internal state.393 394 Calling this method should ensure that the data on the395 output is put into a clean state, that allows appending396 of new fresh data without having to rescan the whole397 stream to recover state.398 399 """400 pass401 402 def seek(self, offset, whence=0):403 self.stream.seek(offset, whence)404 if whence == 0 and offset == 0:405 self.reset()406 407 def __getattr__(self, name,408 getattr=getattr):409 410 """ Inherit all other methods from the underlying stream.411 """412 return getattr(self.stream, name)413 414 def __enter__(self):415 return self416 417 def __exit__(self, type, value, tb):418 self.stream.close()419 420 def __reduce_ex__(self, proto):421 raise TypeError("can't serialize %s" % self.__class__.__name__)422 423###424 425class StreamReader(Codec):426 427 charbuffertype = str428 429 def __init__(self, stream, errors='strict'):430 431 """ Creates a StreamReader instance.432 433 stream must be a file-like object open for reading.434 435 The StreamReader may use different error handling436 schemes by providing the errors keyword argument. These437 parameters are predefined:438 439 'strict' - raise a ValueError (or a subclass)440 'ignore' - ignore the character and continue with the next441 'replace'- replace with a suitable replacement character442 'backslashreplace' - Replace with backslashed escape sequences;443 444 The set of allowed parameter values can be extended via445 register_error.446 """447 self.stream = stream448 self.errors = errors449 self.bytebuffer = b""450 self._empty_charbuffer = self.charbuffertype()451 self.charbuffer = self._empty_charbuffer452 self.linebuffer = None453 454 def decode(self, input, errors='strict'):455 raise NotImplementedError456 457 def read(self, size=-1, chars=-1, firstline=False):458 459 """ Decodes data from the stream self.stream and returns the460 resulting object.461 462 chars indicates the number of decoded code points or bytes to463 return. read() will never return more data than requested,464 but it might return less, if there is not enough available.465 466 size indicates the approximate maximum number of decoded467 bytes or code points to read for decoding. The decoder468 can modify this setting as appropriate. The default value469 -1 indicates to read and decode as much as possible. size470 is intended to prevent having to decode huge files in one471 step.472 473 If firstline is true, and a UnicodeDecodeError happens474 after the first line terminator in the input only the first line475 will be returned, the rest of the input will be kept until the476 next call to read().477 478 The method should use a greedy read strategy, meaning that479 it should read as much data as is allowed within the480 definition of the encoding and the given size, e.g. if481 optional encoding endings or state markers are available482 on the stream, these should be read too.483 """484 # If we have lines cached, first merge them back into characters485 if self.linebuffer:486 self.charbuffer = self._empty_charbuffer.join(self.linebuffer)487 self.linebuffer = None488 489 if chars < 0:490 # For compatibility with other read() methods that take a491 # single argument492 chars = size493 494 # read until we get the required number of characters (if available)495 while True:496 # can the request be satisfied from the character buffer?497 if chars >= 0:498 if len(self.charbuffer) >= chars:499 break500 # we need more data501 if size < 0:502 newdata = self.stream.read()503 else:504 newdata = self.stream.read(size)505 # decode bytes (those remaining from the last call included)506 data = self.bytebuffer + newdata507 if not data:508 break509 try:510 newchars, decodedbytes = self.decode(data, self.errors)511 except UnicodeDecodeError as exc:512 if firstline:513 newchars, decodedbytes = \514 self.decode(data[:exc.start], self.errors)515 lines = newchars.splitlines(keepends=True)516 if len(lines)<=1:517 raise518 else:519 raise520 # keep undecoded bytes until the next call521 self.bytebuffer = data[decodedbytes:]522 # put new characters in the character buffer523 self.charbuffer += newchars524 # there was no data available525 if not newdata:526 break527 if chars < 0:528 # Return everything we've got529 result = self.charbuffer530 self.charbuffer = self._empty_charbuffer531 else:532 # Return the first chars characters533 result = self.charbuffer[:chars]534 self.charbuffer = self.charbuffer[chars:]535 return result536 537 def readline(self, size=None, keepends=True):538 539 """ Read one line from the input stream and return the540 decoded data.541 542 size, if given, is passed as size argument to the543 read() method.544 545 """546 # If we have lines cached from an earlier read, return547 # them unconditionally548 if self.linebuffer:549 line = self.linebuffer[0]550 del self.linebuffer[0]551 if len(self.linebuffer) == 1:552 # revert to charbuffer mode; we might need more data553 # next time554 self.charbuffer = self.linebuffer[0]555 self.linebuffer = None556 if not keepends:557 line = line.splitlines(keepends=False)[0]558 return line559 560 readsize = size or 72561 line = self._empty_charbuffer562 # If size is given, we call read() only once563 while True:564 data = self.read(readsize, firstline=True)565 if data:566 # If we're at a "\r" read one extra character (which might567 # be a "\n") to get a proper line ending. If the stream is568 # temporarily exhausted we return the wrong line ending.569 if (isinstance(data, str) and data.endswith("\r")) or \570 (isinstance(data, bytes) and data.endswith(b"\r")):571 data += self.read(size=1, chars=1)572 573 line += data574 lines = line.splitlines(keepends=True)575 if lines:576 if len(lines) > 1:577 # More than one line result; the first line is a full line578 # to return579 line = lines[0]580 del lines[0]581 if len(lines) > 1:582 # cache the remaining lines583 lines[-1] += self.charbuffer584 self.linebuffer = lines585 self.charbuffer = None586 else:587 # only one remaining line, put it back into charbuffer588 self.charbuffer = lines[0] + self.charbuffer589 if not keepends:590 line = line.splitlines(keepends=False)[0]591 break592 line0withend = lines[0]593 line0withoutend = lines[0].splitlines(keepends=False)[0]594 if line0withend != line0withoutend: # We really have a line end595 # Put the rest back together and keep it until the next call596 self.charbuffer = self._empty_charbuffer.join(lines[1:]) + \597 self.charbuffer598 if keepends:599 line = line0withend600 else:601 line = line0withoutend602 break603 # we didn't get anything or this was our only try604 if not data or size is not None:605 if line and not keepends:606 line = line.splitlines(keepends=False)[0]607 break608 if readsize < 8000:609 readsize *= 2610 return line611 612 def readlines(self, sizehint=None, keepends=True):613 614 """ Read all lines available on the input stream615 and return them as a list.616 617 Line breaks are implemented using the codec's decoder618 method and are included in the list entries.619 620 sizehint, if given, is ignored since there is no efficient621 way of finding the true end-of-line.622 623 """624 data = self.read()625 return data.splitlines(keepends)626 627 def reset(self):628 629 """ Resets the codec buffers used for keeping internal state.630 631 Note that no stream repositioning should take place.632 This method is primarily intended to be able to recover633 from decoding errors.634 635 """636 self.bytebuffer = b""637 self.charbuffer = self._empty_charbuffer638 self.linebuffer = None639 640 def seek(self, offset, whence=0):641 """ Set the input stream's current position.642 643 Resets the codec buffers used for keeping state.644 """645 self.stream.seek(offset, whence)646 self.reset()647 648 def __next__(self):649 650 """ Return the next decoded line from the input stream."""651 line = self.readline()652 if line:653 return line654 raise StopIteration655 656 def __iter__(self):657 return self658 659 def __getattr__(self, name,660 getattr=getattr):661 662 """ Inherit all other methods from the underlying stream.663 """664 return getattr(self.stream, name)665 666 def __enter__(self):667 return self668 669 def __exit__(self, type, value, tb):670 self.stream.close()671 672 def __reduce_ex__(self, proto):673 raise TypeError("can't serialize %s" % self.__class__.__name__)674 675###676 677class StreamReaderWriter:678 679 """ StreamReaderWriter instances allow wrapping streams which680 work in both read and write modes.681 682 The design is such that one can use the factory functions683 returned by the codec.lookup() function to construct the684 instance.685 686 """687 # Optional attributes set by the file wrappers below688 encoding = 'unknown'689 690 def __init__(self, stream, Reader, Writer, errors='strict'):691 692 """ Creates a StreamReaderWriter instance.693 694 stream must be a Stream-like object.695 696 Reader, Writer must be factory functions or classes697 providing the StreamReader, StreamWriter interface resp.698 699 Error handling is done in the same way as defined for the700 StreamWriter/Readers.701 702 """703 self.stream = stream704 self.reader = Reader(stream, errors)705 self.writer = Writer(stream, errors)706 self.errors = errors707 708 def read(self, size=-1):709 710 return self.reader.read(size)711 712 def readline(self, size=None, keepends=True):713 714 return self.reader.readline(size, keepends)715 716 def readlines(self, sizehint=None, keepends=True):717 718 return self.reader.readlines(sizehint, keepends)719 720 def __next__(self):721 722 """ Return the next decoded line from the input stream."""723 return next(self.reader)724 725 def __iter__(self):726 return self727 728 def write(self, data):729 730 return self.writer.write(data)731 732 def writelines(self, list):733 734 return self.writer.writelines(list)735 736 def reset(self):737 738 self.reader.reset()739 self.writer.reset()740 741 def seek(self, offset, whence=0):742 self.stream.seek(offset, whence)743 self.reader.reset()744 if whence == 0 and offset == 0:745 self.writer.reset()746 747 def __getattr__(self, name,748 getattr=getattr):749 750 """ Inherit all other methods from the underlying stream.751 """752 return getattr(self.stream, name)753 754 # these are needed to make "with StreamReaderWriter(...)" work properly755 756 def __enter__(self):757 return self758 759 def __exit__(self, type, value, tb):760 self.stream.close()761 762 def __reduce_ex__(self, proto):763 raise TypeError("can't serialize %s" % self.__class__.__name__)764 765###766 767class StreamRecoder:768 769 """ StreamRecoder instances translate data from one encoding to another.770 771 They use the complete set of APIs returned by the772 codecs.lookup() function to implement their task.773 774 Data written to the StreamRecoder is first decoded into an775 intermediate format (depending on the "decode" codec) and then776 written to the underlying stream using an instance of the provided777 Writer class.778 779 In the other direction, data is read from the underlying stream using780 a Reader instance and then encoded and returned to the caller.781 782 """783 # Optional attributes set by the file wrappers below784 data_encoding = 'unknown'785 file_encoding = 'unknown'786 787 def __init__(self, stream, encode, decode, Reader, Writer,788 errors='strict'):789 790 """ Creates a StreamRecoder instance which implements a two-way791 conversion: encode and decode work on the frontend (the792 data visible to .read() and .write()) while Reader and Writer793 work on the backend (the data in stream).794 795 You can use these objects to do transparent796 transcodings from e.g. latin-1 to utf-8 and back.797 798 stream must be a file-like object.799 800 encode and decode must adhere to the Codec interface; Reader and801 Writer must be factory functions or classes providing the802 StreamReader and StreamWriter interfaces resp.803 804 Error handling is done in the same way as defined for the805 StreamWriter/Readers.806 807 """808 self.stream = stream809 self.encode = encode810 self.decode = decode811 self.reader = Reader(stream, errors)812 self.writer = Writer(stream, errors)813 self.errors = errors814 815 def read(self, size=-1):816 817 data = self.reader.read(size)818 data, bytesencoded = self.encode(data, self.errors)819 return data820 821 def readline(self, size=None):822 823 if size is None:824 data = self.reader.readline()825 else:826 data = self.reader.readline(size)827 data, bytesencoded = self.encode(data, self.errors)828 return data829 830 def readlines(self, sizehint=None):831 832 data = self.reader.read()833 data, bytesencoded = self.encode(data, self.errors)834 return data.splitlines(keepends=True)835 836 def __next__(self):837 838 """ Return the next decoded line from the input stream."""839 data = next(self.reader)840 data, bytesencoded = self.encode(data, self.errors)841 return data842 843 def __iter__(self):844 return self845 846 def write(self, data):847 848 data, bytesdecoded = self.decode(data, self.errors)849 return self.writer.write(data)850 851 def writelines(self, list):852 853 data = b''.join(list)854 data, bytesdecoded = self.decode(data, self.errors)855 return self.writer.write(data)856 857 def reset(self):858 859 self.reader.reset()860 self.writer.reset()861 862 def seek(self, offset, whence=0):863 # Seeks must be propagated to both the readers and writers864 # as they might need to reset their internal buffers.865 self.reader.seek(offset, whence)866 self.writer.seek(offset, whence)867 868 def __getattr__(self, name,869 getattr=getattr):870 871 """ Inherit all other methods from the underlying stream.872 """873 return getattr(self.stream, name)874 875 def __enter__(self):876 return self877 878 def __exit__(self, type, value, tb):879 self.stream.close()880 881 def __reduce_ex__(self, proto):882 raise TypeError("can't serialize %s" % self.__class__.__name__)883 884### Shortcuts885 886def open(filename, mode='r', encoding=None, errors='strict', buffering=-1):887 """ Open an encoded file using the given mode and return888 a wrapped version providing transparent encoding/decoding.889 890 Note: The wrapped version will only accept the object format891 defined by the codecs, i.e. Unicode objects for most builtin892 codecs. Output is also codec dependent and will usually be893 Unicode as well.894 895 If encoding is not None, then the896 underlying encoded files are always opened in binary mode.897 The default file mode is 'r', meaning to open the file in read mode.898 899 encoding specifies the encoding which is to be used for the900 file.901 902 errors may be given to define the error handling. It defaults903 to 'strict' which causes ValueErrors to be raised in case an904 encoding error occurs.905 906 buffering has the same meaning as for the builtin open() API.907 It defaults to -1 which means that the default buffer size will908 be used.909 910 The returned wrapped file object provides an extra attribute911 .encoding which allows querying the used encoding. This912 attribute is only available if an encoding was specified as913 parameter.914 """915 import warnings916 warnings.warn("codecs.open() is deprecated. Use open() instead.",917 DeprecationWarning, stacklevel=2)918 919 if encoding is not None and \920 'b' not in mode:921 # Force opening of the file in binary mode922 mode = mode + 'b'923 file = builtins.open(filename, mode, buffering)924 if encoding is None:925 return file926 927 try:928 info = lookup(encoding)929 srw = StreamReaderWriter(file, info.streamreader, info.streamwriter, errors)930 # Add attributes to simplify introspection931 srw.encoding = encoding932 return srw933 except:934 file.close()935 raise936 937def EncodedFile(file, data_encoding, file_encoding=None, errors='strict'):938 939 """ Return a wrapped version of file which provides transparent940 encoding translation.941 942 Data written to the wrapped file is decoded according943 to the given data_encoding and then encoded to the underlying944 file using file_encoding. The intermediate data type945 will usually be Unicode but depends on the specified codecs.946 947 Bytes read from the file are decoded using file_encoding and then948 passed back to the caller encoded using data_encoding.949 950 If file_encoding is not given, it defaults to data_encoding.951 952 errors may be given to define the error handling. It defaults953 to 'strict' which causes ValueErrors to be raised in case an954 encoding error occurs.955 956 The returned wrapped file object provides two extra attributes957 .data_encoding and .file_encoding which reflect the given958 parameters of the same name. The attributes can be used for959 introspection by Python programs.960 961 """962 if file_encoding is None:963 file_encoding = data_encoding964 data_info = lookup(data_encoding)965 file_info = lookup(file_encoding)966 sr = StreamRecoder(file, data_info.encode, data_info.decode,967 file_info.streamreader, file_info.streamwriter, errors)968 # Add attributes to simplify introspection969 sr.data_encoding = data_encoding970 sr.file_encoding = file_encoding971 return sr972 973### Helpers for codec lookup974 975def getencoder(encoding):976 977 """ Lookup up the codec for the given encoding and return978 its encoder function.979 980 Raises a LookupError in case the encoding cannot be found.981 982 """983 return lookup(encoding).encode984 985def getdecoder(encoding):986 987 """ Lookup up the codec for the given encoding and return988 its decoder function.989 990 Raises a LookupError in case the encoding cannot be found.991 992 """993 return lookup(encoding).decode994 995def getincrementalencoder(encoding):996 997 """ Lookup up the codec for the given encoding and return998 its IncrementalEncoder class or factory function.999 1000 Raises a LookupError in case the encoding cannot be found1001 or the codecs doesn't provide an incremental encoder.1002 1003 """1004 encoder = lookup(encoding).incrementalencoder1005 if encoder is None:1006 raise LookupError(encoding)1007 return encoder1008 1009def getincrementaldecoder(encoding):1010 1011 """ Lookup up the codec for the given encoding and return1012 its IncrementalDecoder class or factory function.1013 1014 Raises a LookupError in case the encoding cannot be found1015 or the codecs doesn't provide an incremental decoder.1016 1017 """1018 decoder = lookup(encoding).incrementaldecoder1019 if decoder is None:1020 raise LookupError(encoding)1021 return decoder1022 1023def getreader(encoding):1024 1025 """ Lookup up the codec for the given encoding and return1026 its StreamReader class or factory function.1027 1028 Raises a LookupError in case the encoding cannot be found.1029 1030 """1031 return lookup(encoding).streamreader1032 1033def getwriter(encoding):1034 1035 """ Lookup up the codec for the given encoding and return1036 its StreamWriter class or factory function.1037 1038 Raises a LookupError in case the encoding cannot be found.1039 1040 """1041 return lookup(encoding).streamwriter1042 1043def iterencode(iterator, encoding, errors='strict', **kwargs):1044 """1045 Encoding iterator.1046 1047 Encodes the input strings from the iterator using an IncrementalEncoder.1048 1049 errors and kwargs are passed through to the IncrementalEncoder1050 constructor.1051 """1052 encoder = getincrementalencoder(encoding)(errors, **kwargs)1053 for input in iterator:1054 output = encoder.encode(input)1055 if output:1056 yield output1057 output = encoder.encode("", True)1058 if output:1059 yield output1060 1061def iterdecode(iterator, encoding, errors='strict', **kwargs):1062 """1063 Decoding iterator.1064 1065 Decodes the input strings from the iterator using an IncrementalDecoder.1066 1067 errors and kwargs are passed through to the IncrementalDecoder1068 constructor.1069 """1070 decoder = getincrementaldecoder(encoding)(errors, **kwargs)1071 for input in iterator:1072 output = decoder.decode(input)1073 if output:1074 yield output1075 output = decoder.decode(b"", True)1076 if output:1077 yield output1078 1079### Helpers for charmap-based codecs1080 1081def make_identity_dict(rng):1082 1083 """ make_identity_dict(rng) -> dict1084 1085 Return a dictionary where elements of the rng sequence are1086 mapped to themselves.1087 1088 """1089 return {i:i for i in rng}1090 1091def make_encoding_map(decoding_map):1092 1093 """ Creates an encoding map from a decoding map.1094 1095 If a target mapping in the decoding map occurs multiple1096 times, then that target is mapped to None (undefined mapping),1097 causing an exception when encountered by the charmap codec1098 during translation.1099 1100 One example where this happens is cp875.py which decodes1101 multiple character to \\u001a.1102 1103 """1104 m = {}1105 for k,v in decoding_map.items():1106 if not v in m:1107 m[v] = k1108 else:1109 m[v] = None1110 return m1111 1112### error handlers1113 1114strict_errors = lookup_error("strict")1115ignore_errors = lookup_error("ignore")1116replace_errors = lookup_error("replace")1117xmlcharrefreplace_errors = lookup_error("xmlcharrefreplace")1118backslashreplace_errors = lookup_error("backslashreplace")1119namereplace_errors = lookup_error("namereplace")1120 1121# Tell modulefinder that using codecs probably needs the encodings1122# package1123_false = 01124if _false:1125 import encodings # noqa: F4011126 