Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
gzip.py718 linesDownload Raw Back to Lib
1"""Functions that read and write gzipped files.2 3The user of the file doesn't have to worry about the compression,4but random access is not allowed."""5 6# based on Andrew Kuchling's minigzip.py distributed with the zlib module7 8import builtins9import io10import os11import struct12import sys13import time14import weakref15import zlib16from compression._common import _streams17 18__all__ = ["BadGzipFile", "GzipFile", "open", "compress", "decompress"]19 20FTEXT, FHCRC, FEXTRA, FNAME, FCOMMENT = 1, 2, 4, 8, 1621 22READ = 'rb'23WRITE = 'wb'24 25_COMPRESS_LEVEL_FAST = 126_COMPRESS_LEVEL_TRADEOFF = 627_COMPRESS_LEVEL_BEST = 928 29READ_BUFFER_SIZE = 128 * 102430_WRITE_BUFFER_SIZE = 4 * io.DEFAULT_BUFFER_SIZE31 32 33def open(filename, mode="rb", compresslevel=_COMPRESS_LEVEL_BEST,34         encoding=None, errors=None, newline=None):35    """Open a gzip-compressed file in binary or text mode.36 37    The filename argument can be an actual filename (a str or bytes object), or38    an existing file object to read from or write to.39 40    The mode argument can be "r", "rb", "w", "wb", "x", "xb", "a" or "ab" for41    binary mode, or "rt", "wt", "xt" or "at" for text mode. The default mode is42    "rb", and the default compresslevel is 9.43 44    For binary mode, this function is equivalent to the GzipFile constructor:45    GzipFile(filename, mode, compresslevel). In this case, the encoding, errors46    and newline arguments must not be provided.47 48    For text mode, a GzipFile object is created, and wrapped in an49    io.TextIOWrapper instance with the specified encoding, error handling50    behavior, and line ending(s).51 52    """53    if "t" in mode:54        if "b" in mode:55            raise ValueError("Invalid mode: %r" % (mode,))56    else:57        if encoding is not None:58            raise ValueError("Argument 'encoding' not supported in binary mode")59        if errors is not None:60            raise ValueError("Argument 'errors' not supported in binary mode")61        if newline is not None:62            raise ValueError("Argument 'newline' not supported in binary mode")63 64    gz_mode = mode.replace("t", "")65    if isinstance(filename, (str, bytes, os.PathLike)):66        binary_file = GzipFile(filename, gz_mode, compresslevel)67    elif hasattr(filename, "read") or hasattr(filename, "write"):68        binary_file = GzipFile(None, gz_mode, compresslevel, filename)69    else:70        raise TypeError("filename must be a str or bytes object, or a file")71 72    if "t" in mode:73        encoding = io.text_encoding(encoding)74        return io.TextIOWrapper(binary_file, encoding, errors, newline)75    else:76        return binary_file77 78def write32u(output, value):79    # The L format writes the bit pattern correctly whether signed80    # or unsigned.81    output.write(struct.pack("<L", value))82 83class _PaddedFile:84    """Minimal read-only file object that prepends a string to the contents85    of an actual file. Shouldn't be used outside of gzip.py, as it lacks86    essential functionality."""87 88    def __init__(self, f, prepend=b''):89        self._buffer = prepend90        self._length = len(prepend)91        self.file = f92        self._read = 093 94    def read(self, size):95        if self._read is None:96            return self.file.read(size)97        if self._read + size <= self._length:98            read = self._read99            self._read += size100            return self._buffer[read:self._read]101        else:102            read = self._read103            self._read = None104            return self._buffer[read:] + \105                   self.file.read(size-self._length+read)106 107    def prepend(self, prepend=b''):108        if self._read is None:109            self._buffer = prepend110        else:  # Assume data was read since the last prepend() call111            self._read -= len(prepend)112            return113        self._length = len(self._buffer)114        self._read = 0115 116    def seek(self, off):117        self._read = None118        self._buffer = None119        return self.file.seek(off)120 121    def seekable(self):122        return True  # Allows fast-forwarding even in unseekable streams123 124 125class BadGzipFile(OSError):126    """Exception raised in some cases for invalid gzip files."""127 128 129class _WriteBufferStream(io.RawIOBase):130    """Minimal object to pass WriteBuffer flushes into GzipFile"""131    def __init__(self, gzip_file):132        self.gzip_file = weakref.ref(gzip_file)133 134    def write(self, data):135        gzip_file = self.gzip_file()136        if gzip_file is None:137            raise RuntimeError("lost gzip_file")138        return gzip_file._write_raw(data)139 140    def seekable(self):141        return False142 143    def writable(self):144        return True145 146 147class GzipFile(_streams.BaseStream):148    """The GzipFile class simulates most of the methods of a file object with149    the exception of the truncate() method.150 151    This class only supports opening files in binary mode. If you need to open a152    compressed file in text mode, use the gzip.open() function.153 154    """155 156    # Overridden with internal file object to be closed, if only a filename157    # is passed in158    myfileobj = None159 160    def __init__(self, filename=None, mode=None,161                 compresslevel=_COMPRESS_LEVEL_BEST, fileobj=None, mtime=None):162        """Constructor for the GzipFile class.163 164        At least one of fileobj and filename must be given a165        non-trivial value.166 167        The new class instance is based on fileobj, which can be a regular168        file, an io.BytesIO object, or any other object which simulates a file.169        It defaults to None, in which case filename is opened to provide170        a file object.171 172        When fileobj is not None, the filename argument is only used to be173        included in the gzip file header, which may include the original174        filename of the uncompressed file.  It defaults to the filename of175        fileobj, if discernible; otherwise, it defaults to the empty string,176        and in this case the original filename is not included in the header.177 178        The mode argument can be any of 'r', 'rb', 'a', 'ab', 'w', 'wb', 'x', or179        'xb' depending on whether the file will be read or written.  The default180        is the mode of fileobj if discernible; otherwise, the default is 'rb'.181        A mode of 'r' is equivalent to one of 'rb', and similarly for 'w' and182        'wb', 'a' and 'ab', and 'x' and 'xb'.183 184        The compresslevel argument is an integer from 0 to 9 controlling the185        level of compression; 1 is fastest and produces the least compression,186        and 9 is slowest and produces the most compression. 0 is no compression187        at all. The default is 9.188 189        The optional mtime argument is the timestamp requested by gzip. The time190        is in Unix format, i.e., seconds since 00:00:00 UTC, January 1, 1970.191        If mtime is omitted or None, the current time is used. Use mtime = 0192        to generate a compressed stream that does not depend on creation time.193 194        """195 196        # Ensure attributes exist at __del__197        self.mode = None198        self.fileobj = None199        self._buffer = None200 201        if mode and ('t' in mode or 'U' in mode):202            raise ValueError("Invalid mode: {!r}".format(mode))203        if mode and 'b' not in mode:204            mode += 'b'205 206        try:207            if fileobj is None:208                fileobj = self.myfileobj = builtins.open(filename, mode or 'rb')209            if filename is None:210                filename = getattr(fileobj, 'name', '')211                if not isinstance(filename, (str, bytes)):212                    filename = ''213            else:214                filename = os.fspath(filename)215            origmode = mode216            if mode is None:217                mode = getattr(fileobj, 'mode', 'rb')218 219 220            if mode.startswith('r'):221                self.mode = READ222                raw = _GzipReader(fileobj)223                self._buffer = io.BufferedReader(raw)224                self.name = filename225 226            elif mode.startswith(('w', 'a', 'x')):227                if origmode is None:228                    import warnings229                    warnings.warn(230                        "GzipFile was opened for writing, but this will "231                        "change in future Python releases.  "232                        "Specify the mode argument for opening it for writing.",233                        FutureWarning, 2)234                self.mode = WRITE235                self._init_write(filename)236                self.compress = zlib.compressobj(compresslevel,237                                                 zlib.DEFLATED,238                                                 -zlib.MAX_WBITS,239                                                 zlib.DEF_MEM_LEVEL,240                                                 0)241                self._write_mtime = mtime242                self._buffer_size = _WRITE_BUFFER_SIZE243                self._buffer = io.BufferedWriter(_WriteBufferStream(self),244                                                 buffer_size=self._buffer_size)245            else:246                raise ValueError("Invalid mode: {!r}".format(mode))247 248            self.fileobj = fileobj249 250            if self.mode == WRITE:251                self._write_gzip_header(compresslevel)252        except:253            # Avoid a ResourceWarning if the write fails,254            # eg read-only file or KeyboardInterrupt255            self._close()256            raise257 258    @property259    def mtime(self):260        """Last modification time read from stream, or None"""261        return self._buffer.raw._last_mtime262 263    def __repr__(self):264        s = repr(self.fileobj)265        return '<gzip ' + s[1:-1] + ' ' + hex(id(self)) + '>'266 267    def _init_write(self, filename):268        self.name = filename269        self.crc = zlib.crc32(b"")270        self.size = 0271        self.writebuf = []272        self.bufsize = 0273        self.offset = 0  # Current file offset for seek(), tell(), etc274 275    def tell(self):276        self._check_not_closed()277        self._buffer.flush()278        return super().tell()279 280    def _write_gzip_header(self, compresslevel):281        self.fileobj.write(b'\037\213')             # magic header282        self.fileobj.write(b'\010')                 # compression method283        try:284            # RFC 1952 requires the FNAME field to be Latin-1. Do not285            # include filenames that cannot be represented that way.286            fname = os.path.basename(self.name)287            if not isinstance(fname, bytes):288                fname = fname.encode('latin-1')289            if fname.endswith(b'.gz'):290                fname = fname[:-3]291        except UnicodeEncodeError:292            fname = b''293        flags = 0294        if fname:295            flags = FNAME296        self.fileobj.write(chr(flags).encode('latin-1'))297        mtime = self._write_mtime298        if mtime is None:299            mtime = time.time()300        write32u(self.fileobj, int(mtime))301        if compresslevel == _COMPRESS_LEVEL_BEST:302            xfl = b'\002'303        elif compresslevel == _COMPRESS_LEVEL_FAST:304            xfl = b'\004'305        else:306            xfl = b'\000'307        self.fileobj.write(xfl)308        self.fileobj.write(b'\377')309        if fname:310            self.fileobj.write(fname + b'\000')311 312    def write(self,data):313        self._check_not_closed()314        if self.mode != WRITE:315            import errno316            raise OSError(errno.EBADF, "write() on read-only GzipFile object")317 318        if self.fileobj is None:319            raise ValueError("write() on closed GzipFile object")320 321        return self._buffer.write(data)322 323    def _write_raw(self, data):324        # Called by our self._buffer underlying WriteBufferStream.325        if isinstance(data, (bytes, bytearray)):326            length = len(data)327        else:328            # accept any data that supports the buffer protocol329            data = memoryview(data)330            length = data.nbytes331 332        if length > 0:333            self.fileobj.write(self.compress.compress(data))334            self.size += length335            self.crc = zlib.crc32(data, self.crc)336            self.offset += length337 338        return length339 340    def _check_read(self, caller):341        if self.mode != READ:342            import errno343            msg = f"{caller}() on write-only GzipFile object"344            raise OSError(errno.EBADF, msg)345 346    def read(self, size=-1):347        self._check_not_closed()348        self._check_read("read")349        return self._buffer.read(size)350 351    def read1(self, size=-1):352        """Implements BufferedIOBase.read1()353 354        Reads up to a buffer's worth of data if size is negative."""355        self._check_not_closed()356        self._check_read("read1")357 358        if size < 0:359            size = io.DEFAULT_BUFFER_SIZE360        return self._buffer.read1(size)361 362    def readinto(self, b):363        self._check_not_closed()364        self._check_read("readinto")365        return self._buffer.readinto(b)366 367    def readinto1(self, b):368        self._check_not_closed()369        self._check_read("readinto1")370        return self._buffer.readinto1(b)371 372    def peek(self, n):373        self._check_not_closed()374        self._check_read("peek")375        return self._buffer.peek(n)376 377    @property378    def closed(self):379        return self.fileobj is None380 381    def close(self):382        fileobj = self.fileobj383        if fileobj is None:384            return385        if self._buffer is None or self._buffer.closed:386            return387        try:388            if self.mode == WRITE:389                self._buffer.flush()390                fileobj.write(self.compress.flush())391                write32u(fileobj, self.crc)392                # self.size may exceed 2 GiB, or even 4 GiB393                write32u(fileobj, self.size & 0xffffffff)394            elif self.mode == READ:395                self._buffer.close()396        finally:397            self._close()398 399    def _close(self):400        self.fileobj = None401        myfileobj = self.myfileobj402        if myfileobj is not None:403            self.myfileobj = None404            myfileobj.close()405 406    def flush(self,zlib_mode=zlib.Z_SYNC_FLUSH):407        self._check_not_closed()408        if self.mode == WRITE:409            self._buffer.flush()410            # Ensure the compressor's buffer is flushed411            self.fileobj.write(self.compress.flush(zlib_mode))412            self.fileobj.flush()413 414    def fileno(self):415        """Invoke the underlying file object's fileno() method.416 417        This will raise AttributeError if the underlying file object418        doesn't support fileno().419        """420        return self.fileobj.fileno()421 422    def rewind(self):423        '''Return the uncompressed stream file position indicator to the424        beginning of the file'''425        if self.mode != READ:426            raise OSError("Can't rewind in write mode")427        self._buffer.seek(0)428 429    def readable(self):430        return self.mode == READ431 432    def writable(self):433        return self.mode == WRITE434 435    def seekable(self):436        return True437 438    def seek(self, offset, whence=io.SEEK_SET):439        if self.mode == WRITE:440            self._check_not_closed()441            # Flush buffer to ensure validity of self.offset442            self._buffer.flush()443            if whence != io.SEEK_SET:444                if whence == io.SEEK_CUR:445                    offset = self.offset + offset446                else:447                    raise ValueError('Seek from end not supported')448            if offset < self.offset:449                raise OSError('Negative seek in write mode')450            count = offset - self.offset451            chunk = b'\0' * self._buffer_size452            for i in range(count // self._buffer_size):453                self.write(chunk)454            self.write(b'\0' * (count % self._buffer_size))455        elif self.mode == READ:456            self._check_not_closed()457            return self._buffer.seek(offset, whence)458 459        return self.offset460 461    def readline(self, size=-1):462        self._check_not_closed()463        return self._buffer.readline(size)464 465    def __del__(self):466        if self.mode == WRITE and not self.closed:467            import warnings468            warnings.warn("unclosed GzipFile",469                          ResourceWarning, source=self, stacklevel=2)470 471        super().__del__()472 473def _read_exact(fp, n):474    '''Read exactly *n* bytes from `fp`475 476    This method is required because fp may be unbuffered,477    i.e. return short reads.478    '''479    data = fp.read(n)480    while len(data) < n:481        b = fp.read(n - len(data))482        if not b:483            raise EOFError("Compressed file ended before the "484                           "end-of-stream marker was reached")485        data += b486    return data487 488 489def _read_gzip_header(fp):490    '''Read a gzip header from `fp` and progress to the end of the header.491 492    Returns last mtime if header was present or None otherwise.493    '''494    magic = fp.read(2)495    if magic == b'':496        return None497 498    if magic != b'\037\213':499        raise BadGzipFile('Not a gzipped file (%r)' % magic)500 501    (method, flag, last_mtime) = struct.unpack("<BBIxx", _read_exact(fp, 8))502    if method != 8:503        raise BadGzipFile('Unknown compression method')504 505    if flag & FEXTRA:506        # Read & discard the extra field, if present507        extra_len, = struct.unpack("<H", _read_exact(fp, 2))508        _read_exact(fp, extra_len)509    if flag & FNAME:510        # Read and discard a null-terminated string containing the filename511        while True:512            s = fp.read(1)513            if not s or s==b'\000':514                break515    if flag & FCOMMENT:516        # Read and discard a null-terminated string containing a comment517        while True:518            s = fp.read(1)519            if not s or s==b'\000':520                break521    if flag & FHCRC:522        _read_exact(fp, 2)     # Read & discard the 16-bit header CRC523    return last_mtime524 525 526class _GzipReader(_streams.DecompressReader):527    def __init__(self, fp):528        super().__init__(_PaddedFile(fp), zlib._ZlibDecompressor,529                         wbits=-zlib.MAX_WBITS)530        # Set flag indicating start of a new member531        self._new_member = True532        self._last_mtime = None533 534    def _init_read(self):535        self._crc = zlib.crc32(b"")536        self._stream_size = 0  # Decompressed size of unconcatenated stream537 538    def _read_gzip_header(self):539        last_mtime = _read_gzip_header(self._fp)540        if last_mtime is None:541            return False542        self._last_mtime = last_mtime543        return True544 545    def read(self, size=-1):546        if size < 0:547            return self.readall()548        # size=0 is special because decompress(max_length=0) is not supported549        if not size:550            return b""551 552        # For certain input data, a single553        # call to decompress() may not return554        # any data. In this case, retry until we get some data or reach EOF.555        while True:556            if self._decompressor.eof:557                # Ending case: we've come to the end of a member in the file,558                # so finish up this member, and read a new gzip header.559                # Check the CRC and file size, and set the flag so we read560                # a new member561                self._read_eof()562                self._new_member = True563                self._decompressor = self._decomp_factory(564                    **self._decomp_args)565 566            if self._new_member:567                # If the _new_member flag is set, we have to568                # jump to the next member, if there is one.569                self._init_read()570                if not self._read_gzip_header():571                    self._size = self._pos572                    return b""573                self._new_member = False574 575            # Read a chunk of data from the file576            if self._decompressor.needs_input:577                buf = self._fp.read(READ_BUFFER_SIZE)578                uncompress = self._decompressor.decompress(buf, size)579            else:580                uncompress = self._decompressor.decompress(b"", size)581 582            if self._decompressor.unused_data != b"":583                # Prepend the already read bytes to the fileobj so they can584                # be seen by _read_eof() and _read_gzip_header()585                self._fp.prepend(self._decompressor.unused_data)586 587            if uncompress != b"":588                break589            if buf == b"":590                raise EOFError("Compressed file ended before the "591                               "end-of-stream marker was reached")592 593        self._crc = zlib.crc32(uncompress, self._crc)594        self._stream_size += len(uncompress)595        self._pos += len(uncompress)596        return uncompress597 598    def _read_eof(self):599        # We've read to the end of the file600        # We check that the computed CRC and size of the601        # uncompressed data matches the stored values.  Note that the size602        # stored is the true file size mod 2**32.603        crc32, isize = struct.unpack("<II", _read_exact(self._fp, 8))604        if crc32 != self._crc:605            raise BadGzipFile("CRC check failed %s != %s" % (hex(crc32),606                                                             hex(self._crc)))607        elif isize != (self._stream_size & 0xffffffff):608            raise BadGzipFile("Incorrect length of data produced")609 610        # Gzip files can be padded with zeroes and still have archives.611        # Consume all zero bytes and set the file position to the first612        # non-zero byte. See http://www.gzip.org/#faq8613        c = b"\x00"614        while c == b"\x00":615            c = self._fp.read(1)616        if c:617            self._fp.prepend(c)618 619    def _rewind(self):620        super()._rewind()621        self._new_member = True622 623 624def compress(data, compresslevel=_COMPRESS_LEVEL_BEST, *, mtime=0):625    """Compress data in one shot and return the compressed string.626 627    compresslevel sets the compression level in range of 0-9.628    mtime can be used to set the modification time.629    The modification time is set to 0 by default, for reproducibility.630    """631    # Wbits=31 automatically includes a gzip header and trailer.632    gzip_data = zlib.compress(data, level=compresslevel, wbits=31)633    if mtime is None:634        mtime = time.time()635    # Reuse gzip header created by zlib, replace mtime and OS byte for636    # consistency.637    header = struct.pack("<4sLBB", gzip_data, int(mtime), gzip_data[8], 255)638    return header + gzip_data[10:]639 640 641def decompress(data):642    """Decompress a gzip compressed string in one shot.643    Return the decompressed string.644    """645    decompressed_members = []646    while True:647        fp = io.BytesIO(data)648        if _read_gzip_header(fp) is None:649            return b"".join(decompressed_members)650        # Use a zlib raw deflate compressor651        do = zlib.decompressobj(wbits=-zlib.MAX_WBITS)652        # Read all the data except the header653        decompressed = do.decompress(data[fp.tell():])654        if not do.eof or len(do.unused_data) < 8:655            raise EOFError("Compressed file ended before the end-of-stream "656                           "marker was reached")657        crc, length = struct.unpack("<II", do.unused_data[:8])658        if crc != zlib.crc32(decompressed):659            raise BadGzipFile("CRC check failed")660        if length != (len(decompressed) & 0xffffffff):661            raise BadGzipFile("Incorrect length of data produced")662        decompressed_members.append(decompressed)663        data = do.unused_data[8:].lstrip(b"\x00")664 665 666def main():667    from argparse import ArgumentParser668    parser = ArgumentParser(description=669        "A simple command line interface for the gzip module: act like gzip, "670        "but do not delete the input file.",671        color=True,672    )673    group = parser.add_mutually_exclusive_group()674    group.add_argument('--fast', action='store_true', help='compress faster')675    group.add_argument('--best', action='store_true', help='compress better')676    group.add_argument("-d", "--decompress", action="store_true",677                        help="act like gunzip instead of gzip")678 679    parser.add_argument("args", nargs="*", default=["-"], metavar='file')680    args = parser.parse_args()681 682    compresslevel = _COMPRESS_LEVEL_TRADEOFF683    if args.fast:684        compresslevel = _COMPRESS_LEVEL_FAST685    elif args.best:686        compresslevel = _COMPRESS_LEVEL_BEST687 688    for arg in args.args:689        if args.decompress:690            if arg == "-":691                f = GzipFile(filename="", mode="rb", fileobj=sys.stdin.buffer)692                g = sys.stdout.buffer693            else:694                if arg[-3:] != ".gz":695                    sys.exit(f"filename doesn't end in .gz: {arg!r}")696                f = open(arg, "rb")697                g = builtins.open(arg[:-3], "wb")698        else:699            if arg == "-":700                f = sys.stdin.buffer701                g = GzipFile(filename="", mode="wb", fileobj=sys.stdout.buffer,702                             compresslevel=compresslevel)703            else:704                f = builtins.open(arg, "rb")705                g = open(arg + ".gz", "wb")706        while True:707            chunk = f.read(READ_BUFFER_SIZE)708            if not chunk:709                break710            g.write(chunk)711        if g is not sys.stdout.buffer:712            g.close()713        if f is not sys.stdin.buffer:714            f.close()715 716if __name__ == '__main__':717    main()718 
codekingpro/portable-devtools · Team Ai