codekingpro/portable-devtools
114k
1"""Functions that read and write gzipped files.2 3The user of the file doesn't have to worry about the compression,4but random access is not allowed."""5 6# based on Andrew Kuchling's minigzip.py distributed with the zlib module7 8import builtins9import io10import os11import struct12import sys13import time14import weakref15import zlib16from compression._common import _streams17 18__all__ = ["BadGzipFile", "GzipFile", "open", "compress", "decompress"]19 20FTEXT, FHCRC, FEXTRA, FNAME, FCOMMENT = 1, 2, 4, 8, 1621 22READ = 'rb'23WRITE = 'wb'24 25_COMPRESS_LEVEL_FAST = 126_COMPRESS_LEVEL_TRADEOFF = 627_COMPRESS_LEVEL_BEST = 928 29READ_BUFFER_SIZE = 128 * 102430_WRITE_BUFFER_SIZE = 4 * io.DEFAULT_BUFFER_SIZE31 32 33def open(filename, mode="rb", compresslevel=_COMPRESS_LEVEL_BEST,34 encoding=None, errors=None, newline=None):35 """Open a gzip-compressed file in binary or text mode.36 37 The filename argument can be an actual filename (a str or bytes object), or38 an existing file object to read from or write to.39 40 The mode argument can be "r", "rb", "w", "wb", "x", "xb", "a" or "ab" for41 binary mode, or "rt", "wt", "xt" or "at" for text mode. The default mode is42 "rb", and the default compresslevel is 9.43 44 For binary mode, this function is equivalent to the GzipFile constructor:45 GzipFile(filename, mode, compresslevel). In this case, the encoding, errors46 and newline arguments must not be provided.47 48 For text mode, a GzipFile object is created, and wrapped in an49 io.TextIOWrapper instance with the specified encoding, error handling50 behavior, and line ending(s).51 52 """53 if "t" in mode:54 if "b" in mode:55 raise ValueError("Invalid mode: %r" % (mode,))56 else:57 if encoding is not None:58 raise ValueError("Argument 'encoding' not supported in binary mode")59 if errors is not None:60 raise ValueError("Argument 'errors' not supported in binary mode")61 if newline is not None:62 raise ValueError("Argument 'newline' not supported in binary mode")63 64 gz_mode = mode.replace("t", "")65 if isinstance(filename, (str, bytes, os.PathLike)):66 binary_file = GzipFile(filename, gz_mode, compresslevel)67 elif hasattr(filename, "read") or hasattr(filename, "write"):68 binary_file = GzipFile(None, gz_mode, compresslevel, filename)69 else:70 raise TypeError("filename must be a str or bytes object, or a file")71 72 if "t" in mode:73 encoding = io.text_encoding(encoding)74 return io.TextIOWrapper(binary_file, encoding, errors, newline)75 else:76 return binary_file77 78def write32u(output, value):79 # The L format writes the bit pattern correctly whether signed80 # or unsigned.81 output.write(struct.pack("<L", value))82 83class _PaddedFile:84 """Minimal read-only file object that prepends a string to the contents85 of an actual file. Shouldn't be used outside of gzip.py, as it lacks86 essential functionality."""87 88 def __init__(self, f, prepend=b''):89 self._buffer = prepend90 self._length = len(prepend)91 self.file = f92 self._read = 093 94 def read(self, size):95 if self._read is None:96 return self.file.read(size)97 if self._read + size <= self._length:98 read = self._read99 self._read += size100 return self._buffer[read:self._read]101 else:102 read = self._read103 self._read = None104 return self._buffer[read:] + \105 self.file.read(size-self._length+read)106 107 def prepend(self, prepend=b''):108 if self._read is None:109 self._buffer = prepend110 else: # Assume data was read since the last prepend() call111 self._read -= len(prepend)112 return113 self._length = len(self._buffer)114 self._read = 0115 116 def seek(self, off):117 self._read = None118 self._buffer = None119 return self.file.seek(off)120 121 def seekable(self):122 return True # Allows fast-forwarding even in unseekable streams123 124 125class BadGzipFile(OSError):126 """Exception raised in some cases for invalid gzip files."""127 128 129class _WriteBufferStream(io.RawIOBase):130 """Minimal object to pass WriteBuffer flushes into GzipFile"""131 def __init__(self, gzip_file):132 self.gzip_file = weakref.ref(gzip_file)133 134 def write(self, data):135 gzip_file = self.gzip_file()136 if gzip_file is None:137 raise RuntimeError("lost gzip_file")138 return gzip_file._write_raw(data)139 140 def seekable(self):141 return False142 143 def writable(self):144 return True145 146 147class GzipFile(_streams.BaseStream):148 """The GzipFile class simulates most of the methods of a file object with149 the exception of the truncate() method.150 151 This class only supports opening files in binary mode. If you need to open a152 compressed file in text mode, use the gzip.open() function.153 154 """155 156 # Overridden with internal file object to be closed, if only a filename157 # is passed in158 myfileobj = None159 160 def __init__(self, filename=None, mode=None,161 compresslevel=_COMPRESS_LEVEL_BEST, fileobj=None, mtime=None):162 """Constructor for the GzipFile class.163 164 At least one of fileobj and filename must be given a165 non-trivial value.166 167 The new class instance is based on fileobj, which can be a regular168 file, an io.BytesIO object, or any other object which simulates a file.169 It defaults to None, in which case filename is opened to provide170 a file object.171 172 When fileobj is not None, the filename argument is only used to be173 included in the gzip file header, which may include the original174 filename of the uncompressed file. It defaults to the filename of175 fileobj, if discernible; otherwise, it defaults to the empty string,176 and in this case the original filename is not included in the header.177 178 The mode argument can be any of 'r', 'rb', 'a', 'ab', 'w', 'wb', 'x', or179 'xb' depending on whether the file will be read or written. The default180 is the mode of fileobj if discernible; otherwise, the default is 'rb'.181 A mode of 'r' is equivalent to one of 'rb', and similarly for 'w' and182 'wb', 'a' and 'ab', and 'x' and 'xb'.183 184 The compresslevel argument is an integer from 0 to 9 controlling the185 level of compression; 1 is fastest and produces the least compression,186 and 9 is slowest and produces the most compression. 0 is no compression187 at all. The default is 9.188 189 The optional mtime argument is the timestamp requested by gzip. The time190 is in Unix format, i.e., seconds since 00:00:00 UTC, January 1, 1970.191 If mtime is omitted or None, the current time is used. Use mtime = 0192 to generate a compressed stream that does not depend on creation time.193 194 """195 196 # Ensure attributes exist at __del__197 self.mode = None198 self.fileobj = None199 self._buffer = None200 201 if mode and ('t' in mode or 'U' in mode):202 raise ValueError("Invalid mode: {!r}".format(mode))203 if mode and 'b' not in mode:204 mode += 'b'205 206 try:207 if fileobj is None:208 fileobj = self.myfileobj = builtins.open(filename, mode or 'rb')209 if filename is None:210 filename = getattr(fileobj, 'name', '')211 if not isinstance(filename, (str, bytes)):212 filename = ''213 else:214 filename = os.fspath(filename)215 origmode = mode216 if mode is None:217 mode = getattr(fileobj, 'mode', 'rb')218 219 220 if mode.startswith('r'):221 self.mode = READ222 raw = _GzipReader(fileobj)223 self._buffer = io.BufferedReader(raw)224 self.name = filename225 226 elif mode.startswith(('w', 'a', 'x')):227 if origmode is None:228 import warnings229 warnings.warn(230 "GzipFile was opened for writing, but this will "231 "change in future Python releases. "232 "Specify the mode argument for opening it for writing.",233 FutureWarning, 2)234 self.mode = WRITE235 self._init_write(filename)236 self.compress = zlib.compressobj(compresslevel,237 zlib.DEFLATED,238 -zlib.MAX_WBITS,239 zlib.DEF_MEM_LEVEL,240 0)241 self._write_mtime = mtime242 self._buffer_size = _WRITE_BUFFER_SIZE243 self._buffer = io.BufferedWriter(_WriteBufferStream(self),244 buffer_size=self._buffer_size)245 else:246 raise ValueError("Invalid mode: {!r}".format(mode))247 248 self.fileobj = fileobj249 250 if self.mode == WRITE:251 self._write_gzip_header(compresslevel)252 except:253 # Avoid a ResourceWarning if the write fails,254 # eg read-only file or KeyboardInterrupt255 self._close()256 raise257 258 @property259 def mtime(self):260 """Last modification time read from stream, or None"""261 return self._buffer.raw._last_mtime262 263 def __repr__(self):264 s = repr(self.fileobj)265 return '<gzip ' + s[1:-1] + ' ' + hex(id(self)) + '>'266 267 def _init_write(self, filename):268 self.name = filename269 self.crc = zlib.crc32(b"")270 self.size = 0271 self.writebuf = []272 self.bufsize = 0273 self.offset = 0 # Current file offset for seek(), tell(), etc274 275 def tell(self):276 self._check_not_closed()277 self._buffer.flush()278 return super().tell()279 280 def _write_gzip_header(self, compresslevel):281 self.fileobj.write(b'\037\213') # magic header282 self.fileobj.write(b'\010') # compression method283 try:284 # RFC 1952 requires the FNAME field to be Latin-1. Do not285 # include filenames that cannot be represented that way.286 fname = os.path.basename(self.name)287 if not isinstance(fname, bytes):288 fname = fname.encode('latin-1')289 if fname.endswith(b'.gz'):290 fname = fname[:-3]291 except UnicodeEncodeError:292 fname = b''293 flags = 0294 if fname:295 flags = FNAME296 self.fileobj.write(chr(flags).encode('latin-1'))297 mtime = self._write_mtime298 if mtime is None:299 mtime = time.time()300 write32u(self.fileobj, int(mtime))301 if compresslevel == _COMPRESS_LEVEL_BEST:302 xfl = b'\002'303 elif compresslevel == _COMPRESS_LEVEL_FAST:304 xfl = b'\004'305 else:306 xfl = b'\000'307 self.fileobj.write(xfl)308 self.fileobj.write(b'\377')309 if fname:310 self.fileobj.write(fname + b'\000')311 312 def write(self,data):313 self._check_not_closed()314 if self.mode != WRITE:315 import errno316 raise OSError(errno.EBADF, "write() on read-only GzipFile object")317 318 if self.fileobj is None:319 raise ValueError("write() on closed GzipFile object")320 321 return self._buffer.write(data)322 323 def _write_raw(self, data):324 # Called by our self._buffer underlying WriteBufferStream.325 if isinstance(data, (bytes, bytearray)):326 length = len(data)327 else:328 # accept any data that supports the buffer protocol329 data = memoryview(data)330 length = data.nbytes331 332 if length > 0:333 self.fileobj.write(self.compress.compress(data))334 self.size += length335 self.crc = zlib.crc32(data, self.crc)336 self.offset += length337 338 return length339 340 def _check_read(self, caller):341 if self.mode != READ:342 import errno343 msg = f"{caller}() on write-only GzipFile object"344 raise OSError(errno.EBADF, msg)345 346 def read(self, size=-1):347 self._check_not_closed()348 self._check_read("read")349 return self._buffer.read(size)350 351 def read1(self, size=-1):352 """Implements BufferedIOBase.read1()353 354 Reads up to a buffer's worth of data if size is negative."""355 self._check_not_closed()356 self._check_read("read1")357 358 if size < 0:359 size = io.DEFAULT_BUFFER_SIZE360 return self._buffer.read1(size)361 362 def readinto(self, b):363 self._check_not_closed()364 self._check_read("readinto")365 return self._buffer.readinto(b)366 367 def readinto1(self, b):368 self._check_not_closed()369 self._check_read("readinto1")370 return self._buffer.readinto1(b)371 372 def peek(self, n):373 self._check_not_closed()374 self._check_read("peek")375 return self._buffer.peek(n)376 377 @property378 def closed(self):379 return self.fileobj is None380 381 def close(self):382 fileobj = self.fileobj383 if fileobj is None:384 return385 if self._buffer is None or self._buffer.closed:386 return387 try:388 if self.mode == WRITE:389 self._buffer.flush()390 fileobj.write(self.compress.flush())391 write32u(fileobj, self.crc)392 # self.size may exceed 2 GiB, or even 4 GiB393 write32u(fileobj, self.size & 0xffffffff)394 elif self.mode == READ:395 self._buffer.close()396 finally:397 self._close()398 399 def _close(self):400 self.fileobj = None401 myfileobj = self.myfileobj402 if myfileobj is not None:403 self.myfileobj = None404 myfileobj.close()405 406 def flush(self,zlib_mode=zlib.Z_SYNC_FLUSH):407 self._check_not_closed()408 if self.mode == WRITE:409 self._buffer.flush()410 # Ensure the compressor's buffer is flushed411 self.fileobj.write(self.compress.flush(zlib_mode))412 self.fileobj.flush()413 414 def fileno(self):415 """Invoke the underlying file object's fileno() method.416 417 This will raise AttributeError if the underlying file object418 doesn't support fileno().419 """420 return self.fileobj.fileno()421 422 def rewind(self):423 '''Return the uncompressed stream file position indicator to the424 beginning of the file'''425 if self.mode != READ:426 raise OSError("Can't rewind in write mode")427 self._buffer.seek(0)428 429 def readable(self):430 return self.mode == READ431 432 def writable(self):433 return self.mode == WRITE434 435 def seekable(self):436 return True437 438 def seek(self, offset, whence=io.SEEK_SET):439 if self.mode == WRITE:440 self._check_not_closed()441 # Flush buffer to ensure validity of self.offset442 self._buffer.flush()443 if whence != io.SEEK_SET:444 if whence == io.SEEK_CUR:445 offset = self.offset + offset446 else:447 raise ValueError('Seek from end not supported')448 if offset < self.offset:449 raise OSError('Negative seek in write mode')450 count = offset - self.offset451 chunk = b'\0' * self._buffer_size452 for i in range(count // self._buffer_size):453 self.write(chunk)454 self.write(b'\0' * (count % self._buffer_size))455 elif self.mode == READ:456 self._check_not_closed()457 return self._buffer.seek(offset, whence)458 459 return self.offset460 461 def readline(self, size=-1):462 self._check_not_closed()463 return self._buffer.readline(size)464 465 def __del__(self):466 if self.mode == WRITE and not self.closed:467 import warnings468 warnings.warn("unclosed GzipFile",469 ResourceWarning, source=self, stacklevel=2)470 471 super().__del__()472 473def _read_exact(fp, n):474 '''Read exactly *n* bytes from `fp`475 476 This method is required because fp may be unbuffered,477 i.e. return short reads.478 '''479 data = fp.read(n)480 while len(data) < n:481 b = fp.read(n - len(data))482 if not b:483 raise EOFError("Compressed file ended before the "484 "end-of-stream marker was reached")485 data += b486 return data487 488 489def _read_gzip_header(fp):490 '''Read a gzip header from `fp` and progress to the end of the header.491 492 Returns last mtime if header was present or None otherwise.493 '''494 magic = fp.read(2)495 if magic == b'':496 return None497 498 if magic != b'\037\213':499 raise BadGzipFile('Not a gzipped file (%r)' % magic)500 501 (method, flag, last_mtime) = struct.unpack("<BBIxx", _read_exact(fp, 8))502 if method != 8:503 raise BadGzipFile('Unknown compression method')504 505 if flag & FEXTRA:506 # Read & discard the extra field, if present507 extra_len, = struct.unpack("<H", _read_exact(fp, 2))508 _read_exact(fp, extra_len)509 if flag & FNAME:510 # Read and discard a null-terminated string containing the filename511 while True:512 s = fp.read(1)513 if not s or s==b'\000':514 break515 if flag & FCOMMENT:516 # Read and discard a null-terminated string containing a comment517 while True:518 s = fp.read(1)519 if not s or s==b'\000':520 break521 if flag & FHCRC:522 _read_exact(fp, 2) # Read & discard the 16-bit header CRC523 return last_mtime524 525 526class _GzipReader(_streams.DecompressReader):527 def __init__(self, fp):528 super().__init__(_PaddedFile(fp), zlib._ZlibDecompressor,529 wbits=-zlib.MAX_WBITS)530 # Set flag indicating start of a new member531 self._new_member = True532 self._last_mtime = None533 534 def _init_read(self):535 self._crc = zlib.crc32(b"")536 self._stream_size = 0 # Decompressed size of unconcatenated stream537 538 def _read_gzip_header(self):539 last_mtime = _read_gzip_header(self._fp)540 if last_mtime is None:541 return False542 self._last_mtime = last_mtime543 return True544 545 def read(self, size=-1):546 if size < 0:547 return self.readall()548 # size=0 is special because decompress(max_length=0) is not supported549 if not size:550 return b""551 552 # For certain input data, a single553 # call to decompress() may not return554 # any data. In this case, retry until we get some data or reach EOF.555 while True:556 if self._decompressor.eof:557 # Ending case: we've come to the end of a member in the file,558 # so finish up this member, and read a new gzip header.559 # Check the CRC and file size, and set the flag so we read560 # a new member561 self._read_eof()562 self._new_member = True563 self._decompressor = self._decomp_factory(564 **self._decomp_args)565 566 if self._new_member:567 # If the _new_member flag is set, we have to568 # jump to the next member, if there is one.569 self._init_read()570 if not self._read_gzip_header():571 self._size = self._pos572 return b""573 self._new_member = False574 575 # Read a chunk of data from the file576 if self._decompressor.needs_input:577 buf = self._fp.read(READ_BUFFER_SIZE)578 uncompress = self._decompressor.decompress(buf, size)579 else:580 uncompress = self._decompressor.decompress(b"", size)581 582 if self._decompressor.unused_data != b"":583 # Prepend the already read bytes to the fileobj so they can584 # be seen by _read_eof() and _read_gzip_header()585 self._fp.prepend(self._decompressor.unused_data)586 587 if uncompress != b"":588 break589 if buf == b"":590 raise EOFError("Compressed file ended before the "591 "end-of-stream marker was reached")592 593 self._crc = zlib.crc32(uncompress, self._crc)594 self._stream_size += len(uncompress)595 self._pos += len(uncompress)596 return uncompress597 598 def _read_eof(self):599 # We've read to the end of the file600 # We check that the computed CRC and size of the601 # uncompressed data matches the stored values. Note that the size602 # stored is the true file size mod 2**32.603 crc32, isize = struct.unpack("<II", _read_exact(self._fp, 8))604 if crc32 != self._crc:605 raise BadGzipFile("CRC check failed %s != %s" % (hex(crc32),606 hex(self._crc)))607 elif isize != (self._stream_size & 0xffffffff):608 raise BadGzipFile("Incorrect length of data produced")609 610 # Gzip files can be padded with zeroes and still have archives.611 # Consume all zero bytes and set the file position to the first612 # non-zero byte. See http://www.gzip.org/#faq8613 c = b"\x00"614 while c == b"\x00":615 c = self._fp.read(1)616 if c:617 self._fp.prepend(c)618 619 def _rewind(self):620 super()._rewind()621 self._new_member = True622 623 624def compress(data, compresslevel=_COMPRESS_LEVEL_BEST, *, mtime=0):625 """Compress data in one shot and return the compressed string.626 627 compresslevel sets the compression level in range of 0-9.628 mtime can be used to set the modification time.629 The modification time is set to 0 by default, for reproducibility.630 """631 # Wbits=31 automatically includes a gzip header and trailer.632 gzip_data = zlib.compress(data, level=compresslevel, wbits=31)633 if mtime is None:634 mtime = time.time()635 # Reuse gzip header created by zlib, replace mtime and OS byte for636 # consistency.637 header = struct.pack("<4sLBB", gzip_data, int(mtime), gzip_data[8], 255)638 return header + gzip_data[10:]639 640 641def decompress(data):642 """Decompress a gzip compressed string in one shot.643 Return the decompressed string.644 """645 decompressed_members = []646 while True:647 fp = io.BytesIO(data)648 if _read_gzip_header(fp) is None:649 return b"".join(decompressed_members)650 # Use a zlib raw deflate compressor651 do = zlib.decompressobj(wbits=-zlib.MAX_WBITS)652 # Read all the data except the header653 decompressed = do.decompress(data[fp.tell():])654 if not do.eof or len(do.unused_data) < 8:655 raise EOFError("Compressed file ended before the end-of-stream "656 "marker was reached")657 crc, length = struct.unpack("<II", do.unused_data[:8])658 if crc != zlib.crc32(decompressed):659 raise BadGzipFile("CRC check failed")660 if length != (len(decompressed) & 0xffffffff):661 raise BadGzipFile("Incorrect length of data produced")662 decompressed_members.append(decompressed)663 data = do.unused_data[8:].lstrip(b"\x00")664 665 666def main():667 from argparse import ArgumentParser668 parser = ArgumentParser(description=669 "A simple command line interface for the gzip module: act like gzip, "670 "but do not delete the input file.",671 color=True,672 )673 group = parser.add_mutually_exclusive_group()674 group.add_argument('--fast', action='store_true', help='compress faster')675 group.add_argument('--best', action='store_true', help='compress better')676 group.add_argument("-d", "--decompress", action="store_true",677 help="act like gunzip instead of gzip")678 679 parser.add_argument("args", nargs="*", default=["-"], metavar='file')680 args = parser.parse_args()681 682 compresslevel = _COMPRESS_LEVEL_TRADEOFF683 if args.fast:684 compresslevel = _COMPRESS_LEVEL_FAST685 elif args.best:686 compresslevel = _COMPRESS_LEVEL_BEST687 688 for arg in args.args:689 if args.decompress:690 if arg == "-":691 f = GzipFile(filename="", mode="rb", fileobj=sys.stdin.buffer)692 g = sys.stdout.buffer693 else:694 if arg[-3:] != ".gz":695 sys.exit(f"filename doesn't end in .gz: {arg!r}")696 f = open(arg, "rb")697 g = builtins.open(arg[:-3], "wb")698 else:699 if arg == "-":700 f = sys.stdin.buffer701 g = GzipFile(filename="", mode="wb", fileobj=sys.stdout.buffer,702 compresslevel=compresslevel)703 else:704 f = builtins.open(arg, "rb")705 g = open(arg + ".gz", "wb")706 while True:707 chunk = f.read(READ_BUFFER_SIZE)708 if not chunk:709 break710 g.write(chunk)711 if g is not sys.stdout.buffer:712 g.close()713 if f is not sys.stdin.buffer:714 f.close()715 716if __name__ == '__main__':717 main()718 