codekingpro/portable-devtools
114k
1#-------------------------------------------------------------------2# tarfile.py3#-------------------------------------------------------------------4# Copyright (C) 2002 Lars Gustaebel <lars@gustaebel.de>5# All rights reserved.6#7# Permission is hereby granted, free of charge, to any person8# obtaining a copy of this software and associated documentation9# files (the "Software"), to deal in the Software without10# restriction, including without limitation the rights to use,11# copy, modify, merge, publish, distribute, sublicense, and/or sell12# copies of the Software, and to permit persons to whom the13# Software is furnished to do so, subject to the following14# conditions:15#16# The above copyright notice and this permission notice shall be17# included in all copies or substantial portions of the Software.18#19# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,20# EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES21# OF MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND22# NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT23# HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY,24# WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING25# FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR26# OTHER DEALINGS IN THE SOFTWARE.27#28"""Read from and write to tar format archives.29"""30 31version = "0.9.0"32__author__ = "Lars Gust\u00e4bel (lars@gustaebel.de)"33__credits__ = "Gustavo Niemeyer, Niels Gust\u00e4bel, Richard Townsend."34 35#---------36# Imports37#---------38from builtins import open as bltn_open39import sys40import os41import io42import shutil43import stat44import time45import struct46import copy47import re48 49try:50 import pwd51except ImportError:52 pwd = None53try:54 import grp55except ImportError:56 grp = None57 58# os.symlink on Windows prior to 6.0 raises NotImplementedError59# OSError (winerror=1314) will be raised if the caller does not hold the60# SeCreateSymbolicLinkPrivilege privilege61symlink_exception = (AttributeError, NotImplementedError, OSError)62 63# from tarfile import *64__all__ = ["TarFile", "TarInfo", "is_tarfile", "TarError", "ReadError",65 "CompressionError", "StreamError", "ExtractError", "HeaderError",66 "ENCODING", "USTAR_FORMAT", "GNU_FORMAT", "PAX_FORMAT",67 "DEFAULT_FORMAT", "open","fully_trusted_filter", "data_filter",68 "tar_filter", "FilterError", "AbsoluteLinkError",69 "OutsideDestinationError", "SpecialFileError", "AbsolutePathError",70 "LinkOutsideDestinationError", "LinkFallbackError"]71 72 73#---------------------------------------------------------74# tar constants75#---------------------------------------------------------76NUL = b"\0" # the null character77BLOCKSIZE = 512 # length of processing blocks78RECORDSIZE = BLOCKSIZE * 20 # length of records79GNU_MAGIC = b"ustar \0" # magic gnu tar string80POSIX_MAGIC = b"ustar\x0000" # magic posix tar string81 82LENGTH_NAME = 100 # maximum length of a filename83LENGTH_LINK = 100 # maximum length of a linkname84LENGTH_PREFIX = 155 # maximum length of the prefix field85 86REGTYPE = b"0" # regular file87AREGTYPE = b"\0" # regular file88LNKTYPE = b"1" # link (inside tarfile)89SYMTYPE = b"2" # symbolic link90CHRTYPE = b"3" # character special device91BLKTYPE = b"4" # block special device92DIRTYPE = b"5" # directory93FIFOTYPE = b"6" # fifo special device94CONTTYPE = b"7" # contiguous file95 96GNUTYPE_LONGNAME = b"L" # GNU tar longname97GNUTYPE_LONGLINK = b"K" # GNU tar longlink98GNUTYPE_SPARSE = b"S" # GNU tar sparse file99 100XHDTYPE = b"x" # POSIX.1-2001 extended header101XGLTYPE = b"g" # POSIX.1-2001 global header102SOLARIS_XHDTYPE = b"X" # Solaris extended header103 104USTAR_FORMAT = 0 # POSIX.1-1988 (ustar) format105GNU_FORMAT = 1 # GNU tar format106PAX_FORMAT = 2 # POSIX.1-2001 (pax) format107DEFAULT_FORMAT = PAX_FORMAT108 109#---------------------------------------------------------110# tarfile constants111#---------------------------------------------------------112# File types that tarfile supports:113SUPPORTED_TYPES = (REGTYPE, AREGTYPE, LNKTYPE,114 SYMTYPE, DIRTYPE, FIFOTYPE,115 CONTTYPE, CHRTYPE, BLKTYPE,116 GNUTYPE_LONGNAME, GNUTYPE_LONGLINK,117 GNUTYPE_SPARSE)118 119# File types that will be treated as a regular file.120REGULAR_TYPES = (REGTYPE, AREGTYPE,121 CONTTYPE, GNUTYPE_SPARSE)122 123# File types that are part of the GNU tar format.124GNU_TYPES = (GNUTYPE_LONGNAME, GNUTYPE_LONGLINK,125 GNUTYPE_SPARSE)126 127# Fields from a pax header that override a TarInfo attribute.128PAX_FIELDS = ("path", "linkpath", "size", "mtime",129 "uid", "gid", "uname", "gname")130 131# Fields from a pax header that are affected by hdrcharset.132PAX_NAME_FIELDS = {"path", "linkpath", "uname", "gname"}133 134# Fields in a pax header that are numbers, all other fields135# are treated as strings.136PAX_NUMBER_FIELDS = {137 "atime": float,138 "ctime": float,139 "mtime": float,140 "uid": int,141 "gid": int,142 "size": int143}144 145#---------------------------------------------------------146# initialization147#---------------------------------------------------------148if os.name == "nt":149 ENCODING = "utf-8"150else:151 ENCODING = sys.getfilesystemencoding()152 153#---------------------------------------------------------154# Some useful functions155#---------------------------------------------------------156 157def stn(s, length, encoding, errors):158 """Convert a string to a null-terminated bytes object.159 """160 if s is None:161 raise ValueError("metadata cannot contain None")162 s = s.encode(encoding, errors)163 return s[:length] + (length - len(s)) * NUL164 165def nts(s, encoding, errors):166 """Convert a null-terminated bytes object to a string.167 """168 p = s.find(b"\0")169 if p != -1:170 s = s[:p]171 return s.decode(encoding, errors)172 173def nti(s):174 """Convert a number field to a python number.175 """176 # There are two possible encodings for a number field, see177 # itn() below.178 if s[0] in (0o200, 0o377):179 n = 0180 for i in range(len(s) - 1):181 n <<= 8182 n += s[i + 1]183 if s[0] == 0o377:184 n = -(256 ** (len(s) - 1) - n)185 else:186 try:187 s = nts(s, "ascii", "strict")188 n = int(s.strip() or "0", 8)189 except ValueError:190 raise InvalidHeaderError("invalid header")191 return n192 193def itn(n, digits=8, format=DEFAULT_FORMAT):194 """Convert a python number to a number field.195 """196 # POSIX 1003.1-1988 requires numbers to be encoded as a string of197 # octal digits followed by a null-byte, this allows values up to198 # (8**(digits-1))-1. GNU tar allows storing numbers greater than199 # that if necessary. A leading 0o200 or 0o377 byte indicate this200 # particular encoding, the following digits-1 bytes are a big-endian201 # base-256 representation. This allows values up to (256**(digits-1))-1.202 # A 0o200 byte indicates a positive number, a 0o377 byte a negative203 # number.204 original_n = n205 n = int(n)206 if 0 <= n < 8 ** (digits - 1):207 s = bytes("%0*o" % (digits - 1, n), "ascii") + NUL208 elif format == GNU_FORMAT and -256 ** (digits - 1) <= n < 256 ** (digits - 1):209 if n >= 0:210 s = bytearray([0o200])211 else:212 s = bytearray([0o377])213 n = 256 ** digits + n214 215 for i in range(digits - 1):216 s.insert(1, n & 0o377)217 n >>= 8218 else:219 raise ValueError("overflow in number field")220 221 return s222 223def calc_chksums(buf):224 """Calculate the checksum for a member's header by summing up all225 characters except for the chksum field which is treated as if226 it was filled with spaces. According to the GNU tar sources,227 some tars (Sun and NeXT) calculate chksum with signed char,228 which will be different if there are chars in the buffer with229 the high bit set. So we calculate two checksums, unsigned and230 signed.231 """232 unsigned_chksum = 256 + sum(struct.unpack_from("148B8x356B", buf))233 signed_chksum = 256 + sum(struct.unpack_from("148b8x356b", buf))234 return unsigned_chksum, signed_chksum235 236def copyfileobj(src, dst, length=None, exception=OSError, bufsize=None):237 """Copy length bytes from fileobj src to fileobj dst.238 If length is None, copy the entire content.239 """240 bufsize = bufsize or 16 * 1024241 if length == 0:242 return243 if length is None:244 shutil.copyfileobj(src, dst, bufsize)245 return246 247 blocks, remainder = divmod(length, bufsize)248 for b in range(blocks):249 buf = src.read(bufsize)250 if len(buf) < bufsize:251 raise exception("unexpected end of data")252 dst.write(buf)253 254 if remainder != 0:255 buf = src.read(remainder)256 if len(buf) < remainder:257 raise exception("unexpected end of data")258 dst.write(buf)259 return260 261def _safe_print(s):262 encoding = getattr(sys.stdout, 'encoding', None)263 if encoding is not None:264 s = s.encode(encoding, 'backslashreplace').decode(encoding)265 print(s, end=' ')266 267 268class TarError(Exception):269 """Base exception."""270 pass271class ExtractError(TarError):272 """General exception for extract errors."""273 pass274class ReadError(TarError):275 """Exception for unreadable tar archives."""276 pass277class CompressionError(TarError):278 """Exception for unavailable compression methods."""279 pass280class StreamError(TarError):281 """Exception for unsupported operations on stream-like TarFiles."""282 pass283class HeaderError(TarError):284 """Base exception for header errors."""285 pass286class EmptyHeaderError(HeaderError):287 """Exception for empty headers."""288 pass289class TruncatedHeaderError(HeaderError):290 """Exception for truncated headers."""291 pass292class EOFHeaderError(HeaderError):293 """Exception for end of file headers."""294 pass295class InvalidHeaderError(HeaderError):296 """Exception for invalid headers."""297 pass298class SubsequentHeaderError(HeaderError):299 """Exception for missing and invalid extended headers."""300 pass301 302#---------------------------303# internal stream interface304#---------------------------305class _LowLevelFile:306 """Low-level file object. Supports reading and writing.307 It is used instead of a regular file object for streaming308 access.309 """310 311 def __init__(self, name, mode):312 mode = {313 "r": os.O_RDONLY,314 "w": os.O_WRONLY | os.O_CREAT | os.O_TRUNC,315 }[mode]316 if hasattr(os, "O_BINARY"):317 mode |= os.O_BINARY318 self.fd = os.open(name, mode, 0o666)319 320 def close(self):321 os.close(self.fd)322 323 def read(self, size):324 return os.read(self.fd, size)325 326 def write(self, s):327 os.write(self.fd, s)328 329class _Stream:330 """Class that serves as an adapter between TarFile and331 a stream-like object. The stream-like object only332 needs to have a read() or write() method that works with bytes,333 and the method is accessed blockwise.334 Use of gzip or bzip2 compression is possible.335 A stream-like object could be for example: sys.stdin.buffer,336 sys.stdout.buffer, a socket, a tape device etc.337 338 _Stream is intended to be used only internally.339 """340 341 def __init__(self, name, mode, comptype, fileobj, bufsize,342 compresslevel, preset):343 """Construct a _Stream object.344 """345 self._extfileobj = True346 if fileobj is None:347 fileobj = _LowLevelFile(name, mode)348 self._extfileobj = False349 350 if comptype == '*':351 # Enable transparent compression detection for the352 # stream interface353 fileobj = _StreamProxy(fileobj)354 comptype = fileobj.getcomptype()355 356 self.name = os.fspath(name) if name is not None else ""357 self.mode = mode358 self.comptype = comptype359 self.fileobj = fileobj360 self.bufsize = bufsize361 self.buf = b""362 self.pos = 0363 self.closed = False364 365 try:366 if comptype == "gz":367 try:368 import zlib369 except ImportError:370 raise CompressionError("zlib module is not available") from None371 self.zlib = zlib372 self.crc = zlib.crc32(b"")373 if mode == "r":374 self.exception = zlib.error375 self._init_read_gz()376 else:377 self._init_write_gz(compresslevel)378 379 elif comptype == "bz2":380 try:381 import bz2382 except ImportError:383 raise CompressionError("bz2 module is not available") from None384 if mode == "r":385 self.dbuf = b""386 self.cmp = bz2.BZ2Decompressor()387 self.exception = OSError388 else:389 self.cmp = bz2.BZ2Compressor(compresslevel)390 391 elif comptype == "xz":392 try:393 import lzma394 except ImportError:395 raise CompressionError("lzma module is not available") from None396 if mode == "r":397 self.dbuf = b""398 self.cmp = lzma.LZMADecompressor()399 self.exception = lzma.LZMAError400 else:401 self.cmp = lzma.LZMACompressor(preset=preset)402 elif comptype == "zst":403 try:404 from compression import zstd405 except ImportError:406 raise CompressionError("compression.zstd module is not available") from None407 if mode == "r":408 self.dbuf = b""409 self.cmp = zstd.ZstdDecompressor()410 self.exception = zstd.ZstdError411 else:412 self.cmp = zstd.ZstdCompressor()413 elif comptype != "tar":414 raise CompressionError("unknown compression type %r" % comptype)415 416 except:417 if not self._extfileobj:418 self.fileobj.close()419 self.closed = True420 raise421 422 def __del__(self):423 if hasattr(self, "closed") and not self.closed:424 self.close()425 426 def _init_write_gz(self, compresslevel):427 """Initialize for writing with gzip compression.428 """429 self.cmp = self.zlib.compressobj(compresslevel,430 self.zlib.DEFLATED,431 -self.zlib.MAX_WBITS,432 self.zlib.DEF_MEM_LEVEL,433 0)434 timestamp = struct.pack("<L", int(time.time()))435 self.__write(b"\037\213\010\010" + timestamp + b"\002\377")436 if self.name.endswith(".gz"):437 self.name = self.name[:-3]438 # Honor "directory components removed" from RFC1952439 self.name = os.path.basename(self.name)440 # RFC1952 says we must use ISO-8859-1 for the FNAME field.441 self.__write(self.name.encode("iso-8859-1", "replace") + NUL)442 443 def write(self, s):444 """Write string s to the stream.445 """446 if self.comptype == "gz":447 self.crc = self.zlib.crc32(s, self.crc)448 self.pos += len(s)449 if self.comptype != "tar":450 s = self.cmp.compress(s)451 self.__write(s)452 453 def __write(self, s):454 """Write string s to the stream if a whole new block455 is ready to be written.456 """457 self.buf += s458 while len(self.buf) > self.bufsize:459 self.fileobj.write(self.buf[:self.bufsize])460 self.buf = self.buf[self.bufsize:]461 462 def close(self):463 """Close the _Stream object. No operation should be464 done on it afterwards.465 """466 if self.closed:467 return468 469 self.closed = True470 try:471 if self.mode == "w" and self.comptype != "tar":472 self.buf += self.cmp.flush()473 474 if self.mode == "w" and self.buf:475 self.fileobj.write(self.buf)476 self.buf = b""477 if self.comptype == "gz":478 self.fileobj.write(struct.pack("<L", self.crc))479 self.fileobj.write(struct.pack("<L", self.pos & 0xffffFFFF))480 finally:481 if not self._extfileobj:482 self.fileobj.close()483 484 def _init_read_gz(self):485 """Initialize for reading a gzip compressed fileobj.486 """487 self.cmp = self.zlib.decompressobj(-self.zlib.MAX_WBITS)488 self.dbuf = b""489 490 # taken from gzip.GzipFile with some alterations491 if self.__read(2) != b"\037\213":492 raise ReadError("not a gzip file")493 if self.__read(1) != b"\010":494 raise CompressionError("unsupported compression method")495 496 flag = ord(self.__read(1))497 self.__read(6)498 499 if flag & 4:500 xlen = ord(self.__read(1)) + 256 * ord(self.__read(1))501 self.read(xlen)502 if flag & 8:503 while True:504 s = self.__read(1)505 if not s or s == NUL:506 break507 if flag & 16:508 while True:509 s = self.__read(1)510 if not s or s == NUL:511 break512 if flag & 2:513 self.__read(2)514 515 def tell(self):516 """Return the stream's file pointer position.517 """518 return self.pos519 520 def seek(self, pos=0):521 """Set the stream's file pointer to pos. Negative seeking522 is forbidden.523 """524 if pos - self.pos >= 0:525 blocks, remainder = divmod(pos - self.pos, self.bufsize)526 for i in range(blocks):527 self.read(self.bufsize)528 self.read(remainder)529 else:530 raise StreamError("seeking backwards is not allowed")531 return self.pos532 533 def read(self, size):534 """Return the next size number of bytes from the stream."""535 assert size is not None536 buf = self._read(size)537 self.pos += len(buf)538 return buf539 540 def _read(self, size):541 """Return size bytes from the stream.542 """543 if self.comptype == "tar":544 return self.__read(size)545 546 c = len(self.dbuf)547 t = [self.dbuf]548 while c < size:549 # Skip underlying buffer to avoid unaligned double buffering.550 if self.buf:551 buf = self.buf552 self.buf = b""553 else:554 buf = self.fileobj.read(self.bufsize)555 if not buf:556 break557 try:558 buf = self.cmp.decompress(buf)559 except self.exception as e:560 raise ReadError("invalid compressed data") from e561 t.append(buf)562 c += len(buf)563 t = b"".join(t)564 self.dbuf = t[size:]565 return t[:size]566 567 def __read(self, size):568 """Return size bytes from stream. If internal buffer is empty,569 read another block from the stream.570 """571 c = len(self.buf)572 t = [self.buf]573 while c < size:574 buf = self.fileobj.read(self.bufsize)575 if not buf:576 break577 t.append(buf)578 c += len(buf)579 t = b"".join(t)580 self.buf = t[size:]581 return t[:size]582# class _Stream583 584class _StreamProxy(object):585 """Small proxy class that enables transparent compression586 detection for the Stream interface (mode 'r|*').587 """588 589 def __init__(self, fileobj):590 self.fileobj = fileobj591 self.buf = self.fileobj.read(BLOCKSIZE)592 593 def read(self, size):594 self.read = self.fileobj.read595 return self.buf596 597 def getcomptype(self):598 if self.buf.startswith(b"\x1f\x8b\x08"):599 return "gz"600 elif self.buf[0:3] == b"BZh" and self.buf[4:10] == b"1AY&SY":601 return "bz2"602 elif self.buf.startswith((b"\x5d\x00\x00\x80", b"\xfd7zXZ")):603 return "xz"604 elif self.buf.startswith(b"\x28\xb5\x2f\xfd"):605 return "zst"606 else:607 return "tar"608 609 def close(self):610 self.fileobj.close()611# class StreamProxy612 613#------------------------614# Extraction file object615#------------------------616class _FileInFile(object):617 """A thin wrapper around an existing file object that618 provides a part of its data as an individual file619 object.620 """621 622 def __init__(self, fileobj, offset, size, name, blockinfo=None):623 self.fileobj = fileobj624 self.offset = offset625 self.size = size626 self.position = 0627 self.name = name628 self.closed = False629 630 if blockinfo is None:631 blockinfo = [(0, size)]632 633 # Construct a map with data and zero blocks.634 self.map_index = 0635 self.map = []636 lastpos = 0637 realpos = self.offset638 for offset, size in blockinfo:639 if offset > lastpos:640 self.map.append((False, lastpos, offset, None))641 self.map.append((True, offset, offset + size, realpos))642 realpos += size643 lastpos = offset + size644 if lastpos < self.size:645 self.map.append((False, lastpos, self.size, None))646 647 def flush(self):648 pass649 650 @property651 def mode(self):652 return 'rb'653 654 def readable(self):655 return True656 657 def writable(self):658 return False659 660 def seekable(self):661 return self.fileobj.seekable()662 663 def tell(self):664 """Return the current file position.665 """666 return self.position667 668 def seek(self, position, whence=io.SEEK_SET):669 """Seek to a position in the file.670 """671 if whence == io.SEEK_SET:672 self.position = min(max(position, 0), self.size)673 elif whence == io.SEEK_CUR:674 if position < 0:675 self.position = max(self.position + position, 0)676 else:677 self.position = min(self.position + position, self.size)678 elif whence == io.SEEK_END:679 self.position = max(min(self.size + position, self.size), 0)680 else:681 raise ValueError("Invalid argument")682 return self.position683 684 def read(self, size=None):685 """Read data from the file.686 """687 if size is None:688 size = self.size - self.position689 else:690 size = min(size, self.size - self.position)691 692 buf = b""693 while size > 0:694 while True:695 data, start, stop, offset = self.map[self.map_index]696 if start <= self.position < stop:697 break698 else:699 self.map_index += 1700 if self.map_index == len(self.map):701 self.map_index = 0702 length = min(size, stop - self.position)703 if data:704 self.fileobj.seek(offset + (self.position - start))705 b = self.fileobj.read(length)706 if len(b) != length:707 raise ReadError("unexpected end of data")708 buf += b709 else:710 buf += NUL * length711 size -= length712 self.position += length713 return buf714 715 def readinto(self, b):716 buf = self.read(len(b))717 b[:len(buf)] = buf718 return len(buf)719 720 def close(self):721 self.closed = True722#class _FileInFile723 724class ExFileObject(io.BufferedReader):725 726 def __init__(self, tarfile, tarinfo):727 fileobj = _FileInFile(tarfile.fileobj, tarinfo.offset_data,728 tarinfo.size, tarinfo.name, tarinfo.sparse)729 super().__init__(fileobj)730#class ExFileObject731 732 733#-----------------------------734# extraction filters (PEP 706)735#-----------------------------736 737class FilterError(TarError):738 pass739 740class AbsolutePathError(FilterError):741 def __init__(self, tarinfo):742 self.tarinfo = tarinfo743 super().__init__(f'member {tarinfo.name!r} has an absolute path')744 745class OutsideDestinationError(FilterError):746 def __init__(self, tarinfo, path):747 self.tarinfo = tarinfo748 self._path = path749 super().__init__(f'{tarinfo.name!r} would be extracted to {path!r}, '750 + 'which is outside the destination')751 752class SpecialFileError(FilterError):753 def __init__(self, tarinfo):754 self.tarinfo = tarinfo755 super().__init__(f'{tarinfo.name!r} is a special file')756 757class AbsoluteLinkError(FilterError):758 def __init__(self, tarinfo):759 self.tarinfo = tarinfo760 super().__init__(f'{tarinfo.name!r} is a link to an absolute path')761 762class LinkOutsideDestinationError(FilterError):763 def __init__(self, tarinfo, path):764 self.tarinfo = tarinfo765 self._path = path766 super().__init__(f'{tarinfo.name!r} would link to {path!r}, '767 + 'which is outside the destination')768 769class LinkFallbackError(FilterError):770 def __init__(self, tarinfo, path):771 self.tarinfo = tarinfo772 self._path = path773 super().__init__(f'link {tarinfo.name!r} would be extracted as a '774 + f'copy of {path!r}, which was rejected')775 776# Errors caused by filters -- both "fatal" and "non-fatal" -- that777# we consider to be issues with the argument, rather than a bug in the778# filter function779_FILTER_ERRORS = (FilterError, OSError, ExtractError)780 781def _get_filtered_attrs(member, dest_path, for_data=True):782 new_attrs = {}783 name = member.name784 dest_path = os.path.realpath(dest_path, strict=os.path.ALLOW_MISSING)785 # Strip leading / (tar's directory separator) from filenames.786 # Include os.sep (target OS directory separator) as well.787 if name.startswith(('/', os.sep)):788 name = new_attrs['name'] = member.path.lstrip('/' + os.sep)789 if os.path.isabs(name):790 # Path is absolute even after stripping.791 # For example, 'C:/foo' on Windows.792 raise AbsolutePathError(member)793 # Ensure we stay in the destination794 target_path = os.path.realpath(os.path.join(dest_path, name),795 strict=os.path.ALLOW_MISSING)796 if os.path.commonpath([target_path, dest_path]) != dest_path:797 raise OutsideDestinationError(member, target_path)798 # Limit permissions (no high bits, and go-w)799 mode = member.mode800 if mode is not None:801 # Strip high bits & group/other write bits802 mode = mode & 0o755803 if for_data:804 # For data, handle permissions & file types805 if member.isreg() or member.islnk():806 if not mode & 0o100:807 # Clear executable bits if not executable by user808 mode &= ~0o111809 # Ensure owner can read & write810 mode |= 0o600811 elif member.isdir() or member.issym():812 # Ignore mode for directories & symlinks813 mode = None814 else:815 # Reject special files816 raise SpecialFileError(member)817 if mode != member.mode:818 new_attrs['mode'] = mode819 if for_data:820 # Ignore ownership for 'data'821 if member.uid is not None:822 new_attrs['uid'] = None823 if member.gid is not None:824 new_attrs['gid'] = None825 if member.uname is not None:826 new_attrs['uname'] = None827 if member.gname is not None:828 new_attrs['gname'] = None829 # Check link destination for 'data'830 if member.islnk() or member.issym():831 if os.path.isabs(member.linkname):832 raise AbsoluteLinkError(member)833 normalized = os.path.normpath(member.linkname)834 if normalized != member.linkname:835 new_attrs['linkname'] = normalized836 if member.issym():837 target_path = os.path.join(dest_path,838 os.path.dirname(name),839 member.linkname)840 else:841 target_path = os.path.join(dest_path,842 member.linkname)843 target_path = os.path.realpath(target_path,844 strict=os.path.ALLOW_MISSING)845 if os.path.commonpath([target_path, dest_path]) != dest_path:846 raise LinkOutsideDestinationError(member, target_path)847 return new_attrs848 849def fully_trusted_filter(member, dest_path):850 return member851 852def tar_filter(member, dest_path):853 new_attrs = _get_filtered_attrs(member, dest_path, False)854 if new_attrs:855 return member.replace(**new_attrs, deep=False)856 return member857 858def data_filter(member, dest_path):859 new_attrs = _get_filtered_attrs(member, dest_path, True)860 if new_attrs:861 return member.replace(**new_attrs, deep=False)862 return member863 864_NAMED_FILTERS = {865 "fully_trusted": fully_trusted_filter,866 "tar": tar_filter,867 "data": data_filter,868}869 870#------------------871# Exported Classes872#------------------873 874# Sentinel for replace() defaults, meaning "don't change the attribute"875_KEEP = object()876 877# Header length is digits followed by a space.878_header_length_prefix_re = re.compile(br"([0-9]{1,20}) ")879 880class TarInfo(object):881 """Informational class which holds the details about an882 archive member given by a tar header block.883 TarInfo objects are returned by TarFile.getmember(),884 TarFile.getmembers() and TarFile.gettarinfo() and are885 usually created internally.886 """887 888 __slots__ = dict(889 name = 'Name of the archive member.',890 mode = 'Permission bits.',891 uid = 'User ID of the user who originally stored this member.',892 gid = 'Group ID of the user who originally stored this member.',893 size = 'Size in bytes.',894 mtime = 'Time of last modification.',895 chksum = 'Header checksum.',896 type = ('File type. type is usually one of these constants: '897 'REGTYPE, AREGTYPE, LNKTYPE, SYMTYPE, DIRTYPE, FIFOTYPE, '898 'CONTTYPE, CHRTYPE, BLKTYPE, GNUTYPE_SPARSE.'),899 linkname = ('Name of the target file name, which is only present '900 'in TarInfo objects of type LNKTYPE and SYMTYPE.'),901 uname = 'User name.',902 gname = 'Group name.',903 devmajor = 'Device major number.',904 devminor = 'Device minor number.',905 offset = 'The tar header starts here.',906 offset_data = "The file's data starts here.",907 pax_headers = ('A dictionary containing key-value pairs of an '908 'associated pax extended header.'),909 sparse = 'Sparse member information.',910 _tarfile = None,911 _sparse_structs = None,912 _link_target = None,913 )914 915 def __init__(self, name=""):916 """Construct a TarInfo object. name is the optional name917 of the member.918 """919 self.name = name # member name920 self.mode = 0o644 # file permissions921 self.uid = 0 # user id922 self.gid = 0 # group id923 self.size = 0 # file size924 self.mtime = 0 # modification time925 self.chksum = 0 # header checksum926 self.type = REGTYPE # member type927 self.linkname = "" # link name928 self.uname = "" # user name929 self.gname = "" # group name930 self.devmajor = 0 # device major number931 self.devminor = 0 # device minor number932 933 self.offset = 0 # the tar header starts here934 self.offset_data = 0 # the file's data starts here935 936 self.sparse = None # sparse member information937 self.pax_headers = {} # pax header information938 939 @property940 def tarfile(self):941 import warnings942 warnings.warn(943 'The undocumented "tarfile" attribute of TarInfo objects '944 + 'is deprecated and will be removed in Python 3.16',945 DeprecationWarning, stacklevel=2)946 return self._tarfile947 948 @tarfile.setter949 def tarfile(self, tarfile):950 import warnings951 warnings.warn(952 'The undocumented "tarfile" attribute of TarInfo objects '953 + 'is deprecated and will be removed in Python 3.16',954 DeprecationWarning, stacklevel=2)955 self._tarfile = tarfile956 957 @property958 def path(self):959 'In pax headers, "name" is called "path".'960 return self.name961 962 @path.setter963 def path(self, name):964 self.name = name965 966 @property967 def linkpath(self):968 'In pax headers, "linkname" is called "linkpath".'969 return self.linkname970 971 @linkpath.setter972 def linkpath(self, linkname):973 self.linkname = linkname974 975 def __repr__(self):976 return "<%s %r at %#x>" % (self.__class__.__name__,self.name,id(self))977 978 def replace(self, *,979 name=_KEEP, mtime=_KEEP, mode=_KEEP, linkname=_KEEP,980 uid=_KEEP, gid=_KEEP, uname=_KEEP, gname=_KEEP,981 deep=True, _KEEP=_KEEP):982 """Return a deep copy of self with the given attributes replaced.983 """984 if deep:985 result = copy.deepcopy(self)986 else:987 result = copy.copy(self)988 if name is not _KEEP:989 result.name = name990 if mtime is not _KEEP:991 result.mtime = mtime992 if mode is not _KEEP:993 result.mode = mode994 if linkname is not _KEEP:995 result.linkname = linkname996 if uid is not _KEEP:997 result.uid = uid998 if gid is not _KEEP:999 result.gid = gid1000 if uname is not _KEEP:1001 result.uname = uname1002 if gname is not _KEEP:1003 result.gname = gname1004 return result1005 1006 def get_info(self):1007 """Return the TarInfo's attributes as a dictionary.1008 """1009 if self.mode is None:1010 mode = None1011 else:1012 mode = self.mode & 0o77771013 info = {1014 "name": self.name,1015 "mode": mode,1016 "uid": self.uid,1017 "gid": self.gid,1018 "size": self.size,1019 "mtime": self.mtime,1020 "chksum": self.chksum,1021 "type": self.type,1022 "linkname": self.linkname,1023 "uname": self.uname,1024 "gname": self.gname,1025 "devmajor": self.devmajor,1026 "devminor": self.devminor1027 }1028 1029 if info["type"] == DIRTYPE and not info["name"].endswith("/"):1030 info["name"] += "/"1031 1032 return info1033 1034 def tobuf(self, format=DEFAULT_FORMAT, encoding=ENCODING, errors="surrogateescape"):1035 """Return a tar header as a string of 512 byte blocks.1036 """1037 info = self.get_info()1038 for name, value in info.items():1039 if value is None:1040 raise ValueError("%s may not be None" % name)1041 1042 if format == USTAR_FORMAT:1043 return self.create_ustar_header(info, encoding, errors)1044 elif format == GNU_FORMAT:1045 return self.create_gnu_header(info, encoding, errors)1046 elif format == PAX_FORMAT:1047 return self.create_pax_header(info, encoding)1048 else:1049 raise ValueError("invalid format")1050 1051 def create_ustar_header(self, info, encoding, errors):1052 """Return the object as a ustar header block.1053 """1054 info["magic"] = POSIX_MAGIC1055 1056 if len(info["linkname"].encode(encoding, errors)) > LENGTH_LINK:1057 raise ValueError("linkname is too long")1058 1059 if len(info["name"].encode(encoding, errors)) > LENGTH_NAME:1060 info["prefix"], info["name"] = self._posix_split_name(info["name"], encoding, errors)1061 1062 return self._create_header(info, USTAR_FORMAT, encoding, errors)1063 1064 def create_gnu_header(self, info, encoding, errors):1065 """Return the object as a GNU header block sequence.1066 """1067 info["magic"] = GNU_MAGIC1068 1069 buf = b""1070 if len(info["linkname"].encode(encoding, errors)) > LENGTH_LINK:1071 buf += self._create_gnu_long_header(info["linkname"], GNUTYPE_LONGLINK, encoding, errors)1072 1073 if len(info["name"].encode(encoding, errors)) > LENGTH_NAME:1074 buf += self._create_gnu_long_header(info["name"], GNUTYPE_LONGNAME, encoding, errors)1075 1076 return buf + self._create_header(info, GNU_FORMAT, encoding, errors)1077 1078 def create_pax_header(self, info, encoding):1079 """Return the object as a ustar header block. If it cannot be1080 represented this way, prepend a pax extended header sequence1081 with supplement information.1082 """1083 info["magic"] = POSIX_MAGIC1084 pax_headers = self.pax_headers.copy()1085 1086 # Test string fields for values that exceed the field length or cannot1087 # be represented in ASCII encoding.1088 for name, hname, length in (1089 ("name", "path", LENGTH_NAME), ("linkname", "linkpath", LENGTH_LINK),1090 ("uname", "uname", 32), ("gname", "gname", 32)):1091 1092 if hname in pax_headers:1093 # The pax header has priority.1094 continue1095 1096 # Try to encode the string as ASCII.1097 try:1098 info[name].encode("ascii", "strict")1099 except UnicodeEncodeError:1100 pax_headers[hname] = info[name]1101 continue1102 1103 if len(info[name]) > length:1104 pax_headers[hname] = info[name]1105 1106 # Test number fields for values that exceed the field limit or values1107 # that like to be stored as float.1108 for name, digits in (("uid", 8), ("gid", 8), ("size", 12), ("mtime", 12)):1109 needs_pax = False1110 1111 val = info[name]1112 val_is_float = isinstance(val, float)1113 val_int = round(val) if val_is_float else val1114 if not 0 <= val_int < 8 ** (digits - 1):1115 # Avoid overflow.1116 info[name] = 01117 needs_pax = True1118 elif val_is_float:1119 # Put rounded value in ustar header, and full1120 # precision value in pax header.1121 info[name] = val_int1122 needs_pax = True1123 1124 # The existing pax header has priority.1125 if needs_pax and name not in pax_headers:1126 pax_headers[name] = str(val)1127 1128 # Create a pax extended header if necessary.1129 if pax_headers:1130 buf = self._create_pax_generic_header(pax_headers, XHDTYPE, encoding)1131 else:1132 buf = b""1133 1134 return buf + self._create_header(info, USTAR_FORMAT, "ascii", "replace")1135 1136 @classmethod1137 def create_pax_global_header(cls, pax_headers):1138 """Return the object as a pax global header block sequence.1139 """1140 return cls._create_pax_generic_header(pax_headers, XGLTYPE, "utf-8")1141 1142 def _posix_split_name(self, name, encoding, errors):1143 """Split a name longer than 100 chars into a prefix1144 and a name part.1145 """1146 components = name.split("/")1147 for i in range(1, len(components)):1148 prefix = "/".join(components[:i])1149 name = "/".join(components[i:])1150 if len(prefix.encode(encoding, errors)) <= LENGTH_PREFIX and \1151 len(name.encode(encoding, errors)) <= LENGTH_NAME:1152 break1153 else:1154 raise ValueError("name is too long")1155 1156 return prefix, name1157 1158 @staticmethod1159 def _create_header(info, format, encoding, errors):1160 """Return a header block. info is a dictionary with file1161 information, format must be one of the *_FORMAT constants.1162 """1163 has_device_fields = info.get("type") in (CHRTYPE, BLKTYPE)1164 if has_device_fields:1165 devmajor = itn(info.get("devmajor", 0), 8, format)1166 devminor = itn(info.get("devminor", 0), 8, format)1167 else:1168 devmajor = stn("", 8, encoding, errors)1169 devminor = stn("", 8, encoding, errors)1170 1171 # None values in metadata should cause ValueError.1172 # itn()/stn() do this for all fields except type.1173 filetype = info.get("type", REGTYPE)1174 if filetype is None:1175 raise ValueError("TarInfo.type must not be None")1176 1177 parts = [1178 stn(info.get("name", ""), 100, encoding, errors),1179 itn(info.get("mode", 0) & 0o7777, 8, format),1180 itn(info.get("uid", 0), 8, format),1181 itn(info.get("gid", 0), 8, format),1182 itn(info.get("size", 0), 12, format),1183 itn(info.get("mtime", 0), 12, format),1184 b" ", # checksum field1185 filetype,1186 stn(info.get("linkname", ""), 100, encoding, errors),1187 info.get("magic", POSIX_MAGIC),1188 stn(info.get("uname", ""), 32, encoding, errors),1189 stn(info.get("gname", ""), 32, encoding, errors),1190 devmajor,1191 devminor,1192 stn(info.get("prefix", ""), 155, encoding, errors)1193 ]1194 1195 buf = struct.pack("%ds" % BLOCKSIZE, b"".join(parts))1196 chksum = calc_chksums(buf[-BLOCKSIZE:])[0]1197 buf = buf[:-364] + bytes("%06o\0" % chksum, "ascii") + buf[-357:]1198 return buf1199 1200 @staticmethod