codekingpro/portable-devtools
114k
1#!/usr/bin/env python2"""3 Patch utility to apply unified diffs4 5 Brute-force line-by-line non-recursive parsing6 7 Copyright (c) 2008-2016 anatoly techtonik8 Available under the terms of MIT license9 10---11 The MIT License (MIT)12 13 Copyright (c) 2019 JFrog LTD14 15 Permission is hereby granted, free of charge, to any person obtaining a copy of this software16 and associated documentation files (the "Software"), to deal in the Software without17 restriction, including without limitation the rights to use, copy, modify, merge, publish,18 distribute, sublicense, and/or sell copies of the Software, and to permit persons to whom the19 Software is furnished to do so, subject to the following conditions:20 21 The above copyright notice and this permission notice shall be included in all copies or22 substantial portions of the Software.23 24 THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR IMPLIED,25 INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS FOR A PARTICULAR26 PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR27 ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,28 ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE29 SOFTWARE.30"""31__author__ = "Conan.io <info@conan.io>"32__version__ = "1.19.1"33__license__ = "MIT"34__url__ = "https://github.com/conan-io/python-patch"35 36import codecs37import copy38import io39import logging40import os41import posixpath42import re43import shutil44import stat45import tempfile46import urllib.request47from os.path import exists, isfile, abspath48 49#------------------------------------------------50# Logging is controlled by logger named after the51# module name (e.g. 'patch' for patch_ng.py module)52 53logger = logging.getLogger("patch_ng")54 55debug = logger.debug56info = logger.info57warning = logger.warning58error = logger.error59 60streamhandler = logging.StreamHandler()61 62# initialize logger itself63logger.addHandler(logging.NullHandler())64 65debugmode = False66 67def setdebug():68 global debugmode, streamhandler69 70 debugmode = True71 loglevel = logging.DEBUG72 logformat = "%(levelname)8s %(message)s"73 logger.setLevel(loglevel)74 75 if streamhandler not in logger.handlers:76 # when used as a library, streamhandler is not added77 # by default78 logger.addHandler(streamhandler)79 80 streamhandler.setFormatter(logging.Formatter(logformat))81 82 83#------------------------------------------------84# Constants for Patch/PatchSet types85 86DIFF = PLAIN = "plain"87GIT = "git"88HG = MERCURIAL = "mercurial"89SVN = SUBVERSION = "svn"90# mixed type is only actual when PatchSet contains91# Patches of different type92MIXED = MIXED = "mixed"93 94 95#------------------------------------------------96# Helpers (these could come with Python stdlib)97 98# x...() function are used to work with paths in99# cross-platform manner - all paths use forward100# slashes even on Windows.101 102def xisabs(filename):103 """ Cross-platform version of `os.path.isabs()`104 Returns True if `filename` is absolute on105 Linux, OS X or Windows.106 """107 if filename.startswith(b'/'): # Linux/Unix108 return True109 elif filename.startswith(b'\\'): # Windows110 return True111 elif re.match(b'\\w:[\\\\/]', filename): # Windows112 return True113 return False114 115def xnormpath(path):116 """ Cross-platform version of os.path.normpath """117 # replace escapes and Windows slashes118 normalized = posixpath.normpath(path).replace(b'\\', b'/')119 # fold the result120 return posixpath.normpath(normalized)121 122def xstrip(filename):123 """ Make relative path out of absolute by stripping124 prefixes used on Linux, OS X and Windows.125 126 This function is critical for security.127 """128 while xisabs(filename):129 # strip windows drive with all slashes130 if re.match(b'\\w:[\\\\/]', filename):131 filename = re.sub(b'^\\w+:[\\\\/]+', b'', filename)132 # strip all slashes133 elif re.match(b'[\\\\/]', filename):134 filename = re.sub(b'^[\\\\/]+', b'', filename)135 return filename136 137 138def safe_unlink(filepath):139 os.chmod(filepath, stat.S_IWUSR | stat.S_IWGRP | stat.S_IWOTH)140 os.unlink(filepath)141 142 143#-----------------------------------------------144# Main API functions145 146def fromfile(filename):147 """ Parse patch file. If successful, returns148 PatchSet() object. Otherwise returns False.149 """150 patchset = PatchSet()151 debug("reading %s" % filename)152 with open(filename, "rb") as fp:153 res = patchset.parse(fp)154 if res == True:155 return patchset156 return False157 158 159def fromstring(s):160 """ Parse text string and return PatchSet()161 object (or False if parsing fails)162 """163 ps = PatchSet( io.BytesIO(s) )164 if ps.errors == 0:165 return ps166 return False167 168 169def fromurl(url):170 """ Parse patch from an URL, return False171 if an error occured. Note that this also172 can throw urlopen() exceptions.173 """174 ps = PatchSet( urllib.request.urlopen(url) )175 if ps.errors == 0:176 return ps177 return False178 179 180# --- Utility functions ---181# [ ] reuse more universal pathsplit()182def pathstrip(path, n):183 """ Strip n leading components from the given path """184 pathlist = [path]185 while os.path.dirname(pathlist[0]) != b'':186 pathlist[0:1] = os.path.split(pathlist[0])187 return b'/'.join(pathlist[n:])188# --- /Utility function ---189 190 191def decode_text(text):192 encodings = {codecs.BOM_UTF8: "utf_8_sig",193 codecs.BOM_UTF16_BE: "utf_16_be",194 codecs.BOM_UTF16_LE: "utf_16_le",195 codecs.BOM_UTF32_BE: "utf_32_be",196 codecs.BOM_UTF32_LE: "utf_32_le",197 b'\x2b\x2f\x76\x38': "utf_7",198 b'\x2b\x2f\x76\x39': "utf_7",199 b'\x2b\x2f\x76\x2b': "utf_7",200 b'\x2b\x2f\x76\x2f': "utf_7",201 b'\x2b\x2f\x76\x38\x2d': "utf_7"}202 for bom in sorted(encodings, key=len, reverse=True):203 if text.startswith(bom):204 try:205 return text[len(bom):].decode(encodings[bom])206 except UnicodeDecodeError:207 continue208 decoders = ["utf-8", "Windows-1252"]209 for decoder in decoders:210 try:211 return text.decode(decoder)212 except UnicodeDecodeError:213 continue214 logger.warning("can't decode %s" % str(text))215 return text.decode("utf-8", "ignore") # Ignore not compatible characters216 217 218def load(path, binary=False):219 """ Loads a file content """220 with open(path, 'rb') as handle:221 tmp = handle.read()222 return tmp if binary else decode_text(tmp)223 224 225def save(path, content, only_if_modified=False):226 """227 Saves a file with given content228 Params:229 path: path to write file to230 content: contents to save in the file231 only_if_modified: file won't be modified if the content hasn't changed232 """233 try:234 os.makedirs(os.path.dirname(path))235 except Exception:236 pass237 238 new_content = content239 if not isinstance(content, bytes):240 new_content = bytes(content, "utf-8")241 242 if only_if_modified and os.path.exists(path):243 old_content = load(path, binary=True)244 if old_content == new_content:245 return246 247 with open(path, "wb") as handle:248 handle.write(new_content)249 250 251class Hunk(object):252 """ Parsed hunk data container (hunk starts with @@ -R +R @@) """253 254 def __init__(self):255 self.startsrc=None #: line count starts with 1256 self.linessrc=None257 self.starttgt=None258 self.linestgt=None259 self.invalid=False260 self.desc=''261 self.text=[]262 263 264class Patch(object):265 """ Patch for a single file.266 If used as an iterable, returns hunks.267 """268 def __init__(self):269 self.source = None270 self.target = None271 self.hunks = []272 self.hunkends = []273 self.header = []274 275 self.type = None276 self.filemode = None277 self.mode = None278 279 def __iter__(self):280 return iter(self.hunks)281 282 283class PatchSet(object):284 """ PatchSet is a patch parser and container.285 When used as an iterable, returns patches.286 """287 288 def __init__(self, stream=None):289 # --- API accessible fields ---290 291 # name of the PatchSet (filename or ...)292 self.name = None293 # patch set type - one of constants294 self.type = None295 self.filemode = None296 297 # list of Patch objects298 self.items = []299 300 self.errors = 0 # fatal parsing errors301 self.warnings = 0 # non-critical warnings302 # --- /API ---303 304 if stream:305 self.parse(stream)306 307 def __len__(self):308 return len(self.items)309 310 def __iter__(self):311 return iter(self.items)312 313 def parse(self, stream):314 """ parse unified diff315 return True on success316 """317 lineends = dict(lf=0, crlf=0, cr=0)318 nexthunkno = 0 #: even if index starts with 0 user messages number hunks from 1319 320 p = None321 hunk = None322 # hunkactual variable is used to calculate hunk lines for comparison323 hunkactual = dict(linessrc=None, linestgt=None)324 325 326 class wrapumerate(enumerate):327 """Enumerate wrapper that uses boolean end of stream status instead of328 StopIteration exception, and properties to access line information.329 """330 331 def __init__(self, *args, **kwargs):332 # we don't call parent, it is magically created by __new__ method333 334 self._exhausted = False335 self._lineno = False # after end of stream equal to the num of lines336 self._line = False # will be reset to False after end of stream337 338 def next(self):339 """Try to read the next line and return True if it is available,340 False if end of stream is reached."""341 if self._exhausted:342 return False343 344 try:345 self._lineno, self._line = super(wrapumerate, self).__next__()346 except StopIteration:347 self._exhausted = True348 self._line = False349 return False350 return True351 352 @property353 def is_empty(self):354 return self._exhausted355 356 @property357 def line(self):358 return self._line359 360 @property361 def lineno(self):362 return self._lineno363 364 # define states (possible file regions) that direct parse flow365 headscan = True # start with scanning header366 filenames = False # lines starting with --- and +++367 368 hunkhead = False # @@ -R +R @@ sequence369 hunkbody = False #370 hunkskip = False # skipping invalid hunk mode371 372 hunkparsed = False # state after successfully parsed hunk373 374 # regexp to match start of hunk, used groups - 1,3,4,6375 re_hunk_start = re.compile(br"^@@ -(\d+)(,(\d+))? \+(\d+)(,(\d+))? @@")376 377 self.errors = 0378 # temp buffers for header and filenames info379 header = []380 srcname = None381 tgtname = None382 rename = False383 384 # start of main cycle385 # each parsing block already has line available in fe.line386 fe = wrapumerate(stream)387 while fe.next():388 389 # -- deciders: these only switch state to decide who should process390 # -- line fetched at the start of this cycle391 if hunkparsed:392 hunkparsed = False393 rename = False394 if re_hunk_start.match(fe.line):395 hunkhead = True396 elif fe.line.startswith(b"--- "):397 filenames = True398 elif fe.line.startswith(b"rename from "):399 filenames = True400 else:401 headscan = True402 # -- ------------------------------------403 404 # read out header405 if headscan:406 while not fe.is_empty and not fe.line.startswith(b"--- ") and not fe.line.startswith(b"rename from "):407 header.append(fe.line)408 fe.next()409 if not fe.is_empty and fe.line.startswith(b"rename from "):410 rename = True411 hunkskip = True412 hunkbody = False413 if fe.is_empty:414 if p is None:415 debug("no patch data found") # error is shown later416 self.errors += 1417 else:418 info("%d unparsed bytes left at the end of stream" % len(b''.join(header)))419 self.warnings += 1420 # TODO check for \No new line at the end..421 # TODO test for unparsed bytes422 # otherwise error += 1423 # this is actually a loop exit424 continue425 426 headscan = False427 # switch to filenames state428 filenames = True429 430 line = fe.line431 lineno = fe.lineno432 433 434 # hunkskip and hunkbody code skipped until definition of hunkhead is parsed435 if hunkbody:436 # [x] treat empty lines inside hunks as containing single space437 # (this happens when diff is saved by copy/pasting to editor438 # that strips trailing whitespace)439 if line.strip(b"\r\n") == b"":440 debug("expanding empty line in a middle of hunk body")441 self.warnings += 1442 line = b' ' + line443 444 # process line first445 if re.match(b"^[- \\+\\\\]", line):446 # gather stats about line endings447 if line.endswith(b"\r\n"):448 p.hunkends["crlf"] += 1449 elif line.endswith(b"\n"):450 p.hunkends["lf"] += 1451 elif line.endswith(b"\r"):452 p.hunkends["cr"] += 1453 454 if line.startswith(b"-"):455 hunkactual["linessrc"] += 1456 elif line.startswith(b"+"):457 hunkactual["linestgt"] += 1458 elif not line.startswith(b"\\"):459 hunkactual["linessrc"] += 1460 hunkactual["linestgt"] += 1461 hunk.text.append(line)462 # todo: handle \ No newline cases463 else:464 warning("invalid hunk no.%d at %d for target file %s" % (nexthunkno, lineno+1, p.target))465 # add hunk status node466 hunk.invalid = True467 p.hunks.append(hunk)468 self.errors += 1469 # switch to hunkskip state470 hunkbody = False471 hunkskip = True472 473 # check exit conditions474 if hunkactual["linessrc"] > hunk.linessrc or hunkactual["linestgt"] > hunk.linestgt:475 warning("extra lines for hunk no.%d at %d for target %s" % (nexthunkno, lineno+1, p.target))476 # add hunk status node477 hunk.invalid = True478 p.hunks.append(hunk)479 self.errors += 1480 # switch to hunkskip state481 hunkbody = False482 hunkskip = True483 elif hunk.linessrc == hunkactual["linessrc"] and hunk.linestgt == hunkactual["linestgt"]:484 # hunk parsed successfully485 p.hunks.append(hunk)486 # switch to hunkparsed state487 hunkbody = False488 hunkparsed = True489 490 # detect mixed window/unix line ends491 ends = p.hunkends492 if ((ends["cr"]!=0) + (ends["crlf"]!=0) + (ends["lf"]!=0)) > 1:493 warning("inconsistent line ends in patch hunks for %s" % p.source)494 self.warnings += 1495 if debugmode:496 debuglines = dict(ends)497 debuglines.update(file=p.target, hunk=nexthunkno)498 debug("crlf: %(crlf)d lf: %(lf)d cr: %(cr)d\t - file: %(file)s hunk: %(hunk)d" % debuglines)499 # fetch next line500 continue501 502 if hunkskip:503 if re_hunk_start.match(line):504 # switch to hunkhead state505 hunkskip = False506 hunkhead = True507 elif line.startswith(b"--- ") or line.startswith(b"rename from "):508 # switch to filenames state509 hunkskip = False510 filenames = True511 if debugmode and len(self.items) > 0:512 debug("- %2d hunks for %s" % (len(p.hunks), p.source))513 514 if filenames:515 if line.startswith(b"--- "):516 if srcname != None:517 # XXX testcase518 warning("skipping false patch for %s" % srcname)519 srcname = None520 # XXX header += srcname521 # double source filename line is encountered522 # attempt to restart from this second line523 524 # Files dated at Unix epoch don't exist, e.g.:525 # '1970-01-01 01:00:00.000000000 +0100'526 # They include timezone offsets.527 # .. which can be parsed (if we remove the nanoseconds)528 # .. by strptime() with:529 # '%Y-%m-%d %H:%M:%S %z'530 # .. but unfortunately this relies on the OSes libc531 # strptime function and %z support is patchy, so we drop532 # everything from the . onwards and group the year and time533 # separately.534 re_filename_date_time = br"^--- ([^\t]+)(?:\s([0-9-]+)\s([0-9:]+)|.*)"535 match = re.match(re_filename_date_time, line)536 # todo: support spaces in filenames537 if match:538 srcname = match.group(1).strip()539 date = match.group(2)540 time = match.group(3)541 if (date == b'1970-01-01' or date == b'1969-12-31') and time.split(b':',1)[1] == b'00:00':542 srcname = b'/dev/null'543 else:544 warning("skipping invalid filename at line %d" % (lineno+1))545 self.errors += 1546 # XXX p.header += line547 # switch back to headscan state548 filenames = False549 headscan = True550 elif rename:551 if line.startswith(b"rename from "):552 re_rename_from = br"^rename from (.+)"553 match = re.match(re_rename_from, line)554 if match:555 srcname = match.group(1).strip()556 else:557 warning("skipping invalid rename from at line %d" % (lineno+1))558 self.errors += 1559 # XXX p.header += line560 # switch back to headscan state561 filenames = False562 headscan = True563 if not fe.is_empty:564 fe.next()565 line = fe.line566 lineno = fe.lineno567 re_rename_to = br"^rename to (.+)"568 match = re.match(re_rename_to, line)569 if match:570 tgtname = match.group(1).strip()571 else:572 warning("skipping invalid rename from at line %d" % (lineno + 1))573 self.errors += 1574 # XXX p.header += line575 # switch back to headscan state576 filenames = False577 headscan = True578 if p: # for the first run p is None579 self.items.append(p)580 p = Patch()581 p.source = srcname582 srcname = None583 p.target = tgtname584 tgtname = None585 p.header = header586 header = []587 # switch to hunkhead state588 filenames = False589 hunkhead = False590 nexthunkno = 0591 p.hunkends = lineends.copy()592 hunkparsed = True593 continue594 elif not line.startswith(b"+++ "):595 if srcname != None:596 warning("skipping invalid patch with no target for %s" % srcname)597 self.errors += 1598 srcname = None599 # XXX header += srcname600 # XXX header += line601 else:602 # this should be unreachable603 warning("skipping invalid target patch")604 filenames = False605 headscan = True606 else:607 if tgtname != None:608 # XXX seems to be a dead branch609 warning("skipping invalid patch - double target at line %d" % (lineno+1))610 self.errors += 1611 srcname = None612 tgtname = None613 # XXX header += srcname614 # XXX header += tgtname615 # XXX header += line616 # double target filename line is encountered617 # switch back to headscan state618 filenames = False619 headscan = True620 else:621 re_filename_date_time = br"^\+\+\+ ([^\t]+)(?:\s([0-9-]+)\s([0-9:]+)|.*)"622 match = re.match(re_filename_date_time, line)623 if not match:624 warning("skipping invalid patch - no target filename at line %d" % (lineno+1))625 self.errors += 1626 srcname = None627 # switch back to headscan state628 filenames = False629 headscan = True630 else:631 tgtname = match.group(1).strip()632 date = match.group(2)633 time = match.group(3)634 if (date == b'1970-01-01' or date == b'1969-12-31') and time.split(b':',1)[1] == b'00:00':635 tgtname = b'/dev/null'636 if p: # for the first run p is None637 self.items.append(p)638 p = Patch()639 p.source = srcname640 srcname = None641 p.target = tgtname642 tgtname = None643 p.header = header644 header = []645 # switch to hunkhead state646 filenames = False647 hunkhead = True648 nexthunkno = 0649 p.hunkends = lineends.copy()650 continue651 652 if hunkhead:653 match = re.match(br"^@@ -(\d+)(,(\d+))? \+(\d+)(,(\d+))? @@(.*)", line)654 if not match:655 if not p.hunks:656 warning("skipping invalid patch with no hunks for file %s" % p.source)657 self.errors += 1658 # XXX review switch659 # switch to headscan state660 hunkhead = False661 headscan = True662 continue663 else:664 # TODO review condition case665 # switch to headscan state666 hunkhead = False667 headscan = True668 else:669 hunk = Hunk()670 hunk.startsrc = int(match.group(1))671 hunk.linessrc = 1672 if match.group(3): hunk.linessrc = int(match.group(3))673 hunk.starttgt = int(match.group(4))674 hunk.linestgt = 1675 if match.group(6): hunk.linestgt = int(match.group(6))676 hunk.invalid = False677 hunk.desc = match.group(7)[1:].rstrip()678 hunk.text = []679 680 hunkactual["linessrc"] = hunkactual["linestgt"] = 0681 682 # switch to hunkbody state683 hunkhead = False684 hunkbody = True685 nexthunkno += 1686 continue687 688 # /while fe.next()689 690 if p:691 self.items.append(p)692 693 if not hunkparsed:694 if hunkskip:695 warning("warning: finished with errors, some hunks may be invalid")696 elif headscan:697 if len(self.items) == 0:698 warning("error: no patch data found!")699 return False700 else: # extra data at the end of file701 pass702 else:703 warning("error: patch stream is incomplete!")704 self.errors += 1705 if len(self.items) == 0:706 return False707 708 if debugmode and len(self.items) > 0:709 debug("- %2d hunks for %s" % (len(p.hunks), p.source))710 711 # XXX fix total hunks calculation712 debug("total files: %d total hunks: %d" % (len(self.items),713 sum(len(p.hunks) for p in self.items)))714 715 # ---- detect patch and patchset types ----716 for idx, p in enumerate(self.items):717 self.items[idx].type = self._detect_type(p)718 if self.items[idx].type == GIT:719 self.items[idx].filemode = self._detect_file_mode(p)720 self.items[idx].mode = self._detect_patch_mode(p)721 722 types = set([p.type for p in self.items])723 if len(types) > 1:724 self.type = MIXED725 else:726 self.type = types.pop()727 # --------728 729 self._normalize_filenames()730 731 return (self.errors == 0)732 733 def _detect_type(self, p):734 """ detect and return type for the specified Patch object735 analyzes header and filenames info736 737 NOTE: must be run before filenames are normalized738 """739 740 # check for SVN741 # - header starts with Index:742 # - next line is ===... delimiter743 # - filename is followed by revision number744 # TODO add SVN revision745 if (len(p.header) > 1 and p.header[-2].startswith(b"Index: ")746 and p.header[-1].startswith(b"="*67)):747 return SVN748 749 # common checks for both HG and GIT750 DVCS = ((p.source.startswith(b'a/') or p.source == b'/dev/null')751 and (p.target.startswith(b'b/') or p.target == b'/dev/null'))752 753 # GIT type check754 # - header[-2] is like "diff --git a/oldname b/newname"755 # - header[-1] is like "index <hash>..<hash> <mode>"756 # TODO add git rename diffs and add/remove diffs757 # add git diff with spaced filename758 # TODO http://www.kernel.org/pub/software/scm/git/docs/git-diff.html759 760 # Git patch header len is 2 min761 if len(p.header) > 1:762 # detect the start of diff header - there might be some comments before763 for idx in reversed(range(len(p.header))):764 if p.header[idx].startswith(b"diff --git"):765 break766 if p.header[idx].startswith(b'diff --git a/'):767 git_indicators = []768 for i in range(idx + 1, len(p.header)):769 git_indicators.append(p.header[i])770 for line in git_indicators:771 if re.match(772 b'(?:index \\w{4,40}\\.\\.\\w{4,40}(?: \\d{6})?|new file mode \\d+|deleted file mode \\d+|old mode \\d+|new mode \\d+)',773 line):774 if DVCS:775 return GIT776 777 # Additional check: look for mode change patterns778 # "old mode XXXXX" followed by "new mode XXXXX"779 has_old_mode = False780 has_new_mode = False781 782 for line in git_indicators:783 if re.match(b'old mode \\d+', line):784 has_old_mode = True785 elif re.match(b'new mode \\d+', line):786 has_new_mode = True787 788 # If we have both old and new mode, it's definitely Git789 if has_old_mode and has_new_mode and DVCS:790 return GIT791 792 # Check for similarity index (Git renames/copies)793 for line in git_indicators:794 if re.match(b'similarity index \\d+%', line):795 return GIT796 797 # HG check798 #799 # - for plain HG format header is like "diff -r b2d9961ff1f5 filename"800 # - for Git-style HG patches it is "diff --git a/oldname b/newname"801 # - filename starts with a/, b/ or is equal to /dev/null802 # - exported changesets also contain the header803 # # HG changeset patch804 # # User name@example.com805 # ...806 # TODO add MQ807 # TODO add revision info808 if len(p.header) > 0:809 if DVCS and re.match(b'diff -r \\w{12} .*', p.header[-1]):810 return HG811 if DVCS and p.header[-1].startswith(b'diff --git a/'):812 if len(p.header) == 1: # native Git patch header len is 2813 return HG814 elif p.header[0].startswith(b'# HG changeset patch'):815 return HG816 817 return PLAIN818 819 def _detect_file_mode(self, p):820 """ Detect the file mode listed in the patch header821 822 INFO: Only working with Git-style patches823 """824 if len(p.header) > 1:825 for idx in reversed(range(len(p.header))):826 if p.header[idx].startswith(b"diff --git"):827 break828 if p.header[idx].startswith(b'diff --git a/'):829 if idx + 1 < len(p.header):830 # new file (e.g)831 # diff --git a/quote.txt b/quote.txt832 # new file mode 100755833 match = re.match(b'new file mode (\\d+)', p.header[idx + 1])834 if match:835 return int(match.group(1), 8)836 # changed mode (e.g)837 # diff --git a/quote.txt b/quote.txt838 # old mode 100755839 # new mode 100644840 if idx + 2 < len(p.header):841 match = re.match(b'new mode (\\d+)', p.header[idx + 2])842 if match:843 return int(match.group(1), 8)844 return None845 846 def _apply_filemode(self, filepath, filemode):847 if filemode is not None and stat.S_ISREG(filemode):848 try:849 only_file_permissions = filemode & 0o777850 os.chmod(filepath, only_file_permissions)851 except Exception as error:852 warning(f"Could not set filemode {oct(filemode)} for {filepath}: {str(error)}")853 854 def _detect_patch_mode(self, p):855 """Detect patch mode - add, delete, rename, etc.856 """857 if len(p.header) > 1:858 for idx in reversed(range(len(p.header))):859 if p.header[idx].startswith(b"diff --git"):860 break861 change_pattern = re.compile(rb"^diff --git a/([^ ]+) b/(.+)")862 match = change_pattern.match(p.header[idx])863 if match:864 if match.group(1) != match.group(2) and not p.hunks and p.source != b'/dev/null' and p.target != b'/dev/null':865 return 'rename'866 return None867 868 def _normalize_filenames(self):869 """ sanitize filenames, normalizing paths, i.e.:870 1. strip a/ and b/ prefixes from GIT and HG style patches871 2. remove all references to parent directories (with warning)872 3. translate any absolute paths to relative (with warning)873 874 [x] always use forward slashes to be crossplatform875 (diff/patch were born as a unix utility after all)876 877 return None878 """879 if debugmode:880 debug("normalize filenames")881 for i,p in enumerate(self.items):882 if debugmode:883 debug(" patch type = %s" % p.type)884 debug(" filemode = %s" % p.filemode)885 debug(" source = %s" % p.source)886 debug(" target = %s" % p.target)887 if p.type in (HG, GIT):888 debug("stripping a/ and b/ prefixes")889 if p.source != b'/dev/null':890 if not p.source.startswith(b"a/"):891 warning("invalid source filename")892 else:893 p.source = p.source[2:]894 if p.target != b'/dev/null':895 if not p.target.startswith(b"b/"):896 warning("invalid target filename")897 else:898 p.target = p.target[2:]899 900 p.source = xnormpath(p.source)901 p.target = xnormpath(p.target)902 903 p.source = p.source.strip(b'"')904 p.target = p.target.strip(b'"')905 906 sep = b'/' # sep value can be hardcoded, but it looks nice this way907 908 # references to parent are not allowed909 if p.source.startswith(b".." + sep):910 warning("error: stripping parent path for source file patch no.%d" % (i+1))911 self.warnings += 1912 while p.source.startswith(b".." + sep):913 p.source = p.source.partition(sep)[2]914 if p.target.startswith(b".." + sep):915 warning("error: stripping parent path for target file patch no.%d" % (i+1))916 self.warnings += 1917 while p.target.startswith(b".." + sep):918 p.target = p.target.partition(sep)[2]919 # absolute paths are not allowed920 if (xisabs(p.source) and p.source != b'/dev/null') or \921 (xisabs(p.target) and p.target != b'/dev/null'):922 warning("error: absolute paths are not allowed - file no.%d" % (i+1))923 self.warnings += 1924 if xisabs(p.source) and p.source != b'/dev/null':925 warning("stripping absolute path from source name '%s'" % p.source)926 p.source = xstrip(p.source)927 if xisabs(p.target) and p.target != b'/dev/null':928 warning("stripping absolute path from target name '%s'" % p.target)929 p.target = xstrip(p.target)930 931 self.items[i].source = p.source932 self.items[i].target = p.target933 934 935 def diffstat(self):936 """ calculate diffstat and return as a string937 Notes:938 - original diffstat ouputs target filename939 - single + or - shouldn't escape histogram940 """941 names = []942 insert = []943 delete = []944 delta = 0 # size change in bytes945 namelen = 0946 maxdiff = 0 # max number of changes for single file947 # (for histogram width calculation)948 for patch in self.items:949 i,d = 0,0950 for hunk in patch.hunks:951 for line in hunk.text:952 if line.startswith(b'+'):953 i += 1954 delta += len(line)-1955 elif line.startswith(b'-'):956 d += 1957 delta -= len(line)-1958 names.append(patch.target)959 insert.append(i)960 delete.append(d)961 namelen = max(namelen, len(patch.target))962 maxdiff = max(maxdiff, i+d)963 output = ''964 statlen = len(str(maxdiff)) # stats column width965 for i,n in enumerate(names):966 # %-19s | %-4d %s967 format = " %-" + str(namelen) + "s | %" + str(statlen) + "s %s\n"968 969 hist = ''970 # -- calculating histogram --971 width = len(format % ('', '', ''))972 histwidth = max(2, 80 - width)973 if maxdiff < histwidth:974 hist = "+"*insert[i] + "-"*delete[i]975 else:976 iratio = (float(insert[i]) / maxdiff) * histwidth977 dratio = (float(delete[i]) / maxdiff) * histwidth978 979 # make sure every entry gets at least one + or -980 iwidth = 1 if 0 < iratio < 1 else int(iratio)981 dwidth = 1 if 0 < dratio < 1 else int(dratio)982 #print(iratio, dratio, iwidth, dwidth, histwidth)983 hist = "+"*int(iwidth) + "-"*int(dwidth)984 # -- /calculating +- histogram --985 output += (format % (names[i].decode('utf-8'), str(insert[i] + delete[i]), hist))986 987 output += (" %d files changed, %d insertions(+), %d deletions(-), %+d bytes"988 % (len(names), sum(insert), sum(delete), delta))989 return output990 991 992 def findfiles(self, old, new):993 """ return tuple of source file, target file """994 if old == b'/dev/null':995 handle, abspath = tempfile.mkstemp(suffix='pypatch')996 abspath = abspath.encode()997 # The source file must contain a line for the hunk matching to succeed.998 os.write(handle, b' ')999 os.close(handle)1000 if not exists(new):1001 handle = open(new, 'wb')1002 handle.close()1003 return abspath, new1004 elif exists(old):1005 return old, old1006 elif exists(new):1007 return new, new1008 elif new == b'/dev/null':1009 return None, None1010 else:1011 # [w] Google Code generates broken patches with its online editor1012 debug("broken patch from Google Code, stripping prefixes..")1013 if old.startswith(b'a/') and new.startswith(b'b/'):1014 old, new = old[2:], new[2:]1015 debug(" %s" % old)1016 debug(" %s" % new)1017 if exists(old):1018 return old, old1019 elif exists(new):1020 return new, new1021 return None, None1022 1023 def _strip_prefix(self, filename):1024 if filename.startswith(b'a/') or filename.startswith(b'b/'):1025 return filename[2:]1026 return filename1027 1028 def decode_clean(self, path, prefix):1029 path = path.decode("utf-8").replace("\\", "/")1030 if path.startswith(prefix):1031 path = path[2:]1032 return path1033 1034 def strip_path(self, path, base_path, strip=0):1035 tokens = path.split("/")1036 if len(tokens) > 1:1037 tokens = tokens[strip:]1038 path = "/".join(tokens)1039 if base_path:1040 path = os.path.join(base_path, path)1041 return path1042 # account for new and deleted files, upstream dep won't fix them1043 1044 1045 1046 1047 def apply(self, strip=0, root=None, fuzz=False):1048 """ Apply parsed patch, optionally stripping leading components1049 from file paths. `root` parameter specifies working dir.1050 :param strip: Strip patch path1051 :param root: Folder to apply the patch1052 :param fuzz: Accept fuzzy patches1053 return True on success1054 """1055 items = []1056 for item in self.items:1057 source = self.decode_clean(item.source, "a/")1058 target = self.decode_clean(item.target, "b/")1059 if "dev/null" in source:1060 target = self.strip_path(target, root, strip)1061 hunks = [s.decode("utf-8") for s in item.hunks[0].text]1062 new_file = "".join(hunk[1:] for hunk in hunks)1063 save(target, new_file)1064 self._apply_filemode(target, item.filemode)1065 elif "dev/null" in target:1066 source = self.strip_path(source, root, strip)1067 safe_unlink(source)1068 elif item.mode == 'rename':1069 source = self.strip_path(source, root, strip)1070 target = self.strip_path(target, root, strip)1071 if exists(source):1072 os.makedirs(os.path.dirname(target), exist_ok=True)1073 shutil.move(source, target)1074 self._apply_filemode(target, item.filemode)1075 else:1076 items.append(item)1077 self.items = items1078 1079 if root:1080 prevdir = os.getcwd()1081 os.chdir(root)1082 1083 total = len(self.items)1084 errors = 01085 if strip:1086 # [ ] test strip level exceeds nesting level1087 # [ ] test the same only for selected files1088 # [ ] test if files end up being on the same level1089 try:1090 strip = int(strip)1091 except ValueError:1092 errors += 11093 warning("error: strip parameter '%s' must be an integer" % strip)1094 strip = 01095 1096 #for fileno, filename in enumerate(self.source):1097 for i,p in enumerate(self.items):1098 if strip:1099 debug("stripping %s leading component(s) from:" % strip)1100 debug(" %s" % p.source)1101 debug(" %s" % p.target)1102 old = p.source if p.source == b'/dev/null' else pathstrip(p.source, strip)1103 new = p.target if p.target == b'/dev/null' else pathstrip(p.target, strip)1104 else:1105 old, new = p.source, p.target1106 1107 filenameo, filenamen = self.findfiles(old, new)1108 1109 if not filenameo or not filenamen:1110 error("source/target file does not exist:\n --- %s\n +++ %s" % (old, new))1111 errors += 11112 continue1113 if not isfile(filenameo):1114 error("not a file - %s" % filenameo)1115 errors += 11116 continue1117 1118 # [ ] check absolute paths security here1119 debug("processing %d/%d:\t %s" % (i+1, total, filenamen))1120 1121 # validate before patching1122 f2fp = open(filenameo, 'rb')1123 hunkno = 01124 hunk = p.hunks[hunkno]1125 hunkfind = []1126 hunkreplace = []1127 validhunks = 01128 canpatch = False1129 for lineno, line in enumerate(f2fp):1130 if lineno+1 < hunk.startsrc:1131 continue1132 elif lineno+1 == hunk.startsrc:1133 hunkfind = [x[1:].rstrip(b"\r\n") for x in hunk.text if x[0] in b" -"]1134 hunkreplace = [x[1:].rstrip(b"\r\n") for x in hunk.text if x[0] in b" +"]1135 #pprint(hunkreplace)1136 hunklineno = 01137 1138 # todo \ No newline at end of file1139 1140 # check hunks in source file1141 if lineno+1 < hunk.startsrc+len(hunkfind):1142 if line.rstrip(b"\r\n") == hunkfind[hunklineno]:1143 hunklineno += 11144 else:1145 warning("file %d/%d:\t %s" % (i+1, total, filenamen))1146 warning(" hunk no.%d doesn't match source file at line %d" % (hunkno+1, lineno+1))1147 warning(" expected: %s" % hunkfind[hunklineno])1148 warning(" actual : %s" % line.rstrip(b"\r\n"))1149 if fuzz:1150 hunklineno += 11151 else:1152 # not counting this as error, because file may already be patched.1153 # check if file is already patched is done after the number of1154 # invalid hunks if found1155 # TODO: check hunks against source/target file in one pass1156 # API - check(stream, srchunks, tgthunks)1157 # return tuple (srcerrs, tgterrs)1158 1159 # continue to check other hunks for completeness1160 hunkno += 11161 if hunkno < len(p.hunks):1162 hunk = p.hunks[hunkno]1163 continue1164 else:1165 break1166 1167 # check if processed line is the last line1168 if len(hunkfind) == 0 or lineno+1 == hunk.startsrc+len(hunkfind)-1:1169 debug(" hunk no.%d for file %s -- is ready to be patched" % (hunkno+1, filenamen))1170 hunkno+=11171 validhunks+=11172 if hunkno < len(p.hunks):1173 hunk = p.hunks[hunkno]1174 else:1175 if validhunks == len(p.hunks):1176 # patch file1177 canpatch = True1178 break1179 else:1180 if hunkno < len(p.hunks):1181 error("premature end of source file %s at hunk %d" % (filenameo, hunkno+1))1182 errors += 11183 1184 f2fp.close()1185 1186 if validhunks < len(p.hunks):1187 if self._match_file_hunks(filenameo, p.hunks):1188 warning("already patched %s" % filenameo)1189 else:1190 if fuzz:1191 warning("source file is different - %s" % filenameo)1192 else:1193 error("source file is different - %s" % filenameo)1194 errors += 11195 if canpatch:1196 backupname = filenamen+b".orig"1197 if exists(backupname):1198 warning("can't backup original file to %s - aborting" % backupname)1199 errors += 11200 else: