codekingpro/portable-devtools
114k
1"""PyPI and direct package downloading."""2 3import sys4import os5import re6import io7import shutil8import socket9import base6410import hashlib11import itertools12import configparser13import html14import http.client15import urllib.parse16import urllib.request17import urllib.error18from functools import wraps19 20import setuptools21from pkg_resources import (22 CHECKOUT_DIST,23 Distribution,24 BINARY_DIST,25 normalize_path,26 SOURCE_DIST,27 Environment,28 find_distributions,29 safe_name,30 safe_version,31 to_filename,32 Requirement,33 DEVELOP_DIST,34 EGG_DIST,35 parse_version,36)37from distutils import log38from distutils.errors import DistutilsError39from fnmatch import translate40from setuptools.wheel import Wheel41from setuptools.extern.more_itertools import unique_everseen42 43 44EGG_FRAGMENT = re.compile(r'^egg=([-A-Za-z0-9_.+!]+)$')45HREF = re.compile(r"""href\s*=\s*['"]?([^'"> ]+)""", re.I)46PYPI_MD5 = re.compile(47 r'<a href="([^"#]+)">([^<]+)</a>\n\s+\(<a (?:title="MD5 hash"\n\s+)'48 r'href="[^?]+\?:action=show_md5&digest=([0-9a-f]{32})">md5</a>\)'49)50URL_SCHEME = re.compile('([-+.a-z0-9]{2,}):', re.I).match51EXTENSIONS = ".tar.gz .tar.bz2 .tar .zip .tgz".split()52 53__all__ = [54 'PackageIndex',55 'distros_for_url',56 'parse_bdist_wininst',57 'interpret_distro_name',58]59 60_SOCKET_TIMEOUT = 1561 62_tmpl = "setuptools/{setuptools.__version__} Python-urllib/{py_major}"63user_agent = _tmpl.format(64 py_major='{}.{}'.format(*sys.version_info), setuptools=setuptools65)66 67 68def parse_requirement_arg(spec):69 try:70 return Requirement.parse(spec)71 except ValueError as e:72 raise DistutilsError(73 "Not a URL, existing file, or requirement spec: %r" % (spec,)74 ) from e75 76 77def parse_bdist_wininst(name):78 """Return (base,pyversion) or (None,None) for possible .exe name"""79 80 lower = name.lower()81 base, py_ver, plat = None, None, None82 83 if lower.endswith('.exe'):84 if lower.endswith('.win32.exe'):85 base = name[:-10]86 plat = 'win32'87 elif lower.startswith('.win32-py', -16):88 py_ver = name[-7:-4]89 base = name[:-16]90 plat = 'win32'91 elif lower.endswith('.win-amd64.exe'):92 base = name[:-14]93 plat = 'win-amd64'94 elif lower.startswith('.win-amd64-py', -20):95 py_ver = name[-7:-4]96 base = name[:-20]97 plat = 'win-amd64'98 return base, py_ver, plat99 100 101def egg_info_for_url(url):102 parts = urllib.parse.urlparse(url)103 scheme, server, path, parameters, query, fragment = parts104 base = urllib.parse.unquote(path.split('/')[-1])105 if server == 'sourceforge.net' and base == 'download': # XXX Yuck106 base = urllib.parse.unquote(path.split('/')[-2])107 if '#' in base:108 base, fragment = base.split('#', 1)109 return base, fragment110 111 112def distros_for_url(url, metadata=None):113 """Yield egg or source distribution objects that might be found at a URL"""114 base, fragment = egg_info_for_url(url)115 yield from distros_for_location(url, base, metadata)116 if fragment:117 match = EGG_FRAGMENT.match(fragment)118 if match:119 yield from interpret_distro_name(120 url, match.group(1), metadata, precedence=CHECKOUT_DIST121 )122 123 124def distros_for_location(location, basename, metadata=None):125 """Yield egg or source distribution objects based on basename"""126 if basename.endswith('.egg.zip'):127 basename = basename[:-4] # strip the .zip128 if basename.endswith('.egg') and '-' in basename:129 # only one, unambiguous interpretation130 return [Distribution.from_location(location, basename, metadata)]131 if basename.endswith('.whl') and '-' in basename:132 wheel = Wheel(basename)133 if not wheel.is_compatible():134 return []135 return [136 Distribution(137 location=location,138 project_name=wheel.project_name,139 version=wheel.version,140 # Increase priority over eggs.141 precedence=EGG_DIST + 1,142 )143 ]144 if basename.endswith('.exe'):145 win_base, py_ver, platform = parse_bdist_wininst(basename)146 if win_base is not None:147 return interpret_distro_name(148 location, win_base, metadata, py_ver, BINARY_DIST, platform149 )150 # Try source distro extensions (.zip, .tgz, etc.)151 #152 for ext in EXTENSIONS:153 if basename.endswith(ext):154 basename = basename[: -len(ext)]155 return interpret_distro_name(location, basename, metadata)156 return [] # no extension matched157 158 159def distros_for_filename(filename, metadata=None):160 """Yield possible egg or source distribution objects based on a filename"""161 return distros_for_location(162 normalize_path(filename), os.path.basename(filename), metadata163 )164 165 166def interpret_distro_name(167 location, basename, metadata, py_version=None, precedence=SOURCE_DIST, platform=None168):169 """Generate the interpretation of a source distro name170 171 Note: if `location` is a filesystem filename, you should call172 ``pkg_resources.normalize_path()`` on it before passing it to this173 routine!174 """175 176 parts = basename.split('-')177 if not py_version and any(re.match(r'py\d\.\d$', p) for p in parts[2:]):178 # it is a bdist_dumb, not an sdist -- bail out179 return180 181 # find the pivot (p) that splits the name from the version.182 # infer the version as the first item that has a digit.183 for p in range(len(parts)):184 if parts[p][:1].isdigit():185 break186 else:187 p = len(parts)188 189 yield Distribution(190 location,191 metadata,192 '-'.join(parts[:p]),193 '-'.join(parts[p:]),194 py_version=py_version,195 precedence=precedence,196 platform=platform,197 )198 199 200def unique_values(func):201 """202 Wrap a function returning an iterable such that the resulting iterable203 only ever yields unique items.204 """205 206 @wraps(func)207 def wrapper(*args, **kwargs):208 return unique_everseen(func(*args, **kwargs))209 210 return wrapper211 212 213REL = re.compile(r"""<([^>]*\srel\s{0,10}=\s{0,10}['"]?([^'" >]+)[^>]*)>""", re.I)214"""215Regex for an HTML tag with 'rel="val"' attributes.216"""217 218 219@unique_values220def find_external_links(url, page):221 """Find rel="homepage" and rel="download" links in `page`, yielding URLs"""222 223 for match in REL.finditer(page):224 tag, rel = match.groups()225 rels = set(map(str.strip, rel.lower().split(',')))226 if 'homepage' in rels or 'download' in rels:227 for match in HREF.finditer(tag):228 yield urllib.parse.urljoin(url, htmldecode(match.group(1)))229 230 for tag in ("<th>Home Page", "<th>Download URL"):231 pos = page.find(tag)232 if pos != -1:233 match = HREF.search(page, pos)234 if match:235 yield urllib.parse.urljoin(url, htmldecode(match.group(1)))236 237 238class ContentChecker:239 """240 A null content checker that defines the interface for checking content241 """242 243 def feed(self, block):244 """245 Feed a block of data to the hash.246 """247 return248 249 def is_valid(self):250 """251 Check the hash. Return False if validation fails.252 """253 return True254 255 def report(self, reporter, template):256 """257 Call reporter with information about the checker (hash name)258 substituted into the template.259 """260 return261 262 263class HashChecker(ContentChecker):264 pattern = re.compile(265 r'(?P<hash_name>sha1|sha224|sha384|sha256|sha512|md5)='266 r'(?P<expected>[a-f0-9]+)'267 )268 269 def __init__(self, hash_name, expected):270 self.hash_name = hash_name271 self.hash = hashlib.new(hash_name)272 self.expected = expected273 274 @classmethod275 def from_url(cls, url):276 "Construct a (possibly null) ContentChecker from a URL"277 fragment = urllib.parse.urlparse(url)[-1]278 if not fragment:279 return ContentChecker()280 match = cls.pattern.search(fragment)281 if not match:282 return ContentChecker()283 return cls(**match.groupdict())284 285 def feed(self, block):286 self.hash.update(block)287 288 def is_valid(self):289 return self.hash.hexdigest() == self.expected290 291 def report(self, reporter, template):292 msg = template % self.hash_name293 return reporter(msg)294 295 296class PackageIndex(Environment):297 """A distribution index that scans web pages for download URLs"""298 299 def __init__(300 self,301 index_url="https://pypi.org/simple/",302 hosts=('*',),303 ca_bundle=None,304 verify_ssl=True,305 *args,306 **kw,307 ):308 super().__init__(*args, **kw)309 self.index_url = index_url + "/"[: not index_url.endswith('/')]310 self.scanned_urls = {}311 self.fetched_urls = {}312 self.package_pages = {}313 self.allows = re.compile('|'.join(map(translate, hosts))).match314 self.to_scan = []315 self.opener = urllib.request.urlopen316 317 def add(self, dist):318 # ignore invalid versions319 try:320 parse_version(dist.version)321 except Exception:322 return None323 return super().add(dist)324 325 # FIXME: 'PackageIndex.process_url' is too complex (14)326 def process_url(self, url, retrieve=False): # noqa: C901327 """Evaluate a URL as a possible download, and maybe retrieve it"""328 if url in self.scanned_urls and not retrieve:329 return330 self.scanned_urls[url] = True331 if not URL_SCHEME(url):332 self.process_filename(url)333 return334 else:335 dists = list(distros_for_url(url))336 if dists:337 if not self.url_ok(url):338 return339 self.debug("Found link: %s", url)340 341 if dists or not retrieve or url in self.fetched_urls:342 list(map(self.add, dists))343 return # don't need the actual page344 345 if not self.url_ok(url):346 self.fetched_urls[url] = True347 return348 349 self.info("Reading %s", url)350 self.fetched_urls[url] = True # prevent multiple fetch attempts351 tmpl = "Download error on %s: %%s -- Some packages may not be found!"352 f = self.open_url(url, tmpl % url)353 if f is None:354 return355 if isinstance(f, urllib.error.HTTPError) and f.code == 401:356 self.info("Authentication error: %s" % f.msg)357 self.fetched_urls[f.url] = True358 if 'html' not in f.headers.get('content-type', '').lower():359 f.close() # not html, we can't process it360 return361 362 base = f.url # handle redirects363 page = f.read()364 if not isinstance(page, str):365 # In Python 3 and got bytes but want str.366 if isinstance(f, urllib.error.HTTPError):367 # Errors have no charset, assume latin1:368 charset = 'latin-1'369 else:370 charset = f.headers.get_param('charset') or 'latin-1'371 page = page.decode(charset, "ignore")372 f.close()373 for match in HREF.finditer(page):374 link = urllib.parse.urljoin(base, htmldecode(match.group(1)))375 self.process_url(link)376 if url.startswith(self.index_url) and getattr(f, 'code', None) != 404:377 page = self.process_index(url, page)378 379 def process_filename(self, fn, nested=False):380 # process filenames or directories381 if not os.path.exists(fn):382 self.warn("Not found: %s", fn)383 return384 385 if os.path.isdir(fn) and not nested:386 path = os.path.realpath(fn)387 for item in os.listdir(path):388 self.process_filename(os.path.join(path, item), True)389 390 dists = distros_for_filename(fn)391 if dists:392 self.debug("Found: %s", fn)393 list(map(self.add, dists))394 395 def url_ok(self, url, fatal=False):396 s = URL_SCHEME(url)397 is_file = s and s.group(1).lower() == 'file'398 if is_file or self.allows(urllib.parse.urlparse(url)[1]):399 return True400 msg = (401 "\nNote: Bypassing %s (disallowed host; see "402 "https://setuptools.pypa.io/en/latest/deprecated/"403 "easy_install.html#restricting-downloads-with-allow-hosts for details).\n"404 )405 if fatal:406 raise DistutilsError(msg % url)407 else:408 self.warn(msg, url)409 return False410 411 def scan_egg_links(self, search_path):412 dirs = filter(os.path.isdir, search_path)413 egg_links = (414 (path, entry)415 for path in dirs416 for entry in os.listdir(path)417 if entry.endswith('.egg-link')418 )419 list(itertools.starmap(self.scan_egg_link, egg_links))420 421 def scan_egg_link(self, path, entry):422 with open(os.path.join(path, entry)) as raw_lines:423 # filter non-empty lines424 lines = list(filter(None, map(str.strip, raw_lines)))425 426 if len(lines) != 2:427 # format is not recognized; punt428 return429 430 egg_path, setup_path = lines431 432 for dist in find_distributions(os.path.join(path, egg_path)):433 dist.location = os.path.join(path, *lines)434 dist.precedence = SOURCE_DIST435 self.add(dist)436 437 def _scan(self, link):438 # Process a URL to see if it's for a package page439 NO_MATCH_SENTINEL = None, None440 if not link.startswith(self.index_url):441 return NO_MATCH_SENTINEL442 443 parts = list(map(urllib.parse.unquote, link[len(self.index_url) :].split('/')))444 if len(parts) != 2 or '#' in parts[1]:445 return NO_MATCH_SENTINEL446 447 # it's a package page, sanitize and index it448 pkg = safe_name(parts[0])449 ver = safe_version(parts[1])450 self.package_pages.setdefault(pkg.lower(), {})[link] = True451 return to_filename(pkg), to_filename(ver)452 453 def process_index(self, url, page):454 """Process the contents of a PyPI page"""455 456 # process an index page into the package-page index457 for match in HREF.finditer(page):458 try:459 self._scan(urllib.parse.urljoin(url, htmldecode(match.group(1))))460 except ValueError:461 pass462 463 pkg, ver = self._scan(url) # ensure this page is in the page index464 if not pkg:465 return "" # no sense double-scanning non-package pages466 467 # process individual package page468 for new_url in find_external_links(url, page):469 # Process the found URL470 base, frag = egg_info_for_url(new_url)471 if base.endswith('.py') and not frag:472 if ver:473 new_url += '#egg=%s-%s' % (pkg, ver)474 else:475 self.need_version_info(url)476 self.scan_url(new_url)477 478 return PYPI_MD5.sub(479 lambda m: '<a href="%s#md5=%s">%s</a>' % m.group(1, 3, 2), page480 )481 482 def need_version_info(self, url):483 self.scan_all(484 "Page at %s links to .py file(s) without version info; an index "485 "scan is required.",486 url,487 )488 489 def scan_all(self, msg=None, *args):490 if self.index_url not in self.fetched_urls:491 if msg:492 self.warn(msg, *args)493 self.info("Scanning index of all packages (this may take a while)")494 self.scan_url(self.index_url)495 496 def find_packages(self, requirement):497 self.scan_url(self.index_url + requirement.unsafe_name + '/')498 499 if not self.package_pages.get(requirement.key):500 # Fall back to safe version of the name501 self.scan_url(self.index_url + requirement.project_name + '/')502 503 if not self.package_pages.get(requirement.key):504 # We couldn't find the target package, so search the index page too505 self.not_found_in_index(requirement)506 507 for url in list(self.package_pages.get(requirement.key, ())):508 # scan each page that might be related to the desired package509 self.scan_url(url)510 511 def obtain(self, requirement, installer=None):512 self.prescan()513 self.find_packages(requirement)514 for dist in self[requirement.key]:515 if dist in requirement:516 return dist517 self.debug("%s does not match %s", requirement, dist)518 return super().obtain(requirement, installer)519 520 def check_hash(self, checker, filename, tfp):521 """522 checker is a ContentChecker523 """524 checker.report(self.debug, "Validating %%s checksum for %s" % filename)525 if not checker.is_valid():526 tfp.close()527 os.unlink(filename)528 raise DistutilsError(529 "%s validation failed for %s; "530 "possible download problem?"531 % (checker.hash.name, os.path.basename(filename))532 )533 534 def add_find_links(self, urls):535 """Add `urls` to the list that will be prescanned for searches"""536 for url in urls:537 if (538 self.to_scan is None # if we have already "gone online"539 or not URL_SCHEME(url) # or it's a local file/directory540 or url.startswith('file:')541 or list(distros_for_url(url)) # or a direct package link542 ):543 # then go ahead and process it now544 self.scan_url(url)545 else:546 # otherwise, defer retrieval till later547 self.to_scan.append(url)548 549 def prescan(self):550 """Scan urls scheduled for prescanning (e.g. --find-links)"""551 if self.to_scan:552 list(map(self.scan_url, self.to_scan))553 self.to_scan = None # from now on, go ahead and process immediately554 555 def not_found_in_index(self, requirement):556 if self[requirement.key]: # we've seen at least one distro557 meth, msg = self.info, "Couldn't retrieve index page for %r"558 else: # no distros seen for this name, might be misspelled559 meth, msg = (560 self.warn,561 "Couldn't find index page for %r (maybe misspelled?)",562 )563 meth(msg, requirement.unsafe_name)564 self.scan_all()565 566 def download(self, spec, tmpdir):567 """Locate and/or download `spec` to `tmpdir`, returning a local path568 569 `spec` may be a ``Requirement`` object, or a string containing a URL,570 an existing local filename, or a project/version requirement spec571 (i.e. the string form of a ``Requirement`` object). If it is the URL572 of a .py file with an unambiguous ``#egg=name-version`` tag (i.e., one573 that escapes ``-`` as ``_`` throughout), a trivial ``setup.py`` is574 automatically created alongside the downloaded file.575 576 If `spec` is a ``Requirement`` object or a string containing a577 project/version requirement spec, this method returns the location of578 a matching distribution (possibly after downloading it to `tmpdir`).579 If `spec` is a locally existing file or directory name, it is simply580 returned unchanged. If `spec` is a URL, it is downloaded to a subpath581 of `tmpdir`, and the local filename is returned. Various errors may be582 raised if a problem occurs during downloading.583 """584 if not isinstance(spec, Requirement):585 scheme = URL_SCHEME(spec)586 if scheme:587 # It's a url, download it to tmpdir588 found = self._download_url(scheme.group(1), spec, tmpdir)589 base, fragment = egg_info_for_url(spec)590 if base.endswith('.py'):591 found = self.gen_setup(found, fragment, tmpdir)592 return found593 elif os.path.exists(spec):594 # Existing file or directory, just return it595 return spec596 else:597 spec = parse_requirement_arg(spec)598 return getattr(self.fetch_distribution(spec, tmpdir), 'location', None)599 600 def fetch_distribution( # noqa: C901 # is too complex (14) # FIXME601 self,602 requirement,603 tmpdir,604 force_scan=False,605 source=False,606 develop_ok=False,607 local_index=None,608 ):609 """Obtain a distribution suitable for fulfilling `requirement`610 611 `requirement` must be a ``pkg_resources.Requirement`` instance.612 If necessary, or if the `force_scan` flag is set, the requirement is613 searched for in the (online) package index as well as the locally614 installed packages. If a distribution matching `requirement` is found,615 the returned distribution's ``location`` is the value you would have616 gotten from calling the ``download()`` method with the matching617 distribution's URL or filename. If no matching distribution is found,618 ``None`` is returned.619 620 If the `source` flag is set, only source distributions and source621 checkout links will be considered. Unless the `develop_ok` flag is622 set, development and system eggs (i.e., those using the ``.egg-info``623 format) will be ignored.624 """625 # process a Requirement626 self.info("Searching for %s", requirement)627 skipped = {}628 dist = None629 630 def find(req, env=None):631 if env is None:632 env = self633 # Find a matching distribution; may be called more than once634 635 for dist in env[req.key]:636 if dist.precedence == DEVELOP_DIST and not develop_ok:637 if dist not in skipped:638 self.warn(639 "Skipping development or system egg: %s",640 dist,641 )642 skipped[dist] = 1643 continue644 645 test = dist in req and (dist.precedence <= SOURCE_DIST or not source)646 if test:647 loc = self.download(dist.location, tmpdir)648 dist.download_location = loc649 if os.path.exists(dist.download_location):650 return dist651 652 return None653 654 if force_scan:655 self.prescan()656 self.find_packages(requirement)657 dist = find(requirement)658 659 if not dist and local_index is not None:660 dist = find(requirement, local_index)661 662 if dist is None:663 if self.to_scan is not None:664 self.prescan()665 dist = find(requirement)666 667 if dist is None and not force_scan:668 self.find_packages(requirement)669 dist = find(requirement)670 671 if dist is None:672 self.warn(673 "No local packages or working download links found for %s%s",674 (source and "a source distribution of " or ""),675 requirement,676 )677 return None678 else:679 self.info("Best match: %s", dist)680 return dist.clone(location=dist.download_location)681 682 def fetch(self, requirement, tmpdir, force_scan=False, source=False):683 """Obtain a file suitable for fulfilling `requirement`684 685 DEPRECATED; use the ``fetch_distribution()`` method now instead. For686 backward compatibility, this routine is identical but returns the687 ``location`` of the downloaded distribution instead of a distribution688 object.689 """690 dist = self.fetch_distribution(requirement, tmpdir, force_scan, source)691 if dist is not None:692 return dist.location693 return None694 695 def gen_setup(self, filename, fragment, tmpdir):696 match = EGG_FRAGMENT.match(fragment)697 dists = (698 match699 and [700 d701 for d in interpret_distro_name(filename, match.group(1), None)702 if d.version703 ]704 or []705 )706 707 if len(dists) == 1: # unambiguous ``#egg`` fragment708 basename = os.path.basename(filename)709 710 # Make sure the file has been downloaded to the temp dir.711 if os.path.dirname(filename) != tmpdir:712 dst = os.path.join(tmpdir, basename)713 if not (os.path.exists(dst) and os.path.samefile(filename, dst)):714 shutil.copy2(filename, dst)715 filename = dst716 717 with open(os.path.join(tmpdir, 'setup.py'), 'w') as file:718 file.write(719 "from setuptools import setup\n"720 "setup(name=%r, version=%r, py_modules=[%r])\n"721 % (722 dists[0].project_name,723 dists[0].version,724 os.path.splitext(basename)[0],725 )726 )727 return filename728 729 elif match:730 raise DistutilsError(731 "Can't unambiguously interpret project/version identifier %r; "732 "any dashes in the name or version should be escaped using "733 "underscores. %r" % (fragment, dists)734 )735 else:736 raise DistutilsError(737 "Can't process plain .py files without an '#egg=name-version'"738 " suffix to enable automatic setup script generation."739 )740 741 dl_blocksize = 8192742 743 def _download_to(self, url, filename):744 self.info("Downloading %s", url)745 # Download the file746 fp = None747 try:748 checker = HashChecker.from_url(url)749 fp = self.open_url(url)750 if isinstance(fp, urllib.error.HTTPError):751 raise DistutilsError(752 "Can't download %s: %s %s" % (url, fp.code, fp.msg)753 )754 headers = fp.info()755 blocknum = 0756 bs = self.dl_blocksize757 size = -1758 if "content-length" in headers:759 # Some servers return multiple Content-Length headers :(760 sizes = headers.get_all('Content-Length')761 size = max(map(int, sizes))762 self.reporthook(url, filename, blocknum, bs, size)763 with open(filename, 'wb') as tfp:764 while True:765 block = fp.read(bs)766 if block:767 checker.feed(block)768 tfp.write(block)769 blocknum += 1770 self.reporthook(url, filename, blocknum, bs, size)771 else:772 break773 self.check_hash(checker, filename, tfp)774 return headers775 finally:776 if fp:777 fp.close()778 779 def reporthook(self, url, filename, blocknum, blksize, size):780 pass # no-op781 782 # FIXME:783 def open_url(self, url, warning=None): # noqa: C901 # is too complex (12)784 if url.startswith('file:'):785 return local_open(url)786 try:787 return open_with_auth(url, self.opener)788 except (ValueError, http.client.InvalidURL) as v:789 msg = ' '.join([str(arg) for arg in v.args])790 if warning:791 self.warn(warning, msg)792 else:793 raise DistutilsError('%s %s' % (url, msg)) from v794 except urllib.error.HTTPError as v:795 return v796 except urllib.error.URLError as v:797 if warning:798 self.warn(warning, v.reason)799 else:800 raise DistutilsError(801 "Download error for %s: %s" % (url, v.reason)802 ) from v803 except http.client.BadStatusLine as v:804 if warning:805 self.warn(warning, v.line)806 else:807 raise DistutilsError(808 '%s returned a bad status line. The server might be '809 'down, %s' % (url, v.line)810 ) from v811 except (http.client.HTTPException, OSError) as v:812 if warning:813 self.warn(warning, v)814 else:815 raise DistutilsError("Download error for %s: %s" % (url, v)) from v816 817 def _download_url(self, scheme, url, tmpdir):818 # Determine download filename819 #820 name, fragment = egg_info_for_url(url)821 if name:822 while '..' in name:823 name = name.replace('..', '.').replace('\\', '_')824 else:825 name = "__downloaded__" # default if URL has no path contents826 827 if name.endswith('.egg.zip'):828 name = name[:-4] # strip the extra .zip before download829 830 filename = os.path.join(tmpdir, name)831 832 # Download the file833 #834 if scheme == 'svn' or scheme.startswith('svn+'):835 return self._download_svn(url, filename)836 elif scheme == 'git' or scheme.startswith('git+'):837 return self._download_git(url, filename)838 elif scheme.startswith('hg+'):839 return self._download_hg(url, filename)840 elif scheme == 'file':841 return urllib.request.url2pathname(urllib.parse.urlparse(url)[2])842 else:843 self.url_ok(url, True) # raises error if not allowed844 return self._attempt_download(url, filename)845 846 def scan_url(self, url):847 self.process_url(url, True)848 849 def _attempt_download(self, url, filename):850 headers = self._download_to(url, filename)851 if 'html' in headers.get('content-type', '').lower():852 return self._invalid_download_html(url, headers, filename)853 else:854 return filename855 856 def _invalid_download_html(self, url, headers, filename):857 os.unlink(filename)858 raise DistutilsError(f"Unexpected HTML page found at {url}")859 860 def _download_svn(self, url, _filename):861 raise DistutilsError(f"Invalid config, SVN download is not supported: {url}")862 863 @staticmethod864 def _vcs_split_rev_from_url(url, pop_prefix=False):865 scheme, netloc, path, query, frag = urllib.parse.urlsplit(url)866 867 scheme = scheme.split('+', 1)[-1]868 869 # Some fragment identification fails870 path = path.split('#', 1)[0]871 872 rev = None873 if '@' in path:874 path, rev = path.rsplit('@', 1)875 876 # Also, discard fragment877 url = urllib.parse.urlunsplit((scheme, netloc, path, query, ''))878 879 return url, rev880 881 def _download_git(self, url, filename):882 filename = filename.split('#', 1)[0]883 url, rev = self._vcs_split_rev_from_url(url, pop_prefix=True)884 885 self.info("Doing git clone from %s to %s", url, filename)886 os.system("git clone --quiet %s %s" % (url, filename))887 888 if rev is not None:889 self.info("Checking out %s", rev)890 os.system(891 "git -C %s checkout --quiet %s"892 % (893 filename,894 rev,895 )896 )897 898 return filename899 900 def _download_hg(self, url, filename):901 filename = filename.split('#', 1)[0]902 url, rev = self._vcs_split_rev_from_url(url, pop_prefix=True)903 904 self.info("Doing hg clone from %s to %s", url, filename)905 os.system("hg clone --quiet %s %s" % (url, filename))906 907 if rev is not None:908 self.info("Updating to %s", rev)909 os.system(910 "hg --cwd %s up -C -r %s -q"911 % (912 filename,913 rev,914 )915 )916 917 return filename918 919 def debug(self, msg, *args):920 log.debug(msg, *args)921 922 def info(self, msg, *args):923 log.info(msg, *args)924 925 def warn(self, msg, *args):926 log.warn(msg, *args)927 928 929# This pattern matches a character entity reference (a decimal numeric930# references, a hexadecimal numeric reference, or a named reference).931entity_sub = re.compile(r'&(#(\d+|x[\da-fA-F]+)|[\w.:-]+);?').sub932 933 934def decode_entity(match):935 what = match.group(0)936 return html.unescape(what)937 938 939def htmldecode(text):940 """941 Decode HTML entities in the given text.942 943 >>> htmldecode(944 ... 'https://../package_name-0.1.2.tar.gz'945 ... '?tokena=A&tokenb=B">package_name-0.1.2.tar.gz')946 'https://../package_name-0.1.2.tar.gz?tokena=A&tokenb=B">package_name-0.1.2.tar.gz'947 """948 return entity_sub(decode_entity, text)949 950 951def socket_timeout(timeout=15):952 def _socket_timeout(func):953 def _socket_timeout(*args, **kwargs):954 old_timeout = socket.getdefaulttimeout()955 socket.setdefaulttimeout(timeout)956 try:957 return func(*args, **kwargs)958 finally:959 socket.setdefaulttimeout(old_timeout)960 961 return _socket_timeout962 963 return _socket_timeout964 965 966def _encode_auth(auth):967 """968 Encode auth from a URL suitable for an HTTP header.969 >>> str(_encode_auth('username%3Apassword'))970 'dXNlcm5hbWU6cGFzc3dvcmQ='971 972 Long auth strings should not cause a newline to be inserted.973 >>> long_auth = 'username:' + 'password'*10974 >>> chr(10) in str(_encode_auth(long_auth))975 False976 """977 auth_s = urllib.parse.unquote(auth)978 # convert to bytes979 auth_bytes = auth_s.encode()980 encoded_bytes = base64.b64encode(auth_bytes)981 # convert back to a string982 encoded = encoded_bytes.decode()983 # strip the trailing carriage return984 return encoded.replace('\n', '')985 986 987class Credential:988 """989 A username/password pair. Use like a namedtuple.990 """991 992 def __init__(self, username, password):993 self.username = username994 self.password = password995 996 def __iter__(self):997 yield self.username998 yield self.password999 1000 def __str__(self):1001 return '%(username)s:%(password)s' % vars(self)1002 1003 1004class PyPIConfig(configparser.RawConfigParser):1005 def __init__(self):1006 """1007 Load from ~/.pypirc1008 """1009 defaults = dict.fromkeys(['username', 'password', 'repository'], '')1010 super().__init__(defaults)1011 1012 rc = os.path.join(os.path.expanduser('~'), '.pypirc')1013 if os.path.exists(rc):1014 self.read(rc)1015 1016 @property1017 def creds_by_repository(self):1018 sections_with_repositories = [1019 section1020 for section in self.sections()1021 if self.get(section, 'repository').strip()1022 ]1023 1024 return dict(map(self._get_repo_cred, sections_with_repositories))1025 1026 def _get_repo_cred(self, section):1027 repo = self.get(section, 'repository').strip()1028 return repo, Credential(1029 self.get(section, 'username').strip(),1030 self.get(section, 'password').strip(),1031 )1032 1033 def find_credential(self, url):1034 """1035 If the URL indicated appears to be a repository defined in this1036 config, return the credential for that repository.1037 """1038 for repository, cred in self.creds_by_repository.items():1039 if url.startswith(repository):1040 return cred1041 return None1042 1043 1044def open_with_auth(url, opener=urllib.request.urlopen):1045 """Open a urllib2 request, handling HTTP authentication"""1046 1047 parsed = urllib.parse.urlparse(url)1048 scheme, netloc, path, params, query, frag = parsed1049 1050 # Double scheme does not raise on macOS as revealed by a1051 # failing test. We would expect "nonnumeric port". Refs #20.1052 if netloc.endswith(':'):1053 raise http.client.InvalidURL("nonnumeric port: ''")1054 1055 if scheme in ('http', 'https'):1056 auth, address = _splituser(netloc)1057 else:1058 auth = None1059 1060 if not auth:1061 cred = PyPIConfig().find_credential(url)1062 if cred:1063 auth = str(cred)1064 info = cred.username, url1065 log.info('Authenticating as %s for %s (from .pypirc)', *info)1066 1067 if auth:1068 auth = "Basic " + _encode_auth(auth)1069 parts = scheme, address, path, params, query, frag1070 new_url = urllib.parse.urlunparse(parts)1071 request = urllib.request.Request(new_url)1072 request.add_header("Authorization", auth)1073 else:1074 request = urllib.request.Request(url)1075 1076 request.add_header('User-Agent', user_agent)1077 fp = opener(request)1078 1079 if auth:1080 # Put authentication info back into request URL if same host,1081 # so that links found on the page will work1082 s2, h2, path2, param2, query2, frag2 = urllib.parse.urlparse(fp.url)1083 if s2 == scheme and h2 == address:1084 parts = s2, netloc, path2, param2, query2, frag21085 fp.url = urllib.parse.urlunparse(parts)1086 1087 return fp1088 1089 1090# copy of urllib.parse._splituser from Python 3.81091def _splituser(host):1092 """splituser('user[:passwd]@host[:port]')1093 --> 'user[:passwd]', 'host[:port]'."""1094 user, delim, host = host.rpartition('@')1095 return (user if delim else None), host1096 1097 1098# adding a timeout to avoid freezing package_index1099open_with_auth = socket_timeout(_SOCKET_TIMEOUT)(open_with_auth)1100 1101 1102def fix_sf_url(url):1103 return url # backward compatibility1104 1105 1106def local_open(url):1107 """Read a local path, with special support for directories"""1108 scheme, server, path, param, query, frag = urllib.parse.urlparse(url)1109 filename = urllib.request.url2pathname(path)1110 if os.path.isfile(filename):1111 return urllib.request.urlopen(url)1112 elif path.endswith('/') and os.path.isdir(filename):1113 files = []1114 for f in os.listdir(filename):1115 filepath = os.path.join(filename, f)1116 if f == 'index.html':1117 with open(filepath, 'r') as fp:1118 body = fp.read()1119 break1120 elif os.path.isdir(filepath):1121 f += '/'1122 files.append('<a href="{name}">{name}</a>'.format(name=f))1123 else:1124 tmpl = (1125 "<html><head><title>{url}</title>" "</head><body>{files}</body></html>"1126 )1127 body = tmpl.format(url=url, files='\n'.join(files))1128 status, message = 200, "OK"1129 else:1130 status, message, body = 404, "Path not found", "Not found"1131 1132 headers = {'content-type': 'text/html'}1133 body_stream = io.StringIO(body)1134 return urllib.error.HTTPError(url, status, message, headers, body_stream)1135 