codekingpro/portable-devtools
114k
1from __future__ import annotations2 3import io4import json5import logging6import os7import threading8import warnings9import weakref10from errno import ESPIPE11from glob import has_magic12from hashlib import sha25613from typing import Any, ClassVar14 15from .callbacks import DEFAULT_CALLBACK16from .config import apply_config, conf17from .dircache import DirCache18from .transaction import Transaction19from .utils import (20 _unstrip_protocol,21 glob_translate,22 isfilelike,23 other_paths,24 read_block,25 stringify_path,26 tokenize,27)28 29logger = logging.getLogger("fsspec")30 31 32def make_instance(cls, args, kwargs):33 return cls(*args, **kwargs)34 35 36class _Cached(type):37 """38 Metaclass for caching file system instances.39 40 Notes41 -----42 Instances are cached according to43 44 * The values of the class attributes listed in `_extra_tokenize_attributes`45 * The arguments passed to ``__init__``.46 47 This creates an additional reference to the filesystem, which prevents the48 filesystem from being garbage collected when all *user* references go away.49 A call to the :meth:`AbstractFileSystem.clear_instance_cache` must *also*50 be made for a filesystem instance to be garbage collected.51 """52 53 def __init__(cls, *args, **kwargs):54 super().__init__(*args, **kwargs)55 # Note: we intentionally create a reference here, to avoid garbage56 # collecting instances when all other references are gone. To really57 # delete a FileSystem, the cache must be cleared.58 if conf.get("weakref_instance_cache"): # pragma: no cover59 # debug option for analysing fork/spawn conditions60 cls._cache = weakref.WeakValueDictionary()61 else:62 cls._cache = {}63 cls._pid = os.getpid()64 65 def __call__(cls, *args, **kwargs):66 kwargs = apply_config(cls, kwargs)67 extra_tokens = tuple(68 getattr(cls, attr, None) for attr in cls._extra_tokenize_attributes69 )70 strip_tokenize_options = {71 k: kwargs.pop(k) for k in cls._strip_tokenize_options if k in kwargs72 }73 token = tokenize(74 cls, cls._pid, threading.get_ident(), *args, *extra_tokens, **kwargs75 )76 skip = kwargs.pop("skip_instance_cache", False)77 if os.getpid() != cls._pid:78 cls._cache.clear()79 cls._pid = os.getpid()80 if not skip and cls.cachable and token in cls._cache:81 cls._latest = token82 return cls._cache[token]83 else:84 obj = super().__call__(*args, **kwargs, **strip_tokenize_options)85 # Setting _fs_token here causes some static linters to complain.86 obj._fs_token_ = token87 obj.storage_args = args88 obj.storage_options = kwargs89 if obj.async_impl and obj.mirror_sync_methods:90 from .asyn import mirror_sync_methods91 92 mirror_sync_methods(obj)93 94 if cls.cachable and not skip:95 cls._latest = token96 cls._cache[token] = obj97 return obj98 99 100class AbstractFileSystem(metaclass=_Cached):101 """102 An abstract super-class for pythonic file-systems103 104 Implementations are expected to be compatible with or, better, subclass105 from here.106 """107 108 cachable = True # this class can be cached, instances reused109 _cached = False110 blocksize = 2**22111 sep = "/"112 protocol: ClassVar[str | tuple[str, ...]] = "abstract"113 _latest = None114 async_impl = False115 mirror_sync_methods = False116 root_marker = "" # For some FSs, may require leading '/' or other character117 transaction_type = Transaction118 119 #: Extra *class attributes* that should be considered when hashing.120 _extra_tokenize_attributes = ()121 #: *storage options* that should not be considered when hashing.122 _strip_tokenize_options = ()123 124 # Set by _Cached metaclass125 storage_args: tuple[Any, ...]126 storage_options: dict[str, Any]127 128 def __init__(self, *args, **storage_options):129 """Create and configure file-system instance130 131 Instances may be cachable, so if similar enough arguments are seen132 a new instance is not required. The token attribute exists to allow133 implementations to cache instances if they wish.134 135 A reasonable default should be provided if there are no arguments.136 137 Subclasses should call this method.138 139 Parameters140 ----------141 use_listings_cache, listings_expiry_time, max_paths:142 passed to ``DirCache``, if the implementation supports143 directory listing caching. Pass use_listings_cache=False144 to disable such caching.145 skip_instance_cache: bool146 If this is a cachable implementation, pass True here to force147 creating a new instance even if a matching instance exists, and prevent148 storing this instance.149 asynchronous: bool150 loop: asyncio-compatible IOLoop or None151 """152 if self._cached:153 # reusing instance, don't change154 return155 self._cached = True156 self._intrans = False157 self._transaction = None158 self._invalidated_caches_in_transaction = []159 self.dircache = DirCache(**storage_options)160 161 if storage_options.pop("add_docs", None):162 warnings.warn("add_docs is no longer supported.", FutureWarning)163 164 if storage_options.pop("add_aliases", None):165 warnings.warn("add_aliases has been removed.", FutureWarning)166 # This is set in _Cached167 self._fs_token_ = None168 169 @property170 def fsid(self):171 """Persistent filesystem id that can be used to compare filesystems172 across sessions.173 """174 raise NotImplementedError175 176 @property177 def _fs_token(self):178 return self._fs_token_179 180 def __dask_tokenize__(self):181 return self._fs_token182 183 def __hash__(self):184 return int(self._fs_token, 16)185 186 def __eq__(self, other):187 return isinstance(other, type(self)) and self._fs_token == other._fs_token188 189 def __reduce__(self):190 return make_instance, (type(self), self.storage_args, self.storage_options)191 192 @classmethod193 def _strip_protocol(cls, path):194 """Turn path from fully-qualified to file-system-specific195 196 May require FS-specific handling, e.g., for relative paths or links.197 """198 if isinstance(path, list):199 return [cls._strip_protocol(p) for p in path]200 path = stringify_path(path)201 protos = (cls.protocol,) if isinstance(cls.protocol, str) else cls.protocol202 for protocol in protos:203 if path.startswith(protocol + "://"):204 path = path[len(protocol) + 3 :]205 elif path.startswith(protocol + "::"):206 path = path[len(protocol) + 2 :]207 path = path.rstrip("/")208 # use of root_marker to make minimum required path, e.g., "/"209 return path or cls.root_marker210 211 def unstrip_protocol(self, name: str) -> str:212 """Format FS-specific path to generic, including protocol"""213 protos = (self.protocol,) if isinstance(self.protocol, str) else self.protocol214 for protocol in protos:215 if name.startswith(f"{protocol}://"):216 return name217 return f"{protos[0]}://{name}"218 219 @staticmethod220 def _get_kwargs_from_urls(path):221 """If kwargs can be encoded in the paths, extract them here222 223 This should happen before instantiation of the class; incoming paths224 then should be amended to strip the options in methods.225 226 Examples may look like an sftp path "sftp://user@host:/my/path", where227 the user and host should become kwargs and later get stripped.228 """229 # by default, nothing happens230 return {}231 232 @classmethod233 def current(cls):234 """Return the most recently instantiated FileSystem235 236 If no instance has been created, then create one with defaults237 """238 if cls._latest in cls._cache:239 return cls._cache[cls._latest]240 return cls()241 242 @property243 def transaction(self):244 """A context within which files are committed together upon exit245 246 Requires the file class to implement `.commit()` and `.discard()`247 for the normal and exception cases.248 """249 if self._transaction is None:250 self._transaction = self.transaction_type(self)251 return self._transaction252 253 def start_transaction(self):254 """Begin write transaction for deferring files, non-context version"""255 self._intrans = True256 self._transaction = self.transaction_type(self)257 return self.transaction258 259 def end_transaction(self):260 """Finish write transaction, non-context version"""261 self.transaction.complete()262 self._transaction = None263 # The invalid cache must be cleared after the transaction is completed.264 for path in self._invalidated_caches_in_transaction:265 self.invalidate_cache(path)266 self._invalidated_caches_in_transaction.clear()267 268 def invalidate_cache(self, path=None):269 """270 Discard any cached directory information271 272 Parameters273 ----------274 path: string or None275 If None, clear all listings cached else listings at or under given276 path.277 """278 # Not necessary to implement invalidation mechanism, may have no cache.279 # But if have, you should call this method of parent class from your280 # subclass to ensure expiring caches after transacations correctly.281 # See the implementation of FTPFileSystem in ftp.py282 if self._intrans:283 self._invalidated_caches_in_transaction.append(path)284 285 def mkdir(self, path, create_parents=True, **kwargs):286 """287 Create directory entry at path288 289 For systems that don't have true directories, may create an for290 this instance only and not touch the real filesystem291 292 Parameters293 ----------294 path: str295 location296 create_parents: bool297 if True, this is equivalent to ``makedirs``298 kwargs:299 may be permissions, etc.300 """301 pass # not necessary to implement, may not have directories302 303 def makedirs(self, path, exist_ok=False):304 """Recursively make directories305 306 Creates directory at path and any intervening required directories.307 Raises exception if, for instance, the path already exists but is a308 file.309 310 Parameters311 ----------312 path: str313 leaf directory name314 exist_ok: bool (False)315 If False, will error if the target already exists316 """317 pass # not necessary to implement, may not have directories318 319 def rmdir(self, path):320 """Remove a directory, if empty"""321 pass # not necessary to implement, may not have directories322 323 def ls(self, path, detail=True, **kwargs):324 """List objects at path.325 326 This should include subdirectories and files at that location. The327 difference between a file and a directory must be clear when details328 are requested.329 330 The specific keys, or perhaps a FileInfo class, or similar, is TBD,331 but must be consistent across implementations.332 Must include:333 334 - full path to the entry (without protocol)335 - size of the entry, in bytes. If the value cannot be determined, will336 be ``None``.337 - type of entry, "file", "directory" or other338 339 Additional information340 may be present, appropriate to the file-system, e.g., generation,341 checksum, etc.342 343 May use refresh=True|False to allow use of self._ls_from_cache to344 check for a saved listing and avoid calling the backend. This would be345 common where listing may be expensive.346 347 Parameters348 ----------349 path: str350 detail: bool351 if True, gives a list of dictionaries, where each is the same as352 the result of ``info(path)``. If False, gives a list of paths353 (str).354 kwargs: may have additional backend-specific options, such as version355 information356 357 Returns358 -------359 List of strings if detail is False, or list of directory information360 dicts if detail is True.361 """362 raise NotImplementedError363 364 def _ls_from_cache(self, path):365 """Check cache for listing366 367 Returns listing, if found (may be empty list for a directly that exists368 but contains nothing), None if not in cache.369 """370 parent = self._parent(path)371 try:372 return self.dircache[path.rstrip("/")]373 except KeyError:374 pass375 try:376 files = [377 f378 for f in self.dircache[parent]379 if f["name"] == path380 or (f["name"] == path.rstrip("/") and f["type"] == "directory")381 ]382 if len(files) == 0:383 # parent dir was listed but did not contain this file384 raise FileNotFoundError(path)385 return files386 except KeyError:387 pass388 389 def walk(self, path, maxdepth=None, topdown=True, on_error="omit", **kwargs):390 """Return all files under the given path.391 392 List all files, recursing into subdirectories; output is iterator-style,393 like ``os.walk()``. For a simple list of files, ``find()`` is available.394 395 When topdown is True, the caller can modify the dirnames list in-place (perhaps396 using del or slice assignment), and walk() will397 only recurse into the subdirectories whose names remain in dirnames;398 this can be used to prune the search, impose a specific order of visiting,399 or even to inform walk() about directories the caller creates or renames before400 it resumes walk() again.401 Modifying dirnames when topdown is False has no effect. (see os.walk)402 403 Note that the "files" outputted will include anything that is not404 a directory, such as links.405 406 Parameters407 ----------408 path: str409 Root to recurse into410 maxdepth: int411 Maximum recursion depth. None means limitless, but not recommended412 on link-based file-systems.413 topdown: bool (True)414 Whether to walk the directory tree from the top downwards or from415 the bottom upwards.416 on_error: "omit", "raise", a callable417 if omit (default), path with exception will simply be empty;418 If raise, an underlying exception will be raised;419 if callable, it will be called with a single OSError instance as argument420 kwargs: passed to ``ls``421 """422 if maxdepth is not None and maxdepth < 1:423 raise ValueError("maxdepth must be at least 1")424 425 path = self._strip_protocol(path)426 full_dirs = {}427 dirs = {}428 files = {}429 430 detail = kwargs.pop("detail", False)431 try:432 listing = self.ls(path, detail=True, **kwargs)433 except (FileNotFoundError, OSError) as e:434 if on_error == "raise":435 raise436 if callable(on_error):437 on_error(e)438 return439 440 for info in listing:441 # each info name must be at least [path]/part , but here442 # we check also for names like [path]/part/443 pathname = info["name"].rstrip("/")444 name = pathname.rsplit("/", 1)[-1]445 if info["type"] == "directory" and pathname != path:446 # do not include "self" path447 full_dirs[name] = pathname448 dirs[name] = info449 elif pathname == path:450 # file-like with same name as give path451 files[""] = info452 else:453 files[name] = info454 455 if not detail:456 dirs = list(dirs)457 files = list(files)458 459 if topdown:460 # Yield before recursion if walking top down461 yield path, dirs, files462 463 if maxdepth is not None:464 maxdepth -= 1465 if maxdepth < 1:466 if not topdown:467 yield path, dirs, files468 return469 470 for d in dirs:471 yield from self.walk(472 full_dirs[d],473 maxdepth=maxdepth,474 detail=detail,475 topdown=topdown,476 **kwargs,477 )478 479 if not topdown:480 # Yield after recursion if walking bottom up481 yield path, dirs, files482 483 def find(self, path, maxdepth=None, withdirs=False, detail=False, **kwargs):484 """List all files below path.485 486 Like posix ``find`` command without conditions487 488 Parameters489 ----------490 path : str491 maxdepth: int or None492 If not None, the maximum number of levels to descend493 withdirs: bool494 Whether to include directory paths in the output. This is True495 when used by glob, but users usually only want files.496 kwargs are passed to ``ls``.497 """498 # TODO: allow equivalent of -name parameter499 path = self._strip_protocol(path)500 out = {}501 502 # Add the root directory if withdirs is requested503 # This is needed for posix glob compliance504 if withdirs and path != "" and self.isdir(path):505 out[path] = self.info(path)506 507 for _, dirs, files in self.walk(path, maxdepth, detail=True, **kwargs):508 if withdirs:509 files.update(dirs)510 out.update({info["name"]: info for name, info in files.items()})511 if not out and self.isfile(path):512 # walk works on directories, but find should also return [path]513 # when path happens to be a file514 out[path] = {}515 names = sorted(out)516 if not detail:517 return names518 else:519 return {name: out[name] for name in names}520 521 def du(self, path, total=True, maxdepth=None, withdirs=False, **kwargs):522 """Space used by files and optionally directories within a path523 524 Directory size does not include the size of its contents.525 526 Parameters527 ----------528 path: str529 total: bool530 Whether to sum all the file sizes531 maxdepth: int or None532 Maximum number of directory levels to descend, None for unlimited.533 withdirs: bool534 Whether to include directory paths in the output.535 kwargs: passed to ``find``536 537 Returns538 -------539 Dict of {path: size} if total=False, or int otherwise, where numbers540 refer to bytes used.541 """542 sizes = {}543 if withdirs and self.isdir(path):544 # Include top-level directory in output545 info = self.info(path)546 sizes[info["name"]] = info["size"]547 for f in self.find(path, maxdepth=maxdepth, withdirs=withdirs, **kwargs):548 info = self.info(f)549 sizes[info["name"]] = info["size"]550 if total:551 return sum(sizes.values())552 else:553 return sizes554 555 def glob(self, path, maxdepth=None, **kwargs):556 """Find files by glob-matching.557 558 Pattern matching capabilities for finding files that match the given pattern.559 560 Parameters561 ----------562 path: str563 The glob pattern to match against564 maxdepth: int or None565 Maximum depth for ``'**'`` patterns. Applied on the first ``'**'`` found.566 Must be at least 1 if provided.567 kwargs:568 Additional arguments passed to ``find`` (e.g., detail=True)569 570 Returns571 -------572 List of matched paths, or dict of paths and their info if detail=True573 574 Notes575 -----576 Supported patterns:577 - '*': Matches any sequence of characters within a single directory level578 - ``'**'``: Matches any number of directory levels (must be an entire path component)579 - '?': Matches exactly one character580 - '[abc]': Matches any character in the set581 - '[a-z]': Matches any character in the range582 - '[!abc]': Matches any character NOT in the set583 584 Special behaviors:585 - If the path ends with '/', only folders are returned586 - Consecutive '*' characters are compressed into a single '*'587 - Empty brackets '[]' never match anything588 - Negated empty brackets '[!]' match any single character589 - Special characters in character classes are escaped properly590 591 Limitations:592 - ``'**'`` must be a complete path component (e.g., ``'a/**/b'``, not ``'a**b'``)593 - No brace expansion ('{a,b}.txt')594 - No extended glob patterns ('+(pattern)', '!(pattern)')595 """596 if maxdepth is not None and maxdepth < 1:597 raise ValueError("maxdepth must be at least 1")598 599 import re600 601 seps = (os.path.sep, os.path.altsep) if os.path.altsep else (os.path.sep,)602 ends_with_sep = path.endswith(seps) # _strip_protocol strips trailing slash603 path = self._strip_protocol(path)604 append_slash_to_dirname = ends_with_sep or path.endswith(605 tuple(sep + "**" for sep in seps)606 )607 idx_star = path.find("*") if path.find("*") >= 0 else len(path)608 idx_qmark = path.find("?") if path.find("?") >= 0 else len(path)609 idx_brace = path.find("[") if path.find("[") >= 0 else len(path)610 611 min_idx = min(idx_star, idx_qmark, idx_brace)612 613 detail = kwargs.pop("detail", False)614 withdirs = kwargs.pop("withdirs", True)615 616 if not has_magic(path):617 if self.exists(path, **kwargs):618 if not detail:619 return [path]620 else:621 return {path: self.info(path, **kwargs)}622 else:623 if not detail:624 return [] # glob of non-existent returns empty625 else:626 return {}627 elif "/" in path[:min_idx]:628 min_idx = path[:min_idx].rindex("/")629 root = path[: min_idx + 1]630 depth = path[min_idx + 1 :].count("/") + 1631 else:632 root = ""633 depth = path[min_idx + 1 :].count("/") + 1634 635 if "**" in path:636 if maxdepth is not None:637 idx_double_stars = path.find("**")638 depth_double_stars = path[idx_double_stars:].count("/") + 1639 depth = depth - depth_double_stars + maxdepth640 else:641 depth = None642 643 allpaths = self.find(644 root, maxdepth=depth, withdirs=withdirs, detail=True, **kwargs645 )646 647 pattern = glob_translate(path + ("/" if ends_with_sep else ""))648 pattern = re.compile(pattern)649 650 out = {651 p: info652 for p, info in sorted(allpaths.items())653 if pattern.match(654 p + "/"655 if append_slash_to_dirname and info["type"] == "directory"656 else p657 )658 }659 660 if detail:661 return out662 else:663 return list(out)664 665 def exists(self, path, **kwargs):666 """Is there a file at the given path"""667 try:668 self.info(path, **kwargs)669 return True670 except: # noqa: E722671 # any exception allowed bar FileNotFoundError?672 return False673 674 def lexists(self, path, **kwargs):675 """If there is a file at the given path (including676 broken links)"""677 return self.exists(path)678 679 def info(self, path, **kwargs):680 """Give details of entry at path681 682 Returns a single dictionary, with exactly the same information as ``ls``683 would with ``detail=True``.684 685 The default implementation calls ls and could be overridden by a686 shortcut. kwargs are passed on to ```ls()``.687 688 Some file systems might not be able to measure the file's size, in689 which case, the returned dict will include ``'size': None``.690 691 Returns692 -------693 dict with keys: name (full path in the FS), size (in bytes), type (file,694 directory, or something else) and other FS-specific keys.695 """696 path = self._strip_protocol(path)697 out = self.ls(self._parent(path), detail=True, **kwargs)698 out = [o for o in out if o["name"].rstrip("/") == path]699 if out:700 return out[0]701 out = self.ls(path, detail=True, **kwargs)702 path = path.rstrip("/")703 out1 = [o for o in out if o["name"].rstrip("/") == path]704 if len(out1) == 1:705 if "size" not in out1[0]:706 out1[0]["size"] = None707 return out1[0]708 elif len(out1) > 1 or out:709 return {"name": path, "size": 0, "type": "directory"}710 else:711 raise FileNotFoundError(path)712 713 def checksum(self, path):714 """Unique value for current version of file715 716 If the checksum is the same from one moment to another, the contents717 are guaranteed to be the same. If the checksum changes, the contents718 *might* have changed.719 720 This should normally be overridden; default will probably capture721 creation/modification timestamp (which would be good) or maybe722 access timestamp (which would be bad)723 """724 return int(tokenize(self.info(path)), 16)725 726 def size(self, path):727 """Size in bytes of file"""728 return self.info(path).get("size", None)729 730 def sizes(self, paths):731 """Size in bytes of each file in a list of paths"""732 return [self.size(p) for p in paths]733 734 def isdir(self, path):735 """Is this entry directory-like?"""736 try:737 return self.info(path)["type"] == "directory"738 except OSError:739 return False740 741 def isfile(self, path):742 """Is this entry file-like?"""743 try:744 return self.info(path)["type"] == "file"745 except: # noqa: E722746 return False747 748 def read_text(self, path, encoding=None, errors=None, newline=None, **kwargs):749 """Get the contents of the file as a string.750 751 Parameters752 ----------753 path: str754 URL of file on this filesystems755 encoding, errors, newline: same as `open`.756 """757 with self.open(758 path,759 mode="r",760 encoding=encoding,761 errors=errors,762 newline=newline,763 **kwargs,764 ) as f:765 return f.read()766 767 def write_text(768 self, path, value, encoding=None, errors=None, newline=None, **kwargs769 ):770 """Write the text to the given file.771 772 An existing file will be overwritten.773 774 Parameters775 ----------776 path: str777 URL of file on this filesystems778 value: str779 Text to write.780 encoding, errors, newline: same as `open`.781 """782 with self.open(783 path,784 mode="w",785 encoding=encoding,786 errors=errors,787 newline=newline,788 **kwargs,789 ) as f:790 return f.write(value)791 792 def cat_file(self, path, start=None, end=None, **kwargs):793 """Get the content of a file794 795 Parameters796 ----------797 path: URL of file on this filesystems798 start, end: int799 Bytes limits of the read. If negative, backwards from end,800 like usual python slices. Either can be None for start or801 end of file, respectively802 kwargs: passed to ``open()``.803 """804 # explicitly set buffering off?805 with self.open(path, "rb", **kwargs) as f:806 if start is not None:807 if start >= 0:808 f.seek(start)809 else:810 f.seek(max(0, f.size + start))811 if end is not None:812 if end < 0:813 end = f.size + end814 return f.read(end - f.tell())815 return f.read()816 817 def pipe_file(self, path, value, mode="overwrite", **kwargs):818 """Set the bytes of given file"""819 if mode == "create" and self.exists(path):820 # non-atomic but simple way; or could use "xb" in open(), which is likely821 # not as well supported822 raise FileExistsError823 with self.open(path, "wb", **kwargs) as f:824 f.write(value)825 826 def pipe(self, path, value=None, **kwargs):827 """Put value into path828 829 (counterpart to ``cat``)830 831 Parameters832 ----------833 path: string or dict(str, bytes)834 If a string, a single remote location to put ``value`` bytes; if a dict,835 a mapping of {path: bytesvalue}.836 value: bytes, optional837 If using a single path, these are the bytes to put there. Ignored if838 ``path`` is a dict839 """840 if isinstance(path, str):841 self.pipe_file(self._strip_protocol(path), value, **kwargs)842 elif isinstance(path, dict):843 for k, v in path.items():844 self.pipe_file(self._strip_protocol(k), v, **kwargs)845 else:846 raise ValueError("path must be str or dict")847 848 def cat_ranges(849 self, paths, starts, ends, max_gap=None, on_error="return", **kwargs850 ):851 """Get the contents of byte ranges from one or more files852 853 Parameters854 ----------855 paths: list856 A list of of filepaths on this filesystems857 starts, ends: int or list858 Bytes limits of the read. If using a single int, the same value will be859 used to read all the specified files.860 """861 if max_gap is not None:862 raise NotImplementedError863 if not isinstance(paths, list):864 raise TypeError865 if not isinstance(starts, list):866 starts = [starts] * len(paths)867 if not isinstance(ends, list):868 ends = [ends] * len(paths)869 if len(starts) != len(paths) or len(ends) != len(paths):870 raise ValueError871 out = []872 for p, s, e in zip(paths, starts, ends):873 try:874 out.append(self.cat_file(p, s, e))875 except Exception as e:876 if on_error == "return":877 out.append(e)878 else:879 raise880 return out881 882 def cat(self, path, recursive=False, on_error="raise", **kwargs):883 """Fetch (potentially multiple) paths' contents884 885 Parameters886 ----------887 recursive: bool888 If True, assume the path(s) are directories, and get all the889 contained files890 on_error : "raise", "omit", "return"891 If raise, an underlying exception will be raised (converted to KeyError892 if the type is in self.missing_exceptions); if omit, keys with exception893 will simply not be included in the output; if "return", all keys are894 included in the output, but the value will be bytes or an exception895 instance.896 kwargs: passed to cat_file897 898 Returns899 -------900 dict of {path: contents} if there are multiple paths901 or the path has been otherwise expanded902 """903 paths = self.expand_path(path, recursive=recursive, **kwargs)904 if (905 len(paths) > 1906 or isinstance(path, list)907 or paths[0] != self._strip_protocol(path)908 ):909 out = {}910 for path in paths:911 try:912 out[path] = self.cat_file(path, **kwargs)913 except Exception as e:914 if on_error == "raise":915 raise916 if on_error == "return":917 out[path] = e918 return out919 else:920 return self.cat_file(paths[0], **kwargs)921 922 def get_file(self, rpath, lpath, callback=DEFAULT_CALLBACK, outfile=None, **kwargs):923 """Copy single remote file to local"""924 from .implementations.local import LocalFileSystem925 926 if isfilelike(lpath):927 outfile = lpath928 elif self.isdir(rpath):929 os.makedirs(lpath, exist_ok=True)930 return None931 932 fs = LocalFileSystem(auto_mkdir=True)933 fs.makedirs(fs._parent(lpath), exist_ok=True)934 935 with self.open(rpath, "rb", **kwargs) as f1:936 if outfile is None:937 outfile = open(lpath, "wb")938 939 try:940 callback.set_size(getattr(f1, "size", None))941 data = True942 while data:943 data = f1.read(self.blocksize)944 segment_len = outfile.write(data)945 if segment_len is None:946 segment_len = len(data)947 callback.relative_update(segment_len)948 finally:949 if not isfilelike(lpath):950 outfile.close()951 952 def get(953 self,954 rpath,955 lpath,956 recursive=False,957 callback=DEFAULT_CALLBACK,958 maxdepth=None,959 **kwargs,960 ):961 """Copy file(s) to local.962 963 Copies a specific file or tree of files (if recursive=True). If lpath964 ends with a "/", it will be assumed to be a directory, and target files965 will go within. Can submit a list of paths, which may be glob-patterns966 and will be expanded.967 968 Calls get_file for each source.969 """970 if isinstance(lpath, list) and isinstance(rpath, list):971 # No need to expand paths when both source and destination972 # are provided as lists973 rpaths = rpath974 lpaths = lpath975 else:976 from .implementations.local import (977 LocalFileSystem,978 make_path_posix,979 trailing_sep,980 )981 982 source_is_str = isinstance(rpath, str)983 rpaths = self.expand_path(984 rpath, recursive=recursive, maxdepth=maxdepth, **kwargs985 )986 if source_is_str and (not recursive or maxdepth is not None):987 # Non-recursive glob does not copy directories988 rpaths = [p for p in rpaths if not (trailing_sep(p) or self.isdir(p))]989 if not rpaths:990 return991 992 if isinstance(lpath, str):993 lpath = make_path_posix(lpath)994 995 source_is_file = len(rpaths) == 1996 dest_is_dir = isinstance(lpath, str) and (997 trailing_sep(lpath) or LocalFileSystem().isdir(lpath)998 )999 1000 exists = source_is_str and (1001 (has_magic(rpath) and source_is_file)1002 or (not has_magic(rpath) and dest_is_dir and not trailing_sep(rpath))1003 )1004 lpaths = other_paths(1005 rpaths,1006 lpath,1007 exists=exists,1008 flatten=not source_is_str,1009 )1010 1011 callback.set_size(len(lpaths))1012 for lpath, rpath in callback.wrap(zip(lpaths, rpaths)):1013 with callback.branched(rpath, lpath) as child:1014 self.get_file(rpath, lpath, callback=child, **kwargs)1015 1016 def put_file(1017 self, lpath, rpath, callback=DEFAULT_CALLBACK, mode="overwrite", **kwargs1018 ):1019 """Copy single file to remote"""1020 if mode == "create" and self.exists(rpath):1021 raise FileExistsError1022 if os.path.isdir(lpath):1023 self.makedirs(rpath, exist_ok=True)1024 return None1025 1026 with open(lpath, "rb") as f1:1027 size = f1.seek(0, 2)1028 callback.set_size(size)1029 f1.seek(0)1030 1031 self.mkdirs(self._parent(os.fspath(rpath)), exist_ok=True)1032 with self.open(rpath, "wb", **kwargs) as f2:1033 while f1.tell() < size:1034 data = f1.read(self.blocksize)1035 segment_len = f2.write(data)1036 if segment_len is None:1037 segment_len = len(data)1038 callback.relative_update(segment_len)1039 1040 def put(1041 self,1042 lpath,1043 rpath,1044 recursive=False,1045 callback=DEFAULT_CALLBACK,1046 maxdepth=None,1047 **kwargs,1048 ):1049 """Copy file(s) from local.1050 1051 Copies a specific file or tree of files (if recursive=True). If rpath1052 ends with a "/", it will be assumed to be a directory, and target files1053 will go within.1054 1055 Calls put_file for each source.1056 """1057 if isinstance(lpath, list) and isinstance(rpath, list):1058 # No need to expand paths when both source and destination1059 # are provided as lists1060 rpaths = rpath1061 lpaths = lpath1062 else:1063 from .implementations.local import (1064 LocalFileSystem,1065 make_path_posix,1066 trailing_sep,1067 )1068 1069 source_is_str = isinstance(lpath, str)1070 if source_is_str:1071 lpath = make_path_posix(lpath)1072 fs = LocalFileSystem()1073 lpaths = fs.expand_path(1074 lpath, recursive=recursive, maxdepth=maxdepth, **kwargs1075 )1076 if source_is_str and (not recursive or maxdepth is not None):1077 # Non-recursive glob does not copy directories1078 lpaths = [p for p in lpaths if not (trailing_sep(p) or fs.isdir(p))]1079 if not lpaths:1080 return1081 1082 source_is_file = len(lpaths) == 11083 dest_is_dir = isinstance(rpath, str) and (1084 trailing_sep(rpath) or self.isdir(rpath)1085 )1086 1087 rpath = (1088 self._strip_protocol(rpath)1089 if isinstance(rpath, str)1090 else [self._strip_protocol(p) for p in rpath]1091 )1092 exists = source_is_str and (1093 (has_magic(lpath) and source_is_file)1094 or (not has_magic(lpath) and dest_is_dir and not trailing_sep(lpath))1095 )1096 rpaths = other_paths(1097 lpaths,1098 rpath,1099 exists=exists,1100 flatten=not source_is_str,1101 )1102 1103 callback.set_size(len(rpaths))1104 for lpath, rpath in callback.wrap(zip(lpaths, rpaths)):1105 with callback.branched(lpath, rpath) as child:1106 self.put_file(lpath, rpath, callback=child, **kwargs)1107 1108 def head(self, path, size=1024):1109 """Get the first ``size`` bytes from file"""1110 with self.open(path, "rb") as f:1111 return f.read(size)1112 1113 def tail(self, path, size=1024):1114 """Get the last ``size`` bytes from file"""1115 with self.open(path, "rb") as f:1116 f.seek(max(-size, -f.size), 2)1117 return f.read()1118 1119 def cp_file(self, path1, path2, **kwargs):1120 raise NotImplementedError1121 1122 def copy(1123 self, path1, path2, recursive=False, maxdepth=None, on_error=None, **kwargs1124 ):1125 """Copy within two locations in the filesystem1126 1127 on_error : "raise", "ignore"1128 If raise, any not-found exceptions will be raised; if ignore any1129 not-found exceptions will cause the path to be skipped; defaults to1130 raise unless recursive is true, where the default is ignore1131 """1132 if on_error is None and recursive:1133 on_error = "ignore"1134 elif on_error is None:1135 on_error = "raise"1136 1137 if isinstance(path1, list) and isinstance(path2, list):1138 # No need to expand paths when both source and destination1139 # are provided as lists1140 paths1 = path11141 paths2 = path21142 else:1143 from .implementations.local import trailing_sep1144 1145 source_is_str = isinstance(path1, str)1146 paths1 = self.expand_path(1147 path1, recursive=recursive, maxdepth=maxdepth, **kwargs1148 )1149 if source_is_str and (not recursive or maxdepth is not None):1150 # Non-recursive glob does not copy directories1151 paths1 = [p for p in paths1 if not (trailing_sep(p) or self.isdir(p))]1152 if not paths1:1153 return1154 1155 source_is_file = len(paths1) == 11156 dest_is_dir = isinstance(path2, str) and (1157 trailing_sep(path2) or self.isdir(path2)1158 )1159 1160 exists = source_is_str and (1161 (has_magic(path1) and source_is_file)1162 or (not has_magic(path1) and dest_is_dir and not trailing_sep(path1))1163 )1164 paths2 = other_paths(1165 paths1,1166 path2,1167 exists=exists,1168 flatten=not source_is_str,1169 )1170 1171 for p1, p2 in zip(paths1, paths2):1172 try:1173 self.cp_file(p1, p2, **kwargs)1174 except FileNotFoundError:1175 if on_error == "raise":1176 raise1177 1178 def expand_path(self, path, recursive=False, maxdepth=None, **kwargs):1179 """Turn one or more globs or directories into a list of all matching paths1180 to files or directories.1181 1182 kwargs are passed to ``glob`` or ``find``, which may in turn call ``ls``1183 """1184 1185 if maxdepth is not None and maxdepth < 1:1186 raise ValueError("maxdepth must be at least 1")1187 1188 if isinstance(path, (str, os.PathLike)):1189 out = self.expand_path([path], recursive, maxdepth, **kwargs)1190 else:1191 out = set()1192 path = [self._strip_protocol(p) for p in path]1193 for p in path:1194 if has_magic(p):1195 bit = set(self.glob(p, maxdepth=maxdepth, **kwargs))1196 out |= bit1197 if recursive:1198 # glob call above expanded one depth so if maxdepth is defined1199 # then decrement it in expand_path call below. If it is zero1200 # after decrementing then avoid expand_path call.