Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
spec.py2285 linesDownload Raw Back to fsspec
1from __future__ import annotations2 3import io4import json5import logging6import os7import threading8import warnings9import weakref10from errno import ESPIPE11from glob import has_magic12from hashlib import sha25613from typing import Any, ClassVar14 15from .callbacks import DEFAULT_CALLBACK16from .config import apply_config, conf17from .dircache import DirCache18from .transaction import Transaction19from .utils import (20    _unstrip_protocol,21    glob_translate,22    isfilelike,23    other_paths,24    read_block,25    stringify_path,26    tokenize,27)28 29logger = logging.getLogger("fsspec")30 31 32def make_instance(cls, args, kwargs):33    return cls(*args, **kwargs)34 35 36class _Cached(type):37    """38    Metaclass for caching file system instances.39 40    Notes41    -----42    Instances are cached according to43 44    * The values of the class attributes listed in `_extra_tokenize_attributes`45    * The arguments passed to ``__init__``.46 47    This creates an additional reference to the filesystem, which prevents the48    filesystem from being garbage collected when all *user* references go away.49    A call to the :meth:`AbstractFileSystem.clear_instance_cache` must *also*50    be made for a filesystem instance to be garbage collected.51    """52 53    def __init__(cls, *args, **kwargs):54        super().__init__(*args, **kwargs)55        # Note: we intentionally create a reference here, to avoid garbage56        # collecting instances when all other references are gone. To really57        # delete a FileSystem, the cache must be cleared.58        if conf.get("weakref_instance_cache"):  # pragma: no cover59            # debug option for analysing fork/spawn conditions60            cls._cache = weakref.WeakValueDictionary()61        else:62            cls._cache = {}63        cls._pid = os.getpid()64 65    def __call__(cls, *args, **kwargs):66        kwargs = apply_config(cls, kwargs)67        extra_tokens = tuple(68            getattr(cls, attr, None) for attr in cls._extra_tokenize_attributes69        )70        strip_tokenize_options = {71            k: kwargs.pop(k) for k in cls._strip_tokenize_options if k in kwargs72        }73        token = tokenize(74            cls, cls._pid, threading.get_ident(), *args, *extra_tokens, **kwargs75        )76        skip = kwargs.pop("skip_instance_cache", False)77        if os.getpid() != cls._pid:78            cls._cache.clear()79            cls._pid = os.getpid()80        if not skip and cls.cachable and token in cls._cache:81            cls._latest = token82            return cls._cache[token]83        else:84            obj = super().__call__(*args, **kwargs, **strip_tokenize_options)85            # Setting _fs_token here causes some static linters to complain.86            obj._fs_token_ = token87            obj.storage_args = args88            obj.storage_options = kwargs89            if obj.async_impl and obj.mirror_sync_methods:90                from .asyn import mirror_sync_methods91 92                mirror_sync_methods(obj)93 94            if cls.cachable and not skip:95                cls._latest = token96                cls._cache[token] = obj97            return obj98 99 100class AbstractFileSystem(metaclass=_Cached):101    """102    An abstract super-class for pythonic file-systems103 104    Implementations are expected to be compatible with or, better, subclass105    from here.106    """107 108    cachable = True  # this class can be cached, instances reused109    _cached = False110    blocksize = 2**22111    sep = "/"112    protocol: ClassVar[str | tuple[str, ...]] = "abstract"113    _latest = None114    async_impl = False115    mirror_sync_methods = False116    root_marker = ""  # For some FSs, may require leading '/' or other character117    transaction_type = Transaction118 119    #: Extra *class attributes* that should be considered when hashing.120    _extra_tokenize_attributes = ()121    #: *storage options* that should not be considered when hashing.122    _strip_tokenize_options = ()123 124    # Set by _Cached metaclass125    storage_args: tuple[Any, ...]126    storage_options: dict[str, Any]127 128    def __init__(self, *args, **storage_options):129        """Create and configure file-system instance130 131        Instances may be cachable, so if similar enough arguments are seen132        a new instance is not required. The token attribute exists to allow133        implementations to cache instances if they wish.134 135        A reasonable default should be provided if there are no arguments.136 137        Subclasses should call this method.138 139        Parameters140        ----------141        use_listings_cache, listings_expiry_time, max_paths:142            passed to ``DirCache``, if the implementation supports143            directory listing caching. Pass use_listings_cache=False144            to disable such caching.145        skip_instance_cache: bool146            If this is a cachable implementation, pass True here to force147            creating a new instance even if a matching instance exists, and prevent148            storing this instance.149        asynchronous: bool150        loop: asyncio-compatible IOLoop or None151        """152        if self._cached:153            # reusing instance, don't change154            return155        self._cached = True156        self._intrans = False157        self._transaction = None158        self._invalidated_caches_in_transaction = []159        self.dircache = DirCache(**storage_options)160 161        if storage_options.pop("add_docs", None):162            warnings.warn("add_docs is no longer supported.", FutureWarning)163 164        if storage_options.pop("add_aliases", None):165            warnings.warn("add_aliases has been removed.", FutureWarning)166        # This is set in _Cached167        self._fs_token_ = None168 169    @property170    def fsid(self):171        """Persistent filesystem id that can be used to compare filesystems172        across sessions.173        """174        raise NotImplementedError175 176    @property177    def _fs_token(self):178        return self._fs_token_179 180    def __dask_tokenize__(self):181        return self._fs_token182 183    def __hash__(self):184        return int(self._fs_token, 16)185 186    def __eq__(self, other):187        return isinstance(other, type(self)) and self._fs_token == other._fs_token188 189    def __reduce__(self):190        return make_instance, (type(self), self.storage_args, self.storage_options)191 192    @classmethod193    def _strip_protocol(cls, path):194        """Turn path from fully-qualified to file-system-specific195 196        May require FS-specific handling, e.g., for relative paths or links.197        """198        if isinstance(path, list):199            return [cls._strip_protocol(p) for p in path]200        path = stringify_path(path)201        protos = (cls.protocol,) if isinstance(cls.protocol, str) else cls.protocol202        for protocol in protos:203            if path.startswith(protocol + "://"):204                path = path[len(protocol) + 3 :]205            elif path.startswith(protocol + "::"):206                path = path[len(protocol) + 2 :]207        path = path.rstrip("/")208        # use of root_marker to make minimum required path, e.g., "/"209        return path or cls.root_marker210 211    def unstrip_protocol(self, name: str) -> str:212        """Format FS-specific path to generic, including protocol"""213        protos = (self.protocol,) if isinstance(self.protocol, str) else self.protocol214        for protocol in protos:215            if name.startswith(f"{protocol}://"):216                return name217        return f"{protos[0]}://{name}"218 219    @staticmethod220    def _get_kwargs_from_urls(path):221        """If kwargs can be encoded in the paths, extract them here222 223        This should happen before instantiation of the class; incoming paths224        then should be amended to strip the options in methods.225 226        Examples may look like an sftp path "sftp://user@host:/my/path", where227        the user and host should become kwargs and later get stripped.228        """229        # by default, nothing happens230        return {}231 232    @classmethod233    def current(cls):234        """Return the most recently instantiated FileSystem235 236        If no instance has been created, then create one with defaults237        """238        if cls._latest in cls._cache:239            return cls._cache[cls._latest]240        return cls()241 242    @property243    def transaction(self):244        """A context within which files are committed together upon exit245 246        Requires the file class to implement `.commit()` and `.discard()`247        for the normal and exception cases.248        """249        if self._transaction is None:250            self._transaction = self.transaction_type(self)251        return self._transaction252 253    def start_transaction(self):254        """Begin write transaction for deferring files, non-context version"""255        self._intrans = True256        self._transaction = self.transaction_type(self)257        return self.transaction258 259    def end_transaction(self):260        """Finish write transaction, non-context version"""261        self.transaction.complete()262        self._transaction = None263        # The invalid cache must be cleared after the transaction is completed.264        for path in self._invalidated_caches_in_transaction:265            self.invalidate_cache(path)266        self._invalidated_caches_in_transaction.clear()267 268    def invalidate_cache(self, path=None):269        """270        Discard any cached directory information271 272        Parameters273        ----------274        path: string or None275            If None, clear all listings cached else listings at or under given276            path.277        """278        # Not necessary to implement invalidation mechanism, may have no cache.279        # But if have, you should call this method of parent class from your280        # subclass to ensure expiring caches after transacations correctly.281        # See the implementation of FTPFileSystem in ftp.py282        if self._intrans:283            self._invalidated_caches_in_transaction.append(path)284 285    def mkdir(self, path, create_parents=True, **kwargs):286        """287        Create directory entry at path288 289        For systems that don't have true directories, may create an for290        this instance only and not touch the real filesystem291 292        Parameters293        ----------294        path: str295            location296        create_parents: bool297            if True, this is equivalent to ``makedirs``298        kwargs:299            may be permissions, etc.300        """301        pass  # not necessary to implement, may not have directories302 303    def makedirs(self, path, exist_ok=False):304        """Recursively make directories305 306        Creates directory at path and any intervening required directories.307        Raises exception if, for instance, the path already exists but is a308        file.309 310        Parameters311        ----------312        path: str313            leaf directory name314        exist_ok: bool (False)315            If False, will error if the target already exists316        """317        pass  # not necessary to implement, may not have directories318 319    def rmdir(self, path):320        """Remove a directory, if empty"""321        pass  # not necessary to implement, may not have directories322 323    def ls(self, path, detail=True, **kwargs):324        """List objects at path.325 326        This should include subdirectories and files at that location. The327        difference between a file and a directory must be clear when details328        are requested.329 330        The specific keys, or perhaps a FileInfo class, or similar, is TBD,331        but must be consistent across implementations.332        Must include:333 334        - full path to the entry (without protocol)335        - size of the entry, in bytes. If the value cannot be determined, will336          be ``None``.337        - type of entry, "file", "directory" or other338 339        Additional information340        may be present, appropriate to the file-system, e.g., generation,341        checksum, etc.342 343        May use refresh=True|False to allow use of self._ls_from_cache to344        check for a saved listing and avoid calling the backend. This would be345        common where listing may be expensive.346 347        Parameters348        ----------349        path: str350        detail: bool351            if True, gives a list of dictionaries, where each is the same as352            the result of ``info(path)``. If False, gives a list of paths353            (str).354        kwargs: may have additional backend-specific options, such as version355            information356 357        Returns358        -------359        List of strings if detail is False, or list of directory information360        dicts if detail is True.361        """362        raise NotImplementedError363 364    def _ls_from_cache(self, path):365        """Check cache for listing366 367        Returns listing, if found (may be empty list for a directly that exists368        but contains nothing), None if not in cache.369        """370        parent = self._parent(path)371        try:372            return self.dircache[path.rstrip("/")]373        except KeyError:374            pass375        try:376            files = [377                f378                for f in self.dircache[parent]379                if f["name"] == path380                or (f["name"] == path.rstrip("/") and f["type"] == "directory")381            ]382            if len(files) == 0:383                # parent dir was listed but did not contain this file384                raise FileNotFoundError(path)385            return files386        except KeyError:387            pass388 389    def walk(self, path, maxdepth=None, topdown=True, on_error="omit", **kwargs):390        """Return all files under the given path.391 392        List all files, recursing into subdirectories; output is iterator-style,393        like ``os.walk()``. For a simple list of files, ``find()`` is available.394 395        When topdown is True, the caller can modify the dirnames list in-place (perhaps396        using del or slice assignment), and walk() will397        only recurse into the subdirectories whose names remain in dirnames;398        this can be used to prune the search, impose a specific order of visiting,399        or even to inform walk() about directories the caller creates or renames before400        it resumes walk() again.401        Modifying dirnames when topdown is False has no effect. (see os.walk)402 403        Note that the "files" outputted will include anything that is not404        a directory, such as links.405 406        Parameters407        ----------408        path: str409            Root to recurse into410        maxdepth: int411            Maximum recursion depth. None means limitless, but not recommended412            on link-based file-systems.413        topdown: bool (True)414            Whether to walk the directory tree from the top downwards or from415            the bottom upwards.416        on_error: "omit", "raise", a callable417            if omit (default), path with exception will simply be empty;418            If raise, an underlying exception will be raised;419            if callable, it will be called with a single OSError instance as argument420        kwargs: passed to ``ls``421        """422        if maxdepth is not None and maxdepth < 1:423            raise ValueError("maxdepth must be at least 1")424 425        path = self._strip_protocol(path)426        full_dirs = {}427        dirs = {}428        files = {}429 430        detail = kwargs.pop("detail", False)431        try:432            listing = self.ls(path, detail=True, **kwargs)433        except (FileNotFoundError, OSError) as e:434            if on_error == "raise":435                raise436            if callable(on_error):437                on_error(e)438            return439 440        for info in listing:441            # each info name must be at least [path]/part , but here442            # we check also for names like [path]/part/443            pathname = info["name"].rstrip("/")444            name = pathname.rsplit("/", 1)[-1]445            if info["type"] == "directory" and pathname != path:446                # do not include "self" path447                full_dirs[name] = pathname448                dirs[name] = info449            elif pathname == path:450                # file-like with same name as give path451                files[""] = info452            else:453                files[name] = info454 455        if not detail:456            dirs = list(dirs)457            files = list(files)458 459        if topdown:460            # Yield before recursion if walking top down461            yield path, dirs, files462 463        if maxdepth is not None:464            maxdepth -= 1465            if maxdepth < 1:466                if not topdown:467                    yield path, dirs, files468                return469 470        for d in dirs:471            yield from self.walk(472                full_dirs[d],473                maxdepth=maxdepth,474                detail=detail,475                topdown=topdown,476                **kwargs,477            )478 479        if not topdown:480            # Yield after recursion if walking bottom up481            yield path, dirs, files482 483    def find(self, path, maxdepth=None, withdirs=False, detail=False, **kwargs):484        """List all files below path.485 486        Like posix ``find`` command without conditions487 488        Parameters489        ----------490        path : str491        maxdepth: int or None492            If not None, the maximum number of levels to descend493        withdirs: bool494            Whether to include directory paths in the output. This is True495            when used by glob, but users usually only want files.496        kwargs are passed to ``ls``.497        """498        # TODO: allow equivalent of -name parameter499        path = self._strip_protocol(path)500        out = {}501 502        # Add the root directory if withdirs is requested503        # This is needed for posix glob compliance504        if withdirs and path != "" and self.isdir(path):505            out[path] = self.info(path)506 507        for _, dirs, files in self.walk(path, maxdepth, detail=True, **kwargs):508            if withdirs:509                files.update(dirs)510            out.update({info["name"]: info for name, info in files.items()})511        if not out and self.isfile(path):512            # walk works on directories, but find should also return [path]513            # when path happens to be a file514            out[path] = {}515        names = sorted(out)516        if not detail:517            return names518        else:519            return {name: out[name] for name in names}520 521    def du(self, path, total=True, maxdepth=None, withdirs=False, **kwargs):522        """Space used by files and optionally directories within a path523 524        Directory size does not include the size of its contents.525 526        Parameters527        ----------528        path: str529        total: bool530            Whether to sum all the file sizes531        maxdepth: int or None532            Maximum number of directory levels to descend, None for unlimited.533        withdirs: bool534            Whether to include directory paths in the output.535        kwargs: passed to ``find``536 537        Returns538        -------539        Dict of {path: size} if total=False, or int otherwise, where numbers540        refer to bytes used.541        """542        sizes = {}543        if withdirs and self.isdir(path):544            # Include top-level directory in output545            info = self.info(path)546            sizes[info["name"]] = info["size"]547        for f in self.find(path, maxdepth=maxdepth, withdirs=withdirs, **kwargs):548            info = self.info(f)549            sizes[info["name"]] = info["size"]550        if total:551            return sum(sizes.values())552        else:553            return sizes554 555    def glob(self, path, maxdepth=None, **kwargs):556        """Find files by glob-matching.557 558        Pattern matching capabilities for finding files that match the given pattern.559 560        Parameters561        ----------562        path: str563            The glob pattern to match against564        maxdepth: int or None565            Maximum depth for ``'**'`` patterns. Applied on the first ``'**'`` found.566            Must be at least 1 if provided.567        kwargs:568            Additional arguments passed to ``find`` (e.g., detail=True)569 570        Returns571        -------572        List of matched paths, or dict of paths and their info if detail=True573 574        Notes575        -----576        Supported patterns:577        - '*': Matches any sequence of characters within a single directory level578        - ``'**'``: Matches any number of directory levels (must be an entire path component)579        - '?': Matches exactly one character580        - '[abc]': Matches any character in the set581        - '[a-z]': Matches any character in the range582        - '[!abc]': Matches any character NOT in the set583 584        Special behaviors:585        - If the path ends with '/', only folders are returned586        - Consecutive '*' characters are compressed into a single '*'587        - Empty brackets '[]' never match anything588        - Negated empty brackets '[!]' match any single character589        - Special characters in character classes are escaped properly590 591        Limitations:592        - ``'**'`` must be a complete path component (e.g., ``'a/**/b'``, not ``'a**b'``)593        - No brace expansion ('{a,b}.txt')594        - No extended glob patterns ('+(pattern)', '!(pattern)')595        """596        if maxdepth is not None and maxdepth < 1:597            raise ValueError("maxdepth must be at least 1")598 599        import re600 601        seps = (os.path.sep, os.path.altsep) if os.path.altsep else (os.path.sep,)602        ends_with_sep = path.endswith(seps)  # _strip_protocol strips trailing slash603        path = self._strip_protocol(path)604        append_slash_to_dirname = ends_with_sep or path.endswith(605            tuple(sep + "**" for sep in seps)606        )607        idx_star = path.find("*") if path.find("*") >= 0 else len(path)608        idx_qmark = path.find("?") if path.find("?") >= 0 else len(path)609        idx_brace = path.find("[") if path.find("[") >= 0 else len(path)610 611        min_idx = min(idx_star, idx_qmark, idx_brace)612 613        detail = kwargs.pop("detail", False)614        withdirs = kwargs.pop("withdirs", True)615 616        if not has_magic(path):617            if self.exists(path, **kwargs):618                if not detail:619                    return [path]620                else:621                    return {path: self.info(path, **kwargs)}622            else:623                if not detail:624                    return []  # glob of non-existent returns empty625                else:626                    return {}627        elif "/" in path[:min_idx]:628            min_idx = path[:min_idx].rindex("/")629            root = path[: min_idx + 1]630            depth = path[min_idx + 1 :].count("/") + 1631        else:632            root = ""633            depth = path[min_idx + 1 :].count("/") + 1634 635        if "**" in path:636            if maxdepth is not None:637                idx_double_stars = path.find("**")638                depth_double_stars = path[idx_double_stars:].count("/") + 1639                depth = depth - depth_double_stars + maxdepth640            else:641                depth = None642 643        allpaths = self.find(644            root, maxdepth=depth, withdirs=withdirs, detail=True, **kwargs645        )646 647        pattern = glob_translate(path + ("/" if ends_with_sep else ""))648        pattern = re.compile(pattern)649 650        out = {651            p: info652            for p, info in sorted(allpaths.items())653            if pattern.match(654                p + "/"655                if append_slash_to_dirname and info["type"] == "directory"656                else p657            )658        }659 660        if detail:661            return out662        else:663            return list(out)664 665    def exists(self, path, **kwargs):666        """Is there a file at the given path"""667        try:668            self.info(path, **kwargs)669            return True670        except:  # noqa: E722671            # any exception allowed bar FileNotFoundError?672            return False673 674    def lexists(self, path, **kwargs):675        """If there is a file at the given path (including676        broken links)"""677        return self.exists(path)678 679    def info(self, path, **kwargs):680        """Give details of entry at path681 682        Returns a single dictionary, with exactly the same information as ``ls``683        would with ``detail=True``.684 685        The default implementation calls ls and could be overridden by a686        shortcut. kwargs are passed on to ```ls()``.687 688        Some file systems might not be able to measure the file's size, in689        which case, the returned dict will include ``'size': None``.690 691        Returns692        -------693        dict with keys: name (full path in the FS), size (in bytes), type (file,694        directory, or something else) and other FS-specific keys.695        """696        path = self._strip_protocol(path)697        out = self.ls(self._parent(path), detail=True, **kwargs)698        out = [o for o in out if o["name"].rstrip("/") == path]699        if out:700            return out[0]701        out = self.ls(path, detail=True, **kwargs)702        path = path.rstrip("/")703        out1 = [o for o in out if o["name"].rstrip("/") == path]704        if len(out1) == 1:705            if "size" not in out1[0]:706                out1[0]["size"] = None707            return out1[0]708        elif len(out1) > 1 or out:709            return {"name": path, "size": 0, "type": "directory"}710        else:711            raise FileNotFoundError(path)712 713    def checksum(self, path):714        """Unique value for current version of file715 716        If the checksum is the same from one moment to another, the contents717        are guaranteed to be the same. If the checksum changes, the contents718        *might* have changed.719 720        This should normally be overridden; default will probably capture721        creation/modification timestamp (which would be good) or maybe722        access timestamp (which would be bad)723        """724        return int(tokenize(self.info(path)), 16)725 726    def size(self, path):727        """Size in bytes of file"""728        return self.info(path).get("size", None)729 730    def sizes(self, paths):731        """Size in bytes of each file in a list of paths"""732        return [self.size(p) for p in paths]733 734    def isdir(self, path):735        """Is this entry directory-like?"""736        try:737            return self.info(path)["type"] == "directory"738        except OSError:739            return False740 741    def isfile(self, path):742        """Is this entry file-like?"""743        try:744            return self.info(path)["type"] == "file"745        except:  # noqa: E722746            return False747 748    def read_text(self, path, encoding=None, errors=None, newline=None, **kwargs):749        """Get the contents of the file as a string.750 751        Parameters752        ----------753        path: str754            URL of file on this filesystems755        encoding, errors, newline: same as `open`.756        """757        with self.open(758            path,759            mode="r",760            encoding=encoding,761            errors=errors,762            newline=newline,763            **kwargs,764        ) as f:765            return f.read()766 767    def write_text(768        self, path, value, encoding=None, errors=None, newline=None, **kwargs769    ):770        """Write the text to the given file.771 772        An existing file will be overwritten.773 774        Parameters775        ----------776        path: str777            URL of file on this filesystems778        value: str779            Text to write.780        encoding, errors, newline: same as `open`.781        """782        with self.open(783            path,784            mode="w",785            encoding=encoding,786            errors=errors,787            newline=newline,788            **kwargs,789        ) as f:790            return f.write(value)791 792    def cat_file(self, path, start=None, end=None, **kwargs):793        """Get the content of a file794 795        Parameters796        ----------797        path: URL of file on this filesystems798        start, end: int799            Bytes limits of the read. If negative, backwards from end,800            like usual python slices. Either can be None for start or801            end of file, respectively802        kwargs: passed to ``open()``.803        """804        # explicitly set buffering off?805        with self.open(path, "rb", **kwargs) as f:806            if start is not None:807                if start >= 0:808                    f.seek(start)809                else:810                    f.seek(max(0, f.size + start))811            if end is not None:812                if end < 0:813                    end = f.size + end814                return f.read(end - f.tell())815            return f.read()816 817    def pipe_file(self, path, value, mode="overwrite", **kwargs):818        """Set the bytes of given file"""819        if mode == "create" and self.exists(path):820            # non-atomic but simple way; or could use "xb" in open(), which is likely821            # not as well supported822            raise FileExistsError823        with self.open(path, "wb", **kwargs) as f:824            f.write(value)825 826    def pipe(self, path, value=None, **kwargs):827        """Put value into path828 829        (counterpart to ``cat``)830 831        Parameters832        ----------833        path: string or dict(str, bytes)834            If a string, a single remote location to put ``value`` bytes; if a dict,835            a mapping of {path: bytesvalue}.836        value: bytes, optional837            If using a single path, these are the bytes to put there. Ignored if838            ``path`` is a dict839        """840        if isinstance(path, str):841            self.pipe_file(self._strip_protocol(path), value, **kwargs)842        elif isinstance(path, dict):843            for k, v in path.items():844                self.pipe_file(self._strip_protocol(k), v, **kwargs)845        else:846            raise ValueError("path must be str or dict")847 848    def cat_ranges(849        self, paths, starts, ends, max_gap=None, on_error="return", **kwargs850    ):851        """Get the contents of byte ranges from one or more files852 853        Parameters854        ----------855        paths: list856            A list of of filepaths on this filesystems857        starts, ends: int or list858            Bytes limits of the read. If using a single int, the same value will be859            used to read all the specified files.860        """861        if max_gap is not None:862            raise NotImplementedError863        if not isinstance(paths, list):864            raise TypeError865        if not isinstance(starts, list):866            starts = [starts] * len(paths)867        if not isinstance(ends, list):868            ends = [ends] * len(paths)869        if len(starts) != len(paths) or len(ends) != len(paths):870            raise ValueError871        out = []872        for p, s, e in zip(paths, starts, ends):873            try:874                out.append(self.cat_file(p, s, e))875            except Exception as e:876                if on_error == "return":877                    out.append(e)878                else:879                    raise880        return out881 882    def cat(self, path, recursive=False, on_error="raise", **kwargs):883        """Fetch (potentially multiple) paths' contents884 885        Parameters886        ----------887        recursive: bool888            If True, assume the path(s) are directories, and get all the889            contained files890        on_error : "raise", "omit", "return"891            If raise, an underlying exception will be raised (converted to KeyError892            if the type is in self.missing_exceptions); if omit, keys with exception893            will simply not be included in the output; if "return", all keys are894            included in the output, but the value will be bytes or an exception895            instance.896        kwargs: passed to cat_file897 898        Returns899        -------900        dict of {path: contents} if there are multiple paths901        or the path has been otherwise expanded902        """903        paths = self.expand_path(path, recursive=recursive, **kwargs)904        if (905            len(paths) > 1906            or isinstance(path, list)907            or paths[0] != self._strip_protocol(path)908        ):909            out = {}910            for path in paths:911                try:912                    out[path] = self.cat_file(path, **kwargs)913                except Exception as e:914                    if on_error == "raise":915                        raise916                    if on_error == "return":917                        out[path] = e918            return out919        else:920            return self.cat_file(paths[0], **kwargs)921 922    def get_file(self, rpath, lpath, callback=DEFAULT_CALLBACK, outfile=None, **kwargs):923        """Copy single remote file to local"""924        from .implementations.local import LocalFileSystem925 926        if isfilelike(lpath):927            outfile = lpath928        elif self.isdir(rpath):929            os.makedirs(lpath, exist_ok=True)930            return None931 932        fs = LocalFileSystem(auto_mkdir=True)933        fs.makedirs(fs._parent(lpath), exist_ok=True)934 935        with self.open(rpath, "rb", **kwargs) as f1:936            if outfile is None:937                outfile = open(lpath, "wb")938 939            try:940                callback.set_size(getattr(f1, "size", None))941                data = True942                while data:943                    data = f1.read(self.blocksize)944                    segment_len = outfile.write(data)945                    if segment_len is None:946                        segment_len = len(data)947                    callback.relative_update(segment_len)948            finally:949                if not isfilelike(lpath):950                    outfile.close()951 952    def get(953        self,954        rpath,955        lpath,956        recursive=False,957        callback=DEFAULT_CALLBACK,958        maxdepth=None,959        **kwargs,960    ):961        """Copy file(s) to local.962 963        Copies a specific file or tree of files (if recursive=True). If lpath964        ends with a "/", it will be assumed to be a directory, and target files965        will go within. Can submit a list of paths, which may be glob-patterns966        and will be expanded.967 968        Calls get_file for each source.969        """970        if isinstance(lpath, list) and isinstance(rpath, list):971            # No need to expand paths when both source and destination972            # are provided as lists973            rpaths = rpath974            lpaths = lpath975        else:976            from .implementations.local import (977                LocalFileSystem,978                make_path_posix,979                trailing_sep,980            )981 982            source_is_str = isinstance(rpath, str)983            rpaths = self.expand_path(984                rpath, recursive=recursive, maxdepth=maxdepth, **kwargs985            )986            if source_is_str and (not recursive or maxdepth is not None):987                # Non-recursive glob does not copy directories988                rpaths = [p for p in rpaths if not (trailing_sep(p) or self.isdir(p))]989                if not rpaths:990                    return991 992            if isinstance(lpath, str):993                lpath = make_path_posix(lpath)994 995            source_is_file = len(rpaths) == 1996            dest_is_dir = isinstance(lpath, str) and (997                trailing_sep(lpath) or LocalFileSystem().isdir(lpath)998            )999 1000            exists = source_is_str and (1001                (has_magic(rpath) and source_is_file)1002                or (not has_magic(rpath) and dest_is_dir and not trailing_sep(rpath))1003            )1004            lpaths = other_paths(1005                rpaths,1006                lpath,1007                exists=exists,1008                flatten=not source_is_str,1009            )1010 1011        callback.set_size(len(lpaths))1012        for lpath, rpath in callback.wrap(zip(lpaths, rpaths)):1013            with callback.branched(rpath, lpath) as child:1014                self.get_file(rpath, lpath, callback=child, **kwargs)1015 1016    def put_file(1017        self, lpath, rpath, callback=DEFAULT_CALLBACK, mode="overwrite", **kwargs1018    ):1019        """Copy single file to remote"""1020        if mode == "create" and self.exists(rpath):1021            raise FileExistsError1022        if os.path.isdir(lpath):1023            self.makedirs(rpath, exist_ok=True)1024            return None1025 1026        with open(lpath, "rb") as f1:1027            size = f1.seek(0, 2)1028            callback.set_size(size)1029            f1.seek(0)1030 1031            self.mkdirs(self._parent(os.fspath(rpath)), exist_ok=True)1032            with self.open(rpath, "wb", **kwargs) as f2:1033                while f1.tell() < size:1034                    data = f1.read(self.blocksize)1035                    segment_len = f2.write(data)1036                    if segment_len is None:1037                        segment_len = len(data)1038                    callback.relative_update(segment_len)1039 1040    def put(1041        self,1042        lpath,1043        rpath,1044        recursive=False,1045        callback=DEFAULT_CALLBACK,1046        maxdepth=None,1047        **kwargs,1048    ):1049        """Copy file(s) from local.1050 1051        Copies a specific file or tree of files (if recursive=True). If rpath1052        ends with a "/", it will be assumed to be a directory, and target files1053        will go within.1054 1055        Calls put_file for each source.1056        """1057        if isinstance(lpath, list) and isinstance(rpath, list):1058            # No need to expand paths when both source and destination1059            # are provided as lists1060            rpaths = rpath1061            lpaths = lpath1062        else:1063            from .implementations.local import (1064                LocalFileSystem,1065                make_path_posix,1066                trailing_sep,1067            )1068 1069            source_is_str = isinstance(lpath, str)1070            if source_is_str:1071                lpath = make_path_posix(lpath)1072            fs = LocalFileSystem()1073            lpaths = fs.expand_path(1074                lpath, recursive=recursive, maxdepth=maxdepth, **kwargs1075            )1076            if source_is_str and (not recursive or maxdepth is not None):1077                # Non-recursive glob does not copy directories1078                lpaths = [p for p in lpaths if not (trailing_sep(p) or fs.isdir(p))]1079                if not lpaths:1080                    return1081 1082            source_is_file = len(lpaths) == 11083            dest_is_dir = isinstance(rpath, str) and (1084                trailing_sep(rpath) or self.isdir(rpath)1085            )1086 1087            rpath = (1088                self._strip_protocol(rpath)1089                if isinstance(rpath, str)1090                else [self._strip_protocol(p) for p in rpath]1091            )1092            exists = source_is_str and (1093                (has_magic(lpath) and source_is_file)1094                or (not has_magic(lpath) and dest_is_dir and not trailing_sep(lpath))1095            )1096            rpaths = other_paths(1097                lpaths,1098                rpath,1099                exists=exists,1100                flatten=not source_is_str,1101            )1102 1103        callback.set_size(len(rpaths))1104        for lpath, rpath in callback.wrap(zip(lpaths, rpaths)):1105            with callback.branched(lpath, rpath) as child:1106                self.put_file(lpath, rpath, callback=child, **kwargs)1107 1108    def head(self, path, size=1024):1109        """Get the first ``size`` bytes from file"""1110        with self.open(path, "rb") as f:1111            return f.read(size)1112 1113    def tail(self, path, size=1024):1114        """Get the last ``size`` bytes from file"""1115        with self.open(path, "rb") as f:1116            f.seek(max(-size, -f.size), 2)1117            return f.read()1118 1119    def cp_file(self, path1, path2, **kwargs):1120        raise NotImplementedError1121 1122    def copy(1123        self, path1, path2, recursive=False, maxdepth=None, on_error=None, **kwargs1124    ):1125        """Copy within two locations in the filesystem1126 1127        on_error : "raise", "ignore"1128            If raise, any not-found exceptions will be raised; if ignore any1129            not-found exceptions will cause the path to be skipped; defaults to1130            raise unless recursive is true, where the default is ignore1131        """1132        if on_error is None and recursive:1133            on_error = "ignore"1134        elif on_error is None:1135            on_error = "raise"1136 1137        if isinstance(path1, list) and isinstance(path2, list):1138            # No need to expand paths when both source and destination1139            # are provided as lists1140            paths1 = path11141            paths2 = path21142        else:1143            from .implementations.local import trailing_sep1144 1145            source_is_str = isinstance(path1, str)1146            paths1 = self.expand_path(1147                path1, recursive=recursive, maxdepth=maxdepth, **kwargs1148            )1149            if source_is_str and (not recursive or maxdepth is not None):1150                # Non-recursive glob does not copy directories1151                paths1 = [p for p in paths1 if not (trailing_sep(p) or self.isdir(p))]1152                if not paths1:1153                    return1154 1155            source_is_file = len(paths1) == 11156            dest_is_dir = isinstance(path2, str) and (1157                trailing_sep(path2) or self.isdir(path2)1158            )1159 1160            exists = source_is_str and (1161                (has_magic(path1) and source_is_file)1162                or (not has_magic(path1) and dest_is_dir and not trailing_sep(path1))1163            )1164            paths2 = other_paths(1165                paths1,1166                path2,1167                exists=exists,1168                flatten=not source_is_str,1169            )1170 1171        for p1, p2 in zip(paths1, paths2):1172            try:1173                self.cp_file(p1, p2, **kwargs)1174            except FileNotFoundError:1175                if on_error == "raise":1176                    raise1177 1178    def expand_path(self, path, recursive=False, maxdepth=None, **kwargs):1179        """Turn one or more globs or directories into a list of all matching paths1180        to files or directories.1181 1182        kwargs are passed to ``glob`` or ``find``, which may in turn call ``ls``1183        """1184 1185        if maxdepth is not None and maxdepth < 1:1186            raise ValueError("maxdepth must be at least 1")1187 1188        if isinstance(path, (str, os.PathLike)):1189            out = self.expand_path([path], recursive, maxdepth, **kwargs)1190        else:1191            out = set()1192            path = [self._strip_protocol(p) for p in path]1193            for p in path:1194                if has_magic(p):1195                    bit = set(self.glob(p, maxdepth=maxdepth, **kwargs))1196                    out |= bit1197                    if recursive:1198                        # glob call above expanded one depth so if maxdepth is defined1199                        # then decrement it in expand_path call below. If it is zero1200                        # after decrementing then avoid expand_path call.

Showing the first 1,200 of 2285 lines. Download the file for the rest.

codekingpro/portable-devtools · Team Ai