Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
hf_file_system.py1447 linesDownload Raw Back to huggingface_hub
1import os2import tempfile3import threading4from collections import deque5from collections.abc import Iterable, Iterator6from contextlib import ExitStack7from copy import deepcopy8from dataclasses import dataclass, field9from datetime import datetime10from itertools import chain11from pathlib import Path, PurePosixPath12from typing import Any, NoReturn, Union13from urllib.parse import quote, unquote14 15import fsspec16import httpx17from fsspec.callbacks import _DEFAULT_CALLBACK, NoOpCallback, TqdmCallback18from fsspec.config import apply_config19from fsspec.utils import isfilelike20 21from . import constants22from ._commit_api import CommitOperationCopy, CommitOperationDelete23from .errors import (24    BucketNotFoundError,25    EntryNotFoundError,26    HfHubHTTPError,27    RepositoryNotFoundError,28    RevisionNotFoundError,29)30from .file_download import hf_hub_url, http_get31from .hf_api import SPECIAL_REFS_REVISION_REGEX, BucketFile, BucketFolder, HfApi, LastCommitInfo, RepoFile, RepoFolder32from .utils import HFValidationError, hf_raise_for_status, http_backoff, http_stream_backoff33from .utils.insecure_hashlib import md534 35 36@dataclass37class HfFileSystemResolvedPath:38    """Top level Data structure containing information about a resolved Hugging Face file system path."""39 40    root: str41    path: str42 43    def unresolve(self) -> str:44        return f"{self.root}/{self.path}".rstrip("/")45 46 47@dataclass48class HfFileSystemResolvedRepositoryPath(HfFileSystemResolvedPath):49    """Data structure containing information about a resolved path in a repository."""50 51    repo_type: str52    repo_id: str53    revision: str54    path_in_repo: str55    root: str = field(init=False)56    path: str = field(init=False)57    # The part placed after '@' in the initial path. It can be a quoted or unquoted refs revision.58    # Used to reconstruct the unresolved path to return to the user.59    _raw_revision: str | None = field(default=None, repr=False)60 61    def __post_init__(self):62        repo_path = constants.REPO_TYPES_URL_PREFIXES.get(self.repo_type, "") + self.repo_id63        if self._raw_revision:64            self.root = f"{repo_path}@{self._raw_revision}"65        elif self.revision != constants.DEFAULT_REVISION:66            self.root = f"{repo_path}@{safe_revision(self.revision)}"67        else:68            self.root = repo_path69        self.path = self.path_in_repo70 71 72@dataclass73class HfFileSystemResolvedBucketPath(HfFileSystemResolvedPath):74    """Data structure containing information about a resolved path in a bucket."""75 76    bucket_id: str77    root: str = field(init=False)78 79    def __post_init__(self):80        self.root = "buckets/" + self.bucket_id81 82 83# We need to improve fsspec.spec._Cached which is AbstractFileSystem's metaclass84_cached_base: Any = type(fsspec.AbstractFileSystem)85 86 87class _Cached(_cached_base):88    """89    Metaclass for caching HfFileSystem instances according to the args.90 91    This creates an additional reference to the filesystem, which prevents the92    filesystem from being garbage collected when all *user* references go away.93    A call to the :meth:`AbstractFileSystem.clear_instance_cache` must *also*94    be made for a filesystem instance to be garbage collected.95 96    This is a slightly modified version of `fsspec.spec._Cached` to improve it.97    In particular in `_tokenize` the pid isn't taken into account for the98    `fs_token` used to identify cached instances. The `fs_token` logic is also99    robust to defaults values and the order of the args. Finally new instances100    reuse the states from sister instances in the main thread.101    """102 103    def __init__(cls, *args, **kwargs):104        # Hack: override https://github.com/fsspec/filesystem_spec/blob/dcb167e8f50e6273d4cfdfc4cab8fc5aa4c958bf/fsspec/spec.py#L53105        super().__init__(*args, **kwargs)106        # Note: we intentionally create a reference here, to avoid garbage107        # collecting instances when all other references are gone. To really108        # delete a FileSystem, the cache must be cleared.109        cls._cache = {}110 111    def __call__(cls, *args, **kwargs):112        # Hack: override https://github.com/fsspec/filesystem_spec/blob/dcb167e8f50e6273d4cfdfc4cab8fc5aa4c958bf/fsspec/spec.py#L65113        # Apply fsspec config (env vars / config files) before tokenizing so that114        # HfFileSystem picks up defaults the same way other fsspec filesystems do.115        kwargs = apply_config(cls, kwargs)116        skip = kwargs.pop("skip_instance_cache", False)117        fs_token = cls._tokenize(cls, threading.get_ident(), *args, **kwargs)118        fs_token_main_thread = cls._tokenize(cls, threading.main_thread().ident, *args, **kwargs)119        if not skip and cls.cachable and fs_token in cls._cache:120            # reuse cached instance121            cls._latest = fs_token122            return cls._cache[fs_token]123        else:124            # create new instance125            obj = type.__call__(cls, *args, **kwargs)126            if not skip and cls.cachable and fs_token_main_thread in cls._cache:127                # reuse the cache from the main thread instance in the new instance128                instance_state = cls._cache[fs_token_main_thread]._get_instance_state()129                for attr, state_value in instance_state.items():130                    setattr(obj, attr, state_value)131            obj._fs_token_ = fs_token132            obj.storage_args = args133            obj.storage_options = kwargs134            if cls.cachable and not skip:135                cls._latest = fs_token136                cls._cache[fs_token] = obj137            return obj138 139 140class HfFileSystem(fsspec.AbstractFileSystem, metaclass=_Cached):  # ty: ignore[conflicting-metaclass]141    """142    Access a remote Hugging Face Hub repository as if were a local file system.143 144    > [!WARNING]145    > [`HfFileSystem`] provides fsspec compatibility, which is useful for libraries that require it (e.g., reading146    >     Hugging Face datasets directly with `pandas`). However, it introduces additional overhead due to this compatibility147    >     layer. For better performance and reliability, it's recommended to use `HfApi` methods when possible.148 149    The file system supports paths for the `hf://` protocol, which follows those URL schemes:150 151    * Models, Datasets and Spaces repositories:152 153        ```154        hf://<repo-id>[@<revision>]/<path/in/repo>155        hf://datasets/<repo-id>[@<revision>]/<path/in/repo>156        hf://spaces/<repo-id>[@<revision>]/<path/in/repo>157        ```158 159    * Buckets (generic storage):160 161        ```162        hf://buckets/<bucket-id>/<path/in/bucket>163        ```164 165    Note: when using the [`HfFileSystem`] directly, passing the `hf://` protocol prefix is optional in paths.166 167    Args:168        endpoint (`str`, *optional*):169                Endpoint of the Hub. Defaults to <https://huggingface.co>.170        token (`bool` or `str`, *optional*):171            A valid user access token (string). Defaults to the locally saved172            token, which is the recommended method for authentication (see173            https://huggingface.co/docs/huggingface_hub/quick-start#authentication).174            To disable authentication, pass `False`.175        block_size (`int`, *optional*):176            Block size for reading and writing files.177        expand_info (`bool`, *optional*):178            Whether to expand the information of the files.179        **storage_options (`dict`, *optional*):180            Additional options for the filesystem. See [fsspec documentation](https://filesystem-spec.readthedocs.io/en/latest/api.html#fsspec.spec.AbstractFileSystem.__init__).181 182    Usage:183 184    ```python185    >>> from huggingface_hub import hffs186 187    >>> # List files188    >>> hffs.glob("my-username/my-model/*.bin")189    ['my-username/my-model/pytorch_model.bin']190    >>> hffs.ls("datasets/my-username/my-dataset", detail=False)191    ['datasets/my-username/my-dataset/.gitattributes', 'datasets/my-username/my-dataset/README.md', 'datasets/my-username/my-dataset/data.json']192 193    >>> # Read/write files194    >>> with hffs.open("my-username/my-model/pytorch_model.bin") as f:195    ...     data = f.read()196    >>> with hffs.open("my-username/my-model/pytorch_model.bin", "wb") as f:197    ...     f.write(data)198    ```199 200    Specify a token for authentication:201    ```python202    >>> from huggingface_hub import HfFileSystem203    >>> hffs = HfFileSystem(token=token)204    ```205    """206 207    root_marker = ""208    protocol = "hf"209 210    def __init__(211        self,212        *args,213        endpoint: str | None = None,214        token: bool | str | None = None,215        block_size: int | None = None,216        expand_info: bool | None = None,217        **storage_options,218    ):219        super().__init__(*args, **storage_options)220        self.endpoint = endpoint or constants.ENDPOINT221        self.token = token222        self._api = HfApi(endpoint=endpoint, token=token)223        self.block_size = block_size224        self.expand_info = expand_info225        # Maps (repo_type, repo_id, revision) to a 2-tuple with:226        #  * the 1st element indicating whether the repository and the revision exist227        #  * the 2nd element being the exception raised if the repository or revision doesn't exist228        self._repo_and_revision_exists_cache: dict[tuple[str, str, str | None], tuple[bool, Exception | None]] = {}229        # Same for buckets230        self._bucket_exists_cache: dict[str, tuple[bool, Exception | None]] = {}231        # Note: special case for buckets: revision is always None232        # Maps parent directory path to path infos233        self.dircache: dict[str, list[dict[str, Any]]] = {}234 235    @classmethod236    def _tokenize(cls, threading_ident: int, *args, **kwargs) -> str:237        """Deterministic token for caching"""238        # make fs_token robust to default values and to kwargs order239        kwargs["endpoint"] = kwargs.get("endpoint") or constants.ENDPOINT240        kwargs["token"] = kwargs.get("token")241        kwargs = {key: kwargs[key] for key in sorted(kwargs)}242        # contrary to fsspec, we don't include pid here243        tokenize_args = (cls, threading_ident, args, kwargs)244        h = md5(str(tokenize_args).encode())245        return h.hexdigest()246 247    def _repo_and_revision_exist(248        self, repo_type: str, repo_id: str, revision: str | None249    ) -> tuple[bool, Exception | None]:250        if (repo_type, repo_id, revision) not in self._repo_and_revision_exists_cache:251            try:252                self._api.repo_info(253                    repo_id, revision=revision, repo_type=repo_type, timeout=constants.HF_HUB_ETAG_TIMEOUT254                )255            except (RepositoryNotFoundError, HFValidationError) as e:256                self._repo_and_revision_exists_cache[(repo_type, repo_id, revision)] = False, e257                self._repo_and_revision_exists_cache[(repo_type, repo_id, None)] = False, e258            except RevisionNotFoundError as e:259                self._repo_and_revision_exists_cache[(repo_type, repo_id, revision)] = False, e260                self._repo_and_revision_exists_cache[(repo_type, repo_id, None)] = True, None261            else:262                self._repo_and_revision_exists_cache[(repo_type, repo_id, revision)] = True, None263                self._repo_and_revision_exists_cache[(repo_type, repo_id, None)] = True, None264        return self._repo_and_revision_exists_cache[(repo_type, repo_id, revision)]265 266    def _bucket_exists(self, bucket_id: str) -> tuple[bool, Exception | None]:267        if bucket_id not in self._bucket_exists_cache:268            try:269                self._api.bucket_info(bucket_id)270            except BucketNotFoundError as e:271                self._bucket_exists_cache[bucket_id] = False, e272            else:273                self._bucket_exists_cache[bucket_id] = True, None274        return self._bucket_exists_cache[bucket_id]275 276    def resolve_path(277        self, path: str, revision: str | None = None278    ) -> HfFileSystemResolvedRepositoryPath | HfFileSystemResolvedBucketPath:279        """280        Resolve a Hugging Face file system path into its components.281 282        Args:283            path (`str`):284                Path to resolve.285            revision (`str`, *optional*):286                The revision of the repo to resolve. Defaults to the revision specified in the path.287 288        Returns:289            [`HfFileSystemResolvedPath`]: Resolved path information containing `repo_type`, `repo_id`, `revision` and `path_in_repo`.290 291        Raises:292            `ValueError`:293                If path contains conflicting revision information.294            `NotImplementedError`:295                If trying to list repositories.296        """297 298        def _align_revision_in_path_with_revision(revision_in_path: str | None, revision: str | None) -> str | None:299            if revision is not None:300                if revision_in_path is not None and revision_in_path != revision:301                    raise ValueError(302                        f'Revision specified in path ("{revision_in_path}") and in `revision` argument ("{revision}")'303                        " are not the same."304                    )305            else:306                revision = revision_in_path307            return revision308 309        path = self._strip_protocol(path)310        if not path:311            # can't list repositories at root312            raise NotImplementedError("Access to buckets and repositories lists is not implemented.")313        elif path.split("/")[0] == "buckets":314            bucket_id = "/".join(path.split("/")[1:3])315            path = "/".join(path.split("/")[3:])316            bucket_exists, err = self._bucket_exists(bucket_id)317            if not bucket_exists:318                _raise_file_not_found(path, err)319            return HfFileSystemResolvedBucketPath(bucket_id=bucket_id, path=path)320        elif path.split("/")[0] + "/" in constants.REPO_TYPES_URL_PREFIXES.values():321            if "/" not in path:322                # can't list repositories at the repository type level323                raise NotImplementedError("Access to repositories lists is not implemented.")324            repo_type, path = path.split("/", 1)325            repo_type = constants.REPO_TYPES_MAPPING[repo_type]326        else:327            repo_type = constants.REPO_TYPE_MODEL328        if path.count("/") > 0:329            if "@" in "/".join(path.split("/")[:2]):330                repo_id, revision_in_path = path.split("@", 1)331                if "/" in revision_in_path:332                    match = SPECIAL_REFS_REVISION_REGEX.search(revision_in_path)333                    if match is not None and revision in (None, match.group()):334                        # Handle `refs/convert/parquet` and PR revisions separately335                        path_in_repo = SPECIAL_REFS_REVISION_REGEX.sub("", revision_in_path).lstrip("/")336                        revision_in_path = match.group()337                    else:338                        revision_in_path, path_in_repo = revision_in_path.split("/", 1)339                else:340                    path_in_repo = ""341                revision = _align_revision_in_path_with_revision(unquote(revision_in_path), revision)342                repo_and_revision_exist, err = self._repo_and_revision_exist(repo_type, repo_id, revision)343                if not repo_and_revision_exist:344                    _raise_file_not_found(path, err)345            else:346                revision_in_path = None347                repo_id_with_namespace = "/".join(path.split("/")[:2])348                path_in_repo_with_namespace = "/".join(path.split("/")[2:])349                repo_id_without_namespace = path.split("/")[0]350                path_in_repo_without_namespace = "/".join(path.split("/")[1:])351                repo_id = repo_id_with_namespace352                path_in_repo = path_in_repo_with_namespace353                repo_and_revision_exist, err = self._repo_and_revision_exist(repo_type, repo_id, revision)354                if not repo_and_revision_exist:355                    if isinstance(err, (RepositoryNotFoundError, HFValidationError)):356                        repo_id = repo_id_without_namespace357                        path_in_repo = path_in_repo_without_namespace358                        repo_and_revision_exist, _ = self._repo_and_revision_exist(repo_type, repo_id, revision)359                        if not repo_and_revision_exist:360                            _raise_file_not_found(path, err)361                    else:362                        _raise_file_not_found(path, err)363        else:364            repo_id = path365            path_in_repo = ""366            if "@" in path:367                repo_id, revision_in_path = path.split("@", 1)368                revision = _align_revision_in_path_with_revision(unquote(revision_in_path), revision)369            else:370                revision_in_path = None371            repo_and_revision_exist, _ = self._repo_and_revision_exist(repo_type, repo_id, revision)372            if not repo_and_revision_exist:373                raise NotImplementedError("Access to repositories lists is not implemented.")374 375        revision = revision if revision is not None else constants.DEFAULT_REVISION376        return HfFileSystemResolvedRepositoryPath(377            repo_type, repo_id, revision, path_in_repo, _raw_revision=revision_in_path378        )379 380    def invalidate_cache(self, path: str | None = None) -> None:381        """382        Clear the cache for a given path.383 384        For more details, refer to [fsspec documentation](https://filesystem-spec.readthedocs.io/en/latest/api.html#fsspec.spec.AbstractFileSystem.invalidate_cache).385 386        Args:387            path (`str`, *optional*):388                Path to clear from cache. If not provided, clear the entire cache.389 390        """391        if not path:392            self.dircache.clear()393            self._repo_and_revision_exists_cache.clear()394        else:395            resolved_path = self.resolve_path(path)396            path = resolved_path.unresolve()397            while path:398                self.dircache.pop(path, None)399                path = self._parent(path)400 401            # Only clear repo cache if path is to repo root402            if not resolved_path.path:403                if isinstance(resolved_path, HfFileSystemResolvedRepositoryPath):404                    self._repo_and_revision_exists_cache.pop(405                        (resolved_path.repo_type, resolved_path.repo_id, None), None406                    )407                    self._repo_and_revision_exists_cache.pop(408                        (resolved_path.repo_type, resolved_path.repo_id, resolved_path.revision), None409                    )410                else:411                    self._bucket_exists_cache.pop(resolved_path.bucket_id, None)412 413    def _open(  # type: ignore414        self,415        path: str,416        mode: str = "rb",417        block_size: int | None = None,418        revision: str | None = None,419        **kwargs,420    ) -> Union["HfFileSystemFile", "HfFileSystemStreamFile"]:421        block_size = block_size if block_size is not None else self.block_size422        if block_size is not None:423            kwargs["block_size"] = block_size424        if "a" in mode:425            raise NotImplementedError("Appending to remote files is not yet supported.")426        if block_size == 0:427            return HfFileSystemStreamFile(self, path, mode=mode, revision=revision, **kwargs)428        else:429            return HfFileSystemFile(self, path, mode=mode, revision=revision, **kwargs)430 431    def _rm(self, path: str, revision: str | None = None, **kwargs) -> None:432        resolved_path = self.resolve_path(path, revision=revision)433        if isinstance(resolved_path, HfFileSystemResolvedBucketPath):434            self._api.batch_bucket_files(resolved_path.bucket_id, delete=[resolved_path.path])435        else:436            self._api.delete_file(437                path_in_repo=resolved_path.path_in_repo,438                repo_id=resolved_path.repo_id,439                token=self.token,440                repo_type=resolved_path.repo_type,441                revision=resolved_path.revision,442                commit_message=kwargs.get("commit_message"),443                commit_description=kwargs.get("commit_description"),444            )445        self.invalidate_cache(path=resolved_path.unresolve())446 447    def rm(448        self,449        path: str,450        recursive: bool = False,451        maxdepth: int | None = None,452        revision: str | None = None,453        **kwargs,454    ) -> None:455        """456        Delete files from a repository.457 458        For more details, refer to [fsspec documentation](https://filesystem-spec.readthedocs.io/en/latest/api.html#fsspec.spec.AbstractFileSystem.rm).459 460        > [!WARNING]461        > Note: When possible, use `HfApi.delete_file()` for better performance.462 463        Args:464            path (`str`):465                Path to delete.466            recursive (`bool`, *optional*):467                If True, delete directory and all its contents. Defaults to False.468            maxdepth (`int`, *optional*):469                Maximum number of subdirectories to visit when deleting recursively.470            revision (`str`, *optional*):471                The git revision to delete from.472 473        """474        resolved_path = self.resolve_path(path, revision=revision)475        paths = self.expand_path(path, recursive=recursive, maxdepth=maxdepth, revision=revision)476        if isinstance(resolved_path, HfFileSystemResolvedBucketPath):477            delete = [self.resolve_path(path).path for path in paths if not self.isdir(path)]478            self._api.batch_bucket_files(resolved_path.bucket_id, delete=delete)479        else:480            paths_in_repo = [self.resolve_path(path).path for path in paths if not self.isdir(path)]481            operations = [CommitOperationDelete(path_in_repo=path_in_repo) for path_in_repo in paths_in_repo]482            commit_message = f"Delete {path} "483            commit_message += "recursively " if recursive else ""484            commit_message += f"up to depth {maxdepth} " if maxdepth is not None else ""485            # TODO: use `commit_description` to list all the deleted paths?486            self._api.create_commit(487                repo_id=resolved_path.repo_id,488                repo_type=resolved_path.repo_type,489                token=self.token,490                operations=operations,491                revision=resolved_path.revision,492                commit_message=kwargs.get("commit_message", commit_message),493                commit_description=kwargs.get("commit_description"),494            )495        self.invalidate_cache(path=resolved_path.unresolve())496 497    def ls(498        self, path: str, detail: bool = True, refresh: bool = False, revision: str | None = None, **kwargs499    ) -> list[str | dict[str, Any]]:500        """501        List the contents of a directory.502 503        For more details, refer to [fsspec documentation](https://filesystem-spec.readthedocs.io/en/latest/api.html#fsspec.spec.AbstractFileSystem.ls).504 505        > [!WARNING]506        > Note: When possible, use `HfApi.list_repo_tree()` for better performance.507 508        Args:509            path (`str`):510                Path to the directory.511            detail (`bool`, *optional*):512                If True, returns a list of dictionaries containing file information. If False,513                returns a list of file paths. Defaults to True.514            refresh (`bool`, *optional*):515                If True, bypass the cache and fetch the latest data. Defaults to False.516            revision (`str`, *optional*):517                The git revision to list from.518 519        Returns:520            `list[Union[str, dict[str, Any]]]`: List of file paths (if detail=False) or list of file information521            dictionaries (if detail=True).522        """523        resolved_path = self.resolve_path(path, revision=revision)524        path = resolved_path.unresolve()525        try:526            out = self._ls_tree(path, refresh=refresh, revision=revision, **kwargs)527        except EntryNotFoundError:528            # Path could be a file529            if not resolved_path.path:530                _raise_file_not_found(path, None)531            try:532                out = self._ls_tree(self._parent(path), refresh=refresh, revision=revision, **kwargs)533            except EntryNotFoundError:534                out = []535            out = [o for o in out if o["name"] == path]536            if len(out) == 0:537                _raise_file_not_found(path, None)538        return out if detail else [o["name"] for o in out]539 540    def _ls_tree(541        self,542        path: str,543        recursive: bool = False,544        refresh: bool = False,545        revision: str | None = None,546        expand_info: bool | None = None,547        maxdepth: int | None = None,548    ):549        expand_info = (550            expand_info if expand_info is not None else (self.expand_info if self.expand_info is not None else False)551        )552        resolved_path = self.resolve_path(path, revision=revision)553        path = resolved_path.unresolve()554        root_path = resolved_path.root555        maxdepth = maxdepth if recursive else 1556 557        out = []558        if path in self.dircache and not refresh:559            cached_path_infos = self.dircache[path]560            out.extend(cached_path_infos)561            dirs_not_in_dircache = []562            if recursive:563                # Use BFS to traverse the cache and build the "recursive "output564                # (The Hub uses a so-called "tree first" strategy for the tree endpoint but we sort the output to follow the spec so the result is (eventually) the same)565                depth = 2566                dirs_to_visit = deque(567                    [(depth, path_info) for path_info in cached_path_infos if path_info["type"] == "directory"]568                )569                while dirs_to_visit:570                    depth, dir_info = dirs_to_visit.popleft()571                    if maxdepth is None or depth <= maxdepth:572                        if dir_info["name"] not in self.dircache:573                            dirs_not_in_dircache.append(dir_info["name"])574                        else:575                            cached_path_infos = self.dircache[dir_info["name"]]576                            out.extend(cached_path_infos)577                            dirs_to_visit.extend(578                                [579                                    (depth + 1, path_info)580                                    for path_info in cached_path_infos581                                    if path_info["type"] == "directory"582                                ]583                            )584 585            dirs_not_expanded = []586            if expand_info and isinstance(resolved_path, HfFileSystemResolvedRepositoryPath):587                # Check if there are directories in repos with non-expanded entries588                dirs_not_expanded = [self._parent(o["name"]) for o in out if o["last_commit"] is None]589 590            if (recursive and dirs_not_in_dircache) or (expand_info and dirs_not_expanded):591                # If the dircache is incomplete, find the common path of the missing and non-expanded entries592                # and extend the output with the result of `_ls_tree(common_path, recursive=True)`593                common_prefix = os.path.commonprefix(dirs_not_in_dircache + dirs_not_expanded)594                # Get the parent directory if the common prefix itself is not a directory595                common_path = (596                    common_prefix.rstrip("/")597                    if common_prefix.endswith("/")598                    or common_prefix == root_path599                    or common_prefix in chain(dirs_not_in_dircache, dirs_not_expanded)600                    else self._parent(common_prefix)601                )602                if maxdepth is not None:603                    common_path_depth = common_path[len(path) :].count("/")604                    maxdepth -= common_path_depth605                out = [o for o in out if not o["name"].startswith(common_path + "/")]606                for cached_path in list(self.dircache):607                    if cached_path.startswith(common_path + "/"):608                        self.dircache.pop(cached_path, None)609                self.dircache.pop(common_path, None)610                out.extend(611                    self._ls_tree(612                        common_path,613                        recursive=recursive,614                        refresh=True,615                        revision=revision,616                        expand_info=expand_info,617                        maxdepth=maxdepth,618                    )619                )620        else:621            tree: Iterable[RepoFile | RepoFolder | BucketFile | BucketFolder]622            if isinstance(resolved_path, HfFileSystemResolvedBucketPath):623                tree = self._list_bucket_tree_with_folders(624                    resolved_path.bucket_id,625                    prefix=resolved_path.path,626                    recursive=recursive,627                )628            else:629                tree = self._api.list_repo_tree(630                    resolved_path.repo_id,631                    resolved_path.path,632                    recursive=recursive,633                    expand=expand_info,634                    revision=resolved_path.revision,635                    repo_type=resolved_path.repo_type,636                )637            for path_info in tree:638                cache_path = root_path + "/" + path_info.path639                if isinstance(path_info, RepoFile):640                    cache_path_info = {641                        "name": cache_path,642                        "size": path_info.size,643                        "type": "file",644                        "blob_id": path_info.blob_id,645                        "lfs": path_info.lfs,646                        "xet_hash": path_info.xet_hash,647                        "last_commit": path_info.last_commit,648                        "security": path_info.security,649                    }650                elif isinstance(path_info, BucketFile):651                    cache_path_info = {652                        "name": cache_path,653                        "size": path_info.size,654                        "type": "file",655                        "xet_hash": path_info.xet_hash,656                        "mtime": path_info.mtime,657                        "uploaded_at": path_info.uploaded_at,658                    }659                elif isinstance(path_info, RepoFolder):660                    cache_path_info = {661                        "name": cache_path,662                        "size": 0,663                        "type": "directory",664                        "tree_id": path_info.tree_id,665                        "last_commit": path_info.last_commit,666                    }667                else:668                    cache_path_info = {669                        "name": cache_path,670                        "size": 0,671                        "type": "directory",672                        "uploaded_at": path_info.uploaded_at,673                    }674                parent_path = self._parent(cache_path_info["name"])675                self.dircache.setdefault(parent_path, []).append(cache_path_info)676                depth = cache_path[len(path) :].count("/")677                if maxdepth is None or depth <= maxdepth:678                    out.append(cache_path_info)679        return out680 681    def _list_bucket_tree_with_folders(682        self, bucket_id: str, prefix: str, recursive: bool683    ) -> Iterable[BucketFile | BucketFolder]:684        """Same as `HfApi.list_bucket_tree` but always includes folders"""685        bucket_files = self._api.list_bucket_tree(bucket_id, prefix, recursive=recursive)686        bucket_folders: dict[str, BucketFolder] = {}687        min_depth = 1 + prefix.count("/") if prefix else 0688        out: list[BucketFile | BucketFolder] = []689 690        for bucket_entry in bucket_files:691            out.append(bucket_entry)692 693            # If recursive=False, both files and folders are returned by the server => nothing to do694            if not recursive:695                continue696 697            # Otherwise, let's rebuild BucketFolders manually698            for parent_bucket_folder_str in list(PurePosixPath(bucket_entry.path).parents)[: -min_depth - 1]:699                parent_bucket_folder = BucketFolder(700                    type="directory", path=str(parent_bucket_folder_str), uploaded_at=bucket_entry.uploaded_at701                )702 703                # If folder not visited yet, add it704                if parent_bucket_folder.path not in bucket_folders:705                    out.append(parent_bucket_folder)706                    bucket_folders[parent_bucket_folder.path] = parent_bucket_folder707                    continue708 709                # Otherwise, get back BucketFolder object and update its 'uploaded_at'710                if parent_bucket_folder.uploaded_at is not None:711                    bucket_folder = bucket_folders[parent_bucket_folder.path]712                    if bucket_folder.uploaded_at is None or (713                        bucket_folder.uploaded_at < parent_bucket_folder.uploaded_at714                    ):715                        bucket_folder.uploaded_at = parent_bucket_folder.uploaded_at716 717        if not out:718            raise EntryNotFoundError(f"File not found in bucket '{bucket_id}': '{prefix}'")719        return out720 721    def walk(self, path: str, *args, **kwargs) -> Iterator[tuple[str, list[str], list[str]]]:722        """723        Return all files below the given path.724 725        For more details, refer to [fsspec documentation](https://filesystem-spec.readthedocs.io/en/latest/api.html#fsspec.spec.AbstractFileSystem.walk).726 727        Args:728            path (`str`):729                Root path to list files from.730 731        Returns:732            `Iterator[tuple[str, list[str], list[str]]]`: An iterator of (path, list of directory names, list of file names) tuples.733        """734        path = self.resolve_path(path, revision=kwargs.get("revision")).unresolve()735        yield from super().walk(path, *args, **kwargs)736 737    def glob(self, path: str, maxdepth: int | None = None, **kwargs) -> list[str]:738        """739        Find files by glob-matching.740 741        For more details, refer to [fsspec documentation](https://filesystem-spec.readthedocs.io/en/latest/api.html#fsspec.spec.AbstractFileSystem.glob).742 743        Args:744            path (`str`):745                Path pattern to match.746 747        Returns:748            `list[str]`: List of paths matching the pattern.749        """750        path = self.resolve_path(path, revision=kwargs.get("revision")).unresolve()751        return super().glob(path, maxdepth=maxdepth, **kwargs)752 753    def find(754        self,755        path: str,756        maxdepth: int | None = None,757        withdirs: bool = False,758        detail: bool = False,759        refresh: bool = False,760        revision: str | None = None,761        **kwargs,762    ) -> list[str] | dict[str, dict[str, Any]]:763        """764        List all files below path.765 766        For more details, refer to [fsspec documentation](https://filesystem-spec.readthedocs.io/en/latest/api.html#fsspec.spec.AbstractFileSystem.find).767 768        Args:769            path (`str`):770                Root path to list files from.771            maxdepth (`int`, *optional*):772                Maximum depth to descend into subdirectories.773            withdirs (`bool`, *optional*):774                Include directory paths in the output. Defaults to False.775            detail (`bool`, *optional*):776                If True, returns a dict mapping paths to file information. Defaults to False.777            refresh (`bool`, *optional*):778                If True, bypass the cache and fetch the latest data. Defaults to False.779            revision (`str`, *optional*):780                The git revision to list from.781 782        Returns:783            `Union[list[str], dict[str, dict[str, Any]]]`: List of paths or dict of file information.784        """785        if maxdepth is not None and maxdepth < 1:786            raise ValueError("maxdepth must be at least 1")787        resolved_path = self.resolve_path(path, revision=revision)788        path = resolved_path.unresolve()789        try:790            out = self._ls_tree(path, recursive=True, refresh=refresh, maxdepth=maxdepth, **kwargs)791        except EntryNotFoundError:792            # Path could be a file793            try:794                if self.info(path, revision=revision, **kwargs)["type"] == "file":795                    out = {path: {}}796                else:797                    out = {}798            except FileNotFoundError:799                out = {}800        else:801            if not withdirs:802                out = [o for o in out if o["type"] != "directory"]803            else:804                # If `withdirs=True`, include the directory itself to be consistent with the spec805                path_info = self.info(path, **kwargs)806                out = [path_info] + out if path_info["type"] == "directory" else out807            out = {o["name"]: o for o in out}808        names = sorted(out)809        if not detail:810            return names811        else:812            return {name: out[name] for name in names}813 814    def cp_file(self, path1: str, path2: str, revision: str | None = None, **kwargs) -> None:815        """816        Copy a file within or between repositories.817 818        > [!WARNING]819        > Note: When possible, use `HfApi.upload_file()` for better performance.820 821        Args:822            path1 (`str`):823                Source path to copy from.824            path2 (`str`):825                Destination path to copy to.826            revision (`str`, *optional*):827                The git revision to copy from.828 829        """830        resolved_path1 = self.resolve_path(path1, revision=revision)831        resolved_path2 = self.resolve_path(path2, revision=revision)832        if isinstance(resolved_path1, HfFileSystemResolvedBucketPath) or isinstance(833            resolved_path2, HfFileSystemResolvedBucketPath834        ):835            raise NotImplementedError("Copy from/to buckets is not available yet")836 837        same_repo = (838            resolved_path1.repo_type == resolved_path2.repo_type and resolved_path1.repo_id == resolved_path2.repo_id839        )840 841        if same_repo:842            commit_message = f"Copy {path1} to {path2}"843            self._api.create_commit(844                repo_id=resolved_path1.repo_id,845                repo_type=resolved_path1.repo_type,846                revision=resolved_path2.revision,847                commit_message=kwargs.get("commit_message", commit_message),848                commit_description=kwargs.get("commit_description", ""),849                operations=[850                    CommitOperationCopy(851                        src_path_in_repo=resolved_path1.path_in_repo,852                        path_in_repo=resolved_path2.path_in_repo,853                        src_revision=resolved_path1.revision,854                    )855                ],856            )857        else:858            with self.open(path1, "rb", revision=resolved_path1.revision) as f:859                content = f.read()860            commit_message = f"Copy {path1} to {path2}"861            self._api.upload_file(862                path_or_fileobj=content,863                path_in_repo=resolved_path2.path_in_repo,864                repo_id=resolved_path2.repo_id,865                token=self.token,866                repo_type=resolved_path2.repo_type,867                revision=resolved_path2.revision,868                commit_message=kwargs.get("commit_message", commit_message),869                commit_description=kwargs.get("commit_description"),870            )871        self.invalidate_cache(path=resolved_path1.unresolve())872        self.invalidate_cache(path=resolved_path2.unresolve())873 874    def modified(self, path: str, **kwargs) -> datetime:875        """876        Get the last modified time of a file.877 878        For more details, refer to [fsspec documentation](https://filesystem-spec.readthedocs.io/en/latest/api.html#fsspec.spec.AbstractFileSystem.modified).879 880        Args:881            path (`str`):882                Path to the file.883 884        Returns:885            `datetime`: Last modified time of the file.886        """887        info = self.info(path, **{**kwargs, "expand_info": True})  # type: ignore888        if "last_commit" in info:889            if info["last_commit"] is None:890                raise NotImplementedError(f"'modified' is not implemented for repository paths like '{path}'")891            return info["last_commit"].date892        elif "mtime" in info and info["mtime"]:893            return info["mtime"]894        elif "uploaded_at" in info and info["uploaded_at"]:895            return info["uploaded_at"]896        else:897            raise NotImplementedError(f"Cannot determined 'modified' for path '{path}' (info: {info})")898 899    def info(self, path: str, refresh: bool = False, revision: str | None = None, **kwargs) -> dict[str, Any]:900        """901        Get information about a file or directory.902 903        For more details, refer to [fsspec documentation](https://filesystem-spec.readthedocs.io/en/latest/api.html#fsspec.spec.AbstractFileSystem.info).904 905        > [!WARNING]906        > Note: When possible, use `HfApi.get_paths_info()` or `HfApi.repo_info()`  for better performance907        > (or `HfApi.get_bucket_paths_info()` or `HfApi.bucket_info()` for buckets)908 909        Args:910            path (`str`):911                Path to get info for.912            refresh (`bool`, *optional*):913                If True, bypass the cache and fetch the latest data. Defaults to False.914            revision (`str`, *optional*):915                The git revision to get info from.916 917        Returns:918            `dict[str, Any]`: Dictionary containing file information (type, size, commit info, etc.).919 920        """921        resolved_path = self.resolve_path(path, revision=revision)922        path = resolved_path.unresolve()923        expand_info = kwargs.get(924            "expand_info", self.expand_info if self.expand_info is not None else False925        )  # don't expose it as a parameter in the public API to follow the spec926        out: dict[str, Any] | None927        if not resolved_path.path:928            # Path is the root directory929            out = {930                "name": path,931                "size": 0,932                "type": "directory",933            }934            if isinstance(resolved_path, HfFileSystemResolvedRepositoryPath):935                out["last_commit"] = None936            if isinstance(resolved_path, HfFileSystemResolvedRepositoryPath) and expand_info:937                last_commit = self._api.list_repo_commits(938                    resolved_path.repo_id, repo_type=resolved_path.repo_type, revision=resolved_path.revision939                )[-1]940                out = {941                    **out,942                    "tree_id": None,  # TODO: tree_id of the root directory?943                    "last_commit": LastCommitInfo(944                        oid=last_commit.commit_id, title=last_commit.title, date=last_commit.created_at945                    ),946                }947        elif isinstance(resolved_path, HfFileSystemResolvedBucketPath):948            parent_path = self._parent(path)949            # Fill the cache with cheap call950            self.ls(parent_path, refresh=refresh)951            out1 = [o for o in self.dircache[parent_path] if o["name"] == path]952            if not out1:953                _raise_file_not_found(path, None)954            out = out1[0]955        else:956            out = None957            parent_path = self._parent(path)958            if not expand_info and parent_path not in self.dircache:959                # Fill the cache with cheap call960                self.ls(parent_path)961            if parent_path in self.dircache:962                # Check if the path is in the cache963                out1 = [o for o in self.dircache[parent_path] if o["name"] == path]964                if not out1:965                    _raise_file_not_found(path, None)966                out = out1[0]967            if refresh or out is None or (expand_info and out and out["last_commit"] is None):968                paths_info = self._api.get_paths_info(969                    resolved_path.repo_id,970                    resolved_path.path_in_repo,971                    expand=expand_info,972                    revision=resolved_path.revision,973                    repo_type=resolved_path.repo_type,974                )975                if not paths_info:976                    _raise_file_not_found(path, None)977                path_info = paths_info[0]978                root_path = HfFileSystemResolvedRepositoryPath(979                    resolved_path.repo_type,980                    resolved_path.repo_id,981                    resolved_path.revision,982                    path_in_repo="",983                    _raw_revision=resolved_path._raw_revision,984                ).unresolve()985                if isinstance(path_info, RepoFile):986                    out = {987                        "name": root_path + "/" + path_info.path,988                        "size": path_info.size,989                        "type": "file",990                        "blob_id": path_info.blob_id,991                        "lfs": path_info.lfs,992                        "xet_hash": path_info.xet_hash,993                        "last_commit": path_info.last_commit,994                        "security": path_info.security,995                    }996                else:997                    out = {998                        "name": root_path + "/" + path_info.path,999                        "size": 0,1000                        "type": "directory",1001                        "tree_id": path_info.tree_id,1002                        "last_commit": path_info.last_commit,1003                    }1004                if not expand_info:1005                    out = {k: out[k] for k in ["name", "size", "type"]}1006        assert out is not None1007        return out1008 1009    def exists(self, path, **kwargs):1010        """1011        Check if a file exists.1012 1013        For more details, refer to [fsspec documentation](https://filesystem-spec.readthedocs.io/en/latest/api.html#fsspec.spec.AbstractFileSystem.exists).1014 1015        > [!WARNING]1016        > Note: When possible, use `HfApi.file_exists()` for better performance.1017 1018        Args:1019            path (`str`):1020                Path to check.1021 1022        Returns:1023            `bool`: True if file exists, False otherwise.1024        """1025        try:1026            if kwargs.get("refresh", False):1027                self.invalidate_cache(path)1028 1029            self.info(path, **kwargs)1030            return True1031        except OSError:1032            return False1033 1034    def isdir(self, path):1035        """1036        Check if a path is a directory.1037 1038        For more details, refer to [fsspec documentation](https://filesystem-spec.readthedocs.io/en/latest/api.html#fsspec.spec.AbstractFileSystem.isdir).1039 1040        Args:1041            path (`str`):1042                Path to check.1043 1044        Returns:1045            `bool`: True if path is a directory, False otherwise.1046        """1047        try:1048            return self.info(path)["type"] == "directory"1049        except OSError:1050            return False1051 1052    def isfile(self, path):1053        """1054        Check if a path is a file.1055 1056        For more details, refer to [fsspec documentation](https://filesystem-spec.readthedocs.io/en/latest/api.html#fsspec.spec.AbstractFileSystem.isfile).1057 1058        Args:1059            path (`str`):1060                Path to check.1061 1062        Returns:1063            `bool`: True if path is a file, False otherwise.1064        """1065        try:1066            return self.info(path)["type"] == "file"1067        except OSError:1068            return False1069 1070    def url(self, path: str) -> str:1071        """1072        Get the HTTP URL of the given path.1073 1074        Args:1075            path (`str`):1076                Path to get URL for.1077 1078        Returns:1079            `str`: HTTP URL to access the file or directory on the Hub.1080        """1081        resolved_path = self.resolve_path(path)1082        if isinstance(resolved_path, HfFileSystemResolvedBucketPath):1083            url = f"{self.endpoint}/buckets/{resolved_path.bucket_id}/resolve/{quote(resolved_path.path)}"1084        else:1085            url = hf_hub_url(1086                resolved_path.repo_id,1087                resolved_path.path_in_repo,1088                repo_type=resolved_path.repo_type,1089                revision=resolved_path.revision,1090                endpoint=self.endpoint,1091            )1092        if self.isdir(path):1093            url = url.replace("/resolve/", "/tree/", 1)1094        return url1095 1096    def get_file(self, rpath, lpath, callback=_DEFAULT_CALLBACK, outfile=None, **kwargs) -> None:1097        """1098        Copy single remote file to local.1099 1100        > [!WARNING]1101        > Note: When possible, use `HfApi.hf_hub_download()` or `HfApi.download_bucket_files` for better performance.1102 1103        Args:1104            rpath (`str`):1105                Remote path to download from.1106            lpath (`str`):1107                Local path to download to.1108            callback (`Callback`, *optional*):1109                Optional callback to track download progress. Defaults to no callback.1110            outfile (`IO`, *optional*):1111                Optional file-like object to write to. If provided, `lpath` is ignored.1112 1113        """1114        revision = kwargs.get("revision")1115        unhandled_kwargs = set(kwargs.keys()) - {"revision"}1116        if not isinstance(callback, (NoOpCallback, TqdmCallback)) or len(unhandled_kwargs) > 0:1117            # for now, let's not handle custom callbacks1118            # and let's not handle custom kwargs1119            return super().get_file(rpath, lpath, callback=callback, outfile=outfile, **kwargs)1120 1121        # Taken from https://github.com/fsspec/filesystem_spec/blob/47b445ae4c284a82dd15e0287b1ffc410e8fc470/fsspec/spec.py#L8831122        if isfilelike(lpath):1123            outfile = lpath1124        elif self.isdir(rpath):1125            os.makedirs(lpath, exist_ok=True)1126            return None1127 1128        if isinstance(lpath, (str, Path)):  # otherwise, let's assume it's a file-like object1129            os.makedirs(os.path.dirname(lpath), exist_ok=True)1130 1131        # Open file if not already open1132        close_file = False1133        if outfile is None:1134            outfile = open(lpath, "wb")1135            close_file = True1136        initial_pos = outfile.tell()1137 1138        # Custom implementation of `get_file` to use `http_get`.1139        resolve_remote_path = self.resolve_path(rpath, revision=revision)1140        expected_size = self.info(rpath, revision=revision)["size"]1141        callback.set_size(expected_size)1142        try:1143            http_get(1144                url=self.url(resolve_remote_path.unresolve()),1145                temp_file=outfile,  # type: ignore1146                displayed_filename=rpath,1147                expected_size=expected_size,1148                resume_size=0,1149                headers=self._api._build_hf_headers(),1150                _tqdm_bar=callback.tqdm if isinstance(callback, TqdmCallback) else None,1151            )1152            outfile.seek(initial_pos)1153        finally:1154            # Close file only if we opened it ourselves1155            if close_file:1156                outfile.close()1157 1158    @property1159    def transaction(self):1160        """A context within which files are committed together upon exit1161 1162        Requires the file class to implement `.commit()` and `.discard()`1163        for the normal and exception cases.1164        """1165        # Taken from https://github.com/fsspec/filesystem_spec/blob/3fbb6fee33b46cccb015607630843dea049d3243/fsspec/spec.py#L2311166        # See https://github.com/huggingface/huggingface_hub/issues/17331167        raise NotImplementedError("Transactional commits are not supported.")1168 1169    def start_transaction(self):1170        """Begin write transaction for deferring files, non-context version"""1171        # Taken from https://github.com/fsspec/filesystem_spec/blob/3fbb6fee33b46cccb015607630843dea049d3243/fsspec/spec.py#L2411172        # See https://github.com/huggingface/huggingface_hub/issues/17331173        raise NotImplementedError("Transactional commits are not supported.")1174 1175    def __reduce__(self):1176        # re-populate the instance cache at HfFileSystem._cache and re-populate the state of every instance1177        return make_instance, (1178            type(self),1179            self.storage_args,1180            self.storage_options,1181            self._get_instance_state(),1182        )1183 1184    def _get_instance_state(self):1185        return {1186            "dircache": deepcopy(self.dircache),1187            "_repo_and_revision_exists_cache": deepcopy(self._repo_and_revision_exists_cache),1188            "_bucket_exists_cache": deepcopy(self._bucket_exists_cache),1189        }1190 1191 1192class HfFileSystemFile(fsspec.spec.AbstractBufferedFile):1193    def __init__(self, fs: HfFileSystem, path: str, revision: str | None = None, **kwargs):1194        try:1195            self.resolved_path = fs.resolve_path(path, revision=revision)1196        except FileNotFoundError as e:1197            if "w" in kwargs.get("mode", ""):1198                raise FileNotFoundError(1199                    f"{e}.\nMake sure the repository and revision exist before writing data."1200                ) from e

Showing the first 1,200 of 1447 lines. Download the file for the rest.

codekingpro/portable-devtools · Team Ai