codekingpro/portable-devtools
114k
1import os2import tempfile3import threading4from collections import deque5from collections.abc import Iterable, Iterator6from contextlib import ExitStack7from copy import deepcopy8from dataclasses import dataclass, field9from datetime import datetime10from itertools import chain11from pathlib import Path, PurePosixPath12from typing import Any, NoReturn, Union13from urllib.parse import quote, unquote14 15import fsspec16import httpx17from fsspec.callbacks import _DEFAULT_CALLBACK, NoOpCallback, TqdmCallback18from fsspec.config import apply_config19from fsspec.utils import isfilelike20 21from . import constants22from ._commit_api import CommitOperationCopy, CommitOperationDelete23from .errors import (24 BucketNotFoundError,25 EntryNotFoundError,26 HfHubHTTPError,27 RepositoryNotFoundError,28 RevisionNotFoundError,29)30from .file_download import hf_hub_url, http_get31from .hf_api import SPECIAL_REFS_REVISION_REGEX, BucketFile, BucketFolder, HfApi, LastCommitInfo, RepoFile, RepoFolder32from .utils import HFValidationError, hf_raise_for_status, http_backoff, http_stream_backoff33from .utils.insecure_hashlib import md534 35 36@dataclass37class HfFileSystemResolvedPath:38 """Top level Data structure containing information about a resolved Hugging Face file system path."""39 40 root: str41 path: str42 43 def unresolve(self) -> str:44 return f"{self.root}/{self.path}".rstrip("/")45 46 47@dataclass48class HfFileSystemResolvedRepositoryPath(HfFileSystemResolvedPath):49 """Data structure containing information about a resolved path in a repository."""50 51 repo_type: str52 repo_id: str53 revision: str54 path_in_repo: str55 root: str = field(init=False)56 path: str = field(init=False)57 # The part placed after '@' in the initial path. It can be a quoted or unquoted refs revision.58 # Used to reconstruct the unresolved path to return to the user.59 _raw_revision: str | None = field(default=None, repr=False)60 61 def __post_init__(self):62 repo_path = constants.REPO_TYPES_URL_PREFIXES.get(self.repo_type, "") + self.repo_id63 if self._raw_revision:64 self.root = f"{repo_path}@{self._raw_revision}"65 elif self.revision != constants.DEFAULT_REVISION:66 self.root = f"{repo_path}@{safe_revision(self.revision)}"67 else:68 self.root = repo_path69 self.path = self.path_in_repo70 71 72@dataclass73class HfFileSystemResolvedBucketPath(HfFileSystemResolvedPath):74 """Data structure containing information about a resolved path in a bucket."""75 76 bucket_id: str77 root: str = field(init=False)78 79 def __post_init__(self):80 self.root = "buckets/" + self.bucket_id81 82 83# We need to improve fsspec.spec._Cached which is AbstractFileSystem's metaclass84_cached_base: Any = type(fsspec.AbstractFileSystem)85 86 87class _Cached(_cached_base):88 """89 Metaclass for caching HfFileSystem instances according to the args.90 91 This creates an additional reference to the filesystem, which prevents the92 filesystem from being garbage collected when all *user* references go away.93 A call to the :meth:`AbstractFileSystem.clear_instance_cache` must *also*94 be made for a filesystem instance to be garbage collected.95 96 This is a slightly modified version of `fsspec.spec._Cached` to improve it.97 In particular in `_tokenize` the pid isn't taken into account for the98 `fs_token` used to identify cached instances. The `fs_token` logic is also99 robust to defaults values and the order of the args. Finally new instances100 reuse the states from sister instances in the main thread.101 """102 103 def __init__(cls, *args, **kwargs):104 # Hack: override https://github.com/fsspec/filesystem_spec/blob/dcb167e8f50e6273d4cfdfc4cab8fc5aa4c958bf/fsspec/spec.py#L53105 super().__init__(*args, **kwargs)106 # Note: we intentionally create a reference here, to avoid garbage107 # collecting instances when all other references are gone. To really108 # delete a FileSystem, the cache must be cleared.109 cls._cache = {}110 111 def __call__(cls, *args, **kwargs):112 # Hack: override https://github.com/fsspec/filesystem_spec/blob/dcb167e8f50e6273d4cfdfc4cab8fc5aa4c958bf/fsspec/spec.py#L65113 # Apply fsspec config (env vars / config files) before tokenizing so that114 # HfFileSystem picks up defaults the same way other fsspec filesystems do.115 kwargs = apply_config(cls, kwargs)116 skip = kwargs.pop("skip_instance_cache", False)117 fs_token = cls._tokenize(cls, threading.get_ident(), *args, **kwargs)118 fs_token_main_thread = cls._tokenize(cls, threading.main_thread().ident, *args, **kwargs)119 if not skip and cls.cachable and fs_token in cls._cache:120 # reuse cached instance121 cls._latest = fs_token122 return cls._cache[fs_token]123 else:124 # create new instance125 obj = type.__call__(cls, *args, **kwargs)126 if not skip and cls.cachable and fs_token_main_thread in cls._cache:127 # reuse the cache from the main thread instance in the new instance128 instance_state = cls._cache[fs_token_main_thread]._get_instance_state()129 for attr, state_value in instance_state.items():130 setattr(obj, attr, state_value)131 obj._fs_token_ = fs_token132 obj.storage_args = args133 obj.storage_options = kwargs134 if cls.cachable and not skip:135 cls._latest = fs_token136 cls._cache[fs_token] = obj137 return obj138 139 140class HfFileSystem(fsspec.AbstractFileSystem, metaclass=_Cached): # ty: ignore[conflicting-metaclass]141 """142 Access a remote Hugging Face Hub repository as if were a local file system.143 144 > [!WARNING]145 > [`HfFileSystem`] provides fsspec compatibility, which is useful for libraries that require it (e.g., reading146 > Hugging Face datasets directly with `pandas`). However, it introduces additional overhead due to this compatibility147 > layer. For better performance and reliability, it's recommended to use `HfApi` methods when possible.148 149 The file system supports paths for the `hf://` protocol, which follows those URL schemes:150 151 * Models, Datasets and Spaces repositories:152 153 ```154 hf://<repo-id>[@<revision>]/<path/in/repo>155 hf://datasets/<repo-id>[@<revision>]/<path/in/repo>156 hf://spaces/<repo-id>[@<revision>]/<path/in/repo>157 ```158 159 * Buckets (generic storage):160 161 ```162 hf://buckets/<bucket-id>/<path/in/bucket>163 ```164 165 Note: when using the [`HfFileSystem`] directly, passing the `hf://` protocol prefix is optional in paths.166 167 Args:168 endpoint (`str`, *optional*):169 Endpoint of the Hub. Defaults to <https://huggingface.co>.170 token (`bool` or `str`, *optional*):171 A valid user access token (string). Defaults to the locally saved172 token, which is the recommended method for authentication (see173 https://huggingface.co/docs/huggingface_hub/quick-start#authentication).174 To disable authentication, pass `False`.175 block_size (`int`, *optional*):176 Block size for reading and writing files.177 expand_info (`bool`, *optional*):178 Whether to expand the information of the files.179 **storage_options (`dict`, *optional*):180 Additional options for the filesystem. See [fsspec documentation](https://filesystem-spec.readthedocs.io/en/latest/api.html#fsspec.spec.AbstractFileSystem.__init__).181 182 Usage:183 184 ```python185 >>> from huggingface_hub import hffs186 187 >>> # List files188 >>> hffs.glob("my-username/my-model/*.bin")189 ['my-username/my-model/pytorch_model.bin']190 >>> hffs.ls("datasets/my-username/my-dataset", detail=False)191 ['datasets/my-username/my-dataset/.gitattributes', 'datasets/my-username/my-dataset/README.md', 'datasets/my-username/my-dataset/data.json']192 193 >>> # Read/write files194 >>> with hffs.open("my-username/my-model/pytorch_model.bin") as f:195 ... data = f.read()196 >>> with hffs.open("my-username/my-model/pytorch_model.bin", "wb") as f:197 ... f.write(data)198 ```199 200 Specify a token for authentication:201 ```python202 >>> from huggingface_hub import HfFileSystem203 >>> hffs = HfFileSystem(token=token)204 ```205 """206 207 root_marker = ""208 protocol = "hf"209 210 def __init__(211 self,212 *args,213 endpoint: str | None = None,214 token: bool | str | None = None,215 block_size: int | None = None,216 expand_info: bool | None = None,217 **storage_options,218 ):219 super().__init__(*args, **storage_options)220 self.endpoint = endpoint or constants.ENDPOINT221 self.token = token222 self._api = HfApi(endpoint=endpoint, token=token)223 self.block_size = block_size224 self.expand_info = expand_info225 # Maps (repo_type, repo_id, revision) to a 2-tuple with:226 # * the 1st element indicating whether the repository and the revision exist227 # * the 2nd element being the exception raised if the repository or revision doesn't exist228 self._repo_and_revision_exists_cache: dict[tuple[str, str, str | None], tuple[bool, Exception | None]] = {}229 # Same for buckets230 self._bucket_exists_cache: dict[str, tuple[bool, Exception | None]] = {}231 # Note: special case for buckets: revision is always None232 # Maps parent directory path to path infos233 self.dircache: dict[str, list[dict[str, Any]]] = {}234 235 @classmethod236 def _tokenize(cls, threading_ident: int, *args, **kwargs) -> str:237 """Deterministic token for caching"""238 # make fs_token robust to default values and to kwargs order239 kwargs["endpoint"] = kwargs.get("endpoint") or constants.ENDPOINT240 kwargs["token"] = kwargs.get("token")241 kwargs = {key: kwargs[key] for key in sorted(kwargs)}242 # contrary to fsspec, we don't include pid here243 tokenize_args = (cls, threading_ident, args, kwargs)244 h = md5(str(tokenize_args).encode())245 return h.hexdigest()246 247 def _repo_and_revision_exist(248 self, repo_type: str, repo_id: str, revision: str | None249 ) -> tuple[bool, Exception | None]:250 if (repo_type, repo_id, revision) not in self._repo_and_revision_exists_cache:251 try:252 self._api.repo_info(253 repo_id, revision=revision, repo_type=repo_type, timeout=constants.HF_HUB_ETAG_TIMEOUT254 )255 except (RepositoryNotFoundError, HFValidationError) as e:256 self._repo_and_revision_exists_cache[(repo_type, repo_id, revision)] = False, e257 self._repo_and_revision_exists_cache[(repo_type, repo_id, None)] = False, e258 except RevisionNotFoundError as e:259 self._repo_and_revision_exists_cache[(repo_type, repo_id, revision)] = False, e260 self._repo_and_revision_exists_cache[(repo_type, repo_id, None)] = True, None261 else:262 self._repo_and_revision_exists_cache[(repo_type, repo_id, revision)] = True, None263 self._repo_and_revision_exists_cache[(repo_type, repo_id, None)] = True, None264 return self._repo_and_revision_exists_cache[(repo_type, repo_id, revision)]265 266 def _bucket_exists(self, bucket_id: str) -> tuple[bool, Exception | None]:267 if bucket_id not in self._bucket_exists_cache:268 try:269 self._api.bucket_info(bucket_id)270 except BucketNotFoundError as e:271 self._bucket_exists_cache[bucket_id] = False, e272 else:273 self._bucket_exists_cache[bucket_id] = True, None274 return self._bucket_exists_cache[bucket_id]275 276 def resolve_path(277 self, path: str, revision: str | None = None278 ) -> HfFileSystemResolvedRepositoryPath | HfFileSystemResolvedBucketPath:279 """280 Resolve a Hugging Face file system path into its components.281 282 Args:283 path (`str`):284 Path to resolve.285 revision (`str`, *optional*):286 The revision of the repo to resolve. Defaults to the revision specified in the path.287 288 Returns:289 [`HfFileSystemResolvedPath`]: Resolved path information containing `repo_type`, `repo_id`, `revision` and `path_in_repo`.290 291 Raises:292 `ValueError`:293 If path contains conflicting revision information.294 `NotImplementedError`:295 If trying to list repositories.296 """297 298 def _align_revision_in_path_with_revision(revision_in_path: str | None, revision: str | None) -> str | None:299 if revision is not None:300 if revision_in_path is not None and revision_in_path != revision:301 raise ValueError(302 f'Revision specified in path ("{revision_in_path}") and in `revision` argument ("{revision}")'303 " are not the same."304 )305 else:306 revision = revision_in_path307 return revision308 309 path = self._strip_protocol(path)310 if not path:311 # can't list repositories at root312 raise NotImplementedError("Access to buckets and repositories lists is not implemented.")313 elif path.split("/")[0] == "buckets":314 bucket_id = "/".join(path.split("/")[1:3])315 path = "/".join(path.split("/")[3:])316 bucket_exists, err = self._bucket_exists(bucket_id)317 if not bucket_exists:318 _raise_file_not_found(path, err)319 return HfFileSystemResolvedBucketPath(bucket_id=bucket_id, path=path)320 elif path.split("/")[0] + "/" in constants.REPO_TYPES_URL_PREFIXES.values():321 if "/" not in path:322 # can't list repositories at the repository type level323 raise NotImplementedError("Access to repositories lists is not implemented.")324 repo_type, path = path.split("/", 1)325 repo_type = constants.REPO_TYPES_MAPPING[repo_type]326 else:327 repo_type = constants.REPO_TYPE_MODEL328 if path.count("/") > 0:329 if "@" in "/".join(path.split("/")[:2]):330 repo_id, revision_in_path = path.split("@", 1)331 if "/" in revision_in_path:332 match = SPECIAL_REFS_REVISION_REGEX.search(revision_in_path)333 if match is not None and revision in (None, match.group()):334 # Handle `refs/convert/parquet` and PR revisions separately335 path_in_repo = SPECIAL_REFS_REVISION_REGEX.sub("", revision_in_path).lstrip("/")336 revision_in_path = match.group()337 else:338 revision_in_path, path_in_repo = revision_in_path.split("/", 1)339 else:340 path_in_repo = ""341 revision = _align_revision_in_path_with_revision(unquote(revision_in_path), revision)342 repo_and_revision_exist, err = self._repo_and_revision_exist(repo_type, repo_id, revision)343 if not repo_and_revision_exist:344 _raise_file_not_found(path, err)345 else:346 revision_in_path = None347 repo_id_with_namespace = "/".join(path.split("/")[:2])348 path_in_repo_with_namespace = "/".join(path.split("/")[2:])349 repo_id_without_namespace = path.split("/")[0]350 path_in_repo_without_namespace = "/".join(path.split("/")[1:])351 repo_id = repo_id_with_namespace352 path_in_repo = path_in_repo_with_namespace353 repo_and_revision_exist, err = self._repo_and_revision_exist(repo_type, repo_id, revision)354 if not repo_and_revision_exist:355 if isinstance(err, (RepositoryNotFoundError, HFValidationError)):356 repo_id = repo_id_without_namespace357 path_in_repo = path_in_repo_without_namespace358 repo_and_revision_exist, _ = self._repo_and_revision_exist(repo_type, repo_id, revision)359 if not repo_and_revision_exist:360 _raise_file_not_found(path, err)361 else:362 _raise_file_not_found(path, err)363 else:364 repo_id = path365 path_in_repo = ""366 if "@" in path:367 repo_id, revision_in_path = path.split("@", 1)368 revision = _align_revision_in_path_with_revision(unquote(revision_in_path), revision)369 else:370 revision_in_path = None371 repo_and_revision_exist, _ = self._repo_and_revision_exist(repo_type, repo_id, revision)372 if not repo_and_revision_exist:373 raise NotImplementedError("Access to repositories lists is not implemented.")374 375 revision = revision if revision is not None else constants.DEFAULT_REVISION376 return HfFileSystemResolvedRepositoryPath(377 repo_type, repo_id, revision, path_in_repo, _raw_revision=revision_in_path378 )379 380 def invalidate_cache(self, path: str | None = None) -> None:381 """382 Clear the cache for a given path.383 384 For more details, refer to [fsspec documentation](https://filesystem-spec.readthedocs.io/en/latest/api.html#fsspec.spec.AbstractFileSystem.invalidate_cache).385 386 Args:387 path (`str`, *optional*):388 Path to clear from cache. If not provided, clear the entire cache.389 390 """391 if not path:392 self.dircache.clear()393 self._repo_and_revision_exists_cache.clear()394 else:395 resolved_path = self.resolve_path(path)396 path = resolved_path.unresolve()397 while path:398 self.dircache.pop(path, None)399 path = self._parent(path)400 401 # Only clear repo cache if path is to repo root402 if not resolved_path.path:403 if isinstance(resolved_path, HfFileSystemResolvedRepositoryPath):404 self._repo_and_revision_exists_cache.pop(405 (resolved_path.repo_type, resolved_path.repo_id, None), None406 )407 self._repo_and_revision_exists_cache.pop(408 (resolved_path.repo_type, resolved_path.repo_id, resolved_path.revision), None409 )410 else:411 self._bucket_exists_cache.pop(resolved_path.bucket_id, None)412 413 def _open( # type: ignore414 self,415 path: str,416 mode: str = "rb",417 block_size: int | None = None,418 revision: str | None = None,419 **kwargs,420 ) -> Union["HfFileSystemFile", "HfFileSystemStreamFile"]:421 block_size = block_size if block_size is not None else self.block_size422 if block_size is not None:423 kwargs["block_size"] = block_size424 if "a" in mode:425 raise NotImplementedError("Appending to remote files is not yet supported.")426 if block_size == 0:427 return HfFileSystemStreamFile(self, path, mode=mode, revision=revision, **kwargs)428 else:429 return HfFileSystemFile(self, path, mode=mode, revision=revision, **kwargs)430 431 def _rm(self, path: str, revision: str | None = None, **kwargs) -> None:432 resolved_path = self.resolve_path(path, revision=revision)433 if isinstance(resolved_path, HfFileSystemResolvedBucketPath):434 self._api.batch_bucket_files(resolved_path.bucket_id, delete=[resolved_path.path])435 else:436 self._api.delete_file(437 path_in_repo=resolved_path.path_in_repo,438 repo_id=resolved_path.repo_id,439 token=self.token,440 repo_type=resolved_path.repo_type,441 revision=resolved_path.revision,442 commit_message=kwargs.get("commit_message"),443 commit_description=kwargs.get("commit_description"),444 )445 self.invalidate_cache(path=resolved_path.unresolve())446 447 def rm(448 self,449 path: str,450 recursive: bool = False,451 maxdepth: int | None = None,452 revision: str | None = None,453 **kwargs,454 ) -> None:455 """456 Delete files from a repository.457 458 For more details, refer to [fsspec documentation](https://filesystem-spec.readthedocs.io/en/latest/api.html#fsspec.spec.AbstractFileSystem.rm).459 460 > [!WARNING]461 > Note: When possible, use `HfApi.delete_file()` for better performance.462 463 Args:464 path (`str`):465 Path to delete.466 recursive (`bool`, *optional*):467 If True, delete directory and all its contents. Defaults to False.468 maxdepth (`int`, *optional*):469 Maximum number of subdirectories to visit when deleting recursively.470 revision (`str`, *optional*):471 The git revision to delete from.472 473 """474 resolved_path = self.resolve_path(path, revision=revision)475 paths = self.expand_path(path, recursive=recursive, maxdepth=maxdepth, revision=revision)476 if isinstance(resolved_path, HfFileSystemResolvedBucketPath):477 delete = [self.resolve_path(path).path for path in paths if not self.isdir(path)]478 self._api.batch_bucket_files(resolved_path.bucket_id, delete=delete)479 else:480 paths_in_repo = [self.resolve_path(path).path for path in paths if not self.isdir(path)]481 operations = [CommitOperationDelete(path_in_repo=path_in_repo) for path_in_repo in paths_in_repo]482 commit_message = f"Delete {path} "483 commit_message += "recursively " if recursive else ""484 commit_message += f"up to depth {maxdepth} " if maxdepth is not None else ""485 # TODO: use `commit_description` to list all the deleted paths?486 self._api.create_commit(487 repo_id=resolved_path.repo_id,488 repo_type=resolved_path.repo_type,489 token=self.token,490 operations=operations,491 revision=resolved_path.revision,492 commit_message=kwargs.get("commit_message", commit_message),493 commit_description=kwargs.get("commit_description"),494 )495 self.invalidate_cache(path=resolved_path.unresolve())496 497 def ls(498 self, path: str, detail: bool = True, refresh: bool = False, revision: str | None = None, **kwargs499 ) -> list[str | dict[str, Any]]:500 """501 List the contents of a directory.502 503 For more details, refer to [fsspec documentation](https://filesystem-spec.readthedocs.io/en/latest/api.html#fsspec.spec.AbstractFileSystem.ls).504 505 > [!WARNING]506 > Note: When possible, use `HfApi.list_repo_tree()` for better performance.507 508 Args:509 path (`str`):510 Path to the directory.511 detail (`bool`, *optional*):512 If True, returns a list of dictionaries containing file information. If False,513 returns a list of file paths. Defaults to True.514 refresh (`bool`, *optional*):515 If True, bypass the cache and fetch the latest data. Defaults to False.516 revision (`str`, *optional*):517 The git revision to list from.518 519 Returns:520 `list[Union[str, dict[str, Any]]]`: List of file paths (if detail=False) or list of file information521 dictionaries (if detail=True).522 """523 resolved_path = self.resolve_path(path, revision=revision)524 path = resolved_path.unresolve()525 try:526 out = self._ls_tree(path, refresh=refresh, revision=revision, **kwargs)527 except EntryNotFoundError:528 # Path could be a file529 if not resolved_path.path:530 _raise_file_not_found(path, None)531 try:532 out = self._ls_tree(self._parent(path), refresh=refresh, revision=revision, **kwargs)533 except EntryNotFoundError:534 out = []535 out = [o for o in out if o["name"] == path]536 if len(out) == 0:537 _raise_file_not_found(path, None)538 return out if detail else [o["name"] for o in out]539 540 def _ls_tree(541 self,542 path: str,543 recursive: bool = False,544 refresh: bool = False,545 revision: str | None = None,546 expand_info: bool | None = None,547 maxdepth: int | None = None,548 ):549 expand_info = (550 expand_info if expand_info is not None else (self.expand_info if self.expand_info is not None else False)551 )552 resolved_path = self.resolve_path(path, revision=revision)553 path = resolved_path.unresolve()554 root_path = resolved_path.root555 maxdepth = maxdepth if recursive else 1556 557 out = []558 if path in self.dircache and not refresh:559 cached_path_infos = self.dircache[path]560 out.extend(cached_path_infos)561 dirs_not_in_dircache = []562 if recursive:563 # Use BFS to traverse the cache and build the "recursive "output564 # (The Hub uses a so-called "tree first" strategy for the tree endpoint but we sort the output to follow the spec so the result is (eventually) the same)565 depth = 2566 dirs_to_visit = deque(567 [(depth, path_info) for path_info in cached_path_infos if path_info["type"] == "directory"]568 )569 while dirs_to_visit:570 depth, dir_info = dirs_to_visit.popleft()571 if maxdepth is None or depth <= maxdepth:572 if dir_info["name"] not in self.dircache:573 dirs_not_in_dircache.append(dir_info["name"])574 else:575 cached_path_infos = self.dircache[dir_info["name"]]576 out.extend(cached_path_infos)577 dirs_to_visit.extend(578 [579 (depth + 1, path_info)580 for path_info in cached_path_infos581 if path_info["type"] == "directory"582 ]583 )584 585 dirs_not_expanded = []586 if expand_info and isinstance(resolved_path, HfFileSystemResolvedRepositoryPath):587 # Check if there are directories in repos with non-expanded entries588 dirs_not_expanded = [self._parent(o["name"]) for o in out if o["last_commit"] is None]589 590 if (recursive and dirs_not_in_dircache) or (expand_info and dirs_not_expanded):591 # If the dircache is incomplete, find the common path of the missing and non-expanded entries592 # and extend the output with the result of `_ls_tree(common_path, recursive=True)`593 common_prefix = os.path.commonprefix(dirs_not_in_dircache + dirs_not_expanded)594 # Get the parent directory if the common prefix itself is not a directory595 common_path = (596 common_prefix.rstrip("/")597 if common_prefix.endswith("/")598 or common_prefix == root_path599 or common_prefix in chain(dirs_not_in_dircache, dirs_not_expanded)600 else self._parent(common_prefix)601 )602 if maxdepth is not None:603 common_path_depth = common_path[len(path) :].count("/")604 maxdepth -= common_path_depth605 out = [o for o in out if not o["name"].startswith(common_path + "/")]606 for cached_path in list(self.dircache):607 if cached_path.startswith(common_path + "/"):608 self.dircache.pop(cached_path, None)609 self.dircache.pop(common_path, None)610 out.extend(611 self._ls_tree(612 common_path,613 recursive=recursive,614 refresh=True,615 revision=revision,616 expand_info=expand_info,617 maxdepth=maxdepth,618 )619 )620 else:621 tree: Iterable[RepoFile | RepoFolder | BucketFile | BucketFolder]622 if isinstance(resolved_path, HfFileSystemResolvedBucketPath):623 tree = self._list_bucket_tree_with_folders(624 resolved_path.bucket_id,625 prefix=resolved_path.path,626 recursive=recursive,627 )628 else:629 tree = self._api.list_repo_tree(630 resolved_path.repo_id,631 resolved_path.path,632 recursive=recursive,633 expand=expand_info,634 revision=resolved_path.revision,635 repo_type=resolved_path.repo_type,636 )637 for path_info in tree:638 cache_path = root_path + "/" + path_info.path639 if isinstance(path_info, RepoFile):640 cache_path_info = {641 "name": cache_path,642 "size": path_info.size,643 "type": "file",644 "blob_id": path_info.blob_id,645 "lfs": path_info.lfs,646 "xet_hash": path_info.xet_hash,647 "last_commit": path_info.last_commit,648 "security": path_info.security,649 }650 elif isinstance(path_info, BucketFile):651 cache_path_info = {652 "name": cache_path,653 "size": path_info.size,654 "type": "file",655 "xet_hash": path_info.xet_hash,656 "mtime": path_info.mtime,657 "uploaded_at": path_info.uploaded_at,658 }659 elif isinstance(path_info, RepoFolder):660 cache_path_info = {661 "name": cache_path,662 "size": 0,663 "type": "directory",664 "tree_id": path_info.tree_id,665 "last_commit": path_info.last_commit,666 }667 else:668 cache_path_info = {669 "name": cache_path,670 "size": 0,671 "type": "directory",672 "uploaded_at": path_info.uploaded_at,673 }674 parent_path = self._parent(cache_path_info["name"])675 self.dircache.setdefault(parent_path, []).append(cache_path_info)676 depth = cache_path[len(path) :].count("/")677 if maxdepth is None or depth <= maxdepth:678 out.append(cache_path_info)679 return out680 681 def _list_bucket_tree_with_folders(682 self, bucket_id: str, prefix: str, recursive: bool683 ) -> Iterable[BucketFile | BucketFolder]:684 """Same as `HfApi.list_bucket_tree` but always includes folders"""685 bucket_files = self._api.list_bucket_tree(bucket_id, prefix, recursive=recursive)686 bucket_folders: dict[str, BucketFolder] = {}687 min_depth = 1 + prefix.count("/") if prefix else 0688 out: list[BucketFile | BucketFolder] = []689 690 for bucket_entry in bucket_files:691 out.append(bucket_entry)692 693 # If recursive=False, both files and folders are returned by the server => nothing to do694 if not recursive:695 continue696 697 # Otherwise, let's rebuild BucketFolders manually698 for parent_bucket_folder_str in list(PurePosixPath(bucket_entry.path).parents)[: -min_depth - 1]:699 parent_bucket_folder = BucketFolder(700 type="directory", path=str(parent_bucket_folder_str), uploaded_at=bucket_entry.uploaded_at701 )702 703 # If folder not visited yet, add it704 if parent_bucket_folder.path not in bucket_folders:705 out.append(parent_bucket_folder)706 bucket_folders[parent_bucket_folder.path] = parent_bucket_folder707 continue708 709 # Otherwise, get back BucketFolder object and update its 'uploaded_at'710 if parent_bucket_folder.uploaded_at is not None:711 bucket_folder = bucket_folders[parent_bucket_folder.path]712 if bucket_folder.uploaded_at is None or (713 bucket_folder.uploaded_at < parent_bucket_folder.uploaded_at714 ):715 bucket_folder.uploaded_at = parent_bucket_folder.uploaded_at716 717 if not out:718 raise EntryNotFoundError(f"File not found in bucket '{bucket_id}': '{prefix}'")719 return out720 721 def walk(self, path: str, *args, **kwargs) -> Iterator[tuple[str, list[str], list[str]]]:722 """723 Return all files below the given path.724 725 For more details, refer to [fsspec documentation](https://filesystem-spec.readthedocs.io/en/latest/api.html#fsspec.spec.AbstractFileSystem.walk).726 727 Args:728 path (`str`):729 Root path to list files from.730 731 Returns:732 `Iterator[tuple[str, list[str], list[str]]]`: An iterator of (path, list of directory names, list of file names) tuples.733 """734 path = self.resolve_path(path, revision=kwargs.get("revision")).unresolve()735 yield from super().walk(path, *args, **kwargs)736 737 def glob(self, path: str, maxdepth: int | None = None, **kwargs) -> list[str]:738 """739 Find files by glob-matching.740 741 For more details, refer to [fsspec documentation](https://filesystem-spec.readthedocs.io/en/latest/api.html#fsspec.spec.AbstractFileSystem.glob).742 743 Args:744 path (`str`):745 Path pattern to match.746 747 Returns:748 `list[str]`: List of paths matching the pattern.749 """750 path = self.resolve_path(path, revision=kwargs.get("revision")).unresolve()751 return super().glob(path, maxdepth=maxdepth, **kwargs)752 753 def find(754 self,755 path: str,756 maxdepth: int | None = None,757 withdirs: bool = False,758 detail: bool = False,759 refresh: bool = False,760 revision: str | None = None,761 **kwargs,762 ) -> list[str] | dict[str, dict[str, Any]]:763 """764 List all files below path.765 766 For more details, refer to [fsspec documentation](https://filesystem-spec.readthedocs.io/en/latest/api.html#fsspec.spec.AbstractFileSystem.find).767 768 Args:769 path (`str`):770 Root path to list files from.771 maxdepth (`int`, *optional*):772 Maximum depth to descend into subdirectories.773 withdirs (`bool`, *optional*):774 Include directory paths in the output. Defaults to False.775 detail (`bool`, *optional*):776 If True, returns a dict mapping paths to file information. Defaults to False.777 refresh (`bool`, *optional*):778 If True, bypass the cache and fetch the latest data. Defaults to False.779 revision (`str`, *optional*):780 The git revision to list from.781 782 Returns:783 `Union[list[str], dict[str, dict[str, Any]]]`: List of paths or dict of file information.784 """785 if maxdepth is not None and maxdepth < 1:786 raise ValueError("maxdepth must be at least 1")787 resolved_path = self.resolve_path(path, revision=revision)788 path = resolved_path.unresolve()789 try:790 out = self._ls_tree(path, recursive=True, refresh=refresh, maxdepth=maxdepth, **kwargs)791 except EntryNotFoundError:792 # Path could be a file793 try:794 if self.info(path, revision=revision, **kwargs)["type"] == "file":795 out = {path: {}}796 else:797 out = {}798 except FileNotFoundError:799 out = {}800 else:801 if not withdirs:802 out = [o for o in out if o["type"] != "directory"]803 else:804 # If `withdirs=True`, include the directory itself to be consistent with the spec805 path_info = self.info(path, **kwargs)806 out = [path_info] + out if path_info["type"] == "directory" else out807 out = {o["name"]: o for o in out}808 names = sorted(out)809 if not detail:810 return names811 else:812 return {name: out[name] for name in names}813 814 def cp_file(self, path1: str, path2: str, revision: str | None = None, **kwargs) -> None:815 """816 Copy a file within or between repositories.817 818 > [!WARNING]819 > Note: When possible, use `HfApi.upload_file()` for better performance.820 821 Args:822 path1 (`str`):823 Source path to copy from.824 path2 (`str`):825 Destination path to copy to.826 revision (`str`, *optional*):827 The git revision to copy from.828 829 """830 resolved_path1 = self.resolve_path(path1, revision=revision)831 resolved_path2 = self.resolve_path(path2, revision=revision)832 if isinstance(resolved_path1, HfFileSystemResolvedBucketPath) or isinstance(833 resolved_path2, HfFileSystemResolvedBucketPath834 ):835 raise NotImplementedError("Copy from/to buckets is not available yet")836 837 same_repo = (838 resolved_path1.repo_type == resolved_path2.repo_type and resolved_path1.repo_id == resolved_path2.repo_id839 )840 841 if same_repo:842 commit_message = f"Copy {path1} to {path2}"843 self._api.create_commit(844 repo_id=resolved_path1.repo_id,845 repo_type=resolved_path1.repo_type,846 revision=resolved_path2.revision,847 commit_message=kwargs.get("commit_message", commit_message),848 commit_description=kwargs.get("commit_description", ""),849 operations=[850 CommitOperationCopy(851 src_path_in_repo=resolved_path1.path_in_repo,852 path_in_repo=resolved_path2.path_in_repo,853 src_revision=resolved_path1.revision,854 )855 ],856 )857 else:858 with self.open(path1, "rb", revision=resolved_path1.revision) as f:859 content = f.read()860 commit_message = f"Copy {path1} to {path2}"861 self._api.upload_file(862 path_or_fileobj=content,863 path_in_repo=resolved_path2.path_in_repo,864 repo_id=resolved_path2.repo_id,865 token=self.token,866 repo_type=resolved_path2.repo_type,867 revision=resolved_path2.revision,868 commit_message=kwargs.get("commit_message", commit_message),869 commit_description=kwargs.get("commit_description"),870 )871 self.invalidate_cache(path=resolved_path1.unresolve())872 self.invalidate_cache(path=resolved_path2.unresolve())873 874 def modified(self, path: str, **kwargs) -> datetime:875 """876 Get the last modified time of a file.877 878 For more details, refer to [fsspec documentation](https://filesystem-spec.readthedocs.io/en/latest/api.html#fsspec.spec.AbstractFileSystem.modified).879 880 Args:881 path (`str`):882 Path to the file.883 884 Returns:885 `datetime`: Last modified time of the file.886 """887 info = self.info(path, **{**kwargs, "expand_info": True}) # type: ignore888 if "last_commit" in info:889 if info["last_commit"] is None:890 raise NotImplementedError(f"'modified' is not implemented for repository paths like '{path}'")891 return info["last_commit"].date892 elif "mtime" in info and info["mtime"]:893 return info["mtime"]894 elif "uploaded_at" in info and info["uploaded_at"]:895 return info["uploaded_at"]896 else:897 raise NotImplementedError(f"Cannot determined 'modified' for path '{path}' (info: {info})")898 899 def info(self, path: str, refresh: bool = False, revision: str | None = None, **kwargs) -> dict[str, Any]:900 """901 Get information about a file or directory.902 903 For more details, refer to [fsspec documentation](https://filesystem-spec.readthedocs.io/en/latest/api.html#fsspec.spec.AbstractFileSystem.info).904 905 > [!WARNING]906 > Note: When possible, use `HfApi.get_paths_info()` or `HfApi.repo_info()` for better performance907 > (or `HfApi.get_bucket_paths_info()` or `HfApi.bucket_info()` for buckets)908 909 Args:910 path (`str`):911 Path to get info for.912 refresh (`bool`, *optional*):913 If True, bypass the cache and fetch the latest data. Defaults to False.914 revision (`str`, *optional*):915 The git revision to get info from.916 917 Returns:918 `dict[str, Any]`: Dictionary containing file information (type, size, commit info, etc.).919 920 """921 resolved_path = self.resolve_path(path, revision=revision)922 path = resolved_path.unresolve()923 expand_info = kwargs.get(924 "expand_info", self.expand_info if self.expand_info is not None else False925 ) # don't expose it as a parameter in the public API to follow the spec926 out: dict[str, Any] | None927 if not resolved_path.path:928 # Path is the root directory929 out = {930 "name": path,931 "size": 0,932 "type": "directory",933 }934 if isinstance(resolved_path, HfFileSystemResolvedRepositoryPath):935 out["last_commit"] = None936 if isinstance(resolved_path, HfFileSystemResolvedRepositoryPath) and expand_info:937 last_commit = self._api.list_repo_commits(938 resolved_path.repo_id, repo_type=resolved_path.repo_type, revision=resolved_path.revision939 )[-1]940 out = {941 **out,942 "tree_id": None, # TODO: tree_id of the root directory?943 "last_commit": LastCommitInfo(944 oid=last_commit.commit_id, title=last_commit.title, date=last_commit.created_at945 ),946 }947 elif isinstance(resolved_path, HfFileSystemResolvedBucketPath):948 parent_path = self._parent(path)949 # Fill the cache with cheap call950 self.ls(parent_path, refresh=refresh)951 out1 = [o for o in self.dircache[parent_path] if o["name"] == path]952 if not out1:953 _raise_file_not_found(path, None)954 out = out1[0]955 else:956 out = None957 parent_path = self._parent(path)958 if not expand_info and parent_path not in self.dircache:959 # Fill the cache with cheap call960 self.ls(parent_path)961 if parent_path in self.dircache:962 # Check if the path is in the cache963 out1 = [o for o in self.dircache[parent_path] if o["name"] == path]964 if not out1:965 _raise_file_not_found(path, None)966 out = out1[0]967 if refresh or out is None or (expand_info and out and out["last_commit"] is None):968 paths_info = self._api.get_paths_info(969 resolved_path.repo_id,970 resolved_path.path_in_repo,971 expand=expand_info,972 revision=resolved_path.revision,973 repo_type=resolved_path.repo_type,974 )975 if not paths_info:976 _raise_file_not_found(path, None)977 path_info = paths_info[0]978 root_path = HfFileSystemResolvedRepositoryPath(979 resolved_path.repo_type,980 resolved_path.repo_id,981 resolved_path.revision,982 path_in_repo="",983 _raw_revision=resolved_path._raw_revision,984 ).unresolve()985 if isinstance(path_info, RepoFile):986 out = {987 "name": root_path + "/" + path_info.path,988 "size": path_info.size,989 "type": "file",990 "blob_id": path_info.blob_id,991 "lfs": path_info.lfs,992 "xet_hash": path_info.xet_hash,993 "last_commit": path_info.last_commit,994 "security": path_info.security,995 }996 else:997 out = {998 "name": root_path + "/" + path_info.path,999 "size": 0,1000 "type": "directory",1001 "tree_id": path_info.tree_id,1002 "last_commit": path_info.last_commit,1003 }1004 if not expand_info:1005 out = {k: out[k] for k in ["name", "size", "type"]}1006 assert out is not None1007 return out1008 1009 def exists(self, path, **kwargs):1010 """1011 Check if a file exists.1012 1013 For more details, refer to [fsspec documentation](https://filesystem-spec.readthedocs.io/en/latest/api.html#fsspec.spec.AbstractFileSystem.exists).1014 1015 > [!WARNING]1016 > Note: When possible, use `HfApi.file_exists()` for better performance.1017 1018 Args:1019 path (`str`):1020 Path to check.1021 1022 Returns:1023 `bool`: True if file exists, False otherwise.1024 """1025 try:1026 if kwargs.get("refresh", False):1027 self.invalidate_cache(path)1028 1029 self.info(path, **kwargs)1030 return True1031 except OSError:1032 return False1033 1034 def isdir(self, path):1035 """1036 Check if a path is a directory.1037 1038 For more details, refer to [fsspec documentation](https://filesystem-spec.readthedocs.io/en/latest/api.html#fsspec.spec.AbstractFileSystem.isdir).1039 1040 Args:1041 path (`str`):1042 Path to check.1043 1044 Returns:1045 `bool`: True if path is a directory, False otherwise.1046 """1047 try:1048 return self.info(path)["type"] == "directory"1049 except OSError:1050 return False1051 1052 def isfile(self, path):1053 """1054 Check if a path is a file.1055 1056 For more details, refer to [fsspec documentation](https://filesystem-spec.readthedocs.io/en/latest/api.html#fsspec.spec.AbstractFileSystem.isfile).1057 1058 Args:1059 path (`str`):1060 Path to check.1061 1062 Returns:1063 `bool`: True if path is a file, False otherwise.1064 """1065 try:1066 return self.info(path)["type"] == "file"1067 except OSError:1068 return False1069 1070 def url(self, path: str) -> str:1071 """1072 Get the HTTP URL of the given path.1073 1074 Args:1075 path (`str`):1076 Path to get URL for.1077 1078 Returns:1079 `str`: HTTP URL to access the file or directory on the Hub.1080 """1081 resolved_path = self.resolve_path(path)1082 if isinstance(resolved_path, HfFileSystemResolvedBucketPath):1083 url = f"{self.endpoint}/buckets/{resolved_path.bucket_id}/resolve/{quote(resolved_path.path)}"1084 else:1085 url = hf_hub_url(1086 resolved_path.repo_id,1087 resolved_path.path_in_repo,1088 repo_type=resolved_path.repo_type,1089 revision=resolved_path.revision,1090 endpoint=self.endpoint,1091 )1092 if self.isdir(path):1093 url = url.replace("/resolve/", "/tree/", 1)1094 return url1095 1096 def get_file(self, rpath, lpath, callback=_DEFAULT_CALLBACK, outfile=None, **kwargs) -> None:1097 """1098 Copy single remote file to local.1099 1100 > [!WARNING]1101 > Note: When possible, use `HfApi.hf_hub_download()` or `HfApi.download_bucket_files` for better performance.1102 1103 Args:1104 rpath (`str`):1105 Remote path to download from.1106 lpath (`str`):1107 Local path to download to.1108 callback (`Callback`, *optional*):1109 Optional callback to track download progress. Defaults to no callback.1110 outfile (`IO`, *optional*):1111 Optional file-like object to write to. If provided, `lpath` is ignored.1112 1113 """1114 revision = kwargs.get("revision")1115 unhandled_kwargs = set(kwargs.keys()) - {"revision"}1116 if not isinstance(callback, (NoOpCallback, TqdmCallback)) or len(unhandled_kwargs) > 0:1117 # for now, let's not handle custom callbacks1118 # and let's not handle custom kwargs1119 return super().get_file(rpath, lpath, callback=callback, outfile=outfile, **kwargs)1120 1121 # Taken from https://github.com/fsspec/filesystem_spec/blob/47b445ae4c284a82dd15e0287b1ffc410e8fc470/fsspec/spec.py#L8831122 if isfilelike(lpath):1123 outfile = lpath1124 elif self.isdir(rpath):1125 os.makedirs(lpath, exist_ok=True)1126 return None1127 1128 if isinstance(lpath, (str, Path)): # otherwise, let's assume it's a file-like object1129 os.makedirs(os.path.dirname(lpath), exist_ok=True)1130 1131 # Open file if not already open1132 close_file = False1133 if outfile is None:1134 outfile = open(lpath, "wb")1135 close_file = True1136 initial_pos = outfile.tell()1137 1138 # Custom implementation of `get_file` to use `http_get`.1139 resolve_remote_path = self.resolve_path(rpath, revision=revision)1140 expected_size = self.info(rpath, revision=revision)["size"]1141 callback.set_size(expected_size)1142 try:1143 http_get(1144 url=self.url(resolve_remote_path.unresolve()),1145 temp_file=outfile, # type: ignore1146 displayed_filename=rpath,1147 expected_size=expected_size,1148 resume_size=0,1149 headers=self._api._build_hf_headers(),1150 _tqdm_bar=callback.tqdm if isinstance(callback, TqdmCallback) else None,1151 )1152 outfile.seek(initial_pos)1153 finally:1154 # Close file only if we opened it ourselves1155 if close_file:1156 outfile.close()1157 1158 @property1159 def transaction(self):1160 """A context within which files are committed together upon exit1161 1162 Requires the file class to implement `.commit()` and `.discard()`1163 for the normal and exception cases.1164 """1165 # Taken from https://github.com/fsspec/filesystem_spec/blob/3fbb6fee33b46cccb015607630843dea049d3243/fsspec/spec.py#L2311166 # See https://github.com/huggingface/huggingface_hub/issues/17331167 raise NotImplementedError("Transactional commits are not supported.")1168 1169 def start_transaction(self):1170 """Begin write transaction for deferring files, non-context version"""1171 # Taken from https://github.com/fsspec/filesystem_spec/blob/3fbb6fee33b46cccb015607630843dea049d3243/fsspec/spec.py#L2411172 # See https://github.com/huggingface/huggingface_hub/issues/17331173 raise NotImplementedError("Transactional commits are not supported.")1174 1175 def __reduce__(self):1176 # re-populate the instance cache at HfFileSystem._cache and re-populate the state of every instance1177 return make_instance, (1178 type(self),1179 self.storage_args,1180 self.storage_options,1181 self._get_instance_state(),1182 )1183 1184 def _get_instance_state(self):1185 return {1186 "dircache": deepcopy(self.dircache),1187 "_repo_and_revision_exists_cache": deepcopy(self._repo_and_revision_exists_cache),1188 "_bucket_exists_cache": deepcopy(self._bucket_exists_cache),1189 }1190 1191 1192class HfFileSystemFile(fsspec.spec.AbstractBufferedFile):1193 def __init__(self, fs: HfFileSystem, path: str, revision: str | None = None, **kwargs):1194 try:1195 self.resolved_path = fs.resolve_path(path, revision=revision)1196 except FileNotFoundError as e:1197 if "w" in kwargs.get("mode", ""):1198 raise FileNotFoundError(1199 f"{e}.\nMake sure the repository and revision exist before writing data."1200 ) from e