Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
_buckets.py1226 linesDownload Raw Back to huggingface_hub
1# Copyright 2026-present, the HuggingFace Inc. team.2#3# Licensed under the Apache License, Version 2.0 (the "License");4# you may not use this file except in compliance with the License.5# You may obtain a copy of the License at6#7#     http://www.apache.org/licenses/LICENSE-2.08#9# Unless required by applicable law or agreed to in writing, software10# distributed under the License is distributed on an "AS IS" BASIS,11# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.12# See the License for the specific language governing permissions and13# limitations under the License.14"""Shared logic for bucket operations.15 16This module contains the core buckets logic used by both the CLI and the Python API.17"""18 19import fnmatch20import json21import mimetypes22import os23import stat24import sys25import time26from collections.abc import Iterator27from dataclasses import dataclass, field28from datetime import datetime, timezone29from pathlib import Path30from typing import TYPE_CHECKING, Any, Literal31 32from . import constants, logging33from .errors import BucketNotFoundError34from .utils import XetFileData, disable_progress_bars, enable_progress_bars, parse_datetime35from .utils._terminal import StatusLine36 37 38if TYPE_CHECKING:39    from .hf_api import HfApi40 41 42logger = logging.get_logger(__name__)43 44 45BUCKET_PREFIX = "hf://buckets/"46_SYNC_TIME_WINDOW_MS = 1000  # 1s safety-window for file modification time comparisons47 48 49# =============================================================================50# Bucket data structures51# =============================================================================52 53 54def _split_bucket_id_and_prefix(path: str) -> tuple[str, str]:55    """Split 'namespace/name(/optional/prefix)' into ('namespace/name', 'prefix').56 57    Returns (bucket_id, prefix) where prefix may be empty string.58    Raises ValueError if path doesn't contain at least namespace/name.59    """60    parts = path.split("/", 2)61    if len(parts) < 2 or not parts[0] or not parts[1]:62        raise ValueError(f"Invalid bucket path: '{path}'. Expected format: namespace/bucket_name")63    bucket_id = f"{parts[0]}/{parts[1]}"64    prefix = parts[2] if len(parts) > 2 else ""65    return bucket_id, prefix66 67 68@dataclass69class BucketInfo:70    """71    Contains information about a bucket on the Hub. This object is returned by [`bucket_info`] and [`list_buckets`].72 73    Attributes:74        id (`str`):75            ID of the bucket.76        private (`bool`):77            Is the bucket private.78        created_at (`datetime`):79            Date of creation of the bucket on the Hub.80        size (`int`):81            Size of the bucket in bytes.82        total_files (`int`):83            Total number of files in the bucket.84    """85 86    id: str87    private: bool88    created_at: datetime89    size: int90    total_files: int91 92    def __init__(self, **kwargs):93        self.id = kwargs.pop("id")94        self.private = kwargs.pop("private")95        self.created_at = parse_datetime(kwargs.pop("createdAt"))96        self.size = kwargs.pop("size")97        self.total_files = kwargs.pop("totalFiles")98        self.__dict__.update(**kwargs)99 100 101@dataclass102class _BucketAddFile:103    source: str | Path | bytes104    destination: str105 106    xet_hash: str | None = field(default=None)107    size: int | None = field(default=None)108    mtime: int = field(init=False)109    content_type: str | None = field(init=False)110 111    def __post_init__(self) -> None:112        self.content_type = None113        if isinstance(self.source, (str, Path)):  # guess content type from source path114            self.content_type = mimetypes.guess_type(self.source)[0]115        if self.content_type is None:  # or default to destination path content type116            self.content_type = mimetypes.guess_type(self.destination)[0]117 118        self.mtime = int(119            os.path.getmtime(self.source) * 1000 if not isinstance(self.source, bytes) else time.time() * 1000120        )121 122 123@dataclass124class _BucketCopyFile:125    destination: str126    xet_hash: str127    source_repo_type: str  # "model", "dataset", "space", "bucket"128    source_repo_id: str129    size: int | None = field(default=None)130    mtime: int = field(init=False)131    content_type: str | None = field(init=False)132 133    def __post_init__(self) -> None:134        self.content_type = mimetypes.guess_type(self.destination)[0]135        self.mtime = int(time.time() * 1000)136 137 138@dataclass139class _BucketDeleteFile:140    path: str141 142 143@dataclass(frozen=True)144class BucketFileMetadata:145    """Data structure containing information about a file in a bucket.146 147    Returned by [`get_bucket_file_metadata`].148 149    Args:150        size (`int`):151            Size of the file in bytes.152        xet_file_data (`XetFileData`):153            Xet information for the file (hash and refresh route).154    """155 156    size: int157    xet_file_data: XetFileData158 159 160@dataclass161class BucketUrl:162    """Describes a bucket URL on the Hub.163 164    `BucketUrl` is returned by [`create_bucket`]. At initialization, the URL is parsed to populate properties:165    - endpoint (`str`)166    - namespace (`str`)167    - bucket_id (`str`)168    - url (`str`)169    - handle (`str`)170 171    Args:172        url (`str`):173            String value of the bucket url.174        endpoint (`str`, *optional*):175            Endpoint of the Hub. Defaults to <https://huggingface.co>.176    """177 178    url: str179    endpoint: str = ""180    namespace: str = field(init=False)181    bucket_id: str = field(init=False)182    handle: str = field(init=False)183 184    def __post_init__(self) -> None:185        self.endpoint = self.endpoint or constants.ENDPOINT186 187        # Parse URL: expected format is `{endpoint}/buckets/{namespace}/{bucket_name}`188        url_path = self.url.replace(self.endpoint, "").strip("/")189        # Remove leading "buckets/" prefix190        if url_path.startswith("buckets/"):191            url_path = url_path[len("buckets/") :]192        bucket_id, prefix = _split_bucket_id_and_prefix(url_path)193        if prefix:194            raise ValueError(f"Unable to parse bucket URL: {self.url}")195        self.namespace = bucket_id.split("/")[0]196        self.bucket_id = bucket_id197 198        self.handle = f"hf://buckets/{self.bucket_id}"199 200 201@dataclass202class BucketFile:203    """204    Contains information about a file in a bucket on the Hub. This object is returned by [`list_bucket_tree`].205 206    Similar to [`RepoFile`] but for files in buckets.207    """208 209    type: Literal["file"]210    path: str211    size: int212    xet_hash: str213    mtime: datetime | None214    uploaded_at: datetime | None215 216    def __init__(self, **kwargs):217        self.type = kwargs.pop("type")218        self.path = kwargs.pop("path")219        self.size = kwargs.pop("size")220        self.xet_hash = kwargs.pop("xetHash")221        mtime = kwargs.pop("mtime", None)222        self.mtime = parse_datetime(mtime) if mtime else None223        uploaded_at = kwargs.pop("uploadedAt", None)224        self.uploaded_at = parse_datetime(uploaded_at) if uploaded_at else None225 226 227@dataclass228class BucketFolder:229    """230    Contains information about a directory in a bucket on the Hub. This object is returned by [`list_bucket_tree`].231 232    Similar to [`RepoFolder`] but for directories in buckets.233    """234 235    type: Literal["directory"]236    path: str237    uploaded_at: datetime | None238 239    def __init__(self, **kwargs):240        self.type = kwargs.pop("type")241        self.path = kwargs.pop("path")242        uploaded_at = kwargs.pop("uploadedAt", None) or kwargs.pop("uploaded_at", None)243        self.uploaded_at = (244            (uploaded_at if isinstance(uploaded_at, datetime) else parse_datetime(uploaded_at))245            if uploaded_at246            else None247        )248 249 250# =============================================================================251# Bucket path parsing252# =============================================================================253 254 255def _parse_bucket_path(path: str) -> tuple[str, str]:256    """Parse a bucket path like hf://buckets/namespace/bucket_name/prefix into (bucket_id, prefix).257 258    Returns:259        tuple: (bucket_id, prefix) where bucket_id is "namespace/bucket_name" and prefix may be empty string.260    """261    if not path.startswith(BUCKET_PREFIX):262        raise ValueError(f"Invalid bucket path: {path}. Must start with {BUCKET_PREFIX}")263    return _split_bucket_id_and_prefix(path.removeprefix(BUCKET_PREFIX))264 265 266def _is_bucket_path(path: str) -> bool:267    """Check if a path is a bucket path."""268    return path.startswith(BUCKET_PREFIX)269 270 271# =============================================================================272# Sync data structures273# =============================================================================274 275 276@dataclass277class SyncOperation:278    """Represents a sync operation to be performed."""279 280    action: Literal["upload", "download", "delete", "skip"]281    path: str282    size: int | None = None283    reason: str = ""284    local_mtime: str | None = None285    remote_mtime: str | None = None286    bucket_file: BucketFile | None = None  # BucketFile when available (not serialized to plan file)287 288 289@dataclass290class SyncPlan:291    """Represents a complete sync plan."""292 293    source: str294    dest: str295    timestamp: str296    operations: list[SyncOperation] = field(default_factory=list)297 298    def summary(self) -> dict[str, int | str]:299        uploads = sum(1 for op in self.operations if op.action == "upload")300        downloads = sum(1 for op in self.operations if op.action == "download")301        deletes = sum(1 for op in self.operations if op.action == "delete")302        skips = sum(1 for op in self.operations if op.action == "skip")303        total_size = sum(op.size or 0 for op in self.operations if op.action in ("upload", "download"))304        return {305            "uploads": uploads,306            "downloads": downloads,307            "deletes": deletes,308            "skips": skips,309            "total_size": total_size,310        }311 312 313# =============================================================================314# Filter matching315# =============================================================================316 317 318class FilterMatcher:319    """Matches file paths against include/exclude patterns."""320 321    def __init__(322        self,323        include_patterns: list[str] | None = None,324        exclude_patterns: list[str] | None = None,325        filter_rules: list[tuple[str, str]] | None = None,326    ):327        """Initialize the filter matcher.328 329        Args:330            include_patterns: Patterns to include (from --include)331            exclude_patterns: Patterns to exclude (from --exclude)332            filter_rules: Rules from filter file as list of ("+"/"-", pattern) tuples333        """334        self.include_patterns = include_patterns or []335        self.exclude_patterns = exclude_patterns or []336        self.filter_rules = filter_rules or []337 338    def matches(self, path: str) -> bool:339        """Check if a path should be included based on the filter rules.340 341        Filtering rules:342        - Filters are evaluated in order, first matching rule decides343        - If no rules match, include by default (unless include patterns are specified)344        """345        # First check filter rules from file (in order)346        for sign, pattern in self.filter_rules:347            if fnmatch.fnmatch(path, pattern):348                return sign == "+"349 350        # Then check CLI patterns351        for pattern in self.exclude_patterns:352            if fnmatch.fnmatch(path, pattern):353                return False354 355        for pattern in self.include_patterns:356            if fnmatch.fnmatch(path, pattern):357                return True358 359        # If include patterns were specified but none matched, exclude360        if self.include_patterns:361            return False362 363        # Default: include364        return True365 366 367def _parse_filter_file(filter_file: str) -> list[tuple[str, str]]:368    """Parse a filter file and return a list of (sign, pattern) tuples.369 370    Filter file format:371    - Lines starting with "+" are include patterns372    - Lines starting with "-" are exclude patterns373    - Empty lines and lines starting with "#" are ignored374    """375    rules = []376    with open(filter_file) as f:377        for line in f:378            line = line.strip()379            if not line or line.startswith("#"):380                continue381            if line.startswith("+"):382                rules.append(("+", line[1:].strip()))383            elif line.startswith("-"):384                rules.append(("-", line[1:].strip()))385            else:386                # Default to include if no prefix387                rules.append(("+", line))388    return rules389 390 391# =============================================================================392# File listing393# =============================================================================394 395 396def _stat_local(path: str) -> tuple[int, float] | None:397    """Stat a local file and return (size, mtime_ms).398 399    Returns None if the path is missing or is a directory. Uses a single400    ``os.stat`` call so callers don't pay for multiple syscalls per file.401    """402    try:403        st = os.stat(path)404    except OSError:405        return None406    if stat.S_ISDIR(st.st_mode):407        return None408    return st.st_size, st.st_mtime * 1000409 410 411def _list_local_files(local_path: str) -> Iterator[tuple[str, int, float]]:412    """List all files in a local directory.413 414    Yields:415        tuple: (relative_path, size, mtime_ms) for each file416    """417    local_path = os.path.abspath(local_path)418    if not os.path.isdir(local_path):419        raise ValueError(f"Local path must be a directory: {local_path}")420 421    for root, _, files in os.walk(local_path):422        for filename in files:423            full_path = os.path.join(root, filename)424            stat_info = _stat_local(full_path)425            if stat_info is None:426                continue427            rel_path = os.path.relpath(full_path, local_path)428            # Normalize to forward slashes for consistency429            rel_path = rel_path.replace(os.sep, "/")430            yield rel_path, stat_info[0], stat_info[1]431 432 433def _list_remote_files(api: "HfApi", bucket_id: str, prefix: str) -> Iterator[tuple[str, int, float, Any]]:434    """List all files in a bucket with a given prefix.435 436    Yields:437        tuple: (relative_path, size, mtime_ms, bucket_file) for each file.438            bucket_file is the BucketFile object from list_bucket_tree.439    """440    for item in api.list_bucket_tree(bucket_id, prefix=prefix or None, recursive=True):441        if isinstance(item, BucketFolder):442            continue443        path = item.path444        # Remove prefix from path to get relative path445        # Only strip prefix if it's followed by "/" (directory boundary) or is exact match446        if prefix:447            if path.startswith(prefix + "/"):448                rel_path = path[len(prefix) + 1 :]449            elif path == prefix:450                # Exact match: the file IS the prefix (e.g., single file download)451                rel_path = path.rsplit("/", 1)[-1] if "/" in path else path452            else:453                # Path doesn't match prefix pattern (e.g., "submarine.txt" for prefix "sub")454                # Skip this file - it was returned by the API but doesn't belong to this prefix455                continue456        else:457            rel_path = path458        mtime_ms = item.mtime.timestamp() * 1000 if item.mtime else 0459        yield rel_path, item.size, mtime_ms, item460 461 462# =============================================================================463# Sync plan computation464# =============================================================================465 466 467def _mtime_to_iso(mtime_ms: float) -> str:468    """Convert mtime in milliseconds to ISO format string."""469    return datetime.fromtimestamp(mtime_ms / 1000, tz=timezone.utc).isoformat()470 471 472def _compare_files_for_sync(473    *,474    path: str,475    action: Literal["upload", "download"],476    source_size: int,477    source_mtime: float,478    dest_size: int,479    dest_mtime: float,480    source_newer_label: str,481    dest_newer_label: str,482    ignore_sizes: bool,483    ignore_times: bool,484    ignore_existing: bool,485    bucket_file: Any | None = None,486) -> SyncOperation:487    """Compare source and dest files and return the appropriate sync operation.488 489    This is a unified helper for both upload and download directions.490 491    Args:492        path: Relative file path493        action: "upload" or "download"494        source_size: Size of the source file (bytes)495        source_mtime: Mtime of the source file (milliseconds)496        dest_size: Size of the destination file (bytes)497        dest_mtime: Mtime of the destination file (milliseconds)498        source_newer_label: Label when source is newer (e.g., "local newer" or "remote newer")499        dest_newer_label: Label when dest is newer (e.g., "remote newer" or "local newer")500        ignore_sizes: Only compare mtime501        ignore_times: Only compare size502        ignore_existing: Skip files that exist on receiver503        bucket_file: BucketFile object (for downloads only)504 505    Returns:506        SyncOperation describing the action to take507    """508    local_mtime_iso = _mtime_to_iso(source_mtime if action == "upload" else dest_mtime)509    remote_mtime_iso = _mtime_to_iso(dest_mtime if action == "upload" else source_mtime)510 511    base_kwargs: dict[str, Any] = {512        "path": path,513        "size": source_size,514        "local_mtime": local_mtime_iso,515        "remote_mtime": remote_mtime_iso,516    }517 518    if ignore_existing:519        return SyncOperation(action="skip", reason="exists on receiver (--ignore-existing)", **base_kwargs)520 521    size_differs = source_size != dest_size522    source_newer = (source_mtime - dest_mtime) > _SYNC_TIME_WINDOW_MS523 524    if ignore_sizes:525        if source_newer:526            return SyncOperation(action=action, reason=source_newer_label, bucket_file=bucket_file, **base_kwargs)527        else:528            dest_newer = (dest_mtime - source_mtime) > _SYNC_TIME_WINDOW_MS529            skip_reason = dest_newer_label if dest_newer else "same mtime"530            return SyncOperation(action="skip", reason=skip_reason, **base_kwargs)531    elif ignore_times:532        if size_differs:533            return SyncOperation(action=action, reason="size differs", bucket_file=bucket_file, **base_kwargs)534        else:535            return SyncOperation(action="skip", reason="same size", **base_kwargs)536    else:537        if size_differs or source_newer:538            reason = "size differs" if size_differs else source_newer_label539            return SyncOperation(action=action, reason=reason, bucket_file=bucket_file, **base_kwargs)540        else:541            return SyncOperation(action="skip", reason="identical", **base_kwargs)542 543 544def _compute_sync_plan(545    source: str,546    dest: str,547    api: "HfApi",548    delete: bool = False,549    ignore_times: bool = False,550    ignore_sizes: bool = False,551    existing: bool = False,552    ignore_existing: bool = False,553    filter_matcher: FilterMatcher | None = None,554    status: Any | None = None,555) -> SyncPlan:556    """Compute the sync plan by comparing source and destination.557 558    Returns:559        SyncPlan with all operations to be performed560    """561    filter_matcher = filter_matcher or FilterMatcher()562    is_upload = not _is_bucket_path(source) and _is_bucket_path(dest)563    is_download = _is_bucket_path(source) and not _is_bucket_path(dest)564 565    if not is_upload and not is_download:566        raise ValueError("One of source or dest must be a bucket path (hf://buckets/...) and the other must be local.")567 568    plan = SyncPlan(569        source=source,570        dest=dest,571        timestamp=datetime.now(timezone.utc).isoformat(),572    )573 574    remote_total: int | None = None575    if is_upload:576        # Local -> Remote577        local_path = os.path.abspath(source)578        bucket_id, prefix = _parse_bucket_path(dest)579 580        if not os.path.isdir(local_path):581            raise ValueError(f"Source must be a directory: {local_path}")582 583        # Get local and remote file lists584        local_files = {}585        for rel_path, size, mtime_ms in _list_local_files(local_path):586            if filter_matcher.matches(rel_path):587                local_files[rel_path] = (size, mtime_ms)588            if status:589                status.update(f"Scanning local directory ({len(local_files)} files)")590        if status:591            status.done(f"Scanning local directory ({len(local_files)} files)")592 593        remote_files = {}594        if status:595            try:596                remote_total = api.bucket_info(bucket_id).total_files597            except Exception:598                pass599        try:600            for rel_path, size, mtime_ms, _ in _list_remote_files(api, bucket_id, prefix):601                if filter_matcher.matches(rel_path):602                    remote_files[rel_path] = (size, mtime_ms)603                if status:604                    total_str = f"/{remote_total}" if remote_total is not None else ""605                    status.update(f"Scanning remote bucket ({len(remote_files)}{total_str} files)")606        except BucketNotFoundError:607            # Bucket doesn't exist yet - this is expected for new uploads608            logger.debug(f"Bucket '{bucket_id}' not found, treating as empty.")609        if status:610            status.done(f"Scanning remote bucket ({len(remote_files)} files)")611 612        # Compare files613        all_paths = set(local_files.keys()) | set(remote_files.keys())614        if status:615            status.done(f"Comparing files ({len(all_paths)} paths)")616        for path in sorted(all_paths):617            local_info = local_files.get(path)618            remote_info = remote_files.get(path)619 620            if local_info and not remote_info:621                # New file622                if existing:623                    # --existing: skip new files624                    plan.operations.append(625                        SyncOperation(626                            action="skip",627                            path=path,628                            size=local_info[0],629                            reason="new file (--existing)",630                            local_mtime=_mtime_to_iso(local_info[1]),631                        )632                    )633                else:634                    plan.operations.append(635                        SyncOperation(636                            action="upload",637                            path=path,638                            size=local_info[0],639                            reason="new file",640                            local_mtime=_mtime_to_iso(local_info[1]),641                        )642                    )643            elif local_info and remote_info:644                # File exists in both - use helper to determine action645                local_size, local_mtime = local_info646                remote_size, remote_mtime = remote_info647                plan.operations.append(648                    _compare_files_for_sync(649                        path=path,650                        action="upload",651                        source_size=local_size,652                        source_mtime=local_mtime,653                        dest_size=remote_size,654                        dest_mtime=remote_mtime,655                        source_newer_label="local newer",656                        dest_newer_label="remote newer",657                        ignore_sizes=ignore_sizes,658                        ignore_times=ignore_times,659                        ignore_existing=ignore_existing,660                    )661                )662            elif not local_info and remote_info and delete:663                # File only in remote and --delete mode664                plan.operations.append(665                    SyncOperation(666                        action="delete",667                        path=path,668                        size=remote_info[0],669                        reason="not in source (--delete)",670                        remote_mtime=_mtime_to_iso(remote_info[1]),671                    )672                )673 674    else:675        # Remote -> Local (download)676        bucket_id, prefix = _parse_bucket_path(source)677        local_path = os.path.abspath(dest)678 679        # Get remote and local file lists680        remote_files = {}681        bucket_file_map: dict[str, Any] = {}682        if status:683            try:684                remote_total = api.bucket_info(bucket_id).total_files685            except Exception:686                pass687        for rel_path, size, mtime_ms, bucket_file in _list_remote_files(api, bucket_id, prefix):688            if filter_matcher.matches(rel_path):689                remote_files[rel_path] = (size, mtime_ms)690                bucket_file_map[rel_path] = bucket_file691            if status:692                total_str = f"/{remote_total}" if remote_total is not None else ""693                status.update(f"Scanning remote bucket ({len(remote_files)}{total_str} files)")694        if status:695            status.done(f"Scanning remote bucket ({len(remote_files)} files)")696 697        local_files = {}698        if os.path.isdir(local_path):699            if delete:700                # Full walk needed to discover local-only files for deletion.701                for rel_path, size, mtime_ms in _list_local_files(local_path):702                    if filter_matcher.matches(rel_path):703                        local_files[rel_path] = (size, mtime_ms)704                    if status:705                        status.update(f"Scanning local directory ({len(local_files)} files)")706            else:707                # Without --delete, the plan only depends on paths that exist708                # remotely. Stat just those instead of walking the whole tree,709                # which can take minutes when dest sits in a large directory710                # like ~/.cache/huggingface/.711                for rel_path in remote_files:712                    local_file = os.path.join(local_path, rel_path)713                    stat_info = _stat_local(local_file)714                    if stat_info is None:715                        continue716                    local_files[rel_path] = stat_info717                    if status:718                        status.update(f"Scanning local directory ({len(local_files)} files)")719        if status:720            status.done(f"Scanning local directory ({len(local_files)} files)")721 722        # Compare files723        all_paths = set(remote_files.keys()) | set(local_files.keys())724        if status:725            status.done(f"Comparing files ({len(all_paths)} paths)")726        for path in sorted(all_paths):727            remote_info = remote_files.get(path)728            local_info = local_files.get(path)729 730            if remote_info and not local_info:731                # New file732                if existing:733                    # --existing: skip new files734                    plan.operations.append(735                        SyncOperation(736                            action="skip",737                            path=path,738                            size=remote_info[0],739                            reason="new file (--existing)",740                            remote_mtime=_mtime_to_iso(remote_info[1]),741                        )742                    )743                else:744                    plan.operations.append(745                        SyncOperation(746                            action="download",747                            path=path,748                            size=remote_info[0],749                            reason="new file",750                            remote_mtime=_mtime_to_iso(remote_info[1]),751                            bucket_file=bucket_file_map.get(path),752                        )753                    )754            elif remote_info and local_info:755                # File exists in both - use helper to determine action756                remote_size, remote_mtime = remote_info757                local_size, local_mtime = local_info758                plan.operations.append(759                    _compare_files_for_sync(760                        path=path,761                        action="download",762                        source_size=remote_size,763                        source_mtime=remote_mtime,764                        dest_size=local_size,765                        dest_mtime=local_mtime,766                        source_newer_label="remote newer",767                        dest_newer_label="local newer",768                        ignore_sizes=ignore_sizes,769                        ignore_times=ignore_times,770                        ignore_existing=ignore_existing,771                        bucket_file=bucket_file_map.get(path),772                    )773                )774            elif not remote_info and local_info and delete:775                # File only in local and --delete mode776                plan.operations.append(777                    SyncOperation(778                        action="delete",779                        path=path,780                        size=local_info[0],781                        reason="not in source (--delete)",782                        local_mtime=_mtime_to_iso(local_info[1]),783                    )784                )785 786    return plan787 788 789# =============================================================================790# Plan serialization791# =============================================================================792 793 794def _write_plan(plan: SyncPlan, f) -> None:795    """Write a sync plan as JSONL to a file-like object."""796    # Write header797    header = {798        "type": "header",799        "source": plan.source,800        "dest": plan.dest,801        "timestamp": plan.timestamp,802        "summary": plan.summary(),803    }804    f.write(json.dumps(header) + "\n")805 806    # Write operations807    for op in plan.operations:808        op_dict: dict[str, Any] = {809            "type": "operation",810            "action": op.action,811            "path": op.path,812            "reason": op.reason,813        }814        if op.size is not None:815            op_dict["size"] = op.size816        if op.local_mtime is not None:817            op_dict["local_mtime"] = op.local_mtime818        if op.remote_mtime is not None:819            op_dict["remote_mtime"] = op.remote_mtime820        f.write(json.dumps(op_dict) + "\n")821 822 823def _save_plan(plan: SyncPlan, plan_file: str) -> None:824    """Save a sync plan to a JSONL file."""825    with open(plan_file, "w") as f:826        _write_plan(plan, f)827 828 829def _load_plan(plan_file: str) -> SyncPlan:830    """Load a sync plan from a JSONL file."""831    with open(plan_file) as f:832        lines = f.readlines()833 834    if not lines:835        raise ValueError(f"Empty plan file: {plan_file}")836 837    # Parse header838    header = json.loads(lines[0])839    if header.get("type") != "header":840        raise ValueError("Invalid plan file: expected header as first line")841 842    plan = SyncPlan(843        source=header["source"],844        dest=header["dest"],845        timestamp=header["timestamp"],846    )847 848    # Parse operations849    for line in lines[1:]:850        op_dict = json.loads(line)851        if op_dict.get("type") != "operation":852            continue853        plan.operations.append(854            SyncOperation(855                action=op_dict["action"],856                path=op_dict["path"],857                size=op_dict.get("size"),858                reason=op_dict.get("reason", ""),859                local_mtime=op_dict.get("local_mtime"),860                remote_mtime=op_dict.get("remote_mtime"),861            )862        )863 864    return plan865 866 867# =============================================================================868# Plan execution869# =============================================================================870 871 872def _execute_plan(plan: SyncPlan, api: "HfApi", verbose: bool = False, status: Any | None = None) -> None:873    """Execute a sync plan."""874    is_upload = not _is_bucket_path(plan.source) and _is_bucket_path(plan.dest)875    is_download = _is_bucket_path(plan.source) and not _is_bucket_path(plan.dest)876 877    if is_upload:878        local_path = os.path.abspath(plan.source)879        bucket_id, prefix = _parse_bucket_path(plan.dest)880        prefix = prefix.rstrip("/")  # Avoid double slashes in remote paths881 882        # Collect operations883        add_files: list[tuple[str | Path | bytes, str]] = []884        delete_paths: list[str] = []885 886        for op in plan.operations:887            match op.action:888                case "upload":889                    local_file = os.path.join(local_path, op.path)890                    remote_path = f"{prefix}/{op.path}" if prefix else op.path891                    if verbose:892                        print(f"  Uploading: {op.path} ({op.reason})")893                    add_files.append((local_file, remote_path))894                case "delete":895                    remote_path = f"{prefix}/{op.path}" if prefix else op.path896                    if verbose:897                        print(f"  Deleting: {op.path} ({op.reason})")898                    delete_paths.append(remote_path)899                case "skip" if verbose:900                    print(f"  Skipping: {op.path} ({op.reason})")901 902        # Execute batch operations903        if add_files or delete_paths:904            if status:905                parts = []906                if add_files:907                    parts.append(f"uploading {len(add_files)} files")908                if delete_paths:909                    parts.append(f"deleting {len(delete_paths)} files")910                status.done(", ".join(parts).capitalize())911            api.batch_bucket_files(912                bucket_id,913                add=add_files or None,914                delete=delete_paths or None,915            )916 917    elif is_download:918        bucket_id, prefix = _parse_bucket_path(plan.source)919        prefix = prefix.rstrip("/")  # Avoid double slashes in remote paths920        local_path = os.path.abspath(plan.dest)921 922        # Ensure local directory exists923        os.makedirs(local_path, exist_ok=True)924 925        # Collect download operations926        download_files: list[tuple[str | BucketFile, str | Path]] = []927        delete_files: list[str] = []928 929        for op in plan.operations:930            if op.action == "download":931                local_file = os.path.join(local_path, op.path)932                # Ensure parent directory exists933                os.makedirs(os.path.dirname(local_file), exist_ok=True)934                if verbose:935                    print(f"  Downloading: {op.path} ({op.reason})")936                # Use BucketFile when available (avoids extra metadata fetch per file)937                if op.bucket_file is not None:938                    download_files.append((op.bucket_file, local_file))939                else:940                    remote_path = f"{prefix}/{op.path}" if prefix else op.path941                    download_files.append((remote_path, local_file))942            elif op.action == "delete":943                local_file = os.path.join(local_path, op.path)944                if verbose:945                    print(f"  Deleting: {op.path} ({op.reason})")946                delete_files.append(local_file)947            elif op.action == "skip" and verbose:948                print(f"  Skipping: {op.path} ({op.reason})")949 950        # Execute downloads951        if len(download_files) > 0:952            if status:953                status.done(f"Downloading {len(download_files)} files")954            api.download_bucket_files(bucket_id, download_files)955 956        # Execute deletes957        if status and delete_files:958            status.done(f"Deleting {len(delete_files)} local files")959        for file_path in delete_files:960            if os.path.exists(file_path):961                os.remove(file_path)962                # Remove empty parent directories963                parent = os.path.dirname(file_path)964                while parent != local_path:965                    try:966                        os.rmdir(parent)967                        parent = os.path.dirname(parent)968                    except OSError:969                        break970 971 972def _print_plan_summary(plan: SyncPlan) -> None:973    """Print a summary of the sync plan."""974    summary = plan.summary()975    print(f"Sync plan: {plan.source} -> {plan.dest}")976    print(f"  Uploads: {summary['uploads']}")977    print(f"  Downloads: {summary['downloads']}")978    print(f"  Deletes: {summary['deletes']}")979    print(f"  Skips: {summary['skips']}")980 981 982# =============================================================================983# Public sync function (Python API)984# =============================================================================985 986 987def sync_bucket_internal(988    source: str | None = None,989    dest: str | None = None,990    *,991    api: "HfApi",992    delete: bool = False,993    ignore_times: bool = False,994    ignore_sizes: bool = False,995    existing: bool = False,996    ignore_existing: bool = False,997    include: list[str] | None = None,998    exclude: list[str] | None = None,999    filter_from: str | None = None,1000    plan: str | None = None,1001    apply: str | None = None,1002    dry_run: bool = False,1003    verbose: bool = False,1004    quiet: bool = False,1005    token: bool | str | None = None,1006) -> SyncPlan:1007    """Sync files between a local directory and a bucket.1008 1009    This is equivalent to the ``hf buckets sync`` CLI command. One of ``source`` or ``dest`` must be a bucket path1010    (``hf://buckets/...``) and the other must be a local directory path.1011 1012    Args:1013        source (`str`, *optional*):1014            Source path: local directory or ``hf://buckets/namespace/bucket_name(/prefix)``.1015            Required unless using ``apply``.1016        dest (`str`, *optional*):1017            Destination path: local directory or ``hf://buckets/namespace/bucket_name(/prefix)``.1018            Required unless using ``apply``.1019        api ([`HfApi`]):1020            The HfApi instance to use for API calls.1021        delete (`bool`, *optional*, defaults to `False`):1022            Delete destination files not present in source.1023        ignore_times (`bool`, *optional*, defaults to `False`):1024            Skip files only based on size, ignoring modification times.1025        ignore_sizes (`bool`, *optional*, defaults to `False`):1026            Skip files only based on modification times, ignoring sizes.1027        existing (`bool`, *optional*, defaults to `False`):1028            Skip creating new files on receiver (only update existing files).1029        ignore_existing (`bool`, *optional*, defaults to `False`):1030            Skip updating files that exist on receiver (only create new files).1031        include (`list[str]`, *optional*):1032            Include files matching patterns (fnmatch-style).1033        exclude (`list[str]`, *optional*):1034            Exclude files matching patterns (fnmatch-style).1035        filter_from (`str`, *optional*):1036            Path to a filter file with include/exclude rules.1037        plan (`str`, *optional*):1038            Save sync plan to this JSONL file instead of executing.1039        apply (`str`, *optional*):1040            Apply a previously saved plan file. When set, ``source`` and ``dest`` are not needed.1041        dry_run (`bool`, *optional*, defaults to `False`):1042            Print sync plan to stdout as JSONL without executing.1043        verbose (`bool`, *optional*, defaults to `False`):1044            Show detailed per-file operations.1045        quiet (`bool`, *optional*, defaults to `False`):1046            Suppress all output and progress bars.1047        token (Union[bool, str, None], optional):1048            A valid user access token. If not provided, the locally saved token will be used.1049 1050    Returns:1051        [`SyncPlan`]: The computed (or loaded) sync plan.1052 1053    Raises:1054        `ValueError`: If arguments are invalid (e.g., both paths are remote, conflicting options).1055 1056    Example:1057        ```python1058        >>> from huggingface_hub import HfApi1059        >>> api = HfApi()1060 1061        # Upload local directory to bucket1062        >>> api.sync_bucket("./data", "hf://buckets/username/my-bucket")1063 1064        # Download bucket to local directory1065        >>> api.sync_bucket("hf://buckets/username/my-bucket", "./data")1066 1067        # Sync with delete and filtering1068        >>> api.sync_bucket(1069        ...     "./data",1070        ...     "hf://buckets/username/my-bucket",1071        ...     delete=True,1072        ...     include=["*.safetensors"],1073        ... )1074 1075        # Dry run: preview what would be synced1076        >>> plan = api.sync_bucket("./data", "hf://buckets/username/my-bucket", dry_run=True)1077        >>> plan.summary()1078        {'uploads': 3, 'downloads': 0, 'deletes': 0, 'skips': 1, 'total_size': 4096}1079 1080        # Save plan for review, then apply1081        >>> api.sync_bucket("./data", "hf://buckets/username/my-bucket", plan="sync-plan.jsonl")1082        >>> api.sync_bucket(apply="sync-plan.jsonl")1083        ```1084    """1085    # Build API with token if needed1086    if token is not None:1087        from .hf_api import HfApi1088 1089        api = HfApi(token=token)1090    # --- Apply mode ---1091    if apply:1092        if source or dest:1093            raise ValueError("Cannot specify source/dest when using apply.")1094        if plan is not None:1095            raise ValueError("Cannot specify both plan and apply.")1096        if delete:1097            raise ValueError("Cannot specify delete when using apply.")1098        if ignore_times:1099            raise ValueError("Cannot specify ignore_times when using apply.")1100        if ignore_sizes:1101            raise ValueError("Cannot specify ignore_sizes when using apply.")1102        if include:1103            raise ValueError("Cannot specify include when using apply.")1104        if exclude:1105            raise ValueError("Cannot specify exclude when using apply.")1106        if filter_from:1107            raise ValueError("Cannot specify filter_from when using apply.")1108        if existing:1109            raise ValueError("Cannot specify existing when using apply.")1110        if ignore_existing:1111            raise ValueError("Cannot specify ignore_existing when using apply.")1112        if dry_run:1113            raise ValueError("Cannot specify dry_run when using apply.")1114 1115        sync_plan = _load_plan(apply)1116        status = StatusLine(enabled=not quiet)1117        if not quiet:1118            _print_plan_summary(sync_plan)1119            print("Executing plan...")1120 1121        if quiet:1122            disable_progress_bars()1123        try:1124            _execute_plan(sync_plan, api, verbose=verbose, status=status)1125        finally:1126            if quiet:1127                enable_progress_bars()1128 1129        if not quiet:1130            print("Sync completed.")1131 1132        return sync_plan1133 1134    # --- Normal mode ---1135    if not source or not dest:1136        raise ValueError("Both source and dest are required (unless using apply).")1137 1138    source_is_bucket = _is_bucket_path(source)1139    dest_is_bucket = _is_bucket_path(dest)1140 1141    if source_is_bucket and dest_is_bucket:1142        raise ValueError("Remote to remote sync is not supported. One path must be local.")1143 1144    if not source_is_bucket and not dest_is_bucket:1145        raise ValueError("One of source or dest must be a bucket path (hf://buckets/...).")1146 1147    if ignore_times and ignore_sizes:1148        raise ValueError("Cannot specify both ignore_times and ignore_sizes.")1149 1150    if existing and ignore_existing:1151        raise ValueError("Cannot specify both existing and ignore_existing.")1152 1153    if dry_run and plan:1154        raise ValueError("Cannot specify both dry_run and plan.")1155 1156    # Validate local path1157    if source_is_bucket:1158        if os.path.exists(dest) and not os.path.isdir(dest):1159            raise ValueError(f"Destination must be a directory: {dest}")1160    else:1161        if not os.path.isdir(source):1162            raise ValueError(f"Source must be an existing directory: {source}")1163 1164    # Build filter matcher1165    filter_rules = None1166    if filter_from:1167        filter_rules = _parse_filter_file(filter_from)1168 1169    filter_matcher = FilterMatcher(1170        include_patterns=include,1171        exclude_patterns=exclude,1172        filter_rules=filter_rules,1173    )1174 1175    # Compute sync plan1176    status = StatusLine(enabled=not quiet and not dry_run)1177    sync_plan = _compute_sync_plan(1178        source=source,1179        dest=dest,1180        api=api,1181        delete=delete,1182        ignore_times=ignore_times,1183        ignore_sizes=ignore_sizes,1184        existing=existing,1185        ignore_existing=ignore_existing,1186        filter_matcher=filter_matcher,1187        status=status,1188    )1189 1190    if dry_run:1191        _write_plan(sync_plan, sys.stdout)1192        return sync_plan1193 1194    if plan:1195        _save_plan(sync_plan, plan)1196        if not quiet:1197            _print_plan_summary(sync_plan)1198            print(f"Plan saved to: {plan}")1199        return sync_plan1200 

Showing the first 1,200 of 1226 lines. Download the file for the rest.

codekingpro/portable-devtools · Team Ai