codekingpro/portable-devtools
114k
1# Copyright 2026-present, the HuggingFace Inc. team.2#3# Licensed under the Apache License, Version 2.0 (the "License");4# you may not use this file except in compliance with the License.5# You may obtain a copy of the License at6#7# http://www.apache.org/licenses/LICENSE-2.08#9# Unless required by applicable law or agreed to in writing, software10# distributed under the License is distributed on an "AS IS" BASIS,11# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.12# See the License for the specific language governing permissions and13# limitations under the License.14"""Shared logic for bucket operations.15 16This module contains the core buckets logic used by both the CLI and the Python API.17"""18 19import fnmatch20import json21import mimetypes22import os23import stat24import sys25import time26from collections.abc import Iterator27from dataclasses import dataclass, field28from datetime import datetime, timezone29from pathlib import Path30from typing import TYPE_CHECKING, Any, Literal31 32from . import constants, logging33from .errors import BucketNotFoundError34from .utils import XetFileData, disable_progress_bars, enable_progress_bars, parse_datetime35from .utils._terminal import StatusLine36 37 38if TYPE_CHECKING:39 from .hf_api import HfApi40 41 42logger = logging.get_logger(__name__)43 44 45BUCKET_PREFIX = "hf://buckets/"46_SYNC_TIME_WINDOW_MS = 1000 # 1s safety-window for file modification time comparisons47 48 49# =============================================================================50# Bucket data structures51# =============================================================================52 53 54def _split_bucket_id_and_prefix(path: str) -> tuple[str, str]:55 """Split 'namespace/name(/optional/prefix)' into ('namespace/name', 'prefix').56 57 Returns (bucket_id, prefix) where prefix may be empty string.58 Raises ValueError if path doesn't contain at least namespace/name.59 """60 parts = path.split("/", 2)61 if len(parts) < 2 or not parts[0] or not parts[1]:62 raise ValueError(f"Invalid bucket path: '{path}'. Expected format: namespace/bucket_name")63 bucket_id = f"{parts[0]}/{parts[1]}"64 prefix = parts[2] if len(parts) > 2 else ""65 return bucket_id, prefix66 67 68@dataclass69class BucketInfo:70 """71 Contains information about a bucket on the Hub. This object is returned by [`bucket_info`] and [`list_buckets`].72 73 Attributes:74 id (`str`):75 ID of the bucket.76 private (`bool`):77 Is the bucket private.78 created_at (`datetime`):79 Date of creation of the bucket on the Hub.80 size (`int`):81 Size of the bucket in bytes.82 total_files (`int`):83 Total number of files in the bucket.84 """85 86 id: str87 private: bool88 created_at: datetime89 size: int90 total_files: int91 92 def __init__(self, **kwargs):93 self.id = kwargs.pop("id")94 self.private = kwargs.pop("private")95 self.created_at = parse_datetime(kwargs.pop("createdAt"))96 self.size = kwargs.pop("size")97 self.total_files = kwargs.pop("totalFiles")98 self.__dict__.update(**kwargs)99 100 101@dataclass102class _BucketAddFile:103 source: str | Path | bytes104 destination: str105 106 xet_hash: str | None = field(default=None)107 size: int | None = field(default=None)108 mtime: int = field(init=False)109 content_type: str | None = field(init=False)110 111 def __post_init__(self) -> None:112 self.content_type = None113 if isinstance(self.source, (str, Path)): # guess content type from source path114 self.content_type = mimetypes.guess_type(self.source)[0]115 if self.content_type is None: # or default to destination path content type116 self.content_type = mimetypes.guess_type(self.destination)[0]117 118 self.mtime = int(119 os.path.getmtime(self.source) * 1000 if not isinstance(self.source, bytes) else time.time() * 1000120 )121 122 123@dataclass124class _BucketCopyFile:125 destination: str126 xet_hash: str127 source_repo_type: str # "model", "dataset", "space", "bucket"128 source_repo_id: str129 size: int | None = field(default=None)130 mtime: int = field(init=False)131 content_type: str | None = field(init=False)132 133 def __post_init__(self) -> None:134 self.content_type = mimetypes.guess_type(self.destination)[0]135 self.mtime = int(time.time() * 1000)136 137 138@dataclass139class _BucketDeleteFile:140 path: str141 142 143@dataclass(frozen=True)144class BucketFileMetadata:145 """Data structure containing information about a file in a bucket.146 147 Returned by [`get_bucket_file_metadata`].148 149 Args:150 size (`int`):151 Size of the file in bytes.152 xet_file_data (`XetFileData`):153 Xet information for the file (hash and refresh route).154 """155 156 size: int157 xet_file_data: XetFileData158 159 160@dataclass161class BucketUrl:162 """Describes a bucket URL on the Hub.163 164 `BucketUrl` is returned by [`create_bucket`]. At initialization, the URL is parsed to populate properties:165 - endpoint (`str`)166 - namespace (`str`)167 - bucket_id (`str`)168 - url (`str`)169 - handle (`str`)170 171 Args:172 url (`str`):173 String value of the bucket url.174 endpoint (`str`, *optional*):175 Endpoint of the Hub. Defaults to <https://huggingface.co>.176 """177 178 url: str179 endpoint: str = ""180 namespace: str = field(init=False)181 bucket_id: str = field(init=False)182 handle: str = field(init=False)183 184 def __post_init__(self) -> None:185 self.endpoint = self.endpoint or constants.ENDPOINT186 187 # Parse URL: expected format is `{endpoint}/buckets/{namespace}/{bucket_name}`188 url_path = self.url.replace(self.endpoint, "").strip("/")189 # Remove leading "buckets/" prefix190 if url_path.startswith("buckets/"):191 url_path = url_path[len("buckets/") :]192 bucket_id, prefix = _split_bucket_id_and_prefix(url_path)193 if prefix:194 raise ValueError(f"Unable to parse bucket URL: {self.url}")195 self.namespace = bucket_id.split("/")[0]196 self.bucket_id = bucket_id197 198 self.handle = f"hf://buckets/{self.bucket_id}"199 200 201@dataclass202class BucketFile:203 """204 Contains information about a file in a bucket on the Hub. This object is returned by [`list_bucket_tree`].205 206 Similar to [`RepoFile`] but for files in buckets.207 """208 209 type: Literal["file"]210 path: str211 size: int212 xet_hash: str213 mtime: datetime | None214 uploaded_at: datetime | None215 216 def __init__(self, **kwargs):217 self.type = kwargs.pop("type")218 self.path = kwargs.pop("path")219 self.size = kwargs.pop("size")220 self.xet_hash = kwargs.pop("xetHash")221 mtime = kwargs.pop("mtime", None)222 self.mtime = parse_datetime(mtime) if mtime else None223 uploaded_at = kwargs.pop("uploadedAt", None)224 self.uploaded_at = parse_datetime(uploaded_at) if uploaded_at else None225 226 227@dataclass228class BucketFolder:229 """230 Contains information about a directory in a bucket on the Hub. This object is returned by [`list_bucket_tree`].231 232 Similar to [`RepoFolder`] but for directories in buckets.233 """234 235 type: Literal["directory"]236 path: str237 uploaded_at: datetime | None238 239 def __init__(self, **kwargs):240 self.type = kwargs.pop("type")241 self.path = kwargs.pop("path")242 uploaded_at = kwargs.pop("uploadedAt", None) or kwargs.pop("uploaded_at", None)243 self.uploaded_at = (244 (uploaded_at if isinstance(uploaded_at, datetime) else parse_datetime(uploaded_at))245 if uploaded_at246 else None247 )248 249 250# =============================================================================251# Bucket path parsing252# =============================================================================253 254 255def _parse_bucket_path(path: str) -> tuple[str, str]:256 """Parse a bucket path like hf://buckets/namespace/bucket_name/prefix into (bucket_id, prefix).257 258 Returns:259 tuple: (bucket_id, prefix) where bucket_id is "namespace/bucket_name" and prefix may be empty string.260 """261 if not path.startswith(BUCKET_PREFIX):262 raise ValueError(f"Invalid bucket path: {path}. Must start with {BUCKET_PREFIX}")263 return _split_bucket_id_and_prefix(path.removeprefix(BUCKET_PREFIX))264 265 266def _is_bucket_path(path: str) -> bool:267 """Check if a path is a bucket path."""268 return path.startswith(BUCKET_PREFIX)269 270 271# =============================================================================272# Sync data structures273# =============================================================================274 275 276@dataclass277class SyncOperation:278 """Represents a sync operation to be performed."""279 280 action: Literal["upload", "download", "delete", "skip"]281 path: str282 size: int | None = None283 reason: str = ""284 local_mtime: str | None = None285 remote_mtime: str | None = None286 bucket_file: BucketFile | None = None # BucketFile when available (not serialized to plan file)287 288 289@dataclass290class SyncPlan:291 """Represents a complete sync plan."""292 293 source: str294 dest: str295 timestamp: str296 operations: list[SyncOperation] = field(default_factory=list)297 298 def summary(self) -> dict[str, int | str]:299 uploads = sum(1 for op in self.operations if op.action == "upload")300 downloads = sum(1 for op in self.operations if op.action == "download")301 deletes = sum(1 for op in self.operations if op.action == "delete")302 skips = sum(1 for op in self.operations if op.action == "skip")303 total_size = sum(op.size or 0 for op in self.operations if op.action in ("upload", "download"))304 return {305 "uploads": uploads,306 "downloads": downloads,307 "deletes": deletes,308 "skips": skips,309 "total_size": total_size,310 }311 312 313# =============================================================================314# Filter matching315# =============================================================================316 317 318class FilterMatcher:319 """Matches file paths against include/exclude patterns."""320 321 def __init__(322 self,323 include_patterns: list[str] | None = None,324 exclude_patterns: list[str] | None = None,325 filter_rules: list[tuple[str, str]] | None = None,326 ):327 """Initialize the filter matcher.328 329 Args:330 include_patterns: Patterns to include (from --include)331 exclude_patterns: Patterns to exclude (from --exclude)332 filter_rules: Rules from filter file as list of ("+"/"-", pattern) tuples333 """334 self.include_patterns = include_patterns or []335 self.exclude_patterns = exclude_patterns or []336 self.filter_rules = filter_rules or []337 338 def matches(self, path: str) -> bool:339 """Check if a path should be included based on the filter rules.340 341 Filtering rules:342 - Filters are evaluated in order, first matching rule decides343 - If no rules match, include by default (unless include patterns are specified)344 """345 # First check filter rules from file (in order)346 for sign, pattern in self.filter_rules:347 if fnmatch.fnmatch(path, pattern):348 return sign == "+"349 350 # Then check CLI patterns351 for pattern in self.exclude_patterns:352 if fnmatch.fnmatch(path, pattern):353 return False354 355 for pattern in self.include_patterns:356 if fnmatch.fnmatch(path, pattern):357 return True358 359 # If include patterns were specified but none matched, exclude360 if self.include_patterns:361 return False362 363 # Default: include364 return True365 366 367def _parse_filter_file(filter_file: str) -> list[tuple[str, str]]:368 """Parse a filter file and return a list of (sign, pattern) tuples.369 370 Filter file format:371 - Lines starting with "+" are include patterns372 - Lines starting with "-" are exclude patterns373 - Empty lines and lines starting with "#" are ignored374 """375 rules = []376 with open(filter_file) as f:377 for line in f:378 line = line.strip()379 if not line or line.startswith("#"):380 continue381 if line.startswith("+"):382 rules.append(("+", line[1:].strip()))383 elif line.startswith("-"):384 rules.append(("-", line[1:].strip()))385 else:386 # Default to include if no prefix387 rules.append(("+", line))388 return rules389 390 391# =============================================================================392# File listing393# =============================================================================394 395 396def _stat_local(path: str) -> tuple[int, float] | None:397 """Stat a local file and return (size, mtime_ms).398 399 Returns None if the path is missing or is a directory. Uses a single400 ``os.stat`` call so callers don't pay for multiple syscalls per file.401 """402 try:403 st = os.stat(path)404 except OSError:405 return None406 if stat.S_ISDIR(st.st_mode):407 return None408 return st.st_size, st.st_mtime * 1000409 410 411def _list_local_files(local_path: str) -> Iterator[tuple[str, int, float]]:412 """List all files in a local directory.413 414 Yields:415 tuple: (relative_path, size, mtime_ms) for each file416 """417 local_path = os.path.abspath(local_path)418 if not os.path.isdir(local_path):419 raise ValueError(f"Local path must be a directory: {local_path}")420 421 for root, _, files in os.walk(local_path):422 for filename in files:423 full_path = os.path.join(root, filename)424 stat_info = _stat_local(full_path)425 if stat_info is None:426 continue427 rel_path = os.path.relpath(full_path, local_path)428 # Normalize to forward slashes for consistency429 rel_path = rel_path.replace(os.sep, "/")430 yield rel_path, stat_info[0], stat_info[1]431 432 433def _list_remote_files(api: "HfApi", bucket_id: str, prefix: str) -> Iterator[tuple[str, int, float, Any]]:434 """List all files in a bucket with a given prefix.435 436 Yields:437 tuple: (relative_path, size, mtime_ms, bucket_file) for each file.438 bucket_file is the BucketFile object from list_bucket_tree.439 """440 for item in api.list_bucket_tree(bucket_id, prefix=prefix or None, recursive=True):441 if isinstance(item, BucketFolder):442 continue443 path = item.path444 # Remove prefix from path to get relative path445 # Only strip prefix if it's followed by "/" (directory boundary) or is exact match446 if prefix:447 if path.startswith(prefix + "/"):448 rel_path = path[len(prefix) + 1 :]449 elif path == prefix:450 # Exact match: the file IS the prefix (e.g., single file download)451 rel_path = path.rsplit("/", 1)[-1] if "/" in path else path452 else:453 # Path doesn't match prefix pattern (e.g., "submarine.txt" for prefix "sub")454 # Skip this file - it was returned by the API but doesn't belong to this prefix455 continue456 else:457 rel_path = path458 mtime_ms = item.mtime.timestamp() * 1000 if item.mtime else 0459 yield rel_path, item.size, mtime_ms, item460 461 462# =============================================================================463# Sync plan computation464# =============================================================================465 466 467def _mtime_to_iso(mtime_ms: float) -> str:468 """Convert mtime in milliseconds to ISO format string."""469 return datetime.fromtimestamp(mtime_ms / 1000, tz=timezone.utc).isoformat()470 471 472def _compare_files_for_sync(473 *,474 path: str,475 action: Literal["upload", "download"],476 source_size: int,477 source_mtime: float,478 dest_size: int,479 dest_mtime: float,480 source_newer_label: str,481 dest_newer_label: str,482 ignore_sizes: bool,483 ignore_times: bool,484 ignore_existing: bool,485 bucket_file: Any | None = None,486) -> SyncOperation:487 """Compare source and dest files and return the appropriate sync operation.488 489 This is a unified helper for both upload and download directions.490 491 Args:492 path: Relative file path493 action: "upload" or "download"494 source_size: Size of the source file (bytes)495 source_mtime: Mtime of the source file (milliseconds)496 dest_size: Size of the destination file (bytes)497 dest_mtime: Mtime of the destination file (milliseconds)498 source_newer_label: Label when source is newer (e.g., "local newer" or "remote newer")499 dest_newer_label: Label when dest is newer (e.g., "remote newer" or "local newer")500 ignore_sizes: Only compare mtime501 ignore_times: Only compare size502 ignore_existing: Skip files that exist on receiver503 bucket_file: BucketFile object (for downloads only)504 505 Returns:506 SyncOperation describing the action to take507 """508 local_mtime_iso = _mtime_to_iso(source_mtime if action == "upload" else dest_mtime)509 remote_mtime_iso = _mtime_to_iso(dest_mtime if action == "upload" else source_mtime)510 511 base_kwargs: dict[str, Any] = {512 "path": path,513 "size": source_size,514 "local_mtime": local_mtime_iso,515 "remote_mtime": remote_mtime_iso,516 }517 518 if ignore_existing:519 return SyncOperation(action="skip", reason="exists on receiver (--ignore-existing)", **base_kwargs)520 521 size_differs = source_size != dest_size522 source_newer = (source_mtime - dest_mtime) > _SYNC_TIME_WINDOW_MS523 524 if ignore_sizes:525 if source_newer:526 return SyncOperation(action=action, reason=source_newer_label, bucket_file=bucket_file, **base_kwargs)527 else:528 dest_newer = (dest_mtime - source_mtime) > _SYNC_TIME_WINDOW_MS529 skip_reason = dest_newer_label if dest_newer else "same mtime"530 return SyncOperation(action="skip", reason=skip_reason, **base_kwargs)531 elif ignore_times:532 if size_differs:533 return SyncOperation(action=action, reason="size differs", bucket_file=bucket_file, **base_kwargs)534 else:535 return SyncOperation(action="skip", reason="same size", **base_kwargs)536 else:537 if size_differs or source_newer:538 reason = "size differs" if size_differs else source_newer_label539 return SyncOperation(action=action, reason=reason, bucket_file=bucket_file, **base_kwargs)540 else:541 return SyncOperation(action="skip", reason="identical", **base_kwargs)542 543 544def _compute_sync_plan(545 source: str,546 dest: str,547 api: "HfApi",548 delete: bool = False,549 ignore_times: bool = False,550 ignore_sizes: bool = False,551 existing: bool = False,552 ignore_existing: bool = False,553 filter_matcher: FilterMatcher | None = None,554 status: Any | None = None,555) -> SyncPlan:556 """Compute the sync plan by comparing source and destination.557 558 Returns:559 SyncPlan with all operations to be performed560 """561 filter_matcher = filter_matcher or FilterMatcher()562 is_upload = not _is_bucket_path(source) and _is_bucket_path(dest)563 is_download = _is_bucket_path(source) and not _is_bucket_path(dest)564 565 if not is_upload and not is_download:566 raise ValueError("One of source or dest must be a bucket path (hf://buckets/...) and the other must be local.")567 568 plan = SyncPlan(569 source=source,570 dest=dest,571 timestamp=datetime.now(timezone.utc).isoformat(),572 )573 574 remote_total: int | None = None575 if is_upload:576 # Local -> Remote577 local_path = os.path.abspath(source)578 bucket_id, prefix = _parse_bucket_path(dest)579 580 if not os.path.isdir(local_path):581 raise ValueError(f"Source must be a directory: {local_path}")582 583 # Get local and remote file lists584 local_files = {}585 for rel_path, size, mtime_ms in _list_local_files(local_path):586 if filter_matcher.matches(rel_path):587 local_files[rel_path] = (size, mtime_ms)588 if status:589 status.update(f"Scanning local directory ({len(local_files)} files)")590 if status:591 status.done(f"Scanning local directory ({len(local_files)} files)")592 593 remote_files = {}594 if status:595 try:596 remote_total = api.bucket_info(bucket_id).total_files597 except Exception:598 pass599 try:600 for rel_path, size, mtime_ms, _ in _list_remote_files(api, bucket_id, prefix):601 if filter_matcher.matches(rel_path):602 remote_files[rel_path] = (size, mtime_ms)603 if status:604 total_str = f"/{remote_total}" if remote_total is not None else ""605 status.update(f"Scanning remote bucket ({len(remote_files)}{total_str} files)")606 except BucketNotFoundError:607 # Bucket doesn't exist yet - this is expected for new uploads608 logger.debug(f"Bucket '{bucket_id}' not found, treating as empty.")609 if status:610 status.done(f"Scanning remote bucket ({len(remote_files)} files)")611 612 # Compare files613 all_paths = set(local_files.keys()) | set(remote_files.keys())614 if status:615 status.done(f"Comparing files ({len(all_paths)} paths)")616 for path in sorted(all_paths):617 local_info = local_files.get(path)618 remote_info = remote_files.get(path)619 620 if local_info and not remote_info:621 # New file622 if existing:623 # --existing: skip new files624 plan.operations.append(625 SyncOperation(626 action="skip",627 path=path,628 size=local_info[0],629 reason="new file (--existing)",630 local_mtime=_mtime_to_iso(local_info[1]),631 )632 )633 else:634 plan.operations.append(635 SyncOperation(636 action="upload",637 path=path,638 size=local_info[0],639 reason="new file",640 local_mtime=_mtime_to_iso(local_info[1]),641 )642 )643 elif local_info and remote_info:644 # File exists in both - use helper to determine action645 local_size, local_mtime = local_info646 remote_size, remote_mtime = remote_info647 plan.operations.append(648 _compare_files_for_sync(649 path=path,650 action="upload",651 source_size=local_size,652 source_mtime=local_mtime,653 dest_size=remote_size,654 dest_mtime=remote_mtime,655 source_newer_label="local newer",656 dest_newer_label="remote newer",657 ignore_sizes=ignore_sizes,658 ignore_times=ignore_times,659 ignore_existing=ignore_existing,660 )661 )662 elif not local_info and remote_info and delete:663 # File only in remote and --delete mode664 plan.operations.append(665 SyncOperation(666 action="delete",667 path=path,668 size=remote_info[0],669 reason="not in source (--delete)",670 remote_mtime=_mtime_to_iso(remote_info[1]),671 )672 )673 674 else:675 # Remote -> Local (download)676 bucket_id, prefix = _parse_bucket_path(source)677 local_path = os.path.abspath(dest)678 679 # Get remote and local file lists680 remote_files = {}681 bucket_file_map: dict[str, Any] = {}682 if status:683 try:684 remote_total = api.bucket_info(bucket_id).total_files685 except Exception:686 pass687 for rel_path, size, mtime_ms, bucket_file in _list_remote_files(api, bucket_id, prefix):688 if filter_matcher.matches(rel_path):689 remote_files[rel_path] = (size, mtime_ms)690 bucket_file_map[rel_path] = bucket_file691 if status:692 total_str = f"/{remote_total}" if remote_total is not None else ""693 status.update(f"Scanning remote bucket ({len(remote_files)}{total_str} files)")694 if status:695 status.done(f"Scanning remote bucket ({len(remote_files)} files)")696 697 local_files = {}698 if os.path.isdir(local_path):699 if delete:700 # Full walk needed to discover local-only files for deletion.701 for rel_path, size, mtime_ms in _list_local_files(local_path):702 if filter_matcher.matches(rel_path):703 local_files[rel_path] = (size, mtime_ms)704 if status:705 status.update(f"Scanning local directory ({len(local_files)} files)")706 else:707 # Without --delete, the plan only depends on paths that exist708 # remotely. Stat just those instead of walking the whole tree,709 # which can take minutes when dest sits in a large directory710 # like ~/.cache/huggingface/.711 for rel_path in remote_files:712 local_file = os.path.join(local_path, rel_path)713 stat_info = _stat_local(local_file)714 if stat_info is None:715 continue716 local_files[rel_path] = stat_info717 if status:718 status.update(f"Scanning local directory ({len(local_files)} files)")719 if status:720 status.done(f"Scanning local directory ({len(local_files)} files)")721 722 # Compare files723 all_paths = set(remote_files.keys()) | set(local_files.keys())724 if status:725 status.done(f"Comparing files ({len(all_paths)} paths)")726 for path in sorted(all_paths):727 remote_info = remote_files.get(path)728 local_info = local_files.get(path)729 730 if remote_info and not local_info:731 # New file732 if existing:733 # --existing: skip new files734 plan.operations.append(735 SyncOperation(736 action="skip",737 path=path,738 size=remote_info[0],739 reason="new file (--existing)",740 remote_mtime=_mtime_to_iso(remote_info[1]),741 )742 )743 else:744 plan.operations.append(745 SyncOperation(746 action="download",747 path=path,748 size=remote_info[0],749 reason="new file",750 remote_mtime=_mtime_to_iso(remote_info[1]),751 bucket_file=bucket_file_map.get(path),752 )753 )754 elif remote_info and local_info:755 # File exists in both - use helper to determine action756 remote_size, remote_mtime = remote_info757 local_size, local_mtime = local_info758 plan.operations.append(759 _compare_files_for_sync(760 path=path,761 action="download",762 source_size=remote_size,763 source_mtime=remote_mtime,764 dest_size=local_size,765 dest_mtime=local_mtime,766 source_newer_label="remote newer",767 dest_newer_label="local newer",768 ignore_sizes=ignore_sizes,769 ignore_times=ignore_times,770 ignore_existing=ignore_existing,771 bucket_file=bucket_file_map.get(path),772 )773 )774 elif not remote_info and local_info and delete:775 # File only in local and --delete mode776 plan.operations.append(777 SyncOperation(778 action="delete",779 path=path,780 size=local_info[0],781 reason="not in source (--delete)",782 local_mtime=_mtime_to_iso(local_info[1]),783 )784 )785 786 return plan787 788 789# =============================================================================790# Plan serialization791# =============================================================================792 793 794def _write_plan(plan: SyncPlan, f) -> None:795 """Write a sync plan as JSONL to a file-like object."""796 # Write header797 header = {798 "type": "header",799 "source": plan.source,800 "dest": plan.dest,801 "timestamp": plan.timestamp,802 "summary": plan.summary(),803 }804 f.write(json.dumps(header) + "\n")805 806 # Write operations807 for op in plan.operations:808 op_dict: dict[str, Any] = {809 "type": "operation",810 "action": op.action,811 "path": op.path,812 "reason": op.reason,813 }814 if op.size is not None:815 op_dict["size"] = op.size816 if op.local_mtime is not None:817 op_dict["local_mtime"] = op.local_mtime818 if op.remote_mtime is not None:819 op_dict["remote_mtime"] = op.remote_mtime820 f.write(json.dumps(op_dict) + "\n")821 822 823def _save_plan(plan: SyncPlan, plan_file: str) -> None:824 """Save a sync plan to a JSONL file."""825 with open(plan_file, "w") as f:826 _write_plan(plan, f)827 828 829def _load_plan(plan_file: str) -> SyncPlan:830 """Load a sync plan from a JSONL file."""831 with open(plan_file) as f:832 lines = f.readlines()833 834 if not lines:835 raise ValueError(f"Empty plan file: {plan_file}")836 837 # Parse header838 header = json.loads(lines[0])839 if header.get("type") != "header":840 raise ValueError("Invalid plan file: expected header as first line")841 842 plan = SyncPlan(843 source=header["source"],844 dest=header["dest"],845 timestamp=header["timestamp"],846 )847 848 # Parse operations849 for line in lines[1:]:850 op_dict = json.loads(line)851 if op_dict.get("type") != "operation":852 continue853 plan.operations.append(854 SyncOperation(855 action=op_dict["action"],856 path=op_dict["path"],857 size=op_dict.get("size"),858 reason=op_dict.get("reason", ""),859 local_mtime=op_dict.get("local_mtime"),860 remote_mtime=op_dict.get("remote_mtime"),861 )862 )863 864 return plan865 866 867# =============================================================================868# Plan execution869# =============================================================================870 871 872def _execute_plan(plan: SyncPlan, api: "HfApi", verbose: bool = False, status: Any | None = None) -> None:873 """Execute a sync plan."""874 is_upload = not _is_bucket_path(plan.source) and _is_bucket_path(plan.dest)875 is_download = _is_bucket_path(plan.source) and not _is_bucket_path(plan.dest)876 877 if is_upload:878 local_path = os.path.abspath(plan.source)879 bucket_id, prefix = _parse_bucket_path(plan.dest)880 prefix = prefix.rstrip("/") # Avoid double slashes in remote paths881 882 # Collect operations883 add_files: list[tuple[str | Path | bytes, str]] = []884 delete_paths: list[str] = []885 886 for op in plan.operations:887 match op.action:888 case "upload":889 local_file = os.path.join(local_path, op.path)890 remote_path = f"{prefix}/{op.path}" if prefix else op.path891 if verbose:892 print(f" Uploading: {op.path} ({op.reason})")893 add_files.append((local_file, remote_path))894 case "delete":895 remote_path = f"{prefix}/{op.path}" if prefix else op.path896 if verbose:897 print(f" Deleting: {op.path} ({op.reason})")898 delete_paths.append(remote_path)899 case "skip" if verbose:900 print(f" Skipping: {op.path} ({op.reason})")901 902 # Execute batch operations903 if add_files or delete_paths:904 if status:905 parts = []906 if add_files:907 parts.append(f"uploading {len(add_files)} files")908 if delete_paths:909 parts.append(f"deleting {len(delete_paths)} files")910 status.done(", ".join(parts).capitalize())911 api.batch_bucket_files(912 bucket_id,913 add=add_files or None,914 delete=delete_paths or None,915 )916 917 elif is_download:918 bucket_id, prefix = _parse_bucket_path(plan.source)919 prefix = prefix.rstrip("/") # Avoid double slashes in remote paths920 local_path = os.path.abspath(plan.dest)921 922 # Ensure local directory exists923 os.makedirs(local_path, exist_ok=True)924 925 # Collect download operations926 download_files: list[tuple[str | BucketFile, str | Path]] = []927 delete_files: list[str] = []928 929 for op in plan.operations:930 if op.action == "download":931 local_file = os.path.join(local_path, op.path)932 # Ensure parent directory exists933 os.makedirs(os.path.dirname(local_file), exist_ok=True)934 if verbose:935 print(f" Downloading: {op.path} ({op.reason})")936 # Use BucketFile when available (avoids extra metadata fetch per file)937 if op.bucket_file is not None:938 download_files.append((op.bucket_file, local_file))939 else:940 remote_path = f"{prefix}/{op.path}" if prefix else op.path941 download_files.append((remote_path, local_file))942 elif op.action == "delete":943 local_file = os.path.join(local_path, op.path)944 if verbose:945 print(f" Deleting: {op.path} ({op.reason})")946 delete_files.append(local_file)947 elif op.action == "skip" and verbose:948 print(f" Skipping: {op.path} ({op.reason})")949 950 # Execute downloads951 if len(download_files) > 0:952 if status:953 status.done(f"Downloading {len(download_files)} files")954 api.download_bucket_files(bucket_id, download_files)955 956 # Execute deletes957 if status and delete_files:958 status.done(f"Deleting {len(delete_files)} local files")959 for file_path in delete_files:960 if os.path.exists(file_path):961 os.remove(file_path)962 # Remove empty parent directories963 parent = os.path.dirname(file_path)964 while parent != local_path:965 try:966 os.rmdir(parent)967 parent = os.path.dirname(parent)968 except OSError:969 break970 971 972def _print_plan_summary(plan: SyncPlan) -> None:973 """Print a summary of the sync plan."""974 summary = plan.summary()975 print(f"Sync plan: {plan.source} -> {plan.dest}")976 print(f" Uploads: {summary['uploads']}")977 print(f" Downloads: {summary['downloads']}")978 print(f" Deletes: {summary['deletes']}")979 print(f" Skips: {summary['skips']}")980 981 982# =============================================================================983# Public sync function (Python API)984# =============================================================================985 986 987def sync_bucket_internal(988 source: str | None = None,989 dest: str | None = None,990 *,991 api: "HfApi",992 delete: bool = False,993 ignore_times: bool = False,994 ignore_sizes: bool = False,995 existing: bool = False,996 ignore_existing: bool = False,997 include: list[str] | None = None,998 exclude: list[str] | None = None,999 filter_from: str | None = None,1000 plan: str | None = None,1001 apply: str | None = None,1002 dry_run: bool = False,1003 verbose: bool = False,1004 quiet: bool = False,1005 token: bool | str | None = None,1006) -> SyncPlan:1007 """Sync files between a local directory and a bucket.1008 1009 This is equivalent to the ``hf buckets sync`` CLI command. One of ``source`` or ``dest`` must be a bucket path1010 (``hf://buckets/...``) and the other must be a local directory path.1011 1012 Args:1013 source (`str`, *optional*):1014 Source path: local directory or ``hf://buckets/namespace/bucket_name(/prefix)``.1015 Required unless using ``apply``.1016 dest (`str`, *optional*):1017 Destination path: local directory or ``hf://buckets/namespace/bucket_name(/prefix)``.1018 Required unless using ``apply``.1019 api ([`HfApi`]):1020 The HfApi instance to use for API calls.1021 delete (`bool`, *optional*, defaults to `False`):1022 Delete destination files not present in source.1023 ignore_times (`bool`, *optional*, defaults to `False`):1024 Skip files only based on size, ignoring modification times.1025 ignore_sizes (`bool`, *optional*, defaults to `False`):1026 Skip files only based on modification times, ignoring sizes.1027 existing (`bool`, *optional*, defaults to `False`):1028 Skip creating new files on receiver (only update existing files).1029 ignore_existing (`bool`, *optional*, defaults to `False`):1030 Skip updating files that exist on receiver (only create new files).1031 include (`list[str]`, *optional*):1032 Include files matching patterns (fnmatch-style).1033 exclude (`list[str]`, *optional*):1034 Exclude files matching patterns (fnmatch-style).1035 filter_from (`str`, *optional*):1036 Path to a filter file with include/exclude rules.1037 plan (`str`, *optional*):1038 Save sync plan to this JSONL file instead of executing.1039 apply (`str`, *optional*):1040 Apply a previously saved plan file. When set, ``source`` and ``dest`` are not needed.1041 dry_run (`bool`, *optional*, defaults to `False`):1042 Print sync plan to stdout as JSONL without executing.1043 verbose (`bool`, *optional*, defaults to `False`):1044 Show detailed per-file operations.1045 quiet (`bool`, *optional*, defaults to `False`):1046 Suppress all output and progress bars.1047 token (Union[bool, str, None], optional):1048 A valid user access token. If not provided, the locally saved token will be used.1049 1050 Returns:1051 [`SyncPlan`]: The computed (or loaded) sync plan.1052 1053 Raises:1054 `ValueError`: If arguments are invalid (e.g., both paths are remote, conflicting options).1055 1056 Example:1057 ```python1058 >>> from huggingface_hub import HfApi1059 >>> api = HfApi()1060 1061 # Upload local directory to bucket1062 >>> api.sync_bucket("./data", "hf://buckets/username/my-bucket")1063 1064 # Download bucket to local directory1065 >>> api.sync_bucket("hf://buckets/username/my-bucket", "./data")1066 1067 # Sync with delete and filtering1068 >>> api.sync_bucket(1069 ... "./data",1070 ... "hf://buckets/username/my-bucket",1071 ... delete=True,1072 ... include=["*.safetensors"],1073 ... )1074 1075 # Dry run: preview what would be synced1076 >>> plan = api.sync_bucket("./data", "hf://buckets/username/my-bucket", dry_run=True)1077 >>> plan.summary()1078 {'uploads': 3, 'downloads': 0, 'deletes': 0, 'skips': 1, 'total_size': 4096}1079 1080 # Save plan for review, then apply1081 >>> api.sync_bucket("./data", "hf://buckets/username/my-bucket", plan="sync-plan.jsonl")1082 >>> api.sync_bucket(apply="sync-plan.jsonl")1083 ```1084 """1085 # Build API with token if needed1086 if token is not None:1087 from .hf_api import HfApi1088 1089 api = HfApi(token=token)1090 # --- Apply mode ---1091 if apply:1092 if source or dest:1093 raise ValueError("Cannot specify source/dest when using apply.")1094 if plan is not None:1095 raise ValueError("Cannot specify both plan and apply.")1096 if delete:1097 raise ValueError("Cannot specify delete when using apply.")1098 if ignore_times:1099 raise ValueError("Cannot specify ignore_times when using apply.")1100 if ignore_sizes:1101 raise ValueError("Cannot specify ignore_sizes when using apply.")1102 if include:1103 raise ValueError("Cannot specify include when using apply.")1104 if exclude:1105 raise ValueError("Cannot specify exclude when using apply.")1106 if filter_from:1107 raise ValueError("Cannot specify filter_from when using apply.")1108 if existing:1109 raise ValueError("Cannot specify existing when using apply.")1110 if ignore_existing:1111 raise ValueError("Cannot specify ignore_existing when using apply.")1112 if dry_run:1113 raise ValueError("Cannot specify dry_run when using apply.")1114 1115 sync_plan = _load_plan(apply)1116 status = StatusLine(enabled=not quiet)1117 if not quiet:1118 _print_plan_summary(sync_plan)1119 print("Executing plan...")1120 1121 if quiet:1122 disable_progress_bars()1123 try:1124 _execute_plan(sync_plan, api, verbose=verbose, status=status)1125 finally:1126 if quiet:1127 enable_progress_bars()1128 1129 if not quiet:1130 print("Sync completed.")1131 1132 return sync_plan1133 1134 # --- Normal mode ---1135 if not source or not dest:1136 raise ValueError("Both source and dest are required (unless using apply).")1137 1138 source_is_bucket = _is_bucket_path(source)1139 dest_is_bucket = _is_bucket_path(dest)1140 1141 if source_is_bucket and dest_is_bucket:1142 raise ValueError("Remote to remote sync is not supported. One path must be local.")1143 1144 if not source_is_bucket and not dest_is_bucket:1145 raise ValueError("One of source or dest must be a bucket path (hf://buckets/...).")1146 1147 if ignore_times and ignore_sizes:1148 raise ValueError("Cannot specify both ignore_times and ignore_sizes.")1149 1150 if existing and ignore_existing:1151 raise ValueError("Cannot specify both existing and ignore_existing.")1152 1153 if dry_run and plan:1154 raise ValueError("Cannot specify both dry_run and plan.")1155 1156 # Validate local path1157 if source_is_bucket:1158 if os.path.exists(dest) and not os.path.isdir(dest):1159 raise ValueError(f"Destination must be a directory: {dest}")1160 else:1161 if not os.path.isdir(source):1162 raise ValueError(f"Source must be an existing directory: {source}")1163 1164 # Build filter matcher1165 filter_rules = None1166 if filter_from:1167 filter_rules = _parse_filter_file(filter_from)1168 1169 filter_matcher = FilterMatcher(1170 include_patterns=include,1171 exclude_patterns=exclude,1172 filter_rules=filter_rules,1173 )1174 1175 # Compute sync plan1176 status = StatusLine(enabled=not quiet and not dry_run)1177 sync_plan = _compute_sync_plan(1178 source=source,1179 dest=dest,1180 api=api,1181 delete=delete,1182 ignore_times=ignore_times,1183 ignore_sizes=ignore_sizes,1184 existing=existing,1185 ignore_existing=ignore_existing,1186 filter_matcher=filter_matcher,1187 status=status,1188 )1189 1190 if dry_run:1191 _write_plan(sync_plan, sys.stdout)1192 return sync_plan1193 1194 if plan:1195 _save_plan(sync_plan, plan)1196 if not quiet:1197 _print_plan_summary(sync_plan)1198 print(f"Plan saved to: {plan}")1199 return sync_plan1200 