Team Ai
Datasetpublic

GD-ML/AndroidCode

sourceHugging Facemitupdated 6mo agoView on Hugging Face
1likes3.4kdownloads
androidcontrol_data_load.py107 linesDownload Raw Back to root
1import os2import requests3from requests.adapters import HTTPAdapter4from urllib3.util.retry import Retry5from tqdm import tqdm6 7# ================= 配置区域 =================8# 【非常重要】请确认这里的 URL 是文件的“直链”9# 如果你在浏览器里点击这个链接能直接开始下载文件,那就是对的。10BASE_URL = "https://storage.googleapis.com/gresearch/android_control/"11 12SAVE_DIR = "./downloads"13# ===========================================14 15if not os.path.exists(SAVE_DIR):16    os.makedirs(SAVE_DIR)17 18# --- 构建下载文件清单 ---19# 1. 加入 20 个数据分片文件20files_to_download = [f"android_control-{i:05d}-of-00020" for i in range(20)]21 22# 2. 加入额外的 JSON 配置文件23files_to_download.extend([24    "splits.json", 25    "test_subsplits.json"26])27# -----------------------28 29# 设置网络请求 session30session = requests.Session()31headers = {32    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/119.0.0.0 Safari/537.36"33}34# 设置自动重试防止网络波动35retries = Retry(total=5, backoff_factor=1, status_forcelist=[500, 502, 503, 504])36session.mount('http://', HTTPAdapter(max_retries=retries))37session.mount('https://', HTTPAdapter(max_retries=retries))38 39print(f"开始下载任务,共 {len(files_to_download)} 个文件")40print(f"保存路径: {SAVE_DIR}\n")41 42for file_name in files_to_download:43    # 构造完整 URL44    url = f"{BASE_URL.rstrip('/')}/{file_name}"45    save_path = os.path.join(SAVE_DIR, file_name)46 47    try:48        # 1. 发送 HEAD 请求获取文件大小49        # 注意:如果是 JSON 小文件,服务器响应会很快50        head_resp = session.head(url, headers=headers, timeout=10)51        52        # 某些服务器对小文件可能不返回 content-length,默认为 053        total_size = int(head_resp.headers.get('content-length', 0))54 55        # 2. 检查本地文件状态(断点续传逻辑)56        first_byte = 057        if os.path.exists(save_path):58            local_size = os.path.getsize(save_path)59            60            # 如果本地大小等于服务器大小(且服务器返回了有效大小),则跳过61            if total_size > 0 and local_size == total_size:62                print(f"✅ {file_name} 已存在且完整,跳过。")63                continue64            elif total_size > 0 and local_size < total_size:65                print(f"⚠️ {file_name} 不完整,尝试续传 ({local_size}/{total_size})...")66                first_byte = local_size67            else:68                # 如果本地文件比服务器大,或者服务器没给大小(通常不会),或者想强制覆盖69                # 这里简单处理:如果大小不对劲就重下,或者如果是第一次下载70                if total_size > 0 and local_size > total_size:71                    print(f"❌ {file_name} 本地文件异常,重新下载。")72                first_byte = 073        74        # 3. 构造请求头 (Range)75        resume_header = headers.copy()76        if first_byte > 0:77            resume_header['Range'] = f"bytes={first_byte}-"78 79        # 4. 下载内容80        response = session.get(url, stream=True, headers=resume_header, timeout=30)81        response.raise_for_status() # 检查 404 等错误82 83        # 写入模式84        mode = 'ab' if first_byte > 0 else 'wb'85        86        # 进度条87        with tqdm(88            total=total_size, 89            initial=first_byte,90            unit='B', 91            unit_scale=True, 92            unit_divisor=1024, 93            desc=file_name,94            ascii=False95        ) as bar:96            with open(save_path, mode) as f:97                for chunk in response.iter_content(chunk_size=8192):98                    if chunk:99                        f.write(chunk)100                        bar.update(len(chunk))101 102    except Exception as e:103        print(f"\n❌ 下载 {file_name} 失败: {e}")104        if "404" in str(e):105            print("   (请检查该文件是否存在于服务器上)")106 107print("\n所有任务处理完毕。")
GD-ML/AndroidCode · Team Ai