Download androidcontrol_data_load.py from GD-ML/AndroidCode: direct link, hf CLI and curl.
- Browser
- Download file 4.11 kB
-
https://huggingface.co/datasets/GD-ML/AndroidCode/resolve/main/androidcontrol_data_load.py
- Command line
-
hf download hf://datasets/GD-ML/AndroidCode/androidcontrol_data_load.py
-
curl -L -o androidcontrol_data_load.py https://huggingface.co/datasets/GD-ML/AndroidCode/resolve/main/androidcontrol_data_load.py
4.11 kB
| import os | |
| import requests | |
| from requests.adapters import HTTPAdapter | |
| from urllib3.util.retry import Retry | |
| from tqdm import tqdm | |
| # ================= 配置区域 ================= | |
| # 【非常重要】请确认这里的 URL 是文件的“直链” | |
| # 如果你在浏览器里点击这个链接能直接开始下载文件,那就是对的。 | |
| BASE_URL = "https://storage.googleapis.com/gresearch/android_control/" | |
| SAVE_DIR = "./downloads" | |
| # =========================================== | |
| if not os.path.exists(SAVE_DIR): | |
| os.makedirs(SAVE_DIR) | |
| # --- 构建下载文件清单 --- | |
| # 1. 加入 20 个数据分片文件 | |
| files_to_download = [f"android_control-{i:05d}-of-00020" for i in range(20)] | |
| # 2. 加入额外的 JSON 配置文件 | |
| files_to_download.extend([ | |
| "splits.json", | |
| "test_subsplits.json" | |
| ]) | |
| # ----------------------- | |
| # 设置网络请求 session | |
| session = requests.Session() | |
| headers = { | |
| "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/119.0.0.0 Safari/537.36" | |
| } | |
| # 设置自动重试防止网络波动 | |
| retries = Retry(total=5, backoff_factor=1, status_forcelist=[500, 502, 503, 504]) | |
| session.mount('http://', HTTPAdapter(max_retries=retries)) | |
| session.mount('https://', HTTPAdapter(max_retries=retries)) | |
| print(f"开始下载任务,共 {len(files_to_download)} 个文件") | |
| print(f"保存路径: {SAVE_DIR}\n") | |
| for file_name in files_to_download: | |
| # 构造完整 URL | |
| url = f"{BASE_URL.rstrip('/')}/{file_name}" | |
| save_path = os.path.join(SAVE_DIR, file_name) | |
| try: | |
| # 1. 发送 HEAD 请求获取文件大小 | |
| # 注意:如果是 JSON 小文件,服务器响应会很快 | |
| head_resp = session.head(url, headers=headers, timeout=10) | |
| # 某些服务器对小文件可能不返回 content-length,默认为 0 | |
| total_size = int(head_resp.headers.get('content-length', 0)) | |
| # 2. 检查本地文件状态(断点续传逻辑) | |
| first_byte = 0 | |
| if os.path.exists(save_path): | |
| local_size = os.path.getsize(save_path) | |
| # 如果本地大小等于服务器大小(且服务器返回了有效大小),则跳过 | |
| if total_size > 0 and local_size == total_size: | |
| print(f"✅ {file_name} 已存在且完整,跳过。") | |
| continue | |
| elif total_size > 0 and local_size < total_size: | |
| print(f"⚠️ {file_name} 不完整,尝试续传 ({local_size}/{total_size})...") | |
| first_byte = local_size | |
| else: | |
| # 如果本地文件比服务器大,或者服务器没给大小(通常不会),或者想强制覆盖 | |
| # 这里简单处理:如果大小不对劲就重下,或者如果是第一次下载 | |
| if total_size > 0 and local_size > total_size: | |
| print(f"❌ {file_name} 本地文件异常,重新下载。") | |
| first_byte = 0 | |
| # 3. 构造请求头 (Range) | |
| resume_header = headers.copy() | |
| if first_byte > 0: | |
| resume_header['Range'] = f"bytes={first_byte}-" | |
| # 4. 下载内容 | |
| response = session.get(url, stream=True, headers=resume_header, timeout=30) | |
| response.raise_for_status() # 检查 404 等错误 | |
| # 写入模式 | |
| mode = 'ab' if first_byte > 0 else 'wb' | |
| # 进度条 | |
| with tqdm( | |
| total=total_size, | |
| initial=first_byte, | |
| unit='B', | |
| unit_scale=True, | |
| unit_divisor=1024, | |
| desc=file_name, | |
| ascii=False | |
| ) as bar: | |
| with open(save_path, mode) as f: | |
| for chunk in response.iter_content(chunk_size=8192): | |
| if chunk: | |
| f.write(chunk) | |
| bar.update(len(chunk)) | |
| except Exception as e: | |
| print(f"\n❌ 下载 {file_name} 失败: {e}") | |
| if "404" in str(e): | |
| print(" (请检查该文件是否存在于服务器上)") | |
| print("\n所有任务处理完毕。") |