Buckets:
| """Server-side copy of Bucket A HF datasets into the bucket (no local disk). | |
| Usage: python code/mirror_hf.py [repo ...]""" | |
| import sys, time, json | |
| from pathlib import Path | |
| from dotenv import load_dotenv | |
| load_dotenv(Path(__file__).resolve().parent.parent / ".env") | |
| from huggingface_hub import HfApi | |
| sys.path.insert(0, str(Path(__file__).parent)) | |
| from hf_sources import SOURCES | |
| BUCKET = "hf://buckets/Mercity/SkillsStorage" | |
| api = HfApi() | |
| repos = sys.argv[1:] or list(SOURCES) | |
| for r in repos: | |
| dst = f"{BUCKET}/raw/hf/{r.replace('/', '__')}/" | |
| t = time.time() | |
| try: | |
| sha = api.dataset_info(r).sha | |
| api.copy_files(f"hf://datasets/{r}/", dst) | |
| print(json.dumps({"repo": r, "sha": sha, "ok": True, "secs": round(time.time() - t, 1)}), flush=True) | |
| except Exception as e: | |
| print(json.dumps({"repo": r, "ok": False, "err": f"{type(e).__name__}: {str(e)[:300]}"}), flush=True) | |
Xet Storage Details
- Size:
- 908 Bytes
- Xet hash:
- 923616f2a26029443917536b0231ccbfe7e308492b18097a213be9cf859e00a8
·
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.