Buckets:
| """ClawHub public API (no auth). 1) full listing -> metadata jsonl for every skill. | |
| 2) zip bundle download for skills created/updated after the HF dump cutoff. | |
| Polite: <=4 req/s, honors 429/Retry-After.""" | |
| import json, time, sys, zipfile, io, requests | |
| from pathlib import Path | |
| OUT = Path(sys.argv[1] if len(sys.argv) > 1 else "data/raw/clawhub_api") | |
| CUTOFF_MS = int(sys.argv[2]) if len(sys.argv) > 2 else 1790573689965 # HF dump created_at_lt | |
| (OUT / "zips").mkdir(parents=True, exist_ok=True) | |
| S = requests.Session(); S.headers["User-Agent"] = "skills-db-research/0.1 (dataset building; contact via HF Mercity)" | |
| B = "https://clawhub.ai/api/v1" | |
| def get(url, **kw): | |
| for i in range(8): | |
| r = S.get(url, timeout=60, **kw) | |
| if r.status_code == 429 or r.status_code >= 500: | |
| time.sleep(int(r.headers.get("retry-after", 2 ** i))); continue | |
| time.sleep(0.25); return r | |
| r.raise_for_status() | |
| # 1) listing (resumable: restarts from the last saved createdAt via a synthetic index cursor; | |
| # the boundary item is re-listed, duplicates are dropped below) | |
| lst = OUT / "listing.jsonl" | |
| if not lst.exists(): | |
| tmp = lst.with_suffix(".tmp"); cur = None; n = 0 | |
| if tmp.exists() and tmp.stat().st_size: | |
| lines = tmp.read_text().splitlines(); n = len(lines) | |
| while lines: | |
| try: last = json.loads(lines[-1]); break | |
| except json.JSONDecodeError: lines.pop() # partial last line from a crash | |
| tmp.write_text("\n".join(lines) + "\n") | |
| X = last["createdAt"] | |
| cur = json.dumps({"v": 1, "index": "by_active_created", "key": [{"__undef": 1}, X, X + 0.9999, "z" * 32]}, separators=(",", ":")) | |
| print("resuming listing after", n, "items at createdAt", X, flush=True) | |
| with open(tmp, "a") as f: | |
| while True: | |
| p = {"sort": "createdAt", "limit": 200, **({"cursor": cur} if cur else {})} | |
| d = get(f"{B}/skills", params=p).json() | |
| for it in d["items"]: f.write(json.dumps(it, ensure_ascii=False) + "\n") | |
| f.flush(); n += len(d["items"]); cur = d.get("nextCursor") | |
| if n % 5000 < 200: print("listed", n, flush=True) | |
| if not cur or not d["items"]: break | |
| tmp.rename(lst); print("listing done", n, flush=True) | |
| # 2) bundles for new/updated skills | |
| items = list({(it["ownerHandle"], it["slug"]): it for it in map(json.loads, open(lst))}.values()) | |
| # new content = skill created, or a new version published, after the dump (updatedAt also moves on stats changes) | |
| todo = [it for it in items if max(it["createdAt"], (it.get("latestVersion") or {}).get("createdAt") or 0) >= CUTOFF_MS] | |
| print(f"{len(items)} listed, {len(todo)} newer than cutoff", flush=True) | |
| done = {p.stem for p in (OUT / "zips").glob("*.zip")} | |
| for i, it in enumerate(todo): | |
| key = f"{it['ownerHandle']}__{it['slug']}" | |
| if key in done: continue | |
| r = get(f"{B}/download", params={"slug": it["slug"]}) | |
| if r.status_code != 200 or not r.content.startswith(b"PK"): | |
| print("fail", it["slug"], r.status_code, flush=True); continue | |
| zipfile.ZipFile(io.BytesIO(r.content)).testzip() | |
| (OUT / "zips" / f"{key}.zip").write_bytes(r.content) | |
| if i % 200 == 0: print("zips", i, "/", len(todo), flush=True) | |
| print("done", flush=True) | |
Xet Storage Details
- Size:
- 3.26 kB
- Xet hash:
- 44f861a7193546076b017219956a9a7dddf1edf55bf2adbfe8077d2c305fab95
·
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.