Buckets:

Mercity/SkillsStorage / code /clawhub_api.py
Pranav2748's picture
download
raw
3.26 kB
"""ClawHub public API (no auth). 1) full listing -> metadata jsonl for every skill.
2) zip bundle download for skills created/updated after the HF dump cutoff.
Polite: <=4 req/s, honors 429/Retry-After."""
import json, time, sys, zipfile, io, requests
from pathlib import Path
OUT = Path(sys.argv[1] if len(sys.argv) > 1 else "data/raw/clawhub_api")
CUTOFF_MS = int(sys.argv[2]) if len(sys.argv) > 2 else 1790573689965 # HF dump created_at_lt
(OUT / "zips").mkdir(parents=True, exist_ok=True)
S = requests.Session(); S.headers["User-Agent"] = "skills-db-research/0.1 (dataset building; contact via HF Mercity)"
B = "https://clawhub.ai/api/v1"
def get(url, **kw):
for i in range(8):
r = S.get(url, timeout=60, **kw)
if r.status_code == 429 or r.status_code >= 500:
time.sleep(int(r.headers.get("retry-after", 2 ** i))); continue
time.sleep(0.25); return r
r.raise_for_status()
# 1) listing (resumable: restarts from the last saved createdAt via a synthetic index cursor;
# the boundary item is re-listed, duplicates are dropped below)
lst = OUT / "listing.jsonl"
if not lst.exists():
tmp = lst.with_suffix(".tmp"); cur = None; n = 0
if tmp.exists() and tmp.stat().st_size:
lines = tmp.read_text().splitlines(); n = len(lines)
while lines:
try: last = json.loads(lines[-1]); break
except json.JSONDecodeError: lines.pop() # partial last line from a crash
tmp.write_text("\n".join(lines) + "\n")
X = last["createdAt"]
cur = json.dumps({"v": 1, "index": "by_active_created", "key": [{"__undef": 1}, X, X + 0.9999, "z" * 32]}, separators=(",", ":"))
print("resuming listing after", n, "items at createdAt", X, flush=True)
with open(tmp, "a") as f:
while True:
p = {"sort": "createdAt", "limit": 200, **({"cursor": cur} if cur else {})}
d = get(f"{B}/skills", params=p).json()
for it in d["items"]: f.write(json.dumps(it, ensure_ascii=False) + "\n")
f.flush(); n += len(d["items"]); cur = d.get("nextCursor")
if n % 5000 < 200: print("listed", n, flush=True)
if not cur or not d["items"]: break
tmp.rename(lst); print("listing done", n, flush=True)
# 2) bundles for new/updated skills
items = list({(it["ownerHandle"], it["slug"]): it for it in map(json.loads, open(lst))}.values())
# new content = skill created, or a new version published, after the dump (updatedAt also moves on stats changes)
todo = [it for it in items if max(it["createdAt"], (it.get("latestVersion") or {}).get("createdAt") or 0) >= CUTOFF_MS]
print(f"{len(items)} listed, {len(todo)} newer than cutoff", flush=True)
done = {p.stem for p in (OUT / "zips").glob("*.zip")}
for i, it in enumerate(todo):
key = f"{it['ownerHandle']}__{it['slug']}"
if key in done: continue
r = get(f"{B}/download", params={"slug": it["slug"]})
if r.status_code != 200 or not r.content.startswith(b"PK"):
print("fail", it["slug"], r.status_code, flush=True); continue
zipfile.ZipFile(io.BytesIO(r.content)).testzip()
(OUT / "zips" / f"{key}.zip").write_bytes(r.content)
if i % 200 == 0: print("zips", i, "/", len(todo), flush=True)
print("done", flush=True)

Xet Storage Details

Size:
3.26 kB
·
Xet hash:
44f861a7193546076b017219956a9a7dddf1edf55bf2adbfe8077d2c305fab95

Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.