Buckets:
| """Convert the ClawHub HF dump (raw copy in our bucket) into the unified processed layout: | |
| processed/clawhub/skills.parquet one row per skill version in the dump: SKILL.md + parsed fields + scanner verdicts | |
| processed/clawhub/files.parquet every other bundle file (skill_bundle_content) | |
| All ClawHub skills are MIT-0. Streams the jsonl; uploads then deletes local copies.""" | |
| import json, subprocess, sys | |
| from pathlib import Path | |
| import pyarrow as pa, pyarrow.parquet as pq | |
| sys.path.insert(0, str(Path(__file__).parent)) | |
| from lib_skill import parse_skill_md, cjk_ratio | |
| ROOT = Path(__file__).resolve().parent.parent | |
| BUCKET = "hf://buckets/Mercity/SkillsStorage" | |
| HF = str(ROOT / ".venv/bin/hf") | |
| WORK = ROOT / "work/claw"; WORK.mkdir(parents=True, exist_ok=True) | |
| DONE = ROOT / "logs/build_clawhub.done" | |
| if DONE.exists(): print("already done", flush=True); sys.exit(0) | |
| src = WORK / "latest.jsonl" | |
| if not src.exists(): | |
| subprocess.run([HF, "buckets", "cp", f"{BUCKET}/raw/hf/OpenClaw__clawhub-security-signals-live/data/latest.jsonl", str(src)], check=True) | |
| SCAL = ["clawscan_verdict", "clawscan_confidence", "clawscan_model", "clawscan_summary", "static_status", "static_finding_count", | |
| "virustotal_status", "virustotal_malicious_count", "virustotal_suspicious_count", "virustotal_harmless_count", | |
| "virustotal_undetected_count", "skillspector_status", "skillspector_score", "skillspector_severity", "skillspector_issue_count"] | |
| sk_w = fi_w = None; sk_rows, fi_rows = [], []; n = nf = 0; fi_bytes = [0] | |
| def flush(final=False): | |
| global sk_w, fi_w | |
| if sk_rows and (final or len(sk_rows) >= 5000): | |
| t = pa.Table.from_pylist(sk_rows); sk_w = sk_w or pq.ParquetWriter(WORK / "skills.parquet.tmp", t.schema, compression="zstd") | |
| sk_w.write_table(t.cast(sk_w.schema)); sk_rows.clear() | |
| if fi_rows and (final or len(fi_rows) >= 20000 or fi_bytes[0] > 64 << 20): | |
| t = pa.Table.from_pylist(fi_rows); fi_w = fi_w or pq.ParquetWriter(WORK / "files.parquet.tmp", t.schema, compression="zstd") | |
| fi_w.write_table(t.cast(fi_w.schema)); fi_rows.clear(); fi_bytes[0] = 0 | |
| FINAL = [WORK / "skills.parquet", WORK / "files.parquet"] | |
| if not all(p.exists() for p in FINAL): # finished outputs are never rebuilt (restart-safe) | |
| with open(src) as f: | |
| for line in f: | |
| r = json.loads(line); md = r.get("skill_md_content") or "" | |
| owner, _, slug = r["skill_slug"].partition("/") | |
| uid = f"clawhub:{r['skill_slug']}@{r['skill_version']}" | |
| ok, name, desc, fm, body = parse_skill_md(md) | |
| bundle = r.get("skill_bundle_content") or [] | |
| sk_rows.append({"skill_uid": uid, "source": "clawhub", "repo": None, "owner": owner, "slug": slug, | |
| "version": r["skill_version"], "dump_id": r["id"], "skill_dir": ".", "skill_md": md, "name": name, | |
| "description": desc, "frontmatter_valid": ok, "body_chars": len(body), "n_bundle_files": len(bundle), | |
| "bundle_bytes": sum(int(b.get("sizeBytes") or 0) for b in bundle), "cjk_ratio": cjk_ratio(md), | |
| "license": "MIT-0", **{k: r.get(k) for k in SCAL}, | |
| "static_reason_codes": r.get("static_reason_codes") or [], "skillspector_issue_codes": r.get("skillspector_issue_codes") or [], | |
| "clawscan_context": json.dumps(r.get("clawscan_context"), ensure_ascii=False) if r.get("clawscan_context") else None, | |
| "snapshot": "2026-09-28"}) | |
| for b in bundle: | |
| if b.get("path") == "SKILL.md": continue | |
| c = b.get("content") | |
| fi_rows.append({"skill_uid": uid, "source": "clawhub", "rel_path": b.get("path"), "size": int(b.get("sizeBytes") or 0), | |
| "sha256": b.get("sha256"), "content": c, "has_content": c is not None, | |
| "skipped": None if c is not None else "not_in_dump"}); nf += 1; fi_bytes[0] += len(c or "") | |
| n += 1; flush() | |
| flush(final=True); sk_w.close(); fi_w and fi_w.close() | |
| for p in FINAL: (WORK / (p.name + ".tmp")).rename(p) # atomic: only complete outputs get final names | |
| print("skills", n, "files", nf, flush=True) | |
| for name in ("skills", "files"): | |
| subprocess.run([HF, "buckets", "cp", str(WORK / f"{name}.parquet"), f"{BUCKET}/processed/clawhub/{name}.parquet"], check=True) | |
| print("uploaded", flush=True) | |
| DONE.write_text("ok\n") | |
| for p in FINAL + [src]: p.unlink() # raw jsonl stays in the bucket (raw/hf/OpenClaw__...) | |
Xet Storage Details
- Size:
- 4.41 kB
- Xet hash:
- 69ae61c4dbd341dd297202643b36e9a1350f2562ab3ad78c95f8b35334d8fc23
·
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.