Buckets:

Mercity/SkillsStorage / code /build_clawhub.py
Pranav2748's picture
download
raw
4.41 kB
"""Convert the ClawHub HF dump (raw copy in our bucket) into the unified processed layout:
processed/clawhub/skills.parquet one row per skill version in the dump: SKILL.md + parsed fields + scanner verdicts
processed/clawhub/files.parquet every other bundle file (skill_bundle_content)
All ClawHub skills are MIT-0. Streams the jsonl; uploads then deletes local copies."""
import json, subprocess, sys
from pathlib import Path
import pyarrow as pa, pyarrow.parquet as pq
sys.path.insert(0, str(Path(__file__).parent))
from lib_skill import parse_skill_md, cjk_ratio
ROOT = Path(__file__).resolve().parent.parent
BUCKET = "hf://buckets/Mercity/SkillsStorage"
HF = str(ROOT / ".venv/bin/hf")
WORK = ROOT / "work/claw"; WORK.mkdir(parents=True, exist_ok=True)
DONE = ROOT / "logs/build_clawhub.done"
if DONE.exists(): print("already done", flush=True); sys.exit(0)
src = WORK / "latest.jsonl"
if not src.exists():
subprocess.run([HF, "buckets", "cp", f"{BUCKET}/raw/hf/OpenClaw__clawhub-security-signals-live/data/latest.jsonl", str(src)], check=True)
SCAL = ["clawscan_verdict", "clawscan_confidence", "clawscan_model", "clawscan_summary", "static_status", "static_finding_count",
"virustotal_status", "virustotal_malicious_count", "virustotal_suspicious_count", "virustotal_harmless_count",
"virustotal_undetected_count", "skillspector_status", "skillspector_score", "skillspector_severity", "skillspector_issue_count"]
sk_w = fi_w = None; sk_rows, fi_rows = [], []; n = nf = 0; fi_bytes = [0]
def flush(final=False):
global sk_w, fi_w
if sk_rows and (final or len(sk_rows) >= 5000):
t = pa.Table.from_pylist(sk_rows); sk_w = sk_w or pq.ParquetWriter(WORK / "skills.parquet.tmp", t.schema, compression="zstd")
sk_w.write_table(t.cast(sk_w.schema)); sk_rows.clear()
if fi_rows and (final or len(fi_rows) >= 20000 or fi_bytes[0] > 64 << 20):
t = pa.Table.from_pylist(fi_rows); fi_w = fi_w or pq.ParquetWriter(WORK / "files.parquet.tmp", t.schema, compression="zstd")
fi_w.write_table(t.cast(fi_w.schema)); fi_rows.clear(); fi_bytes[0] = 0
FINAL = [WORK / "skills.parquet", WORK / "files.parquet"]
if not all(p.exists() for p in FINAL): # finished outputs are never rebuilt (restart-safe)
with open(src) as f:
for line in f:
r = json.loads(line); md = r.get("skill_md_content") or ""
owner, _, slug = r["skill_slug"].partition("/")
uid = f"clawhub:{r['skill_slug']}@{r['skill_version']}"
ok, name, desc, fm, body = parse_skill_md(md)
bundle = r.get("skill_bundle_content") or []
sk_rows.append({"skill_uid": uid, "source": "clawhub", "repo": None, "owner": owner, "slug": slug,
"version": r["skill_version"], "dump_id": r["id"], "skill_dir": ".", "skill_md": md, "name": name,
"description": desc, "frontmatter_valid": ok, "body_chars": len(body), "n_bundle_files": len(bundle),
"bundle_bytes": sum(int(b.get("sizeBytes") or 0) for b in bundle), "cjk_ratio": cjk_ratio(md),
"license": "MIT-0", **{k: r.get(k) for k in SCAL},
"static_reason_codes": r.get("static_reason_codes") or [], "skillspector_issue_codes": r.get("skillspector_issue_codes") or [],
"clawscan_context": json.dumps(r.get("clawscan_context"), ensure_ascii=False) if r.get("clawscan_context") else None,
"snapshot": "2026-09-28"})
for b in bundle:
if b.get("path") == "SKILL.md": continue
c = b.get("content")
fi_rows.append({"skill_uid": uid, "source": "clawhub", "rel_path": b.get("path"), "size": int(b.get("sizeBytes") or 0),
"sha256": b.get("sha256"), "content": c, "has_content": c is not None,
"skipped": None if c is not None else "not_in_dump"}); nf += 1; fi_bytes[0] += len(c or "")
n += 1; flush()
flush(final=True); sk_w.close(); fi_w and fi_w.close()
for p in FINAL: (WORK / (p.name + ".tmp")).rename(p) # atomic: only complete outputs get final names
print("skills", n, "files", nf, flush=True)
for name in ("skills", "files"):
subprocess.run([HF, "buckets", "cp", str(WORK / f"{name}.parquet"), f"{BUCKET}/processed/clawhub/{name}.parquet"], check=True)
print("uploaded", flush=True)
DONE.write_text("ok\n")
for p in FINAL + [src]: p.unlink() # raw jsonl stays in the bucket (raw/hf/OpenClaw__...)

Xet Storage Details

Size:
4.41 kB
·
Xet hash:
69ae61c4dbd341dd297202643b36e9a1350f2562ab3ad78c95f8b35334d8fc23

Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.