Buckets:
| """Bucket A: official / curated vendor skill repos. Shallow clone, keep every skill | |
| folder as a full bundle (SKILL.md + siblings), write one tar.zst per repo + manifest.""" | |
| import subprocess, json, os, sys, tarfile, hashlib, shutil, time | |
| from pathlib import Path | |
| REPOS = """anthropics/skills anthropics/claude-plugins-official anthropics/knowledge-work-plugins | |
| anthropics/financial-services anthropics/claude-for-legal anthropics/life-sciences | |
| anthropics/claude-plugins-community openai/skills openai/plugins google/skills microsoft/skills | |
| aws/agent-toolkit-for-aws huggingface/skills github/awesome-copilot K-Dense-AI/scientific-agent-skills | |
| trailofbits/skills stripe/ai obra/superpowers vercel-labs/agent-skills addyosmani/agent-skills | |
| wshobson/agents MicrosoftDocs/Agent-Skills Orchestra-Research/AI-research-SKILLs | |
| davila7/claude-code-templates alirezarezvani/claude-skills ComposioHQ/awesome-claude-skills | |
| VoltAgent/awesome-agent-skills VoltAgent/awesome-claude-code-subagents davepoon/buildwithclaude | |
| qdhenry/Claude-Command-Suite cursor/plugins""".split() | |
| OUT = Path(sys.argv[1] if len(sys.argv) > 1 else "data/raw/vendor") | |
| WORK = Path("/root/skills-db/work/vendor"); WORK.mkdir(parents=True, exist_ok=True); OUT.mkdir(parents=True, exist_ok=True) | |
| def run(*a, **k): return subprocess.run(a, capture_output=True, text=True, **k) | |
| manifest = [] | |
| for r in REPOS: | |
| d = WORK / r.replace("/", "__"); shutil.rmtree(d, ignore_errors=True); t = time.time() | |
| p = run("git", "clone", "-q", "--depth", "1", f"https://github.com/{r}", str(d), env={**os.environ, "GIT_TERMINAL_PROMPT": "0"}) | |
| if p.returncode: | |
| manifest.append({"repo": r, "ok": False, "err": p.stderr.strip()[:200]}); print(manifest[-1], flush=True); continue | |
| commit = run("git", "-C", str(d), "rev-parse", "HEAD").stdout.strip() | |
| lic = next((f.name for f in d.iterdir() if f.name.upper().startswith(("LICENSE", "COPYING"))), None) | |
| skills = sorted(p.parent for p in d.rglob("*") if p.name in ("SKILL.md", "skill.md") and ".git" not in p.parts) | |
| tar = OUT / f"{r.replace('/', '__')}.tar.gz" | |
| with tarfile.open(tar, "w:gz") as tf: # whole repo minus .git: keeps bundles + plugin/agent/command files | |
| tf.add(d, arcname=r.replace("/", "__"), filter=lambda ti: None if "/.git/" in ti.name + "/" else ti) | |
| rec = {"repo": r, "ok": True, "commit": commit, "license_file": lic, "n_skill_md": len(skills), | |
| "skill_dirs": [str(s.relative_to(d)) for s in skills], "tar_bytes": tar.stat().st_size, "secs": round(time.time() - t, 1)} | |
| manifest.append(rec); print({k: v for k, v in rec.items() if k != "skill_dirs"}, flush=True) | |
| shutil.rmtree(d) | |
| (OUT / "manifest.json").write_text(json.dumps(manifest, indent=1)) | |
Xet Storage Details
- Size:
- 2.72 kB
- Xet hash:
- 3693b2f89e0e51c85617c95ce72b5dffb965ec104ca72a2e6e3cafdf6361eecc
·
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.