Buckets:

Mercity/SkillsStorage / code /vendor_repos.py
Pranav2748's picture
download
raw
2.72 kB
"""Bucket A: official / curated vendor skill repos. Shallow clone, keep every skill
folder as a full bundle (SKILL.md + siblings), write one tar.zst per repo + manifest."""
import subprocess, json, os, sys, tarfile, hashlib, shutil, time
from pathlib import Path
REPOS = """anthropics/skills anthropics/claude-plugins-official anthropics/knowledge-work-plugins
anthropics/financial-services anthropics/claude-for-legal anthropics/life-sciences
anthropics/claude-plugins-community openai/skills openai/plugins google/skills microsoft/skills
aws/agent-toolkit-for-aws huggingface/skills github/awesome-copilot K-Dense-AI/scientific-agent-skills
trailofbits/skills stripe/ai obra/superpowers vercel-labs/agent-skills addyosmani/agent-skills
wshobson/agents MicrosoftDocs/Agent-Skills Orchestra-Research/AI-research-SKILLs
davila7/claude-code-templates alirezarezvani/claude-skills ComposioHQ/awesome-claude-skills
VoltAgent/awesome-agent-skills VoltAgent/awesome-claude-code-subagents davepoon/buildwithclaude
qdhenry/Claude-Command-Suite cursor/plugins""".split()
OUT = Path(sys.argv[1] if len(sys.argv) > 1 else "data/raw/vendor")
WORK = Path("/root/skills-db/work/vendor"); WORK.mkdir(parents=True, exist_ok=True); OUT.mkdir(parents=True, exist_ok=True)
def run(*a, **k): return subprocess.run(a, capture_output=True, text=True, **k)
manifest = []
for r in REPOS:
d = WORK / r.replace("/", "__"); shutil.rmtree(d, ignore_errors=True); t = time.time()
p = run("git", "clone", "-q", "--depth", "1", f"https://github.com/{r}", str(d), env={**os.environ, "GIT_TERMINAL_PROMPT": "0"})
if p.returncode:
manifest.append({"repo": r, "ok": False, "err": p.stderr.strip()[:200]}); print(manifest[-1], flush=True); continue
commit = run("git", "-C", str(d), "rev-parse", "HEAD").stdout.strip()
lic = next((f.name for f in d.iterdir() if f.name.upper().startswith(("LICENSE", "COPYING"))), None)
skills = sorted(p.parent for p in d.rglob("*") if p.name in ("SKILL.md", "skill.md") and ".git" not in p.parts)
tar = OUT / f"{r.replace('/', '__')}.tar.gz"
with tarfile.open(tar, "w:gz") as tf: # whole repo minus .git: keeps bundles + plugin/agent/command files
tf.add(d, arcname=r.replace("/", "__"), filter=lambda ti: None if "/.git/" in ti.name + "/" else ti)
rec = {"repo": r, "ok": True, "commit": commit, "license_file": lic, "n_skill_md": len(skills),
"skill_dirs": [str(s.relative_to(d)) for s in skills], "tar_bytes": tar.stat().st_size, "secs": round(time.time() - t, 1)}
manifest.append(rec); print({k: v for k, v in rec.items() if k != "skill_dirs"}, flush=True)
shutil.rmtree(d)
(OUT / "manifest.json").write_text(json.dumps(manifest, indent=1))

Xet Storage Details

Size:
2.72 kB
·
Xet hash:
3693b2f89e0e51c85617c95ce72b5dffb965ec104ca72a2e6e3cafdf6361eecc

Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.