Buckets:

Mercity/SkillsStorage / code /hf_sources.py
Pranav2748's picture
download
raw
1.7 kB
"""Bucket A: Hugging Face datasets to mirror into the bucket under raw/hf/<owner>__<name>/."""
SOURCES = {
# core corpus
"mvaccargiu/gitskills": "GitHub census Jul 2026; artifacts + siblings (bundle files) + repos",
"OpenClaw/clawhub-security-signals-live": "ClawHub full skills + scanner verdicts (MIT-0)",
# other skill corpora (deltas / labels)
"GokuScraper/clawhub-skills": "ClawHub with stats, categories, zh translations",
"FayeZC/SkillMD-138K": "SHA-dedup GitHub skills Apr 2026",
"huzey/claude-skills": "GPT facets + skills.sh installs",
"LittleDinoC/agent-skills": "61k skills (card MIT not valid for content)",
"tickleliu/all-skills-from-skills-sh": "skills.sh crawl (no-commercial clause)",
"CSeemy/SkillCorpus": "full multi-file packages serialized",
"EverMind-AI/skillcorpus-demo-1k": "quality-scored 1k demo",
"benchflow/skillsbench-data": "SkillsMP metadata + embeddings",
# Shiyu-Lab/Skill-Usage: only skills-34k/, search_index/ and root files copied (58k small files are benchmark run outputs)
"AgentFly/router-data": "SkillsMP rows + embeddings",
# supervision / eval (ledger P0)
"ThakiCloud/SKILLRET": "query->skill train/test",
"donghongjiang/skillreason-bench": "implicit-query benchmark",
"pipizhao/SkillRouter-Eval-Core": "graded relevance eval",
"WeihangSu/SRA-Bench": "skill retrieval benchmark",
"Athekunal/Agent-Skills-Retriever": "question/summary/negatives (noisy)",
"mangopy/ToolRet-Queries": "tool retrieval queries",
"Vinkius/mcp-registry": "MCP servers with prompt_examples",
# adjacent
"AgenticResourceDiscovery/verified-mcp-tools": "MCP tools (hard negatives)",
}

Xet Storage Details

Size:
1.7 kB
·
Xet hash:
78ec769cd525f1d29b7aad7a86f9380a2d12c904963db74c480f9d57cbec8fbb

Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.