Buckets:

Mercity/SkillsStorage / code /github_topic_seeds.py
Pranav2748's picture
download
raw
2.75 kB
"""Bucket B seed list: GitHub repo search (no token: 10 req/min) for skill topics,
sharded by creation day since the GitSkills cutoff so each query stays < 1000 results."""
import json, time, sys, datetime as dt, requests
from pathlib import Path
TOPICS = ["agent-skills", "claude-skills", "agent-skill", "claude-skill", "claude-code-skills",
"codex-skills", "skill-md", "skills", "openclaw-skills", "claude-code-skill", "ai-skills", "skillsmp"]
START = dt.date(2026, 7, 1); END = dt.date.today()
OUT = Path(sys.argv[1] if len(sys.argv) > 1 else "data/seeds/github_topics.jsonl")
done = set()
if OUT.exists():
for l in open(OUT): done.add(json.loads(l)["_shard"])
S = requests.Session(); S.headers.update({"Accept": "application/vnd.github+json", "User-Agent": "skills-db-research"})
def search(q, page):
while True:
r = S.get("https://api.github.com/search/repositories", params={"q": q, "per_page": 100, "page": page}, timeout=60)
if r.status_code in (403, 429):
reset = int(r.headers.get("x-ratelimit-reset", time.time() + 60)); time.sleep(max(5, reset - time.time() + 2)); continue
r.raise_for_status()
rem = int(r.headers.get("x-ratelimit-remaining", 1))
if rem == 0: time.sleep(max(1, int(r.headers["x-ratelimit-reset"]) - time.time() + 2))
return r.json()
def windows(day): # split a day into halves if needed
yield f"{day}T00:00:00Z..{day}T11:59:59Z"; yield f"{day}T12:00:00Z..{day}T23:59:59Z"
with open(OUT, "a") as f:
for t in TOPICS:
d = START
while d <= END:
shards = [f"{d}"]
for sh in shards:
key = f"{t}|{sh}"
if key in done: continue
q = f"topic:{t} created:{sh}"
first = search(q, 1); n = first["total_count"]
if n > 1000 and "T" not in sh:
shards += list(windows(d)); continue
items = first["items"]
for p in range(2, min(10, (n + 99) // 100) + 1): items += search(q, p)["items"]
for it in items:
f.write(json.dumps({"_shard": key, "full_name": it["full_name"], "created_at": it["created_at"],
"pushed_at": it["pushed_at"], "stars": it["stargazers_count"], "fork": it["fork"],
"size_kb": it["size"], "license": (it.get("license") or {}).get("spdx_id"),
"default_branch": it["default_branch"], "topics": it.get("topics")}) + "\n")
if not items: f.write(json.dumps({"_shard": key, "full_name": None}) + "\n")
f.flush()
d += dt.timedelta(days=1)
print("topic done", t, flush=True)
print("done", flush=True)

Xet Storage Details

Size:
2.75 kB
·
Xet hash:
5a7b3aa4733f7eac59da614dd743b99bcdfb9c8e60dfd9f4649da28d45c911be

Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.