Buckets:
| """Bucket B seed list: GitHub repo search (no token: 10 req/min) for skill topics, | |
| sharded by creation day since the GitSkills cutoff so each query stays < 1000 results.""" | |
| import json, time, sys, datetime as dt, requests | |
| from pathlib import Path | |
| TOPICS = ["agent-skills", "claude-skills", "agent-skill", "claude-skill", "claude-code-skills", | |
| "codex-skills", "skill-md", "skills", "openclaw-skills", "claude-code-skill", "ai-skills", "skillsmp"] | |
| START = dt.date(2026, 7, 1); END = dt.date.today() | |
| OUT = Path(sys.argv[1] if len(sys.argv) > 1 else "data/seeds/github_topics.jsonl") | |
| done = set() | |
| if OUT.exists(): | |
| for l in open(OUT): done.add(json.loads(l)["_shard"]) | |
| S = requests.Session(); S.headers.update({"Accept": "application/vnd.github+json", "User-Agent": "skills-db-research"}) | |
| def search(q, page): | |
| while True: | |
| r = S.get("https://api.github.com/search/repositories", params={"q": q, "per_page": 100, "page": page}, timeout=60) | |
| if r.status_code in (403, 429): | |
| reset = int(r.headers.get("x-ratelimit-reset", time.time() + 60)); time.sleep(max(5, reset - time.time() + 2)); continue | |
| r.raise_for_status() | |
| rem = int(r.headers.get("x-ratelimit-remaining", 1)) | |
| if rem == 0: time.sleep(max(1, int(r.headers["x-ratelimit-reset"]) - time.time() + 2)) | |
| return r.json() | |
| def windows(day): # split a day into halves if needed | |
| yield f"{day}T00:00:00Z..{day}T11:59:59Z"; yield f"{day}T12:00:00Z..{day}T23:59:59Z" | |
| with open(OUT, "a") as f: | |
| for t in TOPICS: | |
| d = START | |
| while d <= END: | |
| shards = [f"{d}"] | |
| for sh in shards: | |
| key = f"{t}|{sh}" | |
| if key in done: continue | |
| q = f"topic:{t} created:{sh}" | |
| first = search(q, 1); n = first["total_count"] | |
| if n > 1000 and "T" not in sh: | |
| shards += list(windows(d)); continue | |
| items = first["items"] | |
| for p in range(2, min(10, (n + 99) // 100) + 1): items += search(q, p)["items"] | |
| for it in items: | |
| f.write(json.dumps({"_shard": key, "full_name": it["full_name"], "created_at": it["created_at"], | |
| "pushed_at": it["pushed_at"], "stars": it["stargazers_count"], "fork": it["fork"], | |
| "size_kb": it["size"], "license": (it.get("license") or {}).get("spdx_id"), | |
| "default_branch": it["default_branch"], "topics": it.get("topics")}) + "\n") | |
| if not items: f.write(json.dumps({"_shard": key, "full_name": None}) + "\n") | |
| f.flush() | |
| d += dt.timedelta(days=1) | |
| print("topic done", t, flush=True) | |
| print("done", flush=True) | |
Xet Storage Details
- Size:
- 2.75 kB
- Xet hash:
- 5a7b3aa4733f7eac59da614dd743b99bcdfb9c8e60dfd9f4649da28d45c911be
·
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.