Buckets:
| """Write logs/STATUS.md (progress of every job + machine health) and upload it to the bucket root. | |
| Run by the watchdog every ~30 min; safe to run by hand: .venv/bin/python code/status.py [--no-upload]""" | |
| import json, os, re, subprocess, sys, time, collections | |
| from pathlib import Path | |
| from dotenv import load_dotenv | |
| ROOT = Path(__file__).resolve().parent.parent; os.chdir(ROOT); load_dotenv(ROOT / ".env") | |
| L = []; w = L.append | |
| sh = lambda c: subprocess.run(c, shell=True, capture_output=True, text=True).stdout.strip() | |
| lines = lambda p: sum(1 for _ in open(p)) if Path(p).exists() else 0 | |
| w(f"# SkillsStorage — live status\n\nUpdated **{time.strftime('%Y-%m-%d %H:%M', time.gmtime(time.time() + 19800))} IST** ({time.strftime('%H:%M UTC', time.gmtime())}; auto-generated by `code/status.py`)\n") | |
| mem = sh("free -m | awk '/Mem:/{print $2, $3, $7} /Swap:/{print $2, $3}'").split() | |
| disk = sh("df -BG --output=size,used,avail / | tail -1").split() | |
| w(f"Machine: RAM used {mem[1]}/{mem[0]} MB (available {mem[2]}), swap {mem[4]}/{mem[3]} MB, disk free {disk[2]} of {disk[0]}, load {open('/proc/loadavg').read().split()[0]}\n") | |
| w("## Services\n\n| job | state | restarts | RAM MB (anon) | last log line |\n|---|---|---|---|---|") | |
| for line in open("code/jobs.conf"): | |
| if line.startswith("#") or not line.strip(): continue | |
| name = line.split("|")[0].strip() | |
| st = sh(f"systemctl is-active skills-{name}") or "inactive" | |
| ex = Path(f"logs/{name}.exit").read_text().strip() if Path(f"logs/{name}.exit").exists() else None | |
| if st != "active" and ex == "0": st = "✅ finished" | |
| rs = sh(f"systemctl show -p NRestarts --value skills-{name}") or "-" | |
| stat = Path(f"/sys/fs/cgroup/system.slice/skills-{name}.service/memory.stat") # anon = real use (excl. reclaimable cache) | |
| mb = next((str(int(l.split()[1]) // 1048576) for l in stat.read_text().splitlines() if l.startswith("anon ")), "-") if stat.exists() else "-" | |
| last = sh(f"tail -1 logs/{name}.log 2>/dev/null").replace("|", "/")[:90] | |
| w(f"| {name} | {st} | {rs} | {mb} | `{last}` |") | |
| w("\n## Progress\n") | |
| gs = Path("logs/build_gitskills.done").read_text().split() if Path("logs/build_gitskills.done").exists() else [] | |
| _gl = "".join(Path(f).read_text() for f in ("logs/build_gitskills.log", "logs/gitskills.log") if Path(f).exists()) | |
| gs_rows = sum(int(v) for v in dict(re.findall(r"(artifacts/part-\d+)\.parquet: (\d+) rows", _gl)).values()) # one count per part | |
| w(f"- **GitSkills → processed**: {sum(1 for x in gs if x.startswith('artifacts'))}/31 skill parts ({gs_rows:,} skills), " | |
| f"{sum(1 for x in gs if x.startswith('siblings'))}/45 bundle-file parts") | |
| for run, raw, seeds, shiplog in [("pass1_sourcegraph", "data/raw/github_delta", "data/seeds/sg_new.txt", "logs/ship-pass1.log"), | |
| ("pass2_new", "data/raw/github_delta_p2", "data/seeds/pass2.txt", "logs/ship-pass2.log"), | |
| ("backfill_gitskills", "data/raw/github_backfill", "data/seeds/backfill_gitskills.txt", "logs/ship-backfill.log")]: | |
| sk = fi = sh_n = 0 | |
| if Path(shiplog).exists(): | |
| for m in re.finditer(r"shipped \S+: (\d+) skills, (\d+) files", Path(shiplog).read_text()): sh_n += 1; sk += int(m[1]); fi += int(m[2]) | |
| w(f"- **GitHub {run}**: {lines(raw + '/done.txt'):,}/{lines(seeds):,} repos done; shipped {sh_n} shards = {sk:,} skills, {fi:,} bundle files") | |
| w(f"- **ClawHub dump → processed**: {'done (80,961 skills, 291,125 bundle files)' if Path('logs/build_clawhub.done').exists() else 'pending'}") | |
| w(f"- **ClawHub API**: {lines('data/raw/clawhub_api/listing.jsonl') or lines('data/raw/clawhub_api/listing.tmp'):,} listed; " | |
| f"{len(list(Path('data/raw/clawhub_api/zips').glob('*.zip'))):,} delta bundles; converter {'done' if Path('logs/build_clawhub_api.done').exists() else 'pending'}") | |
| _tl = "".join(Path(f).read_text() for f in ("logs/github_topics.log", "logs/topics.log") if Path(f).exists()) | |
| w(f"- **Topic search**: {len(set(re.findall(r'topic done (\S+)', _tl)))}/12 topics, {lines('data/seeds/github_topics.jsonl'):,} hits") | |
| try: | |
| from huggingface_hub import HfApi | |
| s = collections.Counter() | |
| for f in HfApi().list_bucket_tree("Mercity/SkillsStorage", recursive=True): | |
| if getattr(f, "size", None) is not None: s["/".join(f.path.split("/")[:2])] += f.size | |
| w("\n## Bucket `hf://buckets/Mercity/SkillsStorage`\n\n| prefix | GB |\n|---|---|") | |
| for k in sorted(s): | |
| if not k.startswith("code/"): w(f"| {k} | {s[k] / 1e9:.2f} |") | |
| w(f"| **total** | **{sum(s.values()) / 1e9:.2f}** |") | |
| except Exception as e: | |
| w(f"\n(bucket listing failed: {e})") | |
| # only error lines that appeared since the previous status run (byte offsets per log) | |
| off_f = Path("logs/.status_offsets.json"); offs = json.loads(off_f.read_text()) if off_f.exists() else {} | |
| new_errs = [] | |
| for lf in sorted(Path("logs").glob("*.log")): | |
| size = lf.stat().st_size; start = offs.get(lf.name, 0) if offs.get(lf.name, 0) <= size else 0 | |
| with open(lf, errors="replace") as f: | |
| f.seek(start) | |
| new_errs += [f"{lf.stem}: {l.strip()[:160]}" for l in f if re.search(r"Error|Traceback|GIVING UP", l)] | |
| offs[lf.name] = size | |
| off_f.write_text(json.dumps(offs)) | |
| w(f"\n## New error lines since previous status ({len(new_errs)})\n\n```\n" + ("\n".join(new_errs[-8:]) or "none") + "\n```") | |
| mg = sh("tail -5 logs/memguard.log 2>/dev/null") | |
| w("\n## Memory guard actions (last 5; freezes low-priority jobs when RAM < 350 MB)\n\n```\n" + (mg or "none — RAM never ran low") + "\n```") | |
| wd = sh("tail -8 logs/watchdog.log 2>/dev/null") | |
| w("\n## Watchdog actions (last 8)\n\n```\n" + (wd or "none") + "\n```\n") | |
| Path("logs/STATUS.md").write_text("\n".join(L)) | |
| if "--no-upload" not in sys.argv: | |
| subprocess.run([str(ROOT / ".venv/bin/hf"), "buckets", "cp", "logs/STATUS.md", "hf://buckets/Mercity/SkillsStorage/STATUS.md"], capture_output=True) | |
| print("\n".join(L)) | |
Xet Storage Details
- Size:
- 5.9 kB
- Xet hash:
- e0841d112a8d54714016b714a4ab7a97dd56a0171c9b39fb6a525bfe8fae1b39
·
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.