Buckets:

Pranav2748's picture
download
raw
5.9 kB
"""Write logs/STATUS.md (progress of every job + machine health) and upload it to the bucket root.
Run by the watchdog every ~30 min; safe to run by hand: .venv/bin/python code/status.py [--no-upload]"""
import json, os, re, subprocess, sys, time, collections
from pathlib import Path
from dotenv import load_dotenv
ROOT = Path(__file__).resolve().parent.parent; os.chdir(ROOT); load_dotenv(ROOT / ".env")
L = []; w = L.append
sh = lambda c: subprocess.run(c, shell=True, capture_output=True, text=True).stdout.strip()
lines = lambda p: sum(1 for _ in open(p)) if Path(p).exists() else 0
w(f"# SkillsStorage — live status\n\nUpdated **{time.strftime('%Y-%m-%d %H:%M', time.gmtime(time.time() + 19800))} IST** ({time.strftime('%H:%M UTC', time.gmtime())}; auto-generated by `code/status.py`)\n")
mem = sh("free -m | awk '/Mem:/{print $2, $3, $7} /Swap:/{print $2, $3}'").split()
disk = sh("df -BG --output=size,used,avail / | tail -1").split()
w(f"Machine: RAM used {mem[1]}/{mem[0]} MB (available {mem[2]}), swap {mem[4]}/{mem[3]} MB, disk free {disk[2]} of {disk[0]}, load {open('/proc/loadavg').read().split()[0]}\n")
w("## Services\n\n| job | state | restarts | RAM MB (anon) | last log line |\n|---|---|---|---|---|")
for line in open("code/jobs.conf"):
if line.startswith("#") or not line.strip(): continue
name = line.split("|")[0].strip()
st = sh(f"systemctl is-active skills-{name}") or "inactive"
ex = Path(f"logs/{name}.exit").read_text().strip() if Path(f"logs/{name}.exit").exists() else None
if st != "active" and ex == "0": st = "✅ finished"
rs = sh(f"systemctl show -p NRestarts --value skills-{name}") or "-"
stat = Path(f"/sys/fs/cgroup/system.slice/skills-{name}.service/memory.stat") # anon = real use (excl. reclaimable cache)
mb = next((str(int(l.split()[1]) // 1048576) for l in stat.read_text().splitlines() if l.startswith("anon ")), "-") if stat.exists() else "-"
last = sh(f"tail -1 logs/{name}.log 2>/dev/null").replace("|", "/")[:90]
w(f"| {name} | {st} | {rs} | {mb} | `{last}` |")
w("\n## Progress\n")
gs = Path("logs/build_gitskills.done").read_text().split() if Path("logs/build_gitskills.done").exists() else []
_gl = "".join(Path(f).read_text() for f in ("logs/build_gitskills.log", "logs/gitskills.log") if Path(f).exists())
gs_rows = sum(int(v) for v in dict(re.findall(r"(artifacts/part-\d+)\.parquet: (\d+) rows", _gl)).values()) # one count per part
w(f"- **GitSkills → processed**: {sum(1 for x in gs if x.startswith('artifacts'))}/31 skill parts ({gs_rows:,} skills), "
f"{sum(1 for x in gs if x.startswith('siblings'))}/45 bundle-file parts")
for run, raw, seeds, shiplog in [("pass1_sourcegraph", "data/raw/github_delta", "data/seeds/sg_new.txt", "logs/ship-pass1.log"),
("pass2_new", "data/raw/github_delta_p2", "data/seeds/pass2.txt", "logs/ship-pass2.log"),
("backfill_gitskills", "data/raw/github_backfill", "data/seeds/backfill_gitskills.txt", "logs/ship-backfill.log")]:
sk = fi = sh_n = 0
if Path(shiplog).exists():
for m in re.finditer(r"shipped \S+: (\d+) skills, (\d+) files", Path(shiplog).read_text()): sh_n += 1; sk += int(m[1]); fi += int(m[2])
w(f"- **GitHub {run}**: {lines(raw + '/done.txt'):,}/{lines(seeds):,} repos done; shipped {sh_n} shards = {sk:,} skills, {fi:,} bundle files")
w(f"- **ClawHub dump → processed**: {'done (80,961 skills, 291,125 bundle files)' if Path('logs/build_clawhub.done').exists() else 'pending'}")
w(f"- **ClawHub API**: {lines('data/raw/clawhub_api/listing.jsonl') or lines('data/raw/clawhub_api/listing.tmp'):,} listed; "
f"{len(list(Path('data/raw/clawhub_api/zips').glob('*.zip'))):,} delta bundles; converter {'done' if Path('logs/build_clawhub_api.done').exists() else 'pending'}")
_tl = "".join(Path(f).read_text() for f in ("logs/github_topics.log", "logs/topics.log") if Path(f).exists())
w(f"- **Topic search**: {len(set(re.findall(r'topic done (\S+)', _tl)))}/12 topics, {lines('data/seeds/github_topics.jsonl'):,} hits")
try:
from huggingface_hub import HfApi
s = collections.Counter()
for f in HfApi().list_bucket_tree("Mercity/SkillsStorage", recursive=True):
if getattr(f, "size", None) is not None: s["/".join(f.path.split("/")[:2])] += f.size
w("\n## Bucket `hf://buckets/Mercity/SkillsStorage`\n\n| prefix | GB |\n|---|---|")
for k in sorted(s):
if not k.startswith("code/"): w(f"| {k} | {s[k] / 1e9:.2f} |")
w(f"| **total** | **{sum(s.values()) / 1e9:.2f}** |")
except Exception as e:
w(f"\n(bucket listing failed: {e})")
# only error lines that appeared since the previous status run (byte offsets per log)
off_f = Path("logs/.status_offsets.json"); offs = json.loads(off_f.read_text()) if off_f.exists() else {}
new_errs = []
for lf in sorted(Path("logs").glob("*.log")):
size = lf.stat().st_size; start = offs.get(lf.name, 0) if offs.get(lf.name, 0) <= size else 0
with open(lf, errors="replace") as f:
f.seek(start)
new_errs += [f"{lf.stem}: {l.strip()[:160]}" for l in f if re.search(r"Error|Traceback|GIVING UP", l)]
offs[lf.name] = size
off_f.write_text(json.dumps(offs))
w(f"\n## New error lines since previous status ({len(new_errs)})\n\n```\n" + ("\n".join(new_errs[-8:]) or "none") + "\n```")
mg = sh("tail -5 logs/memguard.log 2>/dev/null")
w("\n## Memory guard actions (last 5; freezes low-priority jobs when RAM < 350 MB)\n\n```\n" + (mg or "none — RAM never ran low") + "\n```")
wd = sh("tail -8 logs/watchdog.log 2>/dev/null")
w("\n## Watchdog actions (last 8)\n\n```\n" + (wd or "none") + "\n```\n")
Path("logs/STATUS.md").write_text("\n".join(L))
if "--no-upload" not in sys.argv:
subprocess.run([str(ROOT / ".venv/bin/hf"), "buckets", "cp", "logs/STATUS.md", "hf://buckets/Mercity/SkillsStorage/STATUS.md"], capture_output=True)
print("\n".join(L))

Xet Storage Details

Size:
5.9 kB
·
Xet hash:
e0841d112a8d54714016b714a4ab7a97dd56a0171c9b39fb6a525bfe8fae1b39

Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.