| |
| """ |
| upload_v12_base.py — push the frozen v12 base + the code needed to load it to HF. |
| |
| Run on the pod from /workspace after `huggingface-cli login` (or pass token=...): |
| python upload_v12_base.py |
| |
| Uploads to Daxamite/V12Emoe: |
| ckpt_v12_190m_best.pt <- out_v12_190m/best.pt (the frozen student) |
| model_hybrid.py <- the GPT/GPTConfig the ckpt deserializes into |
| muon.py <- imported by model_hybrid / training code |
| tok_v9.py <- the v9_chatml_enc tokenizer factory |
| README.md <- minimal load instructions (so the repo is self-describing) |
| |
| These are exactly the files run_hyper.sh / runtime_adapters.py pull to rebuild the |
| base, so a later RuntimeConfig(base_ckpt=...) can point straight at this repo. |
| """ |
|
|
| import os |
| from huggingface_hub import HfApi, create_repo |
|
|
| REPO = "Daxamite/V12Emoe" |
| TOKEN = os.environ.get("HF_TOKEN") |
| CKPT_LOCAL = "out_v12_190m/best.pt" |
| CKPT_REMOTE = "ckpt_v12_190m_best.pt" |
|
|
| |
| FILES = [ |
| (CKPT_LOCAL, CKPT_REMOTE), |
| ("model_hybrid.py", "model_hybrid.py"), |
| ("muon.py", "muon.py"), |
| ("tok_v9.py", "tok_v9.py"), |
| ] |
|
|
| README = f"""--- |
| license: other |
| tags: [v12, emoe, fft-hybrid, frozen-base] |
| --- |
| |
| # V12 eMoE — frozen base (190M FFT-hybrid) |
| |
| The frozen "student" for the v12 ephemeral-MoE system: a 12L / 1024d / 16h |
| StripedHyena-style FFT-spectral hybrid (`attn_every=3`), `block_size=2048`, |
| tokenizer `v9_chatml_enc` (vocab 50264). Trained Muon+AdamW (3.0x ratio), WSD, |
| ~7.5B tokens. **best.pt VAL 1.66.** The hypernetwork specializes this; it is |
| never fine-tuned here. |
| |
| ## Load |
| |
| ```python |
| import torch |
| from dataclasses import fields |
| from model_hybrid import GPT, GPTConfig |
| |
| ckpt = torch.load("ckpt_v12_190m_best.pt", map_location="cpu") |
| margs = ckpt.get("model_args") or ckpt.get("config") or {{}} |
| cfg = GPTConfig(**{{k: v for k, v in margs.items() |
| if k in {{f.name for f in fields(GPTConfig)}}}}) |
| sd = ckpt.get("model") or ckpt.get("state_dict") or ckpt.get("model_state_dict") |
| sd = {{k.replace("_orig_mod.", ""): v for k, v in sd.items()}} |
| model = GPT(cfg); model.load_state_dict(sd, strict=False); model.eval() |
| |
| import tok_v9; tok = tok_v9.build() |
| ``` |
| """ |
|
|
|
|
| def main(): |
| api = HfApi(token=TOKEN) |
| create_repo(REPO, repo_type="model", exist_ok=True, token=TOKEN) |
|
|
| missing = [p for p, _ in FILES if not os.path.exists(p)] |
| if missing: |
| raise SystemExit(f"missing local files (run from /workspace?): {missing}") |
|
|
| for local, remote in FILES: |
| sz = os.path.getsize(local) / 1e6 |
| print(f">>> uploading {local} -> {REPO}/{remote} ({sz:.1f} MB)") |
| api.upload_file(path_or_fileobj=local, path_in_repo=remote, |
| repo_id=REPO, repo_type="model") |
|
|
| api.upload_file(path_or_fileobj=README.encode(), path_in_repo="README.md", |
| repo_id=REPO, repo_type="model") |
| print(f">>> done. https://huggingface.co/{REPO}") |
|
|
|
|
| if __name__ == "__main__": |
| main() |
|
|