File size: 5,197 Bytes
12496fc
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
"""Rebuild parameter/cost scenarios and capture primary-source revisions."""
from pathlib import Path
from datetime import datetime, timezone
import importlib.metadata
import json
import platform
import subprocess
import sys
import urllib.request
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
from nexora.compute import Estimate, kv_cache_bytes, topology
from huggingface_hub import HfApi


def moe(layers, hidden, intermediate, experts, selected, shared, heads, kv, vocab=131072):
    dim = hidden//heads
    attention = layers*(2*hidden*hidden + 2*hidden*kv*dim)
    one_expert = 3*hidden*intermediate
    expert_total = layers*(experts+shared)*one_expert
    routing = layers*hidden*experts
    norms = (2*layers+1)*hidden
    embedding = vocab*hidden
    total = attention+expert_total+routing+norms+embedding
    active = attention+layers*(selected+shared)*one_expert+routing+norms+embedding
    return {"layers": layers, "hidden_size": hidden, "expert_intermediate": intermediate, "experts": experts,
            "selected": selected, "shared": shared, "heads": heads, "kv_heads": kv, "head_dim": dim, "vocab": vocab,
            "attention_parameters": attention, "expert_parameters": expert_total, "router_parameters": routing,
            "norm_parameters": norms, "tied_embedding_parameters": embedding, "total_parameters": total, "active_parameters_upper_convention": active,
            "active_convention": "Counts full tied embedding/output matrix, all attention, router and selected/shared FFNs; actual sparse embedding lookup uses fewer elements",
            "kv_cache_GB_batch1": {str(n): kv_cache_bytes(layers, kv, dim, n)/1e9 for n in [32768, 65536, 131072, 262144]}}


def main():
    out = Path("reports")
    out.mkdir(exist_ok=True)
    proposals = {"B_custom_120B_class": moe(49, 3072, 2048, 128, 4, 1, 24, 8),
                 "C_custom_larger_MoE": moe(80, 4096, 1536, 160, 4, 1, 32, 8)}
    cases = {"A_dense_120B": (120e9, 120e9, 2.4e12, 1024),
             "B_custom_120B_class": (proposals["B_custom_120B_class"]["total_parameters"], proposals["B_custom_120B_class"]["active_parameters_upper_convention"], 2.4e12, 256),
             "C_custom_larger_MoE": (proposals["C_custom_larger_MoE"]["total_parameters"], proposals["C_custom_larger_MoE"]["active_parameters_upper_convention"], 3e12, 512),
             "D_30B_continue": (30.5e9, 3.3e9, 10e9, 8),
             "E_4B_continue": (4e9, 4e9, 1e9, 8)}
    estimates = {}
    for name, args in cases.items():
        estimates[name] = {label: Estimate(*args, mfu=mfu, gpu_hour_usd=rate).calculate()
                           for label, mfu, rate in [("low_cost", .5, 2), ("expected", .35, 3), ("high_cost", .2, 5)]}
    report = {"assumptions": "H100 SXM BF16 dense peak 989 TFLOP/s planning scenario; MFU .2/.35/.5 not measured; price $2/$3/$5 per GPU-hour are editable hypothetical inputs, not vendor quotes; excludes retries/people/tax/storage/power", "architectures": proposals, "estimates": estimates,
              "topologies": [topology(1, 8, 1, 1, 1, 8), topology(8, 8, 2, 2, 1, 16, 8), topology(128, 8, 8, 8, 2, 8)],
              "storage_formula": "3 retained checkpoints * checkpoint_GB + 2 replicated copies of token shards + raw corpus + intermediate dedup data; add validation/holdouts and free-space headroom"}
    (out / "compute.json").write_text(json.dumps(report, indent=2))
    env = {"date_utc": datetime.now(timezone.utc).isoformat(), "os": platform.system(), "python": platform.python_version(),
           "packages": {p: importlib.metadata.version(p) for p in ["torch", "numpy", "transformers", "safetensors", "huggingface_hub", "jsonschema", "pytest"]},
           "hardware": {"cpu": "AMD Ryzen 7 7435HS", "cores": 8, "ram_bytes": 16989736960, "gpu": "NVIDIA RTX 3050 Laptop", "vram_mib": 4096,
                        "cuda_in_installed_torch": False, "free_workspace_drive_bytes_at_discovery": 322162003968},
           "cluster": "No allocated cluster discovered", "cloud": "HF authentication available; no paid jobs launched", "budget": "No spending budget specified; local execution only"}
    (out / "environment.json").write_text(json.dumps(env, indent=2))
    sources = []
    api = HfApi()
    for repo in ["Qwen/Qwen3-0.6B", "Qwen/Qwen3.5-0.8B", "Qwen/Qwen3.5-4B", "Qwen/Qwen3-30B-A3B-Instruct-2507"]:
        info = api.model_info(repo)
        url = f"https://huggingface.co/{repo}/resolve/{info.sha}/config.json"
        with urllib.request.urlopen(url, timeout=30) as r:
            config = json.load(r)
        sources.append({"repo": repo, "revision": info.sha, "license": info.card_data.get("license") if info.card_data else None,
                        "config_url": url, "card_url": f"https://huggingface.co/{repo}/blob/{info.sha}/README.md", "config": config})
    (out / "sources.json").write_text(json.dumps({"accessed_utc": env["date_utc"], "models": sources}, indent=2))
    for name, values in estimates.items():
        e = values["expected"]
        print(name, "total_B", e["total_parameters"]/1e9, "active_B", e["active_parameters"]/1e9, "hours", round(e["hours"], 1), "USD", round(e["compute_usd"]))


if __name__ == "__main__":
    main()