Download web/templates/train.html from SLM-Archive/hyperdex-trainer: direct link, hf CLI and curl.
- Browser
- Download file 12.1 kB
-
https://huggingface.co/SLM-Archive/hyperdex-trainer/resolve/main/web/templates/train.html
- Command line
-
hf download hf://SLM-Archive/hyperdex-trainer/web/templates/train.html
-
curl -L -o train.html https://huggingface.co/SLM-Archive/hyperdex-trainer/resolve/main/web/templates/train.html
12.1 kB
| {% extends "base.html" %} | |
| {% set nav = 'train' %} | |
| {% block title %}New run — NanoDex{% endblock %} | |
| {% block body %} | |
| <div style="margin-bottom:24px"> | |
| <span class="eyebrow">New training run</span> | |
| <h1 style="font-size:28px;margin-top:5px">Build a model from nothing</h1> | |
| </div> | |
| <div class="steps"> | |
| <div class="stp on" data-s="1"><div class="n">Step 1</div><div class="t">Size</div></div> | |
| <div class="stp" data-s="2"><div class="n">Step 2</div><div class="t">Tokens</div></div> | |
| <div class="stp" data-s="3"><div class="n">Step 3</div><div class="t">Name</div></div> | |
| <div class="stp" data-s="4"><div class="n">Step 4</div><div class="t">Review</div></div> | |
| </div> | |
| <!-- STEP 1 ----------------------------------------------------------------> | |
| <div class="panelstep on" data-p="1"> | |
| <h2 style="font-size:19px">How big should it be?</h2> | |
| <p class="sub" style="margin:6px 0 20px">Every tier is the same architecture — | |
| a <span class="mono">LlamaForCausalLM</span> decoder-only transformer with | |
| SiLU MLPs, RMSNorm, rotary embeddings and grouped-query attention — scaled | |
| down in width and depth. The parameter counts below are totals, embeddings | |
| included.</p> | |
| <div class="picks"> | |
| {% for t in tiers %} | |
| <button type="button" class="pick {{ 'on' if t.key == '1m' }}" data-tier="{{ t.key }}" | |
| data-params="{{ t.params }}" data-label="{{ t.label }}"> | |
| <b>{{ t.label }}</b> | |
| <div class="p-n">{{ "{:,}".format(t.params) }} params</div> | |
| <div class="p-d"> | |
| {{ t.layers }} layers · {{ t.hidden }} hidden<br> | |
| {{ t.heads }} heads ({{ t.kv_heads }} KV)<br> | |
| FFN {{ t.ffn }} · ctx {{ seq_len }} | |
| </div> | |
| </button> | |
| {% endfor %} | |
| </div> | |
| <div style="display:flex;justify-content:flex-end;margin-top:24px"> | |
| <button class="btn primary" data-go="2">Continue →</button> | |
| </div> | |
| </div> | |
| <!-- STEP 2 ----------------------------------------------------------------> | |
| <div class="panelstep" data-p="2"> | |
| <h2 style="font-size:19px">How much should it read?</h2> | |
| <p class="sub" style="margin:6px 0 24px">Training data is | |
| <a href="https://huggingface.co/datasets/HuggingFaceFW/fineweb-edu" target="_blank" | |
| rel="noopener" style="color:var(--accent)">fineweb-edu</a> — filtered | |
| educational web text. More tokens means a sharper model and a longer wait.</p> | |
| <div class="card pad-lg"> | |
| <div style="display:flex;justify-content:space-between;align-items:baseline;margin-bottom:16px"> | |
| <label class="fl" style="margin:0">Token budget</label> | |
| <div class="mono" style="font-size:26px;font-weight:680;color:var(--accent)" | |
| id="tokLabel">500M</div> | |
| </div> | |
| <input type="range" id="tok" min="0" max="{{ stops|length - 1 }}" | |
| value="4" step="1"> | |
| <div class="mono muted" | |
| style="display:flex;justify-content:space-between;margin-top:9px;font-size:12px"> | |
| <span>200M</span><span>500M</span><span>1B</span><span>1.5B</span> | |
| </div> | |
| <div class="quickpick"> | |
| {% for m in [200, 500, 1000, 1500] %} | |
| <button type="button" class="btn sm" data-tokens="{{ m }}"> | |
| {{ (m / 1000)|round(1)|string|replace('.0','') ~ 'B' if m >= 1000 | |
| else (m|string ~ 'M') }}</button> | |
| {% endfor %} | |
| </div> | |
| </div> | |
| <div class="grid g4" style="margin-top:16px"> | |
| <div class="stat"><span>Parameters</span><b id="e-params">—</b><small id="e-tier"> </small></div> | |
| <div class="stat"><span>Optimizer steps</span><b id="e-steps">—</b><small>gradient updates</small></div> | |
| <div class="stat"><span>Estimated time</span><b id="e-eta">—</b><small id="e-cal"> </small></div> | |
| <div class="stat"><span>Tokens / param</span><b id="e-ratio">—</b><small id="e-chin"> </small></div> | |
| </div> | |
| <p class="hint" id="scaleNote"></p> | |
| <div style="display:flex;justify-content:space-between;margin-top:24px"> | |
| <button class="btn ghost" data-go="1">← Back</button> | |
| <button class="btn primary" data-go="3">Continue →</button> | |
| </div> | |
| </div> | |
| <!-- STEP 3 ----------------------------------------------------------------> | |
| <div class="panelstep" data-p="3"> | |
| <h2 style="font-size:19px">Give it a name</h2> | |
| <p class="sub" style="margin:6px 0 20px">This is what you'll see in your model | |
| list, and the default repository name when you publish it.</p> | |
| <div class="card pad-lg" style="max-width:560px"> | |
| <label class="fl" for="name">Model name</label> | |
| <input type="text" id="name" placeholder="my-first-lm" maxlength="80" autocomplete="off"> | |
| <p class="hint">Letters, numbers, <span class="mono">-</span>, | |
| <span class="mono">_</span> and <span class="mono">.</span>. Leave it blank | |
| and we'll name it after the size and token budget.</p> | |
| <p class="hint" style="margin-top:14px">Will publish to | |
| <span class="mono" style="color:var(--fg-2)">{{ user.username }}/<span id="namePreview">…</span></span></p> | |
| </div> | |
| <div style="display:flex;justify-content:space-between;margin-top:24px"> | |
| <button class="btn ghost" data-go="2">← Back</button> | |
| <button class="btn primary" data-go="4">Review →</button> | |
| </div> | |
| </div> | |
| <!-- STEP 4 ----------------------------------------------------------------> | |
| <div class="panelstep" data-p="4"> | |
| <h2 style="font-size:19px">Ready to launch</h2> | |
| <p class="sub" style="margin:6px 0 20px">Your run joins the shared queue. You | |
| can watch it — or close the tab and come back; it keeps going either way.</p> | |
| <div class="grid g2"> | |
| <div class="card pad-lg"> | |
| <span class="eyebrow">Model</span> | |
| <div style="margin-top:12px"> | |
| <div class="kv"><span>Name</span><b id="r-name">—</b></div> | |
| <div class="kv"><span>Size</span><b id="r-tier">—</b></div> | |
| <div class="kv"><span>Parameters</span><b id="r-params">—</b></div> | |
| <div class="kv"><span>Architecture</span><b id="r-arch">—</b></div> | |
| <div class="kv"><span>Vocabulary</span><b>{{ "{:,}".format(vocab) }} BPE</b></div> | |
| <div class="kv"><span>Context</span><b>{{ seq_len }} tokens</b></div> | |
| </div> | |
| </div> | |
| <div class="card pad-lg"> | |
| <span class="eyebrow">Training</span> | |
| <div style="margin-top:12px"> | |
| <div class="kv"><span>Dataset</span><b>fineweb-edu</b></div> | |
| <div class="kv"><span>Token budget</span><b id="r-tok">—</b></div> | |
| <div class="kv"><span>Optimizer steps</span><b id="r-steps">—</b></div> | |
| <div class="kv"><span>Optimizer</span><b>AdamW · cosine</b></div> | |
| <div class="kv"><span>Estimated time</span><b id="r-eta">—</b></div> | |
| <div class="kv"><span>Queue limit</span><b>{{ max_active }} runs per person</b></div> | |
| </div> | |
| </div> | |
| </div> | |
| <div style="display:flex;justify-content:space-between;margin-top:24px;gap:12px;flex-wrap:wrap"> | |
| <button class="btn ghost" data-go="3">← Back</button> | |
| <button class="btn primary lg" id="launch">Queue this run →</button> | |
| </div> | |
| </div> | |
| {% endblock %} | |
| {% block scripts %} | |
| <script> | |
| const STOPS = {{ stops | tojson }}; | |
| const state = { tier:'1m', tokens:500, name:'' }; | |
| function labelTokens(m){ | |
| return m >= 1000 ? String(m/1000).replace(/\.0$/,'') + 'B' : m + 'M'; | |
| } | |
| let est = null; | |
| function showStep(n){ | |
| document.querySelectorAll('.panelstep').forEach(p => | |
| p.classList.toggle('on', p.dataset.p === String(n))); | |
| document.querySelectorAll('.stp').forEach(s => { | |
| const i = Number(s.dataset.s); | |
| s.classList.toggle('on', i === n); | |
| s.classList.toggle('ok', i < n); | |
| }); | |
| window.scrollTo({top:0, behavior:'smooth'}); | |
| if(n === 4) fillReview(); | |
| } | |
| document.querySelectorAll('[data-go]').forEach(b => | |
| b.addEventListener('click', () => showStep(Number(b.dataset.go)))); | |
| document.querySelectorAll('.stp').forEach(s => | |
| s.addEventListener('click', () => showStep(Number(s.dataset.s)))); | |
| document.querySelectorAll('.pick[data-tier]').forEach(p => | |
| p.addEventListener('click', () => { | |
| document.querySelectorAll('.pick[data-tier]').forEach(q => q.classList.remove('on')); | |
| p.classList.add('on'); | |
| state.tier = p.dataset.tier; | |
| updateEstimate(); | |
| })); | |
| const tok = document.getElementById('tok'); | |
| function setTokens(m){ | |
| const i = STOPS.indexOf(m); | |
| state.tokens = m; | |
| if(i >= 0) tok.value = i; | |
| document.getElementById('tokLabel').textContent = labelTokens(m); | |
| updateEstimate(); | |
| } | |
| tok.addEventListener('input', () => setTokens(STOPS[Number(tok.value)])); | |
| document.querySelectorAll('[data-tokens]').forEach(b => | |
| b.addEventListener('click', () => setTokens(Number(b.dataset.tokens)))); | |
| const nameEl = document.getElementById('name'); | |
| nameEl.addEventListener('input', () => { | |
| state.name = nameEl.value; | |
| document.getElementById('namePreview').textContent = defaultName(); | |
| }); | |
| function defaultName(){ | |
| const s = state.name.trim().replace(/[^A-Za-z0-9._-]+/g,'-').replace(/^[-._]+|[-._]+$/g,''); | |
| if(s) return s.slice(0,80); | |
| const sel = document.querySelector('.pick.on[data-tier]'); | |
| return (sel ? sel.dataset.label.toLowerCase() : 'nanodex') | |
| + '-' + labelTokens(state.tokens).toLowerCase(); | |
| } | |
| let estTimer; | |
| function updateEstimate(){ | |
| clearTimeout(estTimer); | |
| estTimer = setTimeout(async () => { | |
| try{ | |
| est = await NX.get(`/api/estimate?tier=${state.tier}&tokens=${state.tokens*1e6}`); | |
| }catch(e){ return; } | |
| const sel = document.querySelector('.pick.on[data-tier]'); | |
| document.getElementById('e-params').textContent = Number(est.n_params).toLocaleString(); | |
| document.getElementById('e-tier').textContent = sel ? sel.dataset.label : ''; | |
| document.getElementById('e-steps').textContent = Number(est.steps).toLocaleString(); | |
| document.getElementById('e-eta').textContent = est.eta_human; | |
| document.getElementById('e-cal').textContent = est.calibrated | |
| ? 'from real runs here' : 'prior — self-calibrates'; | |
| document.getElementById('e-ratio').textContent = Math.round(est.ratio) + ':1'; | |
| document.getElementById('e-chin').textContent = | |
| est.chinchilla_x.toFixed(1) + '× Chinchilla'; | |
| const r = est.ratio; | |
| document.getElementById('scaleNote').textContent = r < 60 | |
| ? 'Under ~60 tokens per parameter this model never uses the capacity it has. Either raise the budget or pick a smaller size.' | |
| : (r < 150 | |
| ? 'Workable, but on the thin side for this size — 300:1 or more is where these models get interesting.' | |
| : (r > 2000 | |
| ? 'Very deep into the over-trained regime. Still improves, but the last few hundred million tokens buy little — a larger size would use them better.' | |
| : 'A good range for this size: well past compute-optimal, and it still finishes in reasonable time.')); | |
| document.getElementById('namePreview').textContent = defaultName(); | |
| }, 140); | |
| } | |
| function fillReview(){ | |
| const sel = document.querySelector('.pick.on[data-tier]'); | |
| document.getElementById('r-name').textContent = defaultName(); | |
| document.getElementById('r-tier').textContent = sel ? sel.dataset.label : '—'; | |
| document.getElementById('r-params').textContent = | |
| sel ? Number(sel.dataset.params).toLocaleString() : '—'; | |
| const d = sel ? sel.querySelector('.p-d').textContent.trim().split('·') : []; | |
| document.getElementById('r-arch').textContent = 'LlamaForCausalLM'; | |
| document.getElementById('r-tok').textContent = labelTokens(state.tokens) + ' tokens'; | |
| document.getElementById('r-steps').textContent = | |
| est ? Number(est.steps).toLocaleString() : '—'; | |
| document.getElementById('r-eta').textContent = est ? est.eta_human : '—'; | |
| } | |
| document.getElementById('launch').addEventListener('click', async (ev) => { | |
| const b = ev.currentTarget; | |
| b.disabled = true; b.textContent = 'Queueing…'; | |
| try{ | |
| const r = await NX.post('/api/jobs', { | |
| tier: state.tier, tokens: state.tokens * 1e6, name: state.name | |
| }); | |
| location.href = '/models/' + r.id; | |
| }catch(e){ | |
| NX.toast(NX.esc(e.message), 'err'); | |
| b.disabled = false; b.innerHTML = 'Queue this run →'; | |
| } | |
| }); | |
| setTokens(state.tokens); | |
| document.getElementById('namePreview').textContent = defaultName(); | |
| </script> | |
| {% endblock %} | |