Spaces:
Running
Running
File size: 21,392 Bytes
3365e13 abdc36e 3365e13 7d0b228 1185393 7406205 3365e13 cb2d355 3365e13 7406205 439999f 7406205 439999f 7406205 439999f 7406205 e74320a 3365e13 b50807f 3365e13 7406205 e74320a 7406205 3365e13 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 | <!doctype html><html><head><meta charset=utf8><meta name=viewport content="width=device-width,initial-scale=1,viewport-fit=cover">
<title>Q — private AI, in your browser</title>
<style>
:root{--bg:#0b0e14;--panel:#141922;--ink:#e6e9ef;--dim:#8a94a6;--q:#7c5cff;--u:#1f6feb;--line:#1e2531}
*{box-sizing:border-box}html,body{height:100%}
body{margin:0;font:15px/1.55 -apple-system,Segoe UI,Roboto,system-ui,monospace;background:var(--bg);color:var(--ink);display:flex;flex-direction:column;overscroll-behavior:none}
header{padding:10px 16px;border-bottom:1px solid var(--line);display:flex;align-items:center;gap:10px;flex:0 0 auto}
header b{font-weight:600}header .s{color:var(--dim);font-size:12px;overflow:hidden;text-overflow:ellipsis;white-space:nowrap}
#log{flex:1;overflow:auto;padding:16px;display:flex;flex-direction:column;gap:12px;-webkit-overflow-scrolling:touch}
.msg{max-width:82%;padding:9px 13px;border-radius:12px;white-space:pre-wrap;word-wrap:break-word}
.u{align-self:flex-end;background:var(--u)}.a{align-self:flex-start;background:var(--panel);border:1px solid #232b3a}
.a.think{color:var(--dim);font-style:italic}
footer{padding:12px 16px;border-top:1px solid var(--line);display:flex;gap:8px;flex:0 0 auto;padding-bottom:calc(12px + env(safe-area-inset-bottom))}
#in{flex:1;background:var(--panel);border:1px solid #232b3a;color:var(--ink);border-radius:10px;padding:10px 12px;font:inherit;resize:none;max-height:140px}
button{background:var(--q);color:#fff;border:0;border-radius:10px;padding:0 18px;font:inherit;cursor:pointer}button:disabled{opacity:.4;cursor:default}
.stat{color:var(--dim);font-size:11px;margin-top:3px}
</style></head><body>
<header><b>Q</b> <span class=s id=st>booting…</span></header>
<div id=log></div>
<footer><textarea id=in rows=1 placeholder="Message Q…" disabled></textarea><button id=send disabled>Send</button></footer>
<script type=module>
import { ready, loadModel, MODELS, defaultModelIndex } from "./core/loader.js";
import { createEngine } from "./core/engine.js";
import { selfPersona, selfFacts } from "./core/q-self.mjs"; // ONE grounded self-knowledge, shared with the messenger + voice
const $ = (s) => document.querySelector(s);
const log = $("#log"), input = $("#in"), send = $("#send"), st = $("#st");
const bubble = (cls, text = "") => { const d = document.createElement("div"); d.className = "msg " + cls; d.textContent = text; log.appendChild(d); log.scrollTop = log.scrollHeight; return d; };
const params = new URLSearchParams(location.search);
const pick = params.get("m");
let m = pick ? (MODELS.find((x) => new RegExp(pick, "i").test(x.name)) || MODELS[0]) : MODELS[defaultModelIndex()];
// STREAM FROM HF: a bare link streams the HOLOGRAMTECH BitNet κ-object from Hugging Face; ?hf=<org/repo> or
// ?kappa=<absolute-url> override. The κ-object is content-addressed + pinned, so the host is an UNTRUSTED CDN —
// every block is re-derived (Law L5); a bad byte is rejected. Blocks cache locally after first load (0-net on return).
{
let hf = params.get("hf"), kappa = params.get("kappa");
if (!hf && !kappa && !pick) hf = "HOLOGRAMTECH/q-bitnet-2b"; // bare link → stream BitNet from Hologram's HF repo
if (hf || kappa) {
const base = (kappa || `https://huggingface.co/${hf}/resolve/main`).replace(/\/+$/, "");
const bit = MODELS.find((x) => (x.fam || "").toLowerCase() === "bitnet") || m;
// drop the model's SHA-256 manifest pin: a ?hf repo may carry a different (e.g. BLAKE3) manifest, so the
// stale pin would falsely reject it. Per-block κ verification still applies (untrusted-CDN-safe).
m = { ...bit, kappaUrl: base, manifestKappa: undefined, name: bit.name + " · via " + (hf || new URL(base).host) };
}
}
// ?stream=layer → page the model layer-by-layer instead of resident. For t2 (BitNet) this exercises the DRAFT
// t2-streaming engine path (correctness first; the .qvf remote path adds fast-first-token). Default = resident.
{
const sm = params.get("stream");
if (sm && sm !== "false" && sm !== "resident") m = { ...m, stream: sm };
}
// ?verify=gpu → re-derive each BLAKE3 weight-block κ ENTIRELY on the GPU (2.74 GB/s) instead of pure-JS BLAKE3.
if (params.get("verify") === "gpu") globalThis.__gpuVerify = true;
// ?spec → speculative decode (n-gram draft + batched-K verify). Byte-identical to greedy; big wins on echo-heavy
// text (code/quote/retrieval), no gain on free-form chat. ?bench=spec runs the A/B measurement harness after load.
if (params.get("spec") || params.get("bench") === "spec") globalThis.__spec = true;
// GROUND the model as on-device Q (a base/instruct model has NO self-knowledge — without this it confabulates
// a generic "I run on OpenAI/AWS cloud servers" identity, which is false). Injected as the SYSTEM turn.
function frameSystem() {
const PERSONA = selfPersona({ model: m, engine });
if (m.llama3) return `<|start_header_id|>system<|end_header_id|>\n\n${PERSONA}<|eot_id|>`;
if (m.qwen) return `<|im_start|>system\n${PERSONA}<|im_end|>\n`;
if (m.olmo) return `<|system|>\n${PERSONA}\n`;
return PERSONA + "\n\n";
}
// GROUNDED IDENTITY (the anti-confabulation guard). A 2B model reverts to its training prior — "I'm GPT-3.5 on
// AWS" — when asked what/where it is, no matter the system prompt. But identity is not a guess: it is the TRUTH
// of THIS running instance. So provenance questions are answered DETERMINISTICALLY from the live facts (the real
// resident model + its κ + the real host it streamed from), never from the model. Grounded, not performed.
const IDENTITY_RX = /\b(are|r)\s*(you|u)\b.*\b(gpt|chatgpt|openai|claude|anthropic|gemini|bard|llama|language model|an? ai|running|local|on[- ]?device|in the browser|on (a )?server|in the cloud|hosted)\b|\bwhat( kind of| sort of| type of)?\b.*\b(model|llm|ai|are you|based on|powered by|architecture|run on|running)\b|\bwho\b.*\b(are you|made|built|created|trained|develop)\b|\bwhere\b.*\b(run|running|host|hosted|are you|live|located)\b|\bpowered by\b|\bwhat are you\b|\b(openai|chatgpt|gpt-?\d|aws|amazon web|google cloud|cloud server)\b|\b(local|cloud|server)\b.*\bmodel\b|\bdo you run\b/i;
function groundedIdentity() {
const f = selfFacts({ model: m, engine });
const name = f.model || (m && m.name) || "an on-device model";
const host = (f.weightsFrom && !/^local$/i.test(f.weightsFrom)) ? f.weightsFrom : "Hugging Face";
const q = f.quant ? ` (${f.quant})` : "";
return `I'm Q. I run the ${name}${q} entirely in your browser on WebGPU — not GPT, not OpenAI, and not on any server or cloud. `
+ `My weights streamed from ${host} and are content-addressed: every block is re-derived byte-for-byte as it loads, so nothing can be tampered with and no host has to be trusted. `
+ `Once I'm loaded, nothing you type ever leaves your device.`;
}
let engine = null, convIds = [], busy = false, armed = false, pending = null;
input.disabled = send.disabled = false; input.placeholder = "Message Q… (model loading — will send the moment it's ready)"; input.focus();
async function generate(text, skipUser) {
busy = true; input.disabled = send.disabled = true;
if (!skipUser) bubble("u", text);
// Answer identity/provenance questions from the grounded truth, not the model's confabulation.
if (IDENTITY_RX.test(text)) {
const a = bubble("a", groundedIdentity());
const stat = document.createElement("div"); stat.className = "stat"; stat.textContent = "grounded · from this instance"; a.after(stat);
busy = false; input.disabled = send.disabled = false; input.focus(); return;
}
const a = bubble("a think", "…"); let first = true;
const stat = document.createElement("div"); stat.className = "stat";
try {
let framed = engine.frameTurn(text, convIds.length > 0);
if (convIds.length === 0) framed = frameSystem() + framed;
let turnIds = engine.tokenize(framed);
if (m.bos && engine.bosId != null && convIds.length === 0) turnIds = [engine.bosId, ...turnIds];
const res = await engine.generate(convIds.concat(turnIds), { maxNew: m.cap || 256, onToken: ({ text: t, stats }) => {
if (first && t) { a.classList.remove("think"); a.textContent = ""; first = false; }
a.textContent = t; log.scrollTop = log.scrollHeight;
if (stats) stat.textContent = `${stats.tokps ? stats.tokps.toFixed(0) + " tok/s" : ""}${stats.msExec ? " · " + stats.msExec.toFixed(1) + "ms GPU/tok" : ""}${stats.ttft ? " · TTFT " + Math.round(stats.ttft) + "ms" : ""}`;
} });
if (first) { a.classList.remove("think"); a.textContent = res.text || "(no output)"; }
convIds = res.ids; a.after(stat);
} catch (e) { a.classList.remove("think"); a.textContent = "⚠ " + e.message; }
busy = false; input.disabled = send.disabled = false; input.focus();
}
async function proactiveGreeting() {
busy = true; input.disabled = send.disabled = true;
const a = bubble("a think", "…"); let first = true;
const stat = document.createElement("div"); stat.className = "stat";
const FALLBACK = "Hey — I'm Q, running entirely in your browser, no server. My weights streamed from Hugging Face and are verified by re-derivation. What can I help you with?";
try {
const P = "This is the very first thing you say to the person who just opened you. You are Q — a private AI running entirely in their browser with no server, your weights streamed from Hugging Face and verified by re-derivation. Greet them warmly in one or two sentences and invite them to ask you anything.";
let ids = engine.tokenize(engine.frameTurn(P, false));
if (m.bos && engine.bosId != null) ids = [engine.bosId, ...ids];
await engine.generate(ids, { maxNew: 64, onToken: ({ text: t, stats }) => { if (first && t) { a.classList.remove("think"); a.textContent = ""; first = false; } a.textContent = t; log.scrollTop = log.scrollHeight; if (stats && stats.tokps) stat.textContent = `${stats.tokps.toFixed(0)} tok/s`; } });
if (first || a.textContent.trim().length < 4) { a.classList.remove("think"); a.textContent = FALLBACK; } else a.after(stat);
} catch (e) { a.classList.remove("think"); a.textContent = FALLBACK; }
busy = false; input.disabled = send.disabled = false; input.focus();
}
function onSend() {
const text = input.value.trim(); if (!text || busy) return;
input.value = ""; input.style.height = "auto";
if (!armed) { pending = text; bubble("u", text); const w = bubble("a think", "…starting the model, one moment…"); w.dataset.pending = "1"; return; }
generate(text);
}
send.onclick = onSend;
input.onkeydown = (e) => { if (e.key === "Enter" && !e.shiftKey) { e.preventDefault(); onSend(); } };
input.oninput = () => { input.style.height = "auto"; input.style.height = Math.min(140, input.scrollHeight) + "px"; };
// ── SPEC-DECODE A/B HARNESS (?bench=spec) ── measures baseline greedy vs speculative on the operator's real
// GPU across echo-heavy and free-form prompts: byte-identical check (G1), mean accepted tokens/verify (G2),
// and decode tok/s for both. One load, one table — the honest verdict on whether spec-decode earns its place.
const SPEC_BENCH = [
{ tag: "code / edit (echo-heavy)", text: "Here is a function:\n\nfunction add(a, b) {\n return a + b;\n}\n\nRewrite it exactly the same but rename add to sum." },
{ tag: "retrieval / quote", text: "Passage: \"The quick brown fox jumps over the lazy dog near the river bank at dawn.\" Repeat that passage back to me word for word." },
{ tag: "free-form chat", text: "In one short sentence, why is the sky blue?" },
];
const eqArr = (a, b) => a.length === b.length && a.every((x, i) => x === b[i]);
async function runSpecBench() {
log.innerHTML = ""; input.disabled = send.disabled = true;
if (!engine.specAvailable) { st.textContent = "spec-decode unavailable for this model"; bubble("a", "This model can't use the batched-verify head (specAvailable=false)."); return; }
const N = 192, rep = m.rep ?? 1.3, rows = []; // long gen so DECODE dominates (prefill was the old confound)
let prev = { windows: 0, drafted: 0, accepted: 0 };
// warm up to boost the GPU clock (cold vs boosted differs ~2.4×) — measure only when warm
st.textContent = "warming up (boosting GPU clock)…";
globalThis.__spec = false; engine.reset();
await engine.generate(engine.tokenize(engine.frameTurn("Write one sentence about the sea.", false)), { maxNew: 48, repPenalty: rep });
for (const b of SPEC_BENCH) {
st.textContent = `bench: ${b.tag}…`;
const ids = engine.tokenize(engine.frameTurn(b.text, false)); // NO long system prompt — keep prefill small
globalThis.__spec = false; engine.reset();
const t0 = performance.now(); const r0 = await engine.generate(ids.slice(), { maxNew: N, repPenalty: rep }); const w0 = performance.now() - t0;
globalThis.__spec = true; engine.reset();
const t1 = performance.now(); const r1 = await engine.generate(ids.slice(), { maxNew: N, repPenalty: rep }); const w1 = performance.now() - t1;
globalThis.__spec = false;
const cur = (r1.stats && r1.stats.spec) || prev;
const dd = { windows: cur.windows - prev.windows, drafted: cur.drafted - prev.drafted, accepted: cur.accepted - prev.accepted }; prev = { windows: cur.windows, drafted: cur.drafted, accepted: cur.accepted };
rows.push({
tag: b.tag, same: eqArr(r0.outIds, r1.outIds), nB: r0.outIds.length, nS: r1.outIds.length,
baseTok: r0.outIds.length / (w0 / 1000), specTok: r1.outIds.length / (w1 / 1000),
perVerify: dd.windows ? 1 + dd.accepted / dd.windows : 0, accept: dd.drafted ? dd.accepted / dd.drafted : 0, windows: dd.windows,
});
}
const allSame = rows.every((r) => r.same);
const fmt = (x) => x.toFixed(x < 10 ? 1 : 0);
const tbl = `<div style="font-family:ui-monospace,monospace;font-size:13px;max-width:900px;margin:0 auto;padding:8px">
<div style="font-size:18px;font-weight:700;margin-bottom:4px">Speculative decode — measured on your GPU</div>
<div style="color:var(--dim);margin-bottom:12px">BitNet-2B · n-gram draft + batched-K verify · greedy, byte-exact by construction</div>
<table style="width:100%;border-collapse:collapse">
<tr style="color:var(--dim);text-align:left"><th style="padding:6px 8px">workload</th><th style="padding:6px 8px;text-align:right">baseline</th><th style="padding:6px 8px;text-align:right">spec</th><th style="padding:6px 8px;text-align:right">speedup</th><th style="padding:6px 8px;text-align:right">tok/verify</th><th style="padding:6px 8px;text-align:right">accept</th><th style="padding:6px 8px;text-align:right">byte-exact</th></tr>
${rows.map((r) => `<tr style="border-top:1px solid var(--line)"><td style="padding:6px 8px">${r.tag}</td><td style="padding:6px 8px;text-align:right">${fmt(r.baseTok)} tok/s</td><td style="padding:6px 8px;text-align:right">${fmt(r.specTok)} tok/s</td><td style="padding:6px 8px;text-align:right;color:${r.specTok > r.baseTok * 1.05 ? "#48c26c" : r.specTok < r.baseTok * 0.95 ? "#f0616d" : "var(--dim)"}">${(r.specTok / r.baseTok).toFixed(2)}×</td><td style="padding:6px 8px;text-align:right">${r.perVerify.toFixed(2)}</td><td style="padding:6px 8px;text-align:right">${(r.accept * 100).toFixed(0)}%</td><td style="padding:6px 8px;text-align:right;color:${r.same ? "#48c26c" : "#f0616d"}">${r.same ? "✓ identical" : "✗ DIVERGED"}</td></tr>`).join("")}
</table>
<div style="margin-top:14px;font-weight:600;color:${allSame ? "#48c26c" : "#f0616d"}">${allSame ? "✓ G1 PASS — spec output is byte-identical to greedy on every prompt." : "✗ G1 FAIL — spec diverged from greedy; not shippable until fixed (see console)."}</div>
</div>`;
log.innerHTML = tbl;
st.textContent = "spec-decode bench · done";
console.log("[specbench]", rows);
}
// ── LIVE DECODE PROFILE (?bench=perf) ── warms the GPU to boost clock, then measures the REAL decode path
// (engine.generate) steady-state tok/s at constant clock — separating "is there a lever left" from the boost-clock
// noise that makes cold vs warm runs differ ~2.4×. Compares to the 220 tok/s bandwidth roofline.
async function runPerfBench() {
log.innerHTML = ""; input.disabled = send.disabled = true;
const rep = m.rep ?? 1.3;
const ids = engine.tokenize(frameSystem() + engine.frameTurn("Write a detailed paragraph about how ocean currents move heat around the planet.", false));
globalThis.__spec = false;
st.textContent = "warming up (boosting GPU clock)…";
engine.reset(); await engine.generate(ids.slice(), { maxNew: 64, repPenalty: rep }); // warmup → boost clock + warm caches
const runs = [];
for (let i = 0; i < 3; i++) {
st.textContent = `measuring run ${i + 1}/3…`;
engine.reset();
const t0 = performance.now();
const r = await engine.generate(ids.slice(), { maxNew: 128, repPenalty: rep });
const dt = performance.now() - t0;
runs.push({ n: r.outIds.length, wall: dt, e2e: r.outIds.length / (dt / 1000), steady: (r.stats && r.stats.tokps) || 0, msExec: (r.stats && r.stats.msExec) || 0 });
}
const best = runs.slice().sort((a, b) => b.steady - a.steady)[0];
const ROOF = 220, KERNEL = 158; // measured: bandwidth roofline · boosted sustained single-matmul
const pct = 100 * best.steady / ROOF, msTok = best.steady ? 1000 / best.steady : 0;
const near = best.steady >= 0.6 * KERNEL;
const tbl = `<div style="font-family:ui-monospace,monospace;font-size:13px;max-width:860px;margin:0 auto;padding:8px">
<div style="font-size:18px;font-weight:700;margin-bottom:4px">Live decode — measured at boosted clock</div>
<div style="color:var(--dim);margin-bottom:12px">BitNet-2B · real engine.generate path · warmed then timed ×3 · bandwidth roofline 220 tok/s</div>
<table style="width:100%;border-collapse:collapse">
<tr style="color:var(--dim);text-align:left"><th style="padding:6px 8px">run</th><th style="padding:6px 8px;text-align:right">steady tok/s</th><th style="padding:6px 8px;text-align:right">end-to-end tok/s</th><th style="padding:6px 8px;text-align:right">ms/token</th><th style="padding:6px 8px;text-align:right">GPU ms/tok</th></tr>
${runs.map((r, i) => `<tr style="border-top:1px solid var(--line)"><td style="padding:6px 8px">run ${i + 1}</td><td style="padding:6px 8px;text-align:right">${r.steady.toFixed(0)}</td><td style="padding:6px 8px;text-align:right">${r.e2e.toFixed(0)}</td><td style="padding:6px 8px;text-align:right">${(r.steady ? 1000 / r.steady : 0).toFixed(1)}</td><td style="padding:6px 8px;text-align:right">${r.msExec ? r.msExec.toFixed(1) : "—"}</td></tr>`).join("")}
</table>
<div style="margin-top:12px">Best steady: <b>${best.steady.toFixed(0)} tok/s</b> = <b>${pct.toFixed(0)}%</b> of the 220 bandwidth roofline (sustained single-matmul reference ≈ ${KERNEL} tok/s).</div>
<div style="margin-top:10px;font-weight:600;color:${near ? "#48c26c" : "#e0a94a"}">${near
? "✓ Live decode is near the sustained-kernel rate — little recoverable overhead. The kernel/roofline is the ceiling; further tok/s needs fewer weight-bytes (lower-bit/MoE), spec-decode on echo text, or more bandwidth (discrete GPU)."
: `⚠ Live decode (${best.steady.toFixed(0)}) sits well below the sustained kernel (~${KERNEL}) at the SAME clock — the gap is per-token CPU round-trips (fences / JS embed / detokenize) letting the GPU idle. Decode-loop saturation is the real lever, and it's what also unlocks spec-decode's ~free batched verify.`}</div>
</div>`;
log.innerHTML = tbl; st.textContent = "live decode profile · done"; console.log("[perfbench]", runs);
}
try {
if (!navigator.gpu) throw new Error("This browser has no WebGPU — open in Chrome, Edge, or a recent mobile browser.");
st.textContent = `loading ${m.name} (${m.size})…`;
// The load animation IS the proof: each weight block re-derived on YOUR GPU (Law L5), a live honest GB/s.
// Takes over the status only once blocks start verifying; before that, the loader's own messages show.
(function vTick(){ if (armed) return; const v = globalThis.__vs; if (v && v.n) { const gb = v.bytes/1073741824, gbps = v.ms > 0 ? gb/(v.ms/1000) : 0; st.textContent = `🛡 verifying on your GPU · ${v.n} blocks · ${gb.toFixed(2)} GB${gbps ? " · " + gbps.toFixed(1) + " GB/s" : ""}`; } requestAnimationFrame(vTick); })();
const loaded = await loadModel(m, { onStatus: (s) => { if (s && !(globalThis.__vs && globalThis.__vs.n)) st.textContent = `${m.name}: ${s}`; }, onProgress: (d, t, w) => { if (!(globalThis.__vs && globalThis.__vs.n)) st.textContent = `${m.name}: ${w} ${t ? Math.round(100 * d / t) : 0}%`; } });
if (!loaded || !loaded.gpu) throw new Error("model load failed");
engine = await createEngine(m, loaded);
armed = true;
st.textContent = `${m.name} · ${m.size} · in your browser · ready`;
input.placeholder = "Message Q…";
if (params.get("bench") === "spec") { globalThis.__spec = false; await runSpecBench(); }
else if (params.get("bench") === "perf") { await runPerfBench(); }
else if (pending) { const w = [...log.querySelectorAll(".a")].reverse().find((x) => x.dataset.pending); if (w) w.remove(); const p = pending; pending = null; generate(p, true); }
else await proactiveGreeting();
} catch (e) { st.textContent = "⚠ " + e.message; bubble("a", "Could not start: " + e.message); }
</script></body></html>
|