"""Neural RISC-V Linux PC -- Gradio Space (private). A 512MB rv32ima machine whose datapath methodology is the project's verified neural units. Boots mainline Linux 6.8 to a real shell you can type into, with karpathy's llama2.c and two TinyStories models on the mounted ext2 disk. The audit tab replays a window of live machine state with every datapath op through the verified-exact neural units and proves bit-identity. """ import hashlib, os, time import gradio as gr from huggingface_hub import snapshot_download HERE = os.path.dirname(os.path.abspath(__file__)) os.environ.setdefault("RV32_CFAST_PATH", os.path.join(HERE, "rv32_cfast.so")) ASSETS = snapshot_download("Quazim0t0/neural-riscv", token=os.environ.get("HF_TOKEN")) from rv32_machine import RV32Machine, RAM_SIZE # noqa: E402 from rv32_core import RV32Core # noqa: E402 KERNEL = open(os.path.join(ASSETS, "Image"), "rb").read() DTB = open(os.path.join(ASSETS, "neural.dtb"), "rb").read() DISK = open(os.path.join(ASSETS, "disk.ext2"), "rb").read() TAIL = 12_000 # console chars shown STEP = 50_000_000 # instructions per pump FP32_LABEL = "fp32 (original)" INT8_LABEL = "int8 (quantized, 3.6x smaller)" def _nets(variant): """Raw {name: BitMLP} for a weight variant. fp32 loads the hub originals; int8 dequantizes the bundled models_int8/ — neither path is removed.""" import x86_units if variant.startswith("int8"): from quant_units import build_int8 return build_int8() x86_units.CACHE = os.path.join(ASSETS, "models") from x86_units import build_all return build_all() def build_units(variant): from x86_units import NeuralUnits return NeuralUnits(_nets(variant)) def tail(m): return bytes(m.console_log[-TAIL:]).decode("ascii", "replace") def boot(): m = RV32Machine(KERNEL, dtb=DTB) m.disk_write(DISK) t0 = time.time() while b"~ # " not in bytes(m.console_log[-200:]) and time.time() - t0 < 180: m.run(STEP) yield m, tail(m), "booting..." yield (m, tail(m), f"shell up: {m.instr_count/1e6:.0f}M instructions, " f"{time.time()-t0:.1f}s. Type a command below.") def send_cmd(m, cmd, deadline_s): if m is None: yield None, "", "Boot the machine first." return cmd = (cmd or "").strip() if not cmd: yield m, tail(m), "empty command" return scan = len(m.console_log) + len(cmd) + 8 m.send(cmd + "\n") t0 = time.time() while time.time() - t0 < deadline_s: m.run(STEP) yield m, tail(m), (f"running... {m.instr_count/1e6:.0f}M total instr, " f"{time.time()-t0:.0f}s") log = bytes(m.console_log[scan:]) if log.rstrip().endswith((b"# ", b"$ ", b"#", b"$")): break yield m, tail(m), f"done ({m.instr_count/1e6:.0f}M total instructions)" def run_llm(m): yield from send_cmd(m, "cd /mnt/llm && ./run stories260K.bin -z tok512.bin" " -t 0.8 -n 48 -i 'One day'", 600) def audit(m, variant=FP32_LABEL, progress=gr.Progress()): """Replay 2048 instructions of the LIVE machine state twice -- C core vs the all-neural datapath -- and compare complete world state. The neural datapath runs on the selected weight variant (fp32 or int8).""" if m is None: return "Boot the machine first." progress(0.05, desc=f"loading + verifying the 13 slice units ({variant})") units = build_units(variant) snap = m.snapshot() N = 2048 ref = RV32Machine(KERNEL, dtb=DTB) ref.restore(snap) ref.run(N) class IO: def __init__(self): self.out = bytearray() def load(self, a): if a == 0x10000005: return 0x60 if a == 0x1100bff8: return core.timerl if a == 0x1100bffc: return core.timerh return 0 def store(self, a, v): if a == 0x10000000: self.out.append(v & 0xFF) elif a == 0x11004004: core.timermatchh = v & 0xFFFFFFFF elif a == 0x11004000: core.timermatchl = v & 0xFFFFFFFF elif a == 0x11100000: return v return 0 io = IO() ram = bytearray(snap["ram"]) core = RV32Core(ram, RAM_SIZE, units=units, mmio_load=io.load, mmio_store=io.store) core.regs = list(snap["regs"]) for f, v in snap["csrs"].items(): setattr(core, f, v) last_time = snap["last_time"] t0 = time.time() done = 0 while done < N: cyc = (core.cycleh << 32) | core.cyclel el = cyc - last_time last_time += el r = core.step_block(el, 1024) if r == 1: c2 = ((core.cycleh << 32) | core.cyclel) + 1024 core.cyclel = c2 & 0xFFFFFFFF core.cycleh = (c2 >> 32) & 0xFFFFFFFF done += 1024 progress(0.1 + 0.9 * done / N, desc=f"neural replay {done}/{N} instructions") def digest(regs, csrs, ramb): h = hashlib.sha256() for r in regs: h.update(int(r).to_bytes(4, "little")) for f in snap["csrs"]: h.update(int(csrs[f]).to_bytes(4, "little")) h.update(ramb) return h.hexdigest() d_ref = digest([int(ref.ms.regs[i]) for i in range(32)], {f: int(getattr(ref.ms, f)) for f in snap["csrs"]}, ref.ram) d_neu = digest(core.regs, {f: getattr(core, f) for f in snap["csrs"]}, ram) ok = d_ref == d_neu return (f"### {'✅ BIT-IDENTICAL' if ok else '❌ DIVERGED'}\n" f"{N} instructions of the live Linux machine replayed with every " f"datapath operation through the 13 verified-exact neural units " f"(**{variant}**, {time.time()-t0:.0f}s).\n\n" f"- C core state SHA-256: `{d_ref[:32]}…`\n" f"- neural state SHA-256: `{d_neu[:32]}…`\n\n" f"Digest covers all 32 registers, 16 CSRs, and the full 512MB RAM.") def quant_demo(variant, progress=gr.Progress()): """Bonus: load the chosen weight variant and re-prove it bit-exact over the units' FULL input domains, alongside the on-disk size of each variant.""" from quant_units import verify, dir_size, INT8_DIR fp32_dir = os.path.join(ASSETS, "models") fp32_bytes = dir_size(fp32_dir) int8_bytes = dir_size(INT8_DIR) progress(0.1, desc=f"loading {variant}") nets = _nets(variant) progress(0.4, desc="exhaustively re-verifying 13 units over full domains") ok, total, fails = verify(nets) progress(1.0, desc="done") def mb(n): return f"{n/1e6:.2f} MB" factor = fp32_bytes / int8_bytes if int8_bytes else 0 loaded_bytes = int8_bytes if variant.startswith("int8") else fp32_bytes status = ("✅ all 13 units BIT-EXACT over their full input domains" if ok == total else f"❌ {total-ok} unit(s) diverged: {fails}") return ( f"### Loaded: **{variant}** — {status}\n\n" f"Re-verified live: `(net(x) > 0)` matches the truth table for every " f"input of all 13 slice units (ADC8/SBB8 alone are 131,072 cases each). " f"Accuracy here is binary, so *exact* means the int8 machine is " f"**indistinguishable** from the fp32 one — same bits, every op.\n\n" f"| variant | 13-unit weights on disk | vs fp32 | bit-exact |\n" f"|---|---|---|---|\n" f"| {FP32_LABEL} | {mb(fp32_bytes)} | 1.00x | 13/13 |\n" f"| {INT8_LABEL} | {mb(int8_bytes)} | **{factor:.2f}x smaller** | 13/13 |\n\n" f"You loaded **{mb(loaded_bytes)}** of weights. The int8 build keeps the " f"thin sign-setting layers in fp16 and quantizes only the dominant " f"256×256 hidden weight to per-channel int8 — the one scheme that holds " f"all 13 units exact (full int8 flips a *single* SBB8 logit at 0.52).\n\n" f"Pick a variant in the **Audit** tab to drive the live Linux replay " f"on these same int8 weights." ) QUANT_INTRO = """ ## Bonus: int8 datapath — same machine, 3.6× smaller weights The 13 neural slice units are normally fp32 (~3.7 MB). Because each unit is **verified EXACT** — its output bit pattern must match the truth table over its *entire* input domain — quantization is auditable as a single bit, not a fuzzy score. This tab loads either the original fp32 weights or an int8 build and re-proves bit-exactness on the spot, so you can see the size drop with **zero** loss of correctness. The fp32 originals are untouched; the int8 files live in `models_int8/` and are selectable everywhere. """ ABOUT = """ ## A Linux PC on the neural methodology This machine's CPU is **rv32ima**. The fast path is a C core (mini-rv32ima lineage); the proof path is a Python core whose every datapath operation — adds, compares, shifts, multiplies, divides — is composed from **13 neural units trained and exhaustively verified EXACT over their complete input domains** (the same architecture-neutral 8-bit slices that power the project's i386). Wider operations are composed the way silicon does it: carry-rippled adder slices, MASK8 partial products, restoring division. What you're using right now: - **mainline Linux 6.8** booted to a real interactive shell (~100M instructions) - a mounted **ext2 disk** (`/dev/mtdblock0` → `/mnt`) holding karpathy's llama2.c and two TinyStories models - `./run stories260K.bin -z tok512.bin -i 'One day'` writes a little story **inside the emulated OS** (~10s/token here) Validation: 60/60 fuzz programs bit-identical to the C core; 150M live Linux instructions with identical per-instruction pc traces; and the audit — a 95,912,960-instruction window containing the LLM's token bytes replayed bit-identically (full 512MB state SHA-256), plus the all-neural replay you can run yourself in the Audit tab. Try: `uname -a` · `df -h /mnt` · `ls /mnt/llm` · `cat /proc/cpuinfo` """ # Terminal look is applied purely via CSS on the textarea, so the Textbox keeps # the exact structure of the known-good version and can't collapse. TERMINAL_CSS = """ #term-titlebar { display: flex; align-items: center; gap: 8px; padding: 8px 12px; margin-bottom: -2px; /* sit flush on top of the console */ background: linear-gradient(#3a3a3a, #2a2a2a); border: 1px solid #2b2b2b; border-bottom: none; border-radius: 8px 8px 0 0; user-select: none; } #term-titlebar .dot { width: 12px; height: 12px; border-radius: 50%; display: inline-block; } #term-titlebar .red { background: #ff5f56; box-shadow: inset 0 0 0 1px #e0443e; } #term-titlebar .yellow { background: #ffbd2e; box-shadow: inset 0 0 0 1px #dea123; } #term-titlebar .green { background: #27c93f; box-shadow: inset 0 0 0 1px #1aab29; } #term-titlebar .title { flex: 1; text-align: center; color: #c8c8c8; font-size: 12px; font-family: ui-monospace, "SFMono-Regular", Menlo, Consolas, monospace; letter-spacing: 0.3px; margin-right: 54px; /* balance the 3 dots on the left */ } /* the console textarea itself — the only thing we restyle on the Textbox */ #terminal textarea { background: #0c0c0c !important; color: #33ff66 !important; font-family: ui-monospace, "SFMono-Regular", Menlo, Consolas, "DejaVu Sans Mono", monospace !important; font-size: 13px !important; line-height: 1.45 !important; text-shadow: 0 0 3px rgba(51,255,102,0.35); caret-color: #33ff66; border: 1px solid #2b2b2b !important; border-radius: 0 0 8px 8px !important; box-shadow: 0 10px 30px rgba(0,0,0,0.45), 0 0 22px rgba(0,255,90,0.06); padding: 12px 14px !important; min-height: 480px !important; /* keep the pane visible even before boot */ } /* hide only the caption text — NOT the