File size: 6,060 Bytes
5f8a19a
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
"""ECHO-1-ultranano: a from-scratch educational Gleam model with a live compile check.

The demo generates Gleam from a prompt and verifies the output with the real
`gleam build`, so the result is grounded in the actual compiler, not in trust.
"""

import os
import shutil
import subprocess
import tempfile

import gradio as gr
import torch
from tokenizers import Tokenizer

from config import NanoConfig
from model import NanoLM

HERE = os.path.dirname(os.path.abspath(__file__))
DEVICE = "cpu"

# ---- load model + tokenizer ----
_ck = torch.load(os.path.join(HERE, "model.pt"), map_location=DEVICE, weights_only=False)
CFG = NanoConfig(**_ck["cfg"])
MODEL = NanoLM(CFG).to(DEVICE)
MODEL.load_state_dict(_ck["model"])
MODEL.eval()
TOK = Tokenizer.from_file(os.path.join(HERE, "tok.json"))
EOT = TOK.token_to_id("<|endoftext|>")
N_PARAMS = MODEL.n_params()

# ---- one reusable gleam project for the compile check ----
WORK = os.path.join(tempfile.gettempdir(), "ultranano_gleam")
if os.path.isdir(WORK):
    shutil.rmtree(WORK)
shutil.copytree(os.path.join(HERE, "gleam_template"), WORK)


def balanced_prefix(code: str) -> str:
    """Largest prefix ending at a '}' where every bracket type is balanced."""
    cur = {"{": 0, "(": 0, "[": 0}
    pair = {"}": "{", ")": "(", "]": "["}
    end = 0
    for i, ch in enumerate(code):
        if ch in cur:
            cur[ch] += 1
        elif ch in pair:
            cur[pair[ch]] -= 1
            if ch == "}" and cur["{"] == 0 and cur["("] == 0 and cur["["] == 0:
                end = i + 1
    return code[:end] if end else code


def compile_check(code: str):
    with open(os.path.join(WORK, "src", "sol.gleam"), "w") as f:
        f.write(code)
    st = os.path.join(WORK, "test", "sol_test.gleam")
    if os.path.exists(st):
        os.remove(st)
    try:
        p = subprocess.run(["gleam", "build"], cwd=WORK, capture_output=True,
                           text=True, timeout=30)
        return p.returncode == 0, (p.stderr or p.stdout).strip()
    except Exception as e:
        return False, str(e)


@torch.no_grad()
def generate_one(prompt: str, temperature: float) -> str:
    ids = torch.tensor([TOK.encode(prompt).ids], device=DEVICE)
    out = MODEL.generate(ids, max_new=200, temperature=temperature, top_k=30, eos_id=EOT)
    return balanced_prefix(TOK.decode(out[0].tolist()))


def run(prompt, temperature, attempts):
    """Generate up to `attempts` samples, return the first that compiles."""
    attempts = int(attempts)
    last_code, last_msg = "", ""
    for i in range(attempts):
        code = generate_one(prompt, float(temperature))
        ok, msg = compile_check(code)
        if ok:
            status = (f"### Compiles\n`gleam build` succeeded on attempt {i + 1} "
                      f"of {attempts}.")
            return status, code
        last_code, last_msg = code, msg
    err = last_msg.splitlines()[0] if last_msg else "unknown error"
    status = (f"### Does not compile\nNo output compiled within {attempts} attempts. "
              f"Showing the last one. First error: `{err}`")
    return status, last_code


FRAMING = f"""
ECHO-1-ultranano is a from scratch educational language model of about
{N_PARAMS/1e6:.1f} million parameters, trained on Gleam. It is a research toy, not
a production code model. If you are looking for a model that writes real programs,
this is not it.

What it actually does: it learned the shape of Gleam and writes small self
contained snippets that compile some of the time, roughly 18 percent from an open
prompt. The interesting part is the method, not the capability. Every training
example was generated and filtered by the real Gleam compiler, so the model was
taught to write self contained code that actually builds. The output below is
checked live with `gleam build`, green only if it truly compiles.
"""

ARCH = """
Built from scratch with the modern transformer toolkit, each piece toggleable for
ablation: RMSNorm, RoPE, grouped query attention, multi head latent attention
(DeepSeek V2), SwiGLU, a sparse mixture of experts with shared experts and load
balancing (DeepSeek V3), multi token prediction (DeepSeek V3), QK normalization,
and a KV cache. This checkpoint uses latent attention plus multi token prediction.
"""

HOWTO = """
How to read the demo: pick a prompt and a temperature, then generate. The model
tries a few times and stops at the first output that compiles. Lower temperature
is more conservative and tends to compile more often. Open prompts like
`import gleam/int` let the model write a whole self contained function on its own.
"""

EXAMPLES = [
    ["import gleam/int\n\n", 0.4, 12],
    ["import gleam/list\n\n", 0.4, 12],
    ["import gleam/string\n\npub fn ", 0.3, 12],
]

with gr.Blocks(title="ECHO-1-ultranano", theme=gr.themes.Soft()) as demo:
    gr.Markdown("# ECHO-1-ultranano")
    gr.Markdown("A from scratch educational Gleam model, verified live by the real compiler.")
    with gr.Accordion("What this is, read first", open=True):
        gr.Markdown(FRAMING)
    with gr.Row():
        with gr.Column(scale=1):
            prompt = gr.Textbox(label="Prompt", value="import gleam/int\n\n", lines=3)
            temperature = gr.Slider(0.1, 1.0, value=0.4, step=0.05, label="Temperature")
            attempts = gr.Slider(1, 30, value=12, step=1,
                                 label="Attempts (stop at the first that compiles)")
            btn = gr.Button("Generate Gleam", variant="primary")
            gr.Examples(EXAMPLES, inputs=[prompt, temperature, attempts])
        with gr.Column(scale=1):
            status = gr.Markdown("Generate to see a result.")
            code = gr.Code(label="Generated Gleam", language=None)
    with gr.Accordion("How to read this demo", open=False):
        gr.Markdown(HOWTO)
    with gr.Accordion("Architecture", open=False):
        gr.Markdown(ARCH)
    btn.click(run, [prompt, temperature, attempts], [status, code])

if __name__ == "__main__":
    demo.launch(server_name="0.0.0.0", server_port=int(os.environ.get("PORT", 7860)))