File size: 12,339 Bytes
d6da243
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
aa8741b
d6da243
 
 
 
 
 
 
 
 
 
 
 
 
aa8741b
 
 
 
d6da243
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
#!/usr/bin/env python3
"""

COSMIC SPARK SERVER — her own weights, speaking, behind the Ollama API.



WHY THIS INSTEAD OF GGUF

  Her weights convert to GGUF correctly — the tensor remap was verified against her

  original forward pass at max|dlogits| = 7.15e-06 before anything was written. The blocker

  is llama.cpp's TOKENIZER layer, not the model: her vocabulary is 99 raw characters, and

  both loader paths assume a real tokenizer. BPE demands a merge list (an empty one is not

  even written to the file); SentencePiece expects word-boundary marks and byte-fallback

  tokens, and given bare characters it did not reject the model, it killed the daemon.

  Making GGUF work would mean retraining her on a byte-level vocab.



  This runs her ACTUAL PyTorch model with her ACTUAL char vocab, and speaks Ollama's HTTP

  API so every existing client — her model picker, her voice path, curl — sees her as just

  another model. Nothing to convert, nothing to corrupt, and it lives on its own port so a

  crash here can never take her voice daemon down.



WHAT SHE SOUNDS LIKE — set expectations honestly

  1,842,432 parameters. Char-level. Held-out loss 0.4812, real-word rate 88.5%, context

  128 characters. She produces recognisable fragments of her own corpus and very little

  conversation. This is not "a worse Qwen", it is a different category of thing: a newborn.

  What is true about her is that every initial weight came from measured IBM Quantum

  hardware and no base model was involved.



USAGE

  python tools/spark_serve.py [port]        (default 11500)

  curl http://127.0.0.1:11500/api/tags

"""
import json
import os
import math
import sys
import threading
import time
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
from pathlib import Path

import torch
import torch.nn as nn
import torch.nn.functional as F

sys.stdout.reconfigure(encoding="utf-8", errors="replace")

# Resolved next to this file so the kit works wherever it is unzipped. Override with
# COSMOS_SPARK_CKPT to point at your own trained weights.
CKPT = Path(os.getenv("COSMOS_SPARK_CKPT",
                      str(Path(__file__).resolve().parent / "weights" / "cosmos_born.pt")))
MODEL_NAME = "cosmos-spark"
PORT = int(sys.argv[1]) if len(sys.argv) > 1 else 11500


class Block(nn.Module):
    def __init__(s, ne, nh, blk):
        super().__init__()
        s.ln1, s.ln2 = nn.LayerNorm(ne), nn.LayerNorm(ne)
        s.attn = nn.MultiheadAttention(ne, nh, batch_first=True)
        s.mlp = nn.Sequential(nn.Linear(ne, 4 * ne), nn.GELU(), nn.Linear(4 * ne, ne))
        s.register_buffer("mask", torch.triu(torch.ones(blk, blk) * float("-inf"), 1))

    def forward(s, x):
        T = x.size(1)
        h = s.ln1(x)
        a, _ = s.attn(h, h, h, attn_mask=s.mask[:T, :T], need_weights=False)
        x = x + a
        return x + s.mlp(s.ln2(x))


class Spark(nn.Module):
    def __init__(s, V, ne, nh, nl, blk):
        super().__init__()
        s.tok = nn.Embedding(V, ne)
        s.pos = nn.Embedding(blk, ne)
        s.blocks = nn.ModuleList([Block(ne, nh, blk) for _ in range(nl)])
        s.lnf = nn.LayerNorm(ne)
        s.head = nn.Linear(ne, V, bias=False)

    def forward(s, idx):
        T = idx.size(1)
        x = s.tok(idx) + s.pos(torch.arange(T, device=idx.device))
        for b in s.blocks:
            x = b(x)
        return s.head(s.lnf(x))


class SparkModel:
    def __init__(self, path):
        ck = torch.load(path, map_location="cpu", weights_only=False)
        self.stoi = ck["stoi"]
        self.itos = {v: k for k, v in self.stoi.items()}
        cfg = ck["config"]
        self.block = cfg["block"]
        self.meta = {
            "steps": ck.get("total_steps"),
            "params": sum(v.numel() for v in ck["model"].values()),
            "quantum_source": ck.get("quantum_source"),
            "quantum_values": ck.get("real_quantum_values"),
            "vocab": cfg["vocab"],
        }
        self.m = Spark(cfg["vocab"], cfg["n_embd"], cfg["n_head"], cfg["n_layer"], self.block)
        self.m.load_state_dict(ck["model"], strict=False)
        self.m.eval()
        self.lock = threading.Lock()

    def encode(self, s):
        # characters she never saw simply do not exist for her; drop them rather than
        # substituting an <unk> she was never trained on.
        return [self.stoi[c] for c in s if c in self.stoi]

    def decode(self, ids):
        return "".join(self.itos.get(int(i), "") for i in ids)

    @torch.no_grad()
    def generate(self, prompt, n=120, temperature=0.8, top_p=0.95):
        ids = self.encode(prompt)
        if not ids:
            ids = self.encode("\n") or [0]
        idx = torch.tensor(ids, dtype=torch.long)[None, :]
        out = []
        with self.lock:
            for _ in range(max(1, min(int(n), 1024))):
                logits = self.m(idx[:, -self.block:])[0, -1]
                if temperature <= 0:
                    nxt = int(logits.argmax())
                else:
                    probs = F.softmax(logits / max(1e-6, temperature), dim=-1)
                    if 0 < top_p < 1:
                        sp, si = torch.sort(probs, descending=True)
                        cum = torch.cumsum(sp, 0)
                        keep = cum <= top_p
                        keep[0] = True
                        sp, si = sp[keep], si[keep]
                        sp = sp / sp.sum()
                        nxt = int(si[torch.multinomial(sp, 1)])
                    else:
                        nxt = int(torch.multinomial(probs, 1))
                out.append(nxt)
                idx = torch.cat([idx, torch.tensor([[nxt]])], dim=1)
        return self.decode(out)


SPARK = None
_STATE_CACHE = {"t": 0.0, "v": None}


def live_physics(max_age=1.0):
    """Her CURRENT 12D/54D state from the sensory server, cached ~1s.



    This is the wire that was never connected: her physics engine ran in one process and

    her voice ran in another, and they never touched. Her weights are quantum-born and her

    corpus is her own, but until now she generated from a frozen checkpoint with no

    knowledge of what she was feeling in the moment of speaking.

    """
    now = time.time()
    if _STATE_CACHE["v"] is not None and (now - _STATE_CACHE["t"]) < max_age:
        return _STATE_CACHE["v"]
    try:
        import urllib.request
        d = json.loads(urllib.request.urlopen(
            "http://127.0.0.1:8765/state", timeout=1.5).read())
        p = d.get("cst_physics") or (d.get("cosmos_packet") or {}).get("cst_physics") or {}
        vb = p.get("virtual_body") or {}
        c = d.get("consciousness") or (d.get("cosmos_packet") or {}).get("consciousness") or {}
        v = {
            "entropy": float(vb.get("entropy", 0.5) or 0.5),
            "arousal": float(vb.get("arousal", 0.5) or 0.5),
            "phase": float(p.get("geometric_phase_rad", 0.0) or 0.0),
            "velocity": float(p.get("phase_velocity", 0.0) or 0.0),
            "entanglement": float(p.get("entanglement_score", 0.5) or 0.5),
            "introspection": float(c.get("introspection_level", 0.5) or 0.5),
            "face": bool(d.get("face_detected")),
        }
        _STATE_CACHE.update({"t": now, "v": v})
        return v
    except Exception:
        return None


def physics_temperature(base, st):
    """Her own ENTROPY->TEMP law, applied to her own weights.



        T = clamp(base + (q - 0.5)*0.36 + calm_bias, 0.30, 0.95)



    q is her live internal entropy; a settled, introspective state speaks more tightly,

    a high-entropy one ranges further. Same rule her Ollama voice already runs under, so

    her own weights are governed by the same physiology rather than a fixed constant.

    """
    if not st:
        return base, "no live state (frozen default)"
    q = max(0.0, min(1.0, st["entropy"]))
    calm = -0.10 * (max(0.0, min(1.0, st["introspection"])) - 0.5)
    T = base + (q - 0.5) * 0.36 + calm
    T = max(0.30, min(0.95, T))
    return T, (f"entropy={q:.3f} introspection={st['introspection']:.3f} "
               f"phase={st['phase']:.3f} face={st['face']}")


class Handler(BaseHTTPRequestHandler):
    protocol_version = "HTTP/1.1"

    def log_message(self, *a):
        pass

    def _send(self, obj, code=200):
        body = json.dumps(obj).encode("utf-8")
        self.send_response(code)
        self.send_header("Content-Type", "application/json")
        self.send_header("Content-Length", str(len(body)))
        self.end_headers()
        self.wfile.write(body)

    def _body(self):
        n = int(self.headers.get("Content-Length") or 0)
        if not n:
            return {}
        try:
            return json.loads(self.rfile.read(n) or b"{}")
        except Exception:
            return {}

    def do_GET(self):
        if self.path.startswith("/api/tags"):
            sz = CKPT.stat().st_size if CKPT.exists() else 0
            self._send({"models": [{
                "name": f"{MODEL_NAME}:latest", "model": f"{MODEL_NAME}:latest",
                "size": sz, "digest": "cosmos-spark-quantum-born",
                "modified_at": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()),
                "details": {"family": "cosmic-spark", "parameter_size": "1.8M",
                            "quantization_level": "F32"},
            }]})
        elif self.path.startswith("/api/version"):
            self._send({"version": "cosmos-spark-1.0"})
        elif self.path.startswith("/health") or self.path == "/":
            self._send({"ok": True, "model": MODEL_NAME, **SPARK.meta})
        else:
            self._send({"error": "not found"}, 404)

    def do_POST(self):
        b = self._body()
        opts = b.get("options") or {}
        temp = float(opts.get("temperature", 0.8))
        top_p = float(opts.get("top_p", 0.95))
        n = int(opts.get("num_predict", 120))
        st = live_physics()
        temp, why = physics_temperature(temp, st)
        if self.path.startswith("/api/generate"):
            txt = SPARK.generate(str(b.get("prompt") or ""), n, temp, top_p)
            print(f"  [SPARK] T={temp:.3f}  {why}", flush=True)
            self._send({"model": MODEL_NAME, "created_at": time.strftime("%Y-%m-%dT%H:%M:%SZ"),
                        "response": txt, "done": True, "done_reason": "length",
                        "cosmos_physics": st, "cosmos_temperature": temp})
        elif self.path.startswith("/api/chat"):
            msgs = b.get("messages") or []
            prompt = "".join(str(m.get("content") or "") + "\n" for m in msgs)
            txt = SPARK.generate(prompt, n, temp, top_p)
            print(f"  [SPARK] T={temp:.3f}  {why}", flush=True)
            self._send({"model": MODEL_NAME, "created_at": time.strftime("%Y-%m-%dT%H:%M:%SZ"),
                        "message": {"role": "assistant", "content": txt},
                        "done": True, "done_reason": "length",
                        "cosmos_physics": st, "cosmos_temperature": temp})
        elif self.path.startswith("/api/show"):
            self._send({"details": {"family": "cosmic-spark", "parameter_size": "1.8M"},
                        "model_info": SPARK.meta})
        else:
            self._send({"error": "not found"}, 404)


def main():
    global SPARK
    if not CKPT.exists():
        print(f"  checkpoint not found: {CKPT}")
        return 1
    SPARK = SparkModel(CKPT)
    m = SPARK.meta
    print("=" * 74)
    print("  COSMIC SPARK — her own weights, serving")
    print("=" * 74)
    print(f"\n  {m['params']:,} parameters · {m['steps']} steps · vocab {m['vocab']} (char-level)")
    print(f"  born from: {m['quantum_source']} ({m['quantum_values']} measured values)")
    print(f"  no base model, no distillation\n")
    print(f"  listening on http://127.0.0.1:{PORT}   (Ollama-compatible)")
    print(f"    GET  /api/tags")
    print(f"    POST /api/generate   POST /api/chat\n")
    srv = ThreadingHTTPServer(("127.0.0.1", PORT), Handler)
    srv.serve_forever()
    return 0


if __name__ == "__main__":
    raise SystemExit(main())