botp
/

File size: 1,897 Bytes
1d2de8a
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
"""Measure 512/1024/2048-token prefill on frozen original acceptance documents."""

import argparse
import json
import time
from pathlib import Path

import mlx.core as mx

from solomon_mlx import Solomon

p = argparse.ArgumentParser()
p.add_argument("--model", default="models/quality")
p.add_argument("--output", default="evaluations/chunk-benchmark.json")
a = p.parse_args()
out = Path(a.output)
if out.exists():
    raise FileExistsError("Use a new immutable benchmark output")
model = Solomon.load(a.model)
fixtures = json.loads(Path("evaluations/cuda-acceptance/input/documents.json").read_text())
results = []
for document, parts in fixtures["documents"].items():
    for target in (1242, 2048):
        text = "".join(p["text"] for p in parts)
        while True:
            rendered = model.engine.render([{"text": text}], "X")
            end = rendered.rfind("\n\nX")
            length = len(model.engine.t.encode(rendered[:end], add_special_tokens=False)) - 1
            if length >= target:
                break
            text += fixtures["filler"]
        for chunk in (512, 1024, 2048):
            model.engine.chunk_size = chunk
            mx.reset_peak_memory()
            started = time.perf_counter()
            with model.prefill(text) as state:
                results.append(
                    {
                        "document": document,
                        "chunk": chunk,
                        "tokens": state.prefix_tokens,
                        "seconds": time.perf_counter() - started,
                        "peak_metal_bytes": mx.get_peak_memory(),
                        "cache_bytes": sum(c.nbytes for c in state._data["cache"]),
                    }
                )
            print(results[-1], flush=True)
out.write_text(
    json.dumps({"runtime": model.identity, "measurements": results, "hardware_simulation": False}, indent=2)
)