onw / bench_cpu.py
ryugyosoft's picture
v0.8.2: predicted prefetch while decoding off by default (CPU cost on 8-core Lunar Lake); prompt prefetch stays
e32c612 verified
Raw History Blame Contribute Delete
1.79 kB
"""CPU cost of expert streaming: process CPU time per generated token and average cores busy, per prefetch setting.
usage: python bench_cpu.py MODEL_DIR GB K1,K2,... (each setting in its own process; env such as OPENBLAS_NUM_THREADS applies)"""
import json, os, subprocess, sys
CHILD = r'''
import json, sys, time, psutil
from onw.chat import ChatEngine
e = ChatEngine(sys.argv[1], "NPU", pld=False)
P = psutil.Process()
e.checkpoint = None
list(e.stream_chat([{"role": "user", "content": "こんにちは"}], 8)) # warm up
e.checkpoint = None
c0, t0 = P.cpu_times(), time.time()
st = [x for x in e.stream_chat([{"role": "user", "content": "NPUとGPUの違いを、身近なたとえを使って説明してください。"}], 160)
if isinstance(x, dict)][0]
c1, t1 = P.cpu_times(), time.time()
cpu = (c1.user - c0.user) + (c1.system - c0.system)
n = st["completion_tokens"]
print("RESULT " + json.dumps({"tok_s": round(st["decode_tok_s"], 2), "cpu_ms_per_tok": round(cpu * 1000 / n, 1),
"cores_busy": round(cpu / (t1 - t0), 2), "threads": P.num_threads()}), flush=True)
'''
def main():
d, gb, ks = sys.argv[1], sys.argv[2], sys.argv[3].split(",")
for k in ks:
env = {**os.environ, "PYTHONUTF8": "1", "ONW_PREFETCH_K": k}
if gb != "full":
env["ONW_EXPERT_GB"] = gb
p = subprocess.run([sys.executable, "-c", CHILD, d], capture_output=True, text=True, encoding="utf8", errors="replace", env=env)
line = next((l for l in p.stdout.splitlines() if l.startswith("RESULT ")), None)
tag = {k2: os.environ[k2] for k2 in ("OPENBLAS_NUM_THREADS",) if k2 in os.environ}
print(f"GB={gb} K={k:>2} {tag}: {line[7:] if line else 'FAILED ' + p.stderr[-300:]}", flush=True)
if __name__ == "__main__":
main()