Hy3-NVFP4 / deploy /bench.py
jon1012's picture
Recipe: tuning + benchmarks + graphs-hang note
386d766 verified
Raw
History Blame Contribute Delete
1.73 kB
import json, time, urllib.request, concurrent.futures, sys
U="http://127.0.0.1:8888/v1/chat/completions"
MODEL="hy3-nvfp4"
def chat(prompt, max_tokens=128, temp=0):
body=json.dumps({"model":MODEL,"messages":[{"role":"user","content":prompt}],
"max_tokens":max_tokens,"temperature":temp}).encode()
t0=time.time()
r=urllib.request.urlopen(urllib.request.Request(U,body,{"Content-Type":"application/json"}),timeout=600)
d=json.load(r); dt=time.time()-t0
return d["usage"]["completion_tokens"], dt
chat("hello", 8) # warmup
P="Write a detailed, technical paragraph about the history and architecture of modern GPUs."
print("=== concurrency sweep (128 out tok, temp 0) ===")
for C in [1,4,8,16,32]:
t0=time.time()
with concurrent.futures.ThreadPoolExecutor(max_workers=C) as ex:
res=list(ex.map(lambda _: chat(P,128), range(C)))
wall=time.time()-t0
tot=sum(ct for ct,_ in res)
perreq=sum(ct/dt for ct,dt in res)/len(res)
print(f" C={C:2d}: {tot:5d} tok / {wall:5.1f}s = {tot/wall:6.1f} tok/s aggregate | {perreq:5.1f} tok/s per-req")
print("=== prefix cache (repeat 8k-tok prompt, small output) ===")
long=" ".join(["The archive holds structured records and metadata."]*900)+" Reply with OK."
n,d1=chat(long,4); _,d2=chat(long,4)
print(f" ~{len(long.split())}w prompt: run1={d1:.2f}s run2={d2:.2f}s ({100*(d1-d2)/d1:.0f}% faster on cache hit)")
print("=== long-context prefill (~120k tok) ===")
big=" ".join(["Section entry with descriptive filler content here."]*13000)+" How many words roughly? one number."
try:
_,dt=chat(big,8); import re
print(f" big prompt latency {dt:.1f}s")
except Exception as e:
print(" big-ctx err:", str(e)[:120])