| import json, time, urllib.request, concurrent.futures, sys |
| U="http://127.0.0.1:8888/v1/chat/completions" |
| MODEL="hy3-nvfp4" |
| def chat(prompt, max_tokens=128, temp=0): |
| body=json.dumps({"model":MODEL,"messages":[{"role":"user","content":prompt}], |
| "max_tokens":max_tokens,"temperature":temp}).encode() |
| t0=time.time() |
| r=urllib.request.urlopen(urllib.request.Request(U,body,{"Content-Type":"application/json"}),timeout=600) |
| d=json.load(r); dt=time.time()-t0 |
| return d["usage"]["completion_tokens"], dt |
|
|
| chat("hello", 8) |
| P="Write a detailed, technical paragraph about the history and architecture of modern GPUs." |
| print("=== concurrency sweep (128 out tok, temp 0) ===") |
| for C in [1,4,8,16,32]: |
| t0=time.time() |
| with concurrent.futures.ThreadPoolExecutor(max_workers=C) as ex: |
| res=list(ex.map(lambda _: chat(P,128), range(C))) |
| wall=time.time()-t0 |
| tot=sum(ct for ct,_ in res) |
| perreq=sum(ct/dt for ct,dt in res)/len(res) |
| print(f" C={C:2d}: {tot:5d} tok / {wall:5.1f}s = {tot/wall:6.1f} tok/s aggregate | {perreq:5.1f} tok/s per-req") |
|
|
| print("=== prefix cache (repeat 8k-tok prompt, small output) ===") |
| long=" ".join(["The archive holds structured records and metadata."]*900)+" Reply with OK." |
| n,d1=chat(long,4); _,d2=chat(long,4) |
| print(f" ~{len(long.split())}w prompt: run1={d1:.2f}s run2={d2:.2f}s ({100*(d1-d2)/d1:.0f}% faster on cache hit)") |
|
|
| print("=== long-context prefill (~120k tok) ===") |
| big=" ".join(["Section entry with descriptive filler content here."]*13000)+" How many words roughly? one number." |
| try: |
| _,dt=chat(big,8); import re |
| print(f" big prompt latency {dt:.1f}s") |
| except Exception as e: |
| print(" big-ctx err:", str(e)[:120]) |
|
|