| |
| """Perplexity sweep over GGUF tiers using llama-perplexity, with optional wiki.test.raw reference corpus.""" |
| import glob,os,re,shutil,subprocess,sys,threading,time,urllib.request,zipfile |
|
|
| if os.name=="nt":os.environ.setdefault("KMP_AFFINITY","disabled") |
| os.environ.setdefault("GGML_CUDA_ENABLE_UNIFIED_MEMORY","1") |
|
|
| SCRIPT_DIR=os.path.dirname(os.path.abspath(__file__)) |
| LLAMA_CPP=os.path.join(SCRIPT_DIR,"llama-cpp") |
| WIKI_URL="https://huggingface.co/datasets/ggml-org/ci/resolve/main/wikitext-2-raw-v1.zip?download=true" |
| WIKI_PATH=os.path.join(SCRIPT_DIR,"wiki.test.raw") |
| FALLBACK_TXT=os.path.join(SCRIPT_DIR,"01_create-calibration-dataset-and-imatrix","experimental.txt") |
| CHUNKS=64 |
| CTX=2048 |
| THREADS=max(1,(os.cpu_count() or 4)-1) |
|
|
| def _detect_gpu(): |
| if sys.platform=="win32" or shutil.which("nvidia-smi"): |
| try: |
| out=subprocess.check_output( |
| ["nvidia-smi","--query-gpu=memory.total,memory.used","--format=csv,noheader,nounits"], |
| stderr=subprocess.DEVNULL,text=True,timeout=5 |
| ).strip().splitlines()[0] |
| tot,usd=[int(x.strip()) for x in out.split(",")] |
| return "cuda",tot,max(0,tot-usd) |
| except Exception:pass |
| try: |
| out=subprocess.check_output(["rocm-smi","--showmeminfo","vram","--csv"],stderr=subprocess.DEVNULL,text=True,timeout=5) |
| m=re.search(r"(\d+)\s*MiB",out) |
| if m: |
| tot=int(m.group(1)) |
| return "rocm",tot,tot |
| except Exception:pass |
| return None,0,0 |
|
|
| GPU_BACKEND,GPU_TOTAL_VRAM,GPU_FREE_VRAM=_detect_gpu() |
|
|
| def _ngl_for_model(model_path,vram_free): |
| if vram_free<=0:return 0 |
| model_mib=os.path.getsize(model_path)/1024/1024 |
| if model_mib+768<=vram_free:return 99 |
| headroom=max(0,vram_free-768) |
| return max(1,min(99,int(99*(headroom/model_mib)))) |
|
|
| def find_binary(): |
| names=("llama-perplexity.exe","llama-perplexity") |
| if os.path.isdir(LLAMA_CPP): |
| cand=[] |
| for root,dirs,files in os.walk(LLAMA_CPP): |
| dirs[:]=[d for d in dirs if d.lower() not in(".git","models","sources","vendor")] |
| for n in names: |
| if n in files:cand.append(os.path.join(root,n)) |
| if cand: |
| def score(p): |
| s=p.lower() |
| return(0 if"release"in s else 1 if"debug"not in s else 2,len(s)) |
| cand.sort(key=score) |
| return cand[0] |
| for n in names: |
| p=shutil.which(n) |
| if p:return p |
| return None |
|
|
| def fetch_wiki(): |
| if os.path.isfile(WIKI_PATH) and os.path.getsize(WIKI_PATH)>1_000_000: |
| print(f"[Corpus] Using existing {WIKI_PATH}") |
| return WIKI_PATH |
| print("[Corpus] Downloading wiki.test.raw …") |
| tmp=WIKI_PATH+".zip" |
| try: |
| req=urllib.request.Request(WIKI_URL,headers={"User-Agent":"perplexity-test/1.0"}) |
| with urllib.request.urlopen(req,timeout=300) as r,open(tmp,"wb") as f: |
| shutil.copyfileobj(r,f) |
| with zipfile.ZipFile(tmp) as z: |
| member=next((n for n in z.namelist() if n.endswith("wiki.test.raw")),None) |
| if member is None:raise ValueError("wiki.test.raw absent from ZIP") |
| with z.open(member) as src,open(WIKI_PATH,"wb") as out: |
| shutil.copyfileobj(src,out) |
| print(f"[Corpus] Saved {WIKI_PATH} ({os.path.getsize(WIKI_PATH)/1024:.0f} KiB)") |
| return WIKI_PATH |
| except Exception as e: |
| for junk in(tmp,WIKI_PATH): |
| try:os.remove(junk) |
| except OSError:pass |
| print(f"[Corpus] ✗ Download failed ({e})") |
| if os.path.isfile(FALLBACK_TXT): |
| print(f"[Corpus] Falling back to {FALLBACK_TXT} — Δ between tiers stays valid, absolute PPL differs from published wiki benchmarks.") |
| return FALLBACK_TXT |
| return None |
|
|
| def pick_models(): |
| seen=set();out=[] |
| for p in sorted(glob.glob(os.path.join(SCRIPT_DIR,"*.gguf"))): |
| if p in seen:continue |
| seen.add(p) |
| bn=os.path.basename(p).lower() |
| if any(bn.startswith(x) for x in("mmproj","imatrix")) or any(x in bn for x in("-bench","_bench",".tmp",".bak")):continue |
| out.append(p) |
| return out |
|
|
| def menu(models): |
| print("\nAvailable models:") |
| for i,p in enumerate(models,1): |
| mib=os.path.getsize(p)/1024/1024 |
| ngl=_ngl_for_model(p,GPU_FREE_VRAM) |
| tag=f"ngl={ngl}" if ngl>0 else "CPU" |
| print(f" {i:2d}. {os.path.basename(p):<48} ({mib:5.0f} MiB · {tag})") |
| print(" a. ALL · q. quit") |
| while True: |
| choice=input("Select model(s) [1/a/1,3,5/q]: ").strip().lower() |
| if choice in("q","quit",""):return None |
| if choice=="a":return models |
| try: |
| idx=[int(x) for x in re.split(r"[,\s]+",choice) if x] |
| if idx and all(1<=i<=len(models) for i in idx): |
| return [models[i-1] for i in dict.fromkeys(idx)] |
| except ValueError:pass |
| print(" Invalid selection.") |
|
|
| def parse_output(text): |
| ppl=tps=None |
| m=re.search(r"Final estimate:\s*PPL\s*=\s*([\d.]+)",text,re.I) |
| if not m:m=re.search(r"final estimate of PPL:\s*([\d.]+)",text,re.I) |
| if not m:m=re.search(r"^\s*ppl\s*=\s*([\d.]+)",text,re.M) |
| if m: |
| ppl=float(m.group(1)) |
| else: |
| chunks=re.findall(r"\[\d+\]([\d.]+)",text) |
| if chunks:ppl=float(chunks[-1]) |
| t=re.search(r"([\d.]+)\s*tokens per second",text,re.I) |
| if t:tps=float(t.group(1)) |
| else: |
| p=re.search(r"([\d.]+)\s*seconds per pass",text,re.I) |
| if p:tps=CTX/max(float(p.group(1)),0.001) |
| return ppl,tps |
|
|
| def _get_flags(binary): |
| try:help_txt=subprocess.run([binary,"-h"],capture_output=True,text=True,timeout=5).stdout |
| except Exception:help_txt="" |
| extra=[] |
| if "-fa" in help_txt or "--flash-attn" in help_txt:extra+=["-fa","on"] |
| return extra |
|
|
| def _run_single(binary,model,corpus,ngl,flags,label): |
| cmd=[binary,"-m",model,"-f",corpus,"-c",str(CTX),"-b","512","-ub","512","-t",str(THREADS),"--chunks",str(CHUNKS),"--no-warmup"] |
| if ngl>0:cmd+=["-ngl",str(ngl)] |
| cmd+=flags |
| print(f"\n── {label} ──") |
| print(f" ngl={ngl} · ctx={CTX} · batch=512 · chunks={CHUNKS} · threads={THREADS}"+(" · Flash-Attention" if "-fa" in flags else "")) |
| t0=time.perf_counter() |
| try: |
| proc=subprocess.Popen(cmd,stdout=subprocess.PIPE,stderr=subprocess.PIPE,text=True) |
| except FileNotFoundError: |
| print(" ✗ binary not found");return None |
| out_lines=[];err_lines=[] |
| current_chunk=0 |
| last_ppl="" |
| lock=threading.Lock() |
| def on_text(line): |
| nonlocal current_chunk,last_ppl |
| m=re.search(r"\[(\d+)\]([\d.]+)?",line) |
| if m: |
| current_chunk=int(m.group(1)) |
| if m.group(2):last_ppl=f" PPL={m.group(2)}" |
| with lock: |
| if current_chunk>0: |
| el=time.perf_counter()-t0 |
| speed=current_chunk/el if el>0 else 0 |
| rem=(CHUNKS-current_chunk)/speed if speed>0 else 0 |
| sys.stdout.write(f"\r [{current_chunk:2d}/{CHUNKS}] {current_chunk/CHUNKS*100:4.1f}%{last_ppl} · {el:.0f}s elapsed · ETA {rem:.0f}s ") |
| sys.stdout.flush() |
| def drain(stream,buf): |
| for line in stream: |
| buf.append(line) |
| on_text(line) |
| th_out=threading.Thread(target=drain,args=(proc.stdout,out_lines),daemon=True) |
| th_err=threading.Thread(target=drain,args=(proc.stderr,err_lines),daemon=True) |
| th_out.start();th_err.start() |
| proc.wait() |
| th_out.join(timeout=3);th_err.join(timeout=3) |
| sys.stdout.write("\r"+" "*90+"\r") |
| elapsed=time.perf_counter()-t0 |
| out="".join(out_lines)+"\n"+"".join(err_lines) |
| ppl,tps=parse_output(out) |
| if proc.returncode!=0 or ppl is None: |
| return None |
| tps=tps or (CHUNKS*CTX/elapsed if elapsed>0 else 0) |
| print(f" PPL={ppl:.4f} · {tps:.1f} tok/s · {elapsed:.1f}s") |
| return ppl,tps |
|
|
| def run_ppl(binary,model,corpus,flags,label): |
| ngl=_ngl_for_model(model,GPU_FREE_VRAM) |
| res=_run_single(binary,model,corpus,ngl,flags,label) |
| if res is not None:return res |
| if ngl>0: |
| print(" ⚠ Run failed on GPU — retrying CPU-only …") |
| res=_run_single(binary,model,corpus,0,[],label) |
| if res is not None:return res |
| print(" ✗ Run failed.") |
| return None |
|
|
| def main(): |
| binary=find_binary() |
| if not binary: |
| print("Error: llama-perplexity binary not found under llama-cpp/.");sys.exit(1) |
| print(f"Binary : {binary}") |
| if GPU_BACKEND: |
| print(f"GPU : {GPU_BACKEND} · {GPU_FREE_VRAM}/{GPU_TOTAL_VRAM} MiB VRAM available") |
| else: |
| print("GPU : none detected — CPU only") |
| corpus=fetch_wiki() |
| if not corpus: |
| print("Error: no corpus available.");sys.exit(1) |
| models=pick_models() |
| if not models: |
| print("Error: no model-*.gguf found next to this script.");sys.exit(1) |
| sel=menu(models) |
| if not sel:return |
| flags=_get_flags(binary) |
| results=[] |
| for m in sel: |
| res=run_ppl(binary,m,corpus,flags,os.path.basename(m)) |
| if res:results.append((os.path.basename(m),res[0],res[1])) |
| if len(results)<2: |
| if results:print(f"\n Final result: {results[0][0]} → PPL = {results[0][1]:.4f}") |
| return |
| base=results[0][1] |
| print("\n"+"="*64) |
| print(" PERPLEXITY SUMMARY (lower is better)") |
| print("="*64) |
| print(f" {'Model':<46}{'PPL':>9}{'+Δ':>9}{'Speed':>11}") |
| print(f" {'-'*46}{'-'*9}{'-'*9}{'-'*11}") |
| for name,ppl,tps in results: |
| print(f" {name:<46}{ppl:>9.4f}{ppl-base:>+9.4f}{tps:>8.1f} t/s") |
| print(f"\n Δ relative to {results[0][0]} · corpus: {os.path.basename(corpus)}") |
| if corpus==WIKI_PATH: |
| print(" wiki.test.raw corpus: absolute PPL comparable to llama.cpp published benchmarks.") |
|
|
| if __name__=="__main__": |
| main() |
|
|