#!/usr/bin/env python3 """Perplexity sweep over GGUF tiers using llama-perplexity, with optional wiki.test.raw reference corpus.""" import glob,os,re,shutil,subprocess,sys,threading,time,urllib.request,zipfile if os.name=="nt":os.environ.setdefault("KMP_AFFINITY","disabled") os.environ.setdefault("GGML_CUDA_ENABLE_UNIFIED_MEMORY","1") SCRIPT_DIR=os.path.dirname(os.path.abspath(__file__)) LLAMA_CPP=os.path.join(SCRIPT_DIR,"llama-cpp") WIKI_URL="https://huggingface.co/datasets/ggml-org/ci/resolve/main/wikitext-2-raw-v1.zip?download=true" WIKI_PATH=os.path.join(SCRIPT_DIR,"wiki.test.raw") FALLBACK_TXT=os.path.join(SCRIPT_DIR,"01_create-calibration-dataset-and-imatrix","experimental.txt") CHUNKS=64 CTX=2048 THREADS=max(1,(os.cpu_count() or 4)-1) def _detect_gpu(): if sys.platform=="win32" or shutil.which("nvidia-smi"): try: out=subprocess.check_output( ["nvidia-smi","--query-gpu=memory.total,memory.used","--format=csv,noheader,nounits"], stderr=subprocess.DEVNULL,text=True,timeout=5 ).strip().splitlines()[0] tot,usd=[int(x.strip()) for x in out.split(",")] return "cuda",tot,max(0,tot-usd) except Exception:pass try: out=subprocess.check_output(["rocm-smi","--showmeminfo","vram","--csv"],stderr=subprocess.DEVNULL,text=True,timeout=5) m=re.search(r"(\d+)\s*MiB",out) if m: tot=int(m.group(1)) return "rocm",tot,tot except Exception:pass return None,0,0 GPU_BACKEND,GPU_TOTAL_VRAM,GPU_FREE_VRAM=_detect_gpu() def _ngl_for_model(model_path,vram_free): if vram_free<=0:return 0 model_mib=os.path.getsize(model_path)/1024/1024 if model_mib+768<=vram_free:return 99 headroom=max(0,vram_free-768) return max(1,min(99,int(99*(headroom/model_mib)))) def find_binary(): names=("llama-perplexity.exe","llama-perplexity") if os.path.isdir(LLAMA_CPP): cand=[] for root,dirs,files in os.walk(LLAMA_CPP): dirs[:]=[d for d in dirs if d.lower() not in(".git","models","sources","vendor")] for n in names: if n in files:cand.append(os.path.join(root,n)) if cand: def score(p): s=p.lower() return(0 if"release"in s else 1 if"debug"not in s else 2,len(s)) cand.sort(key=score) return cand[0] for n in names: p=shutil.which(n) if p:return p return None def fetch_wiki(): if os.path.isfile(WIKI_PATH) and os.path.getsize(WIKI_PATH)>1_000_000: print(f"[Corpus] Using existing {WIKI_PATH}") return WIKI_PATH print("[Corpus] Downloading wiki.test.raw …") tmp=WIKI_PATH+".zip" try: req=urllib.request.Request(WIKI_URL,headers={"User-Agent":"perplexity-test/1.0"}) with urllib.request.urlopen(req,timeout=300) as r,open(tmp,"wb") as f: shutil.copyfileobj(r,f) with zipfile.ZipFile(tmp) as z: member=next((n for n in z.namelist() if n.endswith("wiki.test.raw")),None) if member is None:raise ValueError("wiki.test.raw absent from ZIP") with z.open(member) as src,open(WIKI_PATH,"wb") as out: shutil.copyfileobj(src,out) print(f"[Corpus] Saved {WIKI_PATH} ({os.path.getsize(WIKI_PATH)/1024:.0f} KiB)") return WIKI_PATH except Exception as e: for junk in(tmp,WIKI_PATH): try:os.remove(junk) except OSError:pass print(f"[Corpus] ✗ Download failed ({e})") if os.path.isfile(FALLBACK_TXT): print(f"[Corpus] Falling back to {FALLBACK_TXT} — Δ between tiers stays valid, absolute PPL differs from published wiki benchmarks.") return FALLBACK_TXT return None def pick_models(): seen=set();out=[] for p in sorted(glob.glob(os.path.join(SCRIPT_DIR,"*.gguf"))): if p in seen:continue seen.add(p) bn=os.path.basename(p).lower() if any(bn.startswith(x) for x in("mmproj","imatrix")) or any(x in bn for x in("-bench","_bench",".tmp",".bak")):continue out.append(p) return out def menu(models): print("\nAvailable models:") for i,p in enumerate(models,1): mib=os.path.getsize(p)/1024/1024 ngl=_ngl_for_model(p,GPU_FREE_VRAM) tag=f"ngl={ngl}" if ngl>0 else "CPU" print(f" {i:2d}. {os.path.basename(p):<48} ({mib:5.0f} MiB · {tag})") print(" a. ALL · q. quit") while True: choice=input("Select model(s) [1/a/1,3,5/q]: ").strip().lower() if choice in("q","quit",""):return None if choice=="a":return models try: idx=[int(x) for x in re.split(r"[,\s]+",choice) if x] if idx and all(1<=i<=len(models) for i in idx): return [models[i-1] for i in dict.fromkeys(idx)] except ValueError:pass print(" Invalid selection.") def parse_output(text): ppl=tps=None m=re.search(r"Final estimate:\s*PPL\s*=\s*([\d.]+)",text,re.I) if not m:m=re.search(r"final estimate of PPL:\s*([\d.]+)",text,re.I) if not m:m=re.search(r"^\s*ppl\s*=\s*([\d.]+)",text,re.M) if m: ppl=float(m.group(1)) else: chunks=re.findall(r"\[\d+\]([\d.]+)",text) if chunks:ppl=float(chunks[-1]) t=re.search(r"([\d.]+)\s*tokens per second",text,re.I) if t:tps=float(t.group(1)) else: p=re.search(r"([\d.]+)\s*seconds per pass",text,re.I) if p:tps=CTX/max(float(p.group(1)),0.001) return ppl,tps def _get_flags(binary): try:help_txt=subprocess.run([binary,"-h"],capture_output=True,text=True,timeout=5).stdout except Exception:help_txt="" extra=[] if "-fa" in help_txt or "--flash-attn" in help_txt:extra+=["-fa","on"] return extra def _run_single(binary,model,corpus,ngl,flags,label): cmd=[binary,"-m",model,"-f",corpus,"-c",str(CTX),"-b","512","-ub","512","-t",str(THREADS),"--chunks",str(CHUNKS),"--no-warmup"] if ngl>0:cmd+=["-ngl",str(ngl)] cmd+=flags print(f"\n── {label} ──") print(f" ngl={ngl} · ctx={CTX} · batch=512 · chunks={CHUNKS} · threads={THREADS}"+(" · Flash-Attention" if "-fa" in flags else "")) t0=time.perf_counter() try: proc=subprocess.Popen(cmd,stdout=subprocess.PIPE,stderr=subprocess.PIPE,text=True) except FileNotFoundError: print(" ✗ binary not found");return None out_lines=[];err_lines=[] current_chunk=0 last_ppl="" lock=threading.Lock() def on_text(line): nonlocal current_chunk,last_ppl m=re.search(r"\[(\d+)\]([\d.]+)?",line) if m: current_chunk=int(m.group(1)) if m.group(2):last_ppl=f" PPL={m.group(2)}" with lock: if current_chunk>0: el=time.perf_counter()-t0 speed=current_chunk/el if el>0 else 0 rem=(CHUNKS-current_chunk)/speed if speed>0 else 0 sys.stdout.write(f"\r [{current_chunk:2d}/{CHUNKS}] {current_chunk/CHUNKS*100:4.1f}%{last_ppl} · {el:.0f}s elapsed · ETA {rem:.0f}s ") sys.stdout.flush() def drain(stream,buf): for line in stream: buf.append(line) on_text(line) th_out=threading.Thread(target=drain,args=(proc.stdout,out_lines),daemon=True) th_err=threading.Thread(target=drain,args=(proc.stderr,err_lines),daemon=True) th_out.start();th_err.start() proc.wait() th_out.join(timeout=3);th_err.join(timeout=3) sys.stdout.write("\r"+" "*90+"\r") elapsed=time.perf_counter()-t0 out="".join(out_lines)+"\n"+"".join(err_lines) ppl,tps=parse_output(out) if proc.returncode!=0 or ppl is None: return None tps=tps or (CHUNKS*CTX/elapsed if elapsed>0 else 0) print(f" PPL={ppl:.4f} · {tps:.1f} tok/s · {elapsed:.1f}s") return ppl,tps def run_ppl(binary,model,corpus,flags,label): ngl=_ngl_for_model(model,GPU_FREE_VRAM) res=_run_single(binary,model,corpus,ngl,flags,label) if res is not None:return res if ngl>0: print(" ⚠ Run failed on GPU — retrying CPU-only …") res=_run_single(binary,model,corpus,0,[],label) if res is not None:return res print(" ✗ Run failed.") return None def main(): binary=find_binary() if not binary: print("Error: llama-perplexity binary not found under llama-cpp/.");sys.exit(1) print(f"Binary : {binary}") if GPU_BACKEND: print(f"GPU : {GPU_BACKEND} · {GPU_FREE_VRAM}/{GPU_TOTAL_VRAM} MiB VRAM available") else: print("GPU : none detected — CPU only") corpus=fetch_wiki() if not corpus: print("Error: no corpus available.");sys.exit(1) models=pick_models() if not models: print("Error: no model-*.gguf found next to this script.");sys.exit(1) sel=menu(models) if not sel:return flags=_get_flags(binary) results=[] for m in sel: res=run_ppl(binary,m,corpus,flags,os.path.basename(m)) if res:results.append((os.path.basename(m),res[0],res[1])) if len(results)<2: if results:print(f"\n Final result: {results[0][0]} → PPL = {results[0][1]:.4f}") return base=results[0][1] print("\n"+"="*64) print(" PERPLEXITY SUMMARY (lower is better)") print("="*64) print(f" {'Model':<46}{'PPL':>9}{'+Δ':>9}{'Speed':>11}") print(f" {'-'*46}{'-'*9}{'-'*9}{'-'*11}") for name,ppl,tps in results: print(f" {name:<46}{ppl:>9.4f}{ppl-base:>+9.4f}{tps:>8.1f} t/s") print(f"\n Δ relative to {results[0][0]} · corpus: {os.path.basename(corpus)}") if corpus==WIKI_PATH: print(" wiki.test.raw corpus: absolute PPL comparable to llama.cpp published benchmarks.") if __name__=="__main__": main()