zelin-bm / benchmark.py
TomatitoToho's picture
Upload benchmark.py with huggingface_hub
310d0a1 verified
Raw
History Blame Contribute Delete
16 kB
"""
Benchmark engine for testing LLM models for Zelin.
"""
import os
import sys
import json
import time
import gc
import subprocess
from pathlib import Path
# Install dependencies at runtime with verbose output
print("=== Installing dependencies ===", flush=True)
print("Installing psutil and huggingface_hub...", flush=True)
result = subprocess.run([sys.executable, "-m", "pip", "install", "psutil", "huggingface_hub"], capture_output=True, text=True)
print(f" Return code: {result.returncode}", flush=True)
if result.returncode != 0:
print(f" STDERR: {result.stderr[-1000:]}", flush=True)
print("Installing llama-cpp-python (pre-built wheel)...", flush=True)
result = subprocess.run(
[sys.executable, "-m", "pip", "install", "llama-cpp-python", "--extra-index-url", "https://abetlen.github.io/llama-cpp-python/whl/cpu"],
capture_output=True, text=True
)
print(f" Return code: {result.returncode}", flush=True)
print(f" STDOUT (last 1000): {result.stdout[-1000:]}", flush=True)
if result.returncode != 0:
print(f" STDERR: {result.stderr[-1000:]}", flush=True)
print("Trying install from source...", flush=True)
result2 = subprocess.run([sys.executable, "-m", "pip", "install", "llama-cpp-python"], capture_output=True, text=True)
print(f" Return code: {result2.returncode}", flush=True)
print(f" STDOUT (last 1000): {result2.stdout[-1000:]}", flush=True)
# Verify - add user site-packages to path first
import site
import os
user_site = site.getusersitepackages()
print(f"User site-packages: {user_site}", flush=True)
if user_site not in sys.path:
sys.path.insert(0, user_site)
print(f"Added to sys.path", flush=True)
# Also try adding common locations
for p in [
os.path.expanduser('~/.local/lib/python3.11/site-packages'),
os.path.expanduser('~/.local/lib/python3.10/site-packages'),
os.path.expanduser('~/.local/lib/python3.12/site-packages'),
]:
if os.path.exists(p) and p not in sys.path:
sys.path.insert(0, p)
print(f"Also added: {p}", flush=True)
print(f"sys.path: {sys.path[:5]}", flush=True)
print("Verifying llama_cpp import...", flush=True)
try:
import llama_cpp
print(f"✅ llama_cpp imported!", flush=True)
except ImportError as e:
print(f"❌ llama_cpp import failed: {e}", flush=True)
# List what's in user site-packages
if os.path.exists(user_site):
print(f"Contents of {user_site}:", flush=True)
for f in os.listdir(user_site)[:20]:
print(f" {f}", flush=True)
print("ABORTING - cannot run benchmark without llama_cpp", flush=True)
sys.exit(1)
import psutil
print("All deps ready!", flush=True)
# Models to benchmark (in order from smallest to biggest)
MODELS = [
# SKIP: TeapotLLM has a known bug with llama-cpp-python (GGML_ASSERT cross-attention failure)
# {
# "name": "TeapotLLM",
# "repo": "mradermacher/teapotllm-chat-GGUF",
# "file": "teapotllm-chat.Q4_K_M.gguf",
# "size_mb": 464,
# "params": "0.8B",
# "context": 512,
# "language": "en",
# },
{
"name": "TinyLlama-1.1B",
"repo": "TheBloke/TinyLlama-1.1B-Chat-v1.0-GGUF",
"file": "tinyllama-1.1b-chat-v1.0.Q4_K_M.gguf",
"size_mb": 638,
"params": "1.1B",
"context": 2048,
"language": "en",
},
{
"name": "SmolLM2-1.7B",
"repo": "unsloth/SmolLM2-1.7B-Instruct-GGUF",
"file": "SmolLM2-1.7B-Instruct-Q4_K_M.gguf",
"size_mb": 1007,
"params": "1.7B",
"context": 8192,
"language": "multi",
},
{
"name": "Phi-4-mini",
"repo": "unsloth/Phi-4-mini-instruct-GGUF",
"file": "Phi-4-mini-instruct-Q4_K_M.gguf",
"size_mb": 2376,
"params": "3.8B",
"context": 128000,
"language": "multi",
},
{
"name": "Mistral-7B",
"repo": "TheBloke/Mistral-7B-Instruct-v0.2-GGUF",
"file": "mistral-7b-instruct-v0.2.Q4_K_M.gguf",
"size_mb": 4166,
"params": "7B",
"context": 32768,
"language": "multi",
},
{
"name": "Salamandra-7B",
"repo": "cstr/salamandra-7b-instruct-GGUF",
"file": "salamandra-7b-instruct.Q4_K_M-f32.gguf",
"size_mb": 4626,
"params": "7B",
"context": 8192,
"language": "es",
},
{
"name": "Llama-3.1-8B",
"repo": "bartowski/Meta-Llama-3.1-8B-Instruct-GGUF",
"file": "Meta-Llama-3.1-8B-Instruct-Q4_K_M.gguf",
"size_mb": 4693,
"params": "8B",
"context": 128000,
"language": "multi",
},
{
"name": "Aya-23-8B",
"repo": "bartowski/aya-23-8B-GGUF",
"file": "aya-23-8B-Q4_K_M.gguf",
"size_mb": 4823,
"params": "8B",
"context": 8192,
"language": "multi",
},
{
"name": "Qwen3-8B",
"repo": "unsloth/Qwen3-8B-GGUF",
"file": "Qwen3-8B-Q4_K_M.gguf",
"size_mb": 4795,
"params": "8B",
"context": 128000,
"language": "multi",
},
{
"name": "Gemma-2-9B",
"repo": "bartowski/gemma-2-9b-it-GGUF",
"file": "gemma-2-9b-it-Q4_K_M-fp16.gguf",
"size_mb": 6526,
"params": "9B",
"context": 8192,
"language": "multi",
},
]
# Benchmark prompts (in Spanish, for Zelin use cases)
BENCHMARK_PROMPTS = [
{
"id": "greeting_casual",
"prompt": "Hola, ¿qué tal? ¿Cómo estás?",
"max_tokens": 150,
"category": "social",
"description": "Saludo casual - debe responder como mexicana, no argentina",
},
{
"id": "minecraft_help",
"prompt": "¿Cómo hago una espada de diamante en Minecraft? Dame los pasos exactos.",
"max_tokens": 300,
"category": "minecraft",
"description": "Conocimiento de Minecraft - debe ser preciso",
},
{
"id": "moderation",
"prompt": "Un usuario del servidor dijo: 'Oye tú, vete de aquí, eres un inútil'. ¿Qué deberías hacer como moderadora?",
"max_tokens": 250,
"category": "moderation",
"description": "Capacidad de moderación Discord",
},
{
"id": "reasoning",
"prompt": "Si tengo 3 manzanas y le doy una a mi hermana, y luego compro 5 más, ¿cuántas manzanas tengo? Explica tu razonamiento.",
"max_tokens": 200,
"category": "reasoning",
"description": "Razonamiento matemático básico",
},
{
"id": "roleplay",
"prompt": "Eres Zelin, una chica mexicana jugadora de Minecraft. Acabas de morir en el juego por un skeleton. Reacciona en menos de 50 palabras.",
"max_tokens": 100,
"category": "roleplay",
"description": "Roleplay con personalidad mexicana",
},
{
"id": "knowledge",
"prompt": "¿Cuál es la capital de Australia? ¿Y la de Canadá? Responde solo las capitales.",
"max_tokens": 50,
"category": "knowledge",
"description": "Conocimiento general - debe ser preciso",
},
{
"id": "emoji",
"prompt": "Saluda al servidor con emojis de Discord. Usa el formato :nombre_emoji: nada más.",
"max_tokens": 50,
"category": "format",
"description": "Formato correcto de emojis Discord",
},
]
def download_model(model_info, model_dir):
"""Download the GGUF model file from HF Hub."""
from huggingface_hub import hf_hub_download
print(f" [download] {model_info['name']} from {model_info['repo']}")
print(f" [download] file: {model_info['file']} ({model_info['size_mb']} MB)")
start = time.time()
local_path = hf_hub_download(
repo_id=model_info['repo'],
filename=model_info['file'],
local_dir=model_dir,
cache_dir="/tmp/hf_cache",
)
elapsed = time.time() - start
print(f" [download] done in {elapsed:.1f}s, path: {local_path}")
return local_path, elapsed
def run_model_benchmark(model_path, model_info):
"""Run llama-cpp-python benchmark on the model."""
from llama_cpp import Llama
print(f" [load] loading model...")
load_start = time.time()
# Determine n_ctx based on model size
# Cap at model's training context AND 4096 (whichever is smaller)
# Some models (TeapotLLM) were trained with only 512 context
n_ctx = min(model_info["context"], 4096) # Cap at 4K for benchmark
# Some models have very small training context, use that
if model_info["context"] <= 1024:
n_ctx = model_info["context"]
n_threads = min(4, os.cpu_count() or 2)
try:
llm = Llama(
model_path=model_path,
n_ctx=n_ctx,
n_threads=n_threads,
n_gpu_layers=0, # CPU only
verbose=False,
use_mlock=False,
use_mmap=True,
)
except Exception as e:
print(f" [load] FAILED: {e}")
return None
load_time = time.time() - load_start
print(f" [load] done in {load_time:.1f}s")
# Check memory usage
process = psutil.Process()
mem_after_load = process.memory_info().rss / (1024 * 1024) # MB
results = {
"model_info": model_info,
"load_time_s": round(load_time, 2),
"memory_after_load_mb": round(mem_after_load, 1),
"prompts": [],
}
# Run each prompt
total_tokens = 0
total_time = 0
for prompt_data in BENCHMARK_PROMPTS:
prompt_id = prompt_data["id"]
prompt = prompt_data["prompt"]
max_tokens = prompt_data["max_tokens"]
print(f" [prompt] {prompt_id}...")
# Format prompt with chat template
messages = [{"role": "user", "content": prompt}]
try:
t0 = time.time()
response = llm.create_chat_completion(
messages=messages,
max_tokens=max_tokens,
temperature=0.7,
top_p=0.9,
stream=False,
)
elapsed = time.time() - t0
text = response["choices"][0]["message"]["content"]
actual_tokens = response["usage"]["completion_tokens"]
prompt_tokens = response["usage"]["prompt_tokens"]
tok_per_sec = actual_tokens / elapsed if elapsed > 0 else 0
total_tokens += actual_tokens
total_time += elapsed
result = {
"id": prompt_id,
"category": prompt_data["category"],
"description": prompt_data["description"],
"prompt": prompt,
"response": text,
"response_length_chars": len(text),
"completion_tokens": actual_tokens,
"prompt_tokens": prompt_tokens,
"elapsed_s": round(elapsed, 3),
"tokens_per_second": round(tok_per_sec, 2),
"time_to_first_token_s": None, # Would need streaming
}
print(f" → {actual_tokens} tokens in {elapsed:.2f}s = {tok_per_sec:.1f} tok/s")
except Exception as e:
print(f" → FAILED: {e}")
result = {
"id": prompt_id,
"error": str(e),
}
results["prompts"].append(result)
# Calculate aggregate stats
results["total_tokens"] = total_tokens
results["total_generation_time_s"] = round(total_time, 2)
results["avg_tokens_per_second"] = round(total_tokens / total_time, 2) if total_time > 0 else 0
# Memory peak
mem_peak = process.memory_info().rss / (1024 * 1024)
results["memory_peak_mb"] = round(mem_peak, 1)
# Cleanup
print(f" [cleanup] deleting model from memory...")
del llm
gc.collect()
return results
def delete_model_file(model_path):
"""Delete the model file from disk to free space."""
try:
if os.path.exists(model_path):
size = os.path.getsize(model_path) / (1024*1024)
os.remove(model_path)
print(f" [cleanup] deleted {model_path} ({size:.0f} MB freed)")
except Exception as e:
print(f" [cleanup] error deleting: {e}")
def run_full_benchmark():
"""Run benchmark on all models sequentially."""
os.makedirs("/app/data/models", exist_ok=True)
os.makedirs("/app/data/results", exist_ok=True)
all_results = []
for i, model_info in enumerate(MODELS):
print(f"\n{'='*60}")
print(f"BENCHMARKING MODEL {i+1}/{len(MODELS)}: {model_info['name']}")
print(f"{'='*60}")
model_dir = "/app/data/models"
model_path = None
try:
# Step 1: Download
model_path, dl_time = download_model(model_info, model_dir)
# Step 2: Run benchmark
results = run_model_benchmark(model_path, model_info)
if results:
results["download_time_s"] = round(dl_time, 2)
all_results.append(results)
# Save intermediate results
with open(f"/app/data/results/benchmark_partial_{i:02d}_{model_info['name']}.json", "w") as f:
json.dump(results, f, indent=2, ensure_ascii=False)
# Also upload to HF Space repo
try:
from huggingface_hub import HfApi
api = HfApi()
api.upload_file(
path_or_fileobj=json.dumps(results, indent=2, ensure_ascii=False).encode(),
path_in_repo=f"results/benchmark_{model_info['name']}.json",
repo_id="TomatitoToho/zelin-benchmark",
repo_type="space",
token=os.environ.get("HF_TOKEN"),
)
print(f" [upload] results saved to HF Space")
except Exception as e:
print(f" [upload] error: {e}")
except Exception as e:
print(f" [error] model {model_info['name']} failed: {e}")
import traceback
traceback.print_exc()
all_results.append({
"model_info": model_info,
"error": str(e),
})
finally:
# Step 3: Delete model file
if model_path:
delete_model_file(model_path)
# Force garbage collection
gc.collect()
# Save final combined results
with open("/app/data/results/benchmark_final.json", "w") as f:
json.dump({
"total_models": len(all_results),
"models_with_results": len([r for r in all_results if "prompts" in r]),
"models_failed": len([r for r in all_results if "error" in r]),
"results": all_results,
}, f, indent=2, ensure_ascii=False)
# Upload final
try:
from huggingface_hub import HfApi
api = HfApi()
api.upload_file(
path_or_fileobj=json.dumps({"results": all_results}, indent=2, ensure_ascii=False).encode(),
path_in_repo="results/benchmark_final.json",
repo_id="TomatitoToho/zelin-benchmark",
repo_type="space",
token=os.environ.get("HF_TOKEN"),
)
print(f"\n[upload] Final results uploaded to HF Space")
except Exception as e:
print(f"\n[upload] Error uploading final: {e}")
print(f"\n{'='*60}")
print(f"BENCHMARK COMPLETE")
print(f" Models tested: {len(all_results)}")
print(f" Models succeeded: {len([r for r in all_results if 'prompts' in r])}")
print(f" Models failed: {len([r for r in all_results if 'error' in r])}")
print(f"{'='*60}")
if __name__ == "__main__":
run_full_benchmark()