import hashlib import json import logging import math import os import re from collections import defaultdict from datetime import datetime, timezone from html import escape from pathlib import Path import spaces import gradio as gr from dotenv import load_dotenv from huggingface_hub import CommitOperationAdd, HfApi, hf_hub_download load_dotenv(Path(__file__).with_name(".env")) REPO_ID = "gamemeine/pllum-slurm-dataset" DATA_FILE = "data/evaluation_dataset.gt.json" TOLERANCE = 1e-6 ERROR_TYPES = ("wrong_answer", "parsing_error", "tool_error", "call_budget_exceeded", "context_limit", "missing_prediction", "evaluation_error") RECENT_LIMIT = 50 api = HfApi(token=os.environ.get("HF_TOKEN")) @spaces.GPU(duration=1) def _zero_gpu_startup_marker(): """Register with ZeroGPU's startup check; the app never calls this function.""" return None def reject_constant(value): raise ValueError(f"Invalid JSON value: {value}") def read_json(path): return json.loads(Path(path).read_text(encoding="utf-8"), parse_constant=reject_constant) def download(filename, revision): return hf_hub_download(REPO_ID, filename, repo_type="dataset", revision=revision, token=api.token) def load_benchmark(revision): content = Path(download(DATA_FILE, revision)).read_bytes() rows = json.loads(content.decode("utf-8"), parse_constant=reject_constant) if not isinstance(rows, list) or not all(isinstance(row, dict) for row in rows): raise ValueError("The benchmark must contain a list of question objects.") if not rows: raise ValueError("The benchmark is empty.") return rows, hashlib.sha256(content).hexdigest() def number(value): if isinstance(value, bool) or not isinstance(value, (int, float, str)): raise ValueError("Prediction must be a number or null.") if isinstance(value, str): value = value.replace(",", ".") value = float(value) if not math.isfinite(value): raise ValueError("Numbers must be finite.") return value def classify_error(error): """Classify evaluator messages""" text = str(error).lower() if "tool call budget exhausted" in text: return "call_budget_exceeded" if "exceeds max-model-len" in text: return "context_limit" if "unknown tool:" in text or " requires arguments " in text or re.match( r"^(add|subtract|multiply|divide|modulo|power|absolute_value|square_root):", text ): return "tool_error" if any(marker in text for marker in ( "expected only a final number", "final answer does not contain a number", "expected exactly one tagged", "tool call must contain", "tool name must", "tool arguments must", "expected a finite real number", "floating-point range", "expecting value", "expecting property name", "expecting ',' delimiter", "expecting ':' delimiter", "unterminated string", "extra data", "invalid control character", "invalid escape", )): return "parsing_error" return "evaluation_error" def score(samples, benchmark): if not isinstance(samples, list) or not all(isinstance(sample, dict) for sample in samples): raise ValueError("samples.json must contain a list of evaluator samples.") def index_rows(rows, label): indexed = {} for row in rows: sample_id = row.get("id") if isinstance(sample_id, bool) or not isinstance(sample_id, (str, int)) or (isinstance(sample_id, str) and not sample_id.strip()): raise ValueError(f"Each {label} row must have a nonempty string or integer id.") if sample_id in indexed: raise ValueError(f"Duplicate id in {label}: {sample_id!r}.") indexed[sample_id] = row return indexed references = index_rows(benchmark, "benchmark") submitted = index_rows(samples, "submission") if not references: raise ValueError("The benchmark is empty.") if submitted.keys() != references.keys(): missing = references.keys() - submitted.keys() unknown = submitted.keys() - references.keys() raise ValueError(f"Submission ids must match the full benchmark: {len(missing)} missing, {len(unknown)} unknown.") categories = defaultdict(lambda: {"total": 0, "correct": 0}) error_counts = dict.fromkeys(ERROR_TYPES, 0) details = [] for index, sample in enumerate(samples): reference = references[sample["id"]] answer = number(reference["answer"]) category_name = reference["category"] if not isinstance(category_name, str) or not category_name: raise ValueError("Each benchmark row must have a nonempty category.") prediction, absolute_error = None, None prediction_error = False if sample.get("prediction") is not None: try: prediction = number(sample["prediction"]) absolute_error = abs(prediction - answer) if not math.isfinite(absolute_error): absolute_error = None except (ValueError, TypeError, OverflowError): prediction_error = True if sample.get("error"): outcome = classify_error(sample["error"]) elif prediction_error: outcome = "parsing_error" elif prediction is None: outcome = "missing_prediction" elif abs(prediction - answer) <= TOLERANCE: outcome = "correct" else: outcome = "wrong_answer" correct = outcome == "correct" if not correct: error_counts[outcome] += 1 details.append({ "id": sample["id"], "index": index, "prompt": reference.get("prompt", reference.get("task")), "category": category_name, "expected_answer": answer, "prediction": prediction, "absolute_error": absolute_error, "outcome": outcome, "error": sample.get("error"), }) category = categories[category_name] category["total"] += 1 category["correct"] += int(correct) for category in categories.values(): category["accuracy"] = category["correct"] / category["total"] correct = sum(category["correct"] for category in categories.values()) return { "total": len(samples), "correct": correct, "accuracy": correct / len(samples), "matching": "id", "by_category": dict(sorted(categories.items())), "error_counts": error_counts, "sample_results": details, } def tables(results, categories): difficulty_order = {"easy": 0, "medium": 1, "challanging": 2, "hard": 3} categories = sorted(categories, key=lambda category: (difficulty_order.get(category.lower(), 4), category)) headers = ["Login", "Accuracy (%)"] + [ f"{category.capitalize()} (%)" for category in categories ] + ["Submission"] def row(result): return [result["login"], round(100 * result["accuracy"], 2)] + [ round(100 * result["by_category"][category]["accuracy"], 2) for category in categories ] + [result["submission_id"]] recent = sorted(results, key=lambda result: result["created_at"], reverse=True)[:RECENT_LIMIT] submissions = gr.update( headers=headers + ["Submitted (UTC)", "Correct / total"] + list(ERROR_TYPES), value=[ row(result) + [result["created_at"][:19].replace("T", " "), f"{result['correct']} / {result['total']}"] + [result["error_counts"][error] for error in ERROR_TYPES] for result in recent ], ) best = {} # Highest accuracy wins; the earliest submission wins an exact tie. for result in sorted(results, key=lambda result: (-result["accuracy"], result["created_at"], result["submission_id"])): best.setdefault(result["login"], result) leaderboard = gr.update( headers=["#"] + headers, value=[[rank] + row(result) for rank, result in enumerate(best.values(), 1)], ) return submissions, leaderboard def refresh_tables(): try: revision = api.dataset_info(REPO_ID).sha benchmark, digest = load_benchmark(revision) categories = sorted({row["category"] for row in benchmark}) results = [] for filename in api.list_repo_files(REPO_ID, repo_type="dataset", revision=revision): if filename.startswith("submissions/") and filename.endswith("/results.json"): result = read_json(download(filename, revision)) if result.get("dataset_sha256") == digest: # Rescore older submissions using the current ID matching rules. if "error_counts" not in result or result.get("matching") != "id": samples_file = filename.removesuffix("results.json") + "samples.json" result.update(score(read_json(download(samples_file, revision)), benchmark)) results.append(result) return tables(results, categories) except Exception as exc: logging.exception("Could not load submissions") raise gr.Error("Could not load results. Check HF_TOKEN access and the benchmark file in the dataset.") from exc def submission_card(result): errors = "".join( f'
No errors
' ) return f"""{escape(result['submission_id'])}
Results are temporarily unavailable — click Refresh.
', gr.skip(), gr.skip() CSS = Path(__file__).with_name("styles.css").read_text(encoding="utf-8") THEME = gr.themes.Base( primary_hue="teal", secondary_hue="purple", neutral_hue="slate", ).set( body_background_fill="#f5f5fa", body_background_fill_dark="#19172e", body_text_color="#2b2661", body_text_color_dark="#eeecfa", body_text_color_subdued="#69667e", body_text_color_subdued_dark="#bab5d1", block_background_fill="#ffffff", block_background_fill_dark="#24213d", panel_background_fill="#ffffff", panel_background_fill_dark="#24213d", block_border_color="#e4e2ee", block_border_color_dark="#403b5e", input_background_fill="#fafafe", input_background_fill_dark="#19172e", input_border_color="#d8d5e5", input_border_color_dark="#504969", input_border_color_focus="#17bcaf", input_border_color_focus_dark="#17bcaf", button_primary_background_fill="#17bcaf", button_primary_background_fill_dark="#17bcaf", button_primary_background_fill_hover="#38cec2", button_primary_background_fill_hover_dark="#38cec2", button_primary_text_color="#172943", button_primary_text_color_dark="#172943", button_primary_text_color_hover="#172943", button_primary_text_color_hover_dark="#172943", button_primary_border_color="#17bcaf", button_primary_border_color_dark="#17bcaf", button_secondary_background_fill="#f3f0f8", button_secondary_background_fill_dark="#352c4e", button_secondary_text_color="#59358c", button_secondary_text_color_dark="#e0cfef", table_even_background_fill="#f8f7fc", table_even_background_fill_dark="#2b2746", table_odd_background_fill="#ffffff", table_odd_background_fill_dark="#24213d", table_border_color="#e4e2ee", table_border_color_dark="#403b5e", table_row_focus="#e4f7f4", table_row_focus_dark="#234340", link_text_color="#59358c", link_text_color_dark="#70dfd5", ) with gr.Blocks(title="Evaluating LLMs on Athena: a practical introduction to Slurm") as demo: with gr.Column(elem_id="workshop"): gr.HTML("""MLinPL Tutorial · PLLuM
Run an experiment. Evaluate your model. Compare the results.