Spaces:
Running
Running
| """Persisted leaderboard rows: the results-CSV schema and its Hugging Face IO. | |
| Kept out of ``app.py`` so the schema has one owner and so the baseline publisher | |
| (``benchmark/public_benchmark/baselines.py``) can append rows without importing | |
| Gradio. ``huggingface_hub`` is imported lazily, so the unit tests touch no | |
| network. | |
| ``is_baseline`` marks a reference submission we produced ourselves (a random | |
| embedding, a PCA of log-CPM) rather than a model somebody sent us. Baselines are | |
| ranked in place, never pinned: the point of showing them is that a foundation | |
| model can lose to a PCA, and a row pushed to the bottom of the table would hide | |
| exactly that. | |
| ``hf_username`` records who submitted a name, because the board keeps each | |
| name's LATEST rows: without an owner, anyone could supersede another team's | |
| ranked entry by sending a one-task file under their model name. The Submit form | |
| refuses a name somebody else already claimed. It is stored in the PRIVATE | |
| results dataset and never rendered -- it is a lock on the name, not a credit. | |
| Rows written before either column existed carry no column at all, so every | |
| reader goes through ``with_baseline_flag`` and ``with_owner``: a missing or | |
| unparsable flag means "a submitted model", and a missing owner means "unclaimed", | |
| so the first submitter of a legacy name takes it. | |
| """ | |
| import io | |
| import pandas as pd | |
| from evaluator import RESULTS_REPO | |
| RESULTS_FILE = "task_results.csv" | |
| SUBMISSIONS_FILE = "submissions.csv" | |
| MODEL_NAME = "model_name" | |
| IS_BASELINE = "is_baseline" | |
| OWNER = "hf_username" | |
| RESULT_COLUMNS = [MODEL_NAME, "task_id", "score", "submitted_at", IS_BASELINE, OWNER] | |
| SUBMISSION_COLUMNS = [ | |
| MODEL_NAME, | |
| "submitted_at", | |
| OWNER, | |
| "email", | |
| "paper_link", | |
| "hf_model_link", | |
| "notes", | |
| ] | |
| TRUTHY = ("true", "1") | |
| BASELINE_TAG = "(baseline)" | |
| def with_baseline_flag(df: pd.DataFrame) -> pd.DataFrame: | |
| """Guarantee a boolean ``is_baseline`` column, whatever the CSV held. | |
| Read back from CSV the column can be bool, the strings ``True``/``False``, or | |
| absent on rows written before baselines existed; all of those must collapse | |
| to a real boolean before anything ranks on it. | |
| """ | |
| if IS_BASELINE not in df.columns: | |
| return df.assign(**{IS_BASELINE: False}) | |
| flags = df[IS_BASELINE].astype(str).str.strip().str.lower().isin(TRUTHY) | |
| return df.assign(**{IS_BASELINE: flags}) | |
| def with_owner(df: pd.DataFrame) -> pd.DataFrame: | |
| """Guarantee a string ``hf_username`` column; an absent one means unclaimed.""" | |
| if OWNER not in df.columns: | |
| return df.assign(**{OWNER: ""}) | |
| return df.assign(**{OWNER: df[OWNER].fillna("").astype(str).str.strip()}) | |
| def owner_of(df: pd.DataFrame, model: str) -> str: | |
| """Who claimed this submitted model name, or ``""`` if it is free. | |
| Baselines are ignored: they live in their own namespace (``is_baseline``), so | |
| publishing ``pca-50`` never stops somebody submitting a model of that name. | |
| """ | |
| if df.empty: | |
| return "" | |
| claimed = with_owner(with_baseline_flag(df)) | |
| rows = claimed[ | |
| (claimed[MODEL_NAME] == model) & ~claimed[IS_BASELINE] & (claimed[OWNER] != "") | |
| ] | |
| return "" if rows.empty else str(rows.iloc[-1][OWNER]) | |
| def display_name(model: str, is_baseline: bool) -> str: | |
| """Leaderboard label: a baseline says so, in the one column everybody reads.""" | |
| return f"{model} {BASELINE_TAG}" if is_baseline else model | |
| def read_csv(filename: str, columns: list[str], token: str | None) -> pd.DataFrame: | |
| """One CSV from the private results dataset; an absent file is an empty table.""" | |
| from huggingface_hub import hf_hub_download | |
| from huggingface_hub.utils import EntryNotFoundError, RepositoryNotFoundError | |
| try: | |
| path = hf_hub_download(RESULTS_REPO, filename, repo_type="dataset", token=token) | |
| except (RepositoryNotFoundError, EntryNotFoundError): | |
| return pd.DataFrame(columns=columns) | |
| return pd.read_csv(path) | |
| def upload_csv(filename: str, df: pd.DataFrame, token: str | None) -> None: | |
| """Overwrite one CSV in the private results dataset.""" | |
| from huggingface_hub import HfApi | |
| api = HfApi(token=token) | |
| api.create_repo(RESULTS_REPO, repo_type="dataset", private=True, exist_ok=True) | |
| buffer = io.BytesIO() | |
| df.to_csv(buffer, index=False) | |
| buffer.seek(0) | |
| api.upload_file( | |
| path_or_fileobj=buffer, | |
| path_in_repo=filename, | |
| repo_id=RESULTS_REPO, | |
| repo_type="dataset", | |
| ) | |
| def read_results(token: str | None) -> pd.DataFrame: | |
| """Every persisted task score, with the baseline flag and owner normalised.""" | |
| return with_owner(with_baseline_flag(read_csv(RESULTS_FILE, RESULT_COLUMNS, token))) | |
| def append_results(rows: list[dict], token: str | None) -> None: | |
| """Append scored rows to the leaderboard, keeping the history intact.""" | |
| if not rows: | |
| return | |
| df = pd.concat([read_results(token), pd.DataFrame(rows)], ignore_index=True) | |
| upload_csv(RESULTS_FILE, with_owner(with_baseline_flag(df)), token) | |
| def append_submission(meta: dict, token: str | None) -> None: | |
| """Persist a submitter's contact metadata, never shown on a public page.""" | |
| df = pd.concat( | |
| [read_csv(SUBMISSIONS_FILE, SUBMISSION_COLUMNS, token), pd.DataFrame([meta])], | |
| ignore_index=True, | |
| ) | |
| upload_csv(SUBMISSIONS_FILE, df, token) | |