primo-eval / render.py
HungryTorch's picture
fix(primo): refine leaderboard labels
75e0906 verified
Raw
History Blame Contribute Delete
13.8 kB
"""The PRIMO pages, rendered as ``pm-*`` markup for ``gr.HTML`` blocks.
Gradio has no card or rich-table component, so every page here is a string of
HTML styled entirely by ``primo.css`` -- the rail, the board grid, the
leaderboard tables and the tasks table. The functions are pure: they take the
data the app already fetched (``Board`` objects, the results ``DataFrame``, the
registry) and return a string, so they unit-test without Gradio or a network.
Two rules hold everywhere:
* Every string that comes from the registry or from a submission goes through
``html.escape`` -- board names, blurbs, disease names and, above all, the
user-chosen model names that become table cells and column headers.
* The private ``hf_username`` is never rendered. It is a name-ownership lock in
the results dataset, not a public credit, so no table here carries a
"submitted by" column.
Navigation is reload-based: the rail is a column of ``<a href="?board=slug">``
and ``<a href="?tab=name">`` links, which costs a page load per click but buys
shareable per-board URLs and needs no JavaScript, matching how the board cards
have always worked.
"""
import math
import numbers
from html import escape
import pandas as pd
from boards import (
AREA_GROUP,
CATEGORY_GROUP,
GROUP_NOTE,
MODALITY_GROUP,
Board,
OpenBoard,
in_group,
metric_label,
modality_label,
open_in_group,
)
from leaderboard import per_task_table, ranked_table, tasks_table, top_models
SECTIONS = (MODALITY_GROUP, AREA_GROUP, CATEGORY_GROUP)
N_TOP_MODELS = 3
BRAND = (
'<a class="pm-brand" target="_self" href="?tab=boards">'
'<img src="/gradio_api/file=assets/primo-mark.webp" alt="">'
"<b>PRIMO</b><span>benchmark</span></a>"
)
FOOT_LINKS = (
("tasks", "Tasks"),
("submit", "Submit a model"),
("contribute", "Contribute"),
("method", "Method"),
)
NAME_COLUMNS = frozenset({"Model", "Best"})
STRONG_COLUMNS = frozenset({"Task"})
METRIC_COLUMNS = frozenset({"Metric"})
RIGHT_COLUMNS = frozenset({"Rank", "Patients"})
# --------------------------------------------------------------------- rail
def rail_html(
boards: list[Board], active_slug: str | None, active_tab: str | None
) -> str:
"""The left navigation: brand, one link per board grouped by facet, foot links.
``active_slug`` highlights the board a visitor is on; ``active_tab`` highlights
a foot link. Open boards link to Contribute, mirroring the overview cards.
"""
parts = ['<div class="pm-rail-inner">', BRAND]
for group in SECTIONS:
cards, opens = in_group(boards, group), open_in_group(boards, group)
if not cards and not opens:
continue
parts.append(
f'<div class="pm-rail-group"><p class="pm-rail-label">{escape(group)}</p>'
)
for board in cards:
cls = "pm-link pm-active" if board.slug == active_slug else "pm-link"
parts.append(
f'<a class="{cls}" target="_self" href="?board={escape(board.slug)}">'
f'<span class="pm-label">{escape(board.name)}</span>'
f'<span class="pm-count">{board.n_tasks}</span></a>'
)
for board in opens:
parts.append(
'<a class="pm-link pm-link--open" target="_self" href="?tab=contribute">'
f'<span class="pm-label">{escape(board.name)}</span>'
'<span class="pm-badge pm-badge--quiet">open</span></a>'
)
parts.append("</div>")
parts.append('<div class="pm-rail-foot">')
for tab, label in FOOT_LINKS:
cls = "pm-foot-link pm-active" if tab == active_tab else "pm-foot-link"
parts.append(
f'<a class="{cls}" target="_self" href="?tab={tab}">{escape(label)}</a>'
)
parts.append("</div></div>")
return "".join(parts)
# ------------------------------------------------------------------ boards
def _leading(top) -> str:
"""The card's mini-ranking, with the same empty/baseline states as before.
An empty board says "be the first"; a board held only by our baselines says
"beat the baseline" instead, because "be the first" misleads once a PCA
already holds a score.
"""
parts = ['<p class="pm-over" style="margin-top:14px">Leading</p>']
if not top:
parts.append(
'<p class="pm-leader pm-leader--2">No ranked model yet. Be the first.</p>'
)
return "".join(parts)
for row in top:
badge = (
' <span class="pm-badge pm-badge--marine">baseline</span>'
if row.is_baseline
else ""
)
score = "n/a" if row.score is None else f"{row.score:.3f}"
parts.append(
f'<p class="pm-leader">{escape(row.name)}{badge}'
f' <span class="pm-num--dim">{score}</span></p>'
)
return "".join(parts)
def _live_card(board: Board, df: pd.DataFrame, by_id: dict[str, dict]) -> str:
top = top_models(df, by_id, board, N_TOP_MODELS)
return (
f'<a class="pm-card pm-card--live" target="_self" href="?board={escape(board.slug)}">'
f"<h3>{escape(board.name)}</h3>"
f'<p class="pm-meta">{board.n_tasks} tasks · {board.n_cohorts} cohorts<br>'
f"{board.n_patients:,} patients · {board.n_diseases} diseases</p>"
f'<p class="pm-asks">{escape(board.blurb)}</p>'
f"{_leading(top)}</a>"
)
def _open_card(board: OpenBoard) -> str:
return (
'<a class="pm-card pm-card--open" target="_self" href="?tab=contribute">'
f"<h3>{escape(board.name)}</h3>"
'<p class="pm-meta"><span class="pm-badge pm-badge--quiet">OPEN · no cohort yet</span></p>'
f'<p class="pm-asks">{escape(board.blurb)}</p>'
'<p class="pm-cta">Propose a cohort →</p></a>'
)
def render_boards(boards: list[Board], df: pd.DataFrame, by_id: dict[str, dict]) -> str:
"""The Boards overview: every board as a card, grouped by facet.
Live cards link to their board and teaser their leaders; open cards state a
gap and link to Contribute. A section counts its open slices so the page
reads as showing its own gaps, not the registry as the whole territory.
"""
if not boards:
return (
'<div class="pm-body"><p class="pm-caption">The task registry is '
"unavailable right now. Please retry in a moment.</p></div>"
)
out = [
'<div class="pm-head"><div><h1>Leaderboards</h1>',
"<p>PRIMO evaluates representations of omics samples through "
"drug-development-related tasks. Benchmarks are organized by data "
"modality, therapeutic area, or task category.</p></div></div>",
'<div class="pm-body">',
]
for group in SECTIONS:
cards, opens = in_group(boards, group), open_in_group(boards, group)
if not cards and not opens:
continue
note = escape(GROUP_NOTE.get(group, ""))
counter = f" · +{len(opens)} open" if opens else ""
out.append(
'<div class="pm-group"><div class="pm-group-head">'
f'<p class="pm-over">{escape(group)}</p>'
f'<p class="pm-note">{note}{counter}</p></div><div class="pm-grid">'
)
out += [_live_card(b, df, by_id) for b in cards]
out += [_open_card(b) for b in opens]
out.append("</div></div>")
out.append("</div>")
return "".join(out)
# ------------------------------------------------------------- html tables
def _fmt_value(value, is_score: bool) -> str | None:
"""Display text for one cell; ``None`` marks a blank (``n/a``) cell."""
try:
if pd.isna(value):
return None
except (TypeError, ValueError):
pass
if isinstance(value, str):
return value
if isinstance(value, numbers.Integral):
return str(int(value))
if isinstance(value, numbers.Real):
if not math.isfinite(float(value)):
return None
return f"{float(value):.3f}" if is_score else str(value)
return str(value)
def _score_columns(df: pd.DataFrame) -> list[str]:
"""Numeric columns to format and bold -- ``Rank`` is an index, not a score."""
return [c for c in df.select_dtypes("number").columns if c not in RIGHT_COLUMNS]
def _bold_cells(df: pd.DataFrame, score_cols: list[str], axis: int) -> set:
"""Which ``(row, col)`` cells hold the best value.
``axis=0`` bolds the best model per column (the ranked table, read down);
``axis=1`` bolds the best model per row (the per-task table, read across).
"""
bold = set()
if not score_cols:
return bold
if axis == 0:
for col in score_cols:
best = df[col].max(skipna=True)
if pd.notna(best):
for idx, value in df[col].items():
if pd.notna(value) and value == best:
bold.add((idx, col))
else:
for idx, row in df.iterrows():
present = {c: row[c] for c in score_cols if pd.notna(row[c])}
if present:
best = max(present.values())
bold.update((idx, c) for c, v in present.items() if v == best)
return bold
def _cell_class(col: str, is_score: bool, blank: bool) -> str:
if col in NAME_COLUMNS:
return "pm-name"
if col in STRONG_COLUMNS:
return "pm-strong"
if col in METRIC_COLUMNS:
return "pm-metric"
if is_score or col in RIGHT_COLUMNS:
return "pm-num pm-num--dim" if blank else "pm-num"
return ""
def _df_to_table(df: pd.DataFrame, bold_axis: int | None, empty: str) -> str:
"""Render a DataFrame as a ``pm-table``, escaping every header and cell."""
if df.empty:
return f'<p class="pm-caption">{escape(empty)}</p>'
score_cols = _score_columns(df)
bold = _bold_cells(df, score_cols, bold_axis) if bold_axis is not None else set()
head = []
for index, col in enumerate(df.columns):
align = (
' style="text-align:right"'
if col in score_cols or col in RIGHT_COLUMNS
else ""
)
head.append(
f'<th class="pm-sort" data-sort-index="{index}" tabindex="0" '
f'role="button" aria-sort="none" title="Sort by {escape(str(col))}"'
f"{align}>{escape(str(col))}</th>"
)
body = []
for idx, row in df.iterrows():
cells = []
for col in df.columns:
is_score = col in score_cols
text = _fmt_value(row[col], is_score)
cls = _cell_class(col, is_score, text is None)
weight = ' style="font-weight:700"' if (idx, col) in bold else ""
shown = "n/a" if text is None else escape(text)
cells.append(f'<td class="{cls}"{weight}>{shown}</td>')
body.append(f"<tr>{''.join(cells)}</tr>")
return (
'<div class="pm-table-wrap"><table class="pm-table" style="table-layout:auto">'
f"<thead><tr>{''.join(head)}</tr></thead>"
f"<tbody>{''.join(body)}</tbody></table></div>"
)
# ------------------------------------------------------------------- board
def _board_meta(board: Board) -> str:
metrics = " · ".join(metric_label(m) for m in board.metrics)
tail = f" · {metrics}" if metrics else ""
return (
f"{escape(board.blurb)} · {board.n_tasks} tasks · {board.n_cohorts} cohorts · "
f"{board.n_patients:,} patients · {escape(modality_label(board.modality))}{tail}"
)
def render_board(board: Board | None, df: pd.DataFrame, by_id: dict[str, dict]) -> str:
"""One board: title strip, the ranked leaderboard, then the per-task table."""
if board is None:
return (
'<div class="pm-head"><div><h1>No board available</h1>'
"<p>The task registry could not be loaded. Please retry shortly.</p>"
"</div></div>"
)
ranked = _df_to_table(
ranked_table(df, by_id, board),
bold_axis=0,
empty="No model has covered every task of this board yet. Be the first to submit.",
)
per_task = _df_to_table(
per_task_table(df, by_id, board),
bold_axis=1,
empty="No submission has scored on this board yet.",
)
return (
'<div class="pm-head"><div><p class="pm-over">Board</p>'
f"<h1>{escape(board.name)}</h1><p>{_board_meta(board)}</p></div></div>"
'<div class="pm-body">'
'<div class="pm-group"><div class="pm-group-head">'
'<p class="pm-over pm-over--marine">Ranked</p>'
f'<p class="pm-note">Only models that covered all {board.n_tasks} tasks are '
"ranked. Mean averages the family columns and mixes metrics; a tie-break, "
"not a score.</p>"
f"</div>{ranked}</div>"
'<div class="pm-group"><div class="pm-group-head">'
'<p class="pm-over pm-over--marine">Per task</p>'
'<p class="pm-note">Every submission, partial ones included. Read across a row.</p>'
f"</div>{per_task}</div>"
"</div>"
)
# ------------------------------------------------------------------- tasks
def render_tasks(by_id: dict[str, dict]) -> str:
"""The Tasks page: one scannable row per hidden target. Provenance is never shown."""
df = tasks_table(list(by_id.values()))
table = _df_to_table(
df, bold_axis=None, empty="The task registry is unavailable right now."
)
return (
'<div class="pm-head"><div><h1>Tasks</h1>'
f"<p>{len(df)} hidden clinical targets. One fixed linear probe reads each "
"one out of your embedding; the cohorts stay anonymous, the biology does "
"not.</p></div></div>"
f'<div class="pm-body">{table}</div>'
)