Spaces:
Running
Running
File size: 13,755 Bytes
c9d0ebc 4cb9990 c9d0ebc 6fbeb8c c9d0ebc 6fbeb8c c9d0ebc 65e3518 c9d0ebc 6fbeb8c c9d0ebc 75e0906 4cb9990 c9d0ebc 6fbeb8c c9d0ebc 65e3518 6fbeb8c 65e3518 c9d0ebc 4cb9990 c9d0ebc 65e3518 c9d0ebc 65e3518 c9d0ebc | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274 275 276 277 278 279 280 281 282 283 284 285 286 287 288 289 290 291 292 293 294 295 296 297 298 299 300 301 302 303 304 305 306 307 308 309 310 311 312 313 314 315 316 317 318 319 320 321 322 323 324 325 326 327 328 329 330 331 332 333 334 335 336 337 338 339 340 341 342 343 344 345 346 347 348 349 350 351 | """The PRIMO pages, rendered as ``pm-*`` markup for ``gr.HTML`` blocks.
Gradio has no card or rich-table component, so every page here is a string of
HTML styled entirely by ``primo.css`` -- the rail, the board grid, the
leaderboard tables and the tasks table. The functions are pure: they take the
data the app already fetched (``Board`` objects, the results ``DataFrame``, the
registry) and return a string, so they unit-test without Gradio or a network.
Two rules hold everywhere:
* Every string that comes from the registry or from a submission goes through
``html.escape`` -- board names, blurbs, disease names and, above all, the
user-chosen model names that become table cells and column headers.
* The private ``hf_username`` is never rendered. It is a name-ownership lock in
the results dataset, not a public credit, so no table here carries a
"submitted by" column.
Navigation is reload-based: the rail is a column of ``<a href="?board=slug">``
and ``<a href="?tab=name">`` links, which costs a page load per click but buys
shareable per-board URLs and needs no JavaScript, matching how the board cards
have always worked.
"""
import math
import numbers
from html import escape
import pandas as pd
from boards import (
AREA_GROUP,
CATEGORY_GROUP,
GROUP_NOTE,
MODALITY_GROUP,
Board,
OpenBoard,
in_group,
metric_label,
modality_label,
open_in_group,
)
from leaderboard import per_task_table, ranked_table, tasks_table, top_models
SECTIONS = (MODALITY_GROUP, AREA_GROUP, CATEGORY_GROUP)
N_TOP_MODELS = 3
BRAND = (
'<a class="pm-brand" target="_self" href="?tab=boards">'
'<img src="/gradio_api/file=assets/primo-mark.webp" alt="">'
"<b>PRIMO</b><span>benchmark</span></a>"
)
FOOT_LINKS = (
("tasks", "Tasks"),
("submit", "Submit a model"),
("contribute", "Contribute"),
("method", "Method"),
)
NAME_COLUMNS = frozenset({"Model", "Best"})
STRONG_COLUMNS = frozenset({"Task"})
METRIC_COLUMNS = frozenset({"Metric"})
RIGHT_COLUMNS = frozenset({"Rank", "Patients"})
# --------------------------------------------------------------------- rail
def rail_html(
boards: list[Board], active_slug: str | None, active_tab: str | None
) -> str:
"""The left navigation: brand, one link per board grouped by facet, foot links.
``active_slug`` highlights the board a visitor is on; ``active_tab`` highlights
a foot link. Open boards link to Contribute, mirroring the overview cards.
"""
parts = ['<div class="pm-rail-inner">', BRAND]
for group in SECTIONS:
cards, opens = in_group(boards, group), open_in_group(boards, group)
if not cards and not opens:
continue
parts.append(
f'<div class="pm-rail-group"><p class="pm-rail-label">{escape(group)}</p>'
)
for board in cards:
cls = "pm-link pm-active" if board.slug == active_slug else "pm-link"
parts.append(
f'<a class="{cls}" target="_self" href="?board={escape(board.slug)}">'
f'<span class="pm-label">{escape(board.name)}</span>'
f'<span class="pm-count">{board.n_tasks}</span></a>'
)
for board in opens:
parts.append(
'<a class="pm-link pm-link--open" target="_self" href="?tab=contribute">'
f'<span class="pm-label">{escape(board.name)}</span>'
'<span class="pm-badge pm-badge--quiet">open</span></a>'
)
parts.append("</div>")
parts.append('<div class="pm-rail-foot">')
for tab, label in FOOT_LINKS:
cls = "pm-foot-link pm-active" if tab == active_tab else "pm-foot-link"
parts.append(
f'<a class="{cls}" target="_self" href="?tab={tab}">{escape(label)}</a>'
)
parts.append("</div></div>")
return "".join(parts)
# ------------------------------------------------------------------ boards
def _leading(top) -> str:
"""The card's mini-ranking, with the same empty/baseline states as before.
An empty board says "be the first"; a board held only by our baselines says
"beat the baseline" instead, because "be the first" misleads once a PCA
already holds a score.
"""
parts = ['<p class="pm-over" style="margin-top:14px">Leading</p>']
if not top:
parts.append(
'<p class="pm-leader pm-leader--2">No ranked model yet. Be the first.</p>'
)
return "".join(parts)
for row in top:
badge = (
' <span class="pm-badge pm-badge--marine">baseline</span>'
if row.is_baseline
else ""
)
score = "n/a" if row.score is None else f"{row.score:.3f}"
parts.append(
f'<p class="pm-leader">{escape(row.name)}{badge}'
f' <span class="pm-num--dim">{score}</span></p>'
)
return "".join(parts)
def _live_card(board: Board, df: pd.DataFrame, by_id: dict[str, dict]) -> str:
top = top_models(df, by_id, board, N_TOP_MODELS)
return (
f'<a class="pm-card pm-card--live" target="_self" href="?board={escape(board.slug)}">'
f"<h3>{escape(board.name)}</h3>"
f'<p class="pm-meta">{board.n_tasks} tasks · {board.n_cohorts} cohorts<br>'
f"{board.n_patients:,} patients · {board.n_diseases} diseases</p>"
f'<p class="pm-asks">{escape(board.blurb)}</p>'
f"{_leading(top)}</a>"
)
def _open_card(board: OpenBoard) -> str:
return (
'<a class="pm-card pm-card--open" target="_self" href="?tab=contribute">'
f"<h3>{escape(board.name)}</h3>"
'<p class="pm-meta"><span class="pm-badge pm-badge--quiet">OPEN · no cohort yet</span></p>'
f'<p class="pm-asks">{escape(board.blurb)}</p>'
'<p class="pm-cta">Propose a cohort →</p></a>'
)
def render_boards(boards: list[Board], df: pd.DataFrame, by_id: dict[str, dict]) -> str:
"""The Boards overview: every board as a card, grouped by facet.
Live cards link to their board and teaser their leaders; open cards state a
gap and link to Contribute. A section counts its open slices so the page
reads as showing its own gaps, not the registry as the whole territory.
"""
if not boards:
return (
'<div class="pm-body"><p class="pm-caption">The task registry is '
"unavailable right now. Please retry in a moment.</p></div>"
)
out = [
'<div class="pm-head"><div><h1>Leaderboards</h1>',
"<p>PRIMO evaluates representations of omics samples through "
"drug-development-related tasks. Benchmarks are organized by data "
"modality, therapeutic area, or task category.</p></div></div>",
'<div class="pm-body">',
]
for group in SECTIONS:
cards, opens = in_group(boards, group), open_in_group(boards, group)
if not cards and not opens:
continue
note = escape(GROUP_NOTE.get(group, ""))
counter = f" · +{len(opens)} open" if opens else ""
out.append(
'<div class="pm-group"><div class="pm-group-head">'
f'<p class="pm-over">{escape(group)}</p>'
f'<p class="pm-note">{note}{counter}</p></div><div class="pm-grid">'
)
out += [_live_card(b, df, by_id) for b in cards]
out += [_open_card(b) for b in opens]
out.append("</div></div>")
out.append("</div>")
return "".join(out)
# ------------------------------------------------------------- html tables
def _fmt_value(value, is_score: bool) -> str | None:
"""Display text for one cell; ``None`` marks a blank (``n/a``) cell."""
try:
if pd.isna(value):
return None
except (TypeError, ValueError):
pass
if isinstance(value, str):
return value
if isinstance(value, numbers.Integral):
return str(int(value))
if isinstance(value, numbers.Real):
if not math.isfinite(float(value)):
return None
return f"{float(value):.3f}" if is_score else str(value)
return str(value)
def _score_columns(df: pd.DataFrame) -> list[str]:
"""Numeric columns to format and bold -- ``Rank`` is an index, not a score."""
return [c for c in df.select_dtypes("number").columns if c not in RIGHT_COLUMNS]
def _bold_cells(df: pd.DataFrame, score_cols: list[str], axis: int) -> set:
"""Which ``(row, col)`` cells hold the best value.
``axis=0`` bolds the best model per column (the ranked table, read down);
``axis=1`` bolds the best model per row (the per-task table, read across).
"""
bold = set()
if not score_cols:
return bold
if axis == 0:
for col in score_cols:
best = df[col].max(skipna=True)
if pd.notna(best):
for idx, value in df[col].items():
if pd.notna(value) and value == best:
bold.add((idx, col))
else:
for idx, row in df.iterrows():
present = {c: row[c] for c in score_cols if pd.notna(row[c])}
if present:
best = max(present.values())
bold.update((idx, c) for c, v in present.items() if v == best)
return bold
def _cell_class(col: str, is_score: bool, blank: bool) -> str:
if col in NAME_COLUMNS:
return "pm-name"
if col in STRONG_COLUMNS:
return "pm-strong"
if col in METRIC_COLUMNS:
return "pm-metric"
if is_score or col in RIGHT_COLUMNS:
return "pm-num pm-num--dim" if blank else "pm-num"
return ""
def _df_to_table(df: pd.DataFrame, bold_axis: int | None, empty: str) -> str:
"""Render a DataFrame as a ``pm-table``, escaping every header and cell."""
if df.empty:
return f'<p class="pm-caption">{escape(empty)}</p>'
score_cols = _score_columns(df)
bold = _bold_cells(df, score_cols, bold_axis) if bold_axis is not None else set()
head = []
for index, col in enumerate(df.columns):
align = (
' style="text-align:right"'
if col in score_cols or col in RIGHT_COLUMNS
else ""
)
head.append(
f'<th class="pm-sort" data-sort-index="{index}" tabindex="0" '
f'role="button" aria-sort="none" title="Sort by {escape(str(col))}"'
f"{align}>{escape(str(col))}</th>"
)
body = []
for idx, row in df.iterrows():
cells = []
for col in df.columns:
is_score = col in score_cols
text = _fmt_value(row[col], is_score)
cls = _cell_class(col, is_score, text is None)
weight = ' style="font-weight:700"' if (idx, col) in bold else ""
shown = "n/a" if text is None else escape(text)
cells.append(f'<td class="{cls}"{weight}>{shown}</td>')
body.append(f"<tr>{''.join(cells)}</tr>")
return (
'<div class="pm-table-wrap"><table class="pm-table" style="table-layout:auto">'
f"<thead><tr>{''.join(head)}</tr></thead>"
f"<tbody>{''.join(body)}</tbody></table></div>"
)
# ------------------------------------------------------------------- board
def _board_meta(board: Board) -> str:
metrics = " · ".join(metric_label(m) for m in board.metrics)
tail = f" · {metrics}" if metrics else ""
return (
f"{escape(board.blurb)} · {board.n_tasks} tasks · {board.n_cohorts} cohorts · "
f"{board.n_patients:,} patients · {escape(modality_label(board.modality))}{tail}"
)
def render_board(board: Board | None, df: pd.DataFrame, by_id: dict[str, dict]) -> str:
"""One board: title strip, the ranked leaderboard, then the per-task table."""
if board is None:
return (
'<div class="pm-head"><div><h1>No board available</h1>'
"<p>The task registry could not be loaded. Please retry shortly.</p>"
"</div></div>"
)
ranked = _df_to_table(
ranked_table(df, by_id, board),
bold_axis=0,
empty="No model has covered every task of this board yet. Be the first to submit.",
)
per_task = _df_to_table(
per_task_table(df, by_id, board),
bold_axis=1,
empty="No submission has scored on this board yet.",
)
return (
'<div class="pm-head"><div><p class="pm-over">Board</p>'
f"<h1>{escape(board.name)}</h1><p>{_board_meta(board)}</p></div></div>"
'<div class="pm-body">'
'<div class="pm-group"><div class="pm-group-head">'
'<p class="pm-over pm-over--marine">Ranked</p>'
f'<p class="pm-note">Only models that covered all {board.n_tasks} tasks are '
"ranked. Mean averages the family columns and mixes metrics; a tie-break, "
"not a score.</p>"
f"</div>{ranked}</div>"
'<div class="pm-group"><div class="pm-group-head">'
'<p class="pm-over pm-over--marine">Per task</p>'
'<p class="pm-note">Every submission, partial ones included. Read across a row.</p>'
f"</div>{per_task}</div>"
"</div>"
)
# ------------------------------------------------------------------- tasks
def render_tasks(by_id: dict[str, dict]) -> str:
"""The Tasks page: one scannable row per hidden target. Provenance is never shown."""
df = tasks_table(list(by_id.values()))
table = _df_to_table(
df, bold_axis=None, empty="The task registry is unavailable right now."
)
return (
'<div class="pm-head"><div><h1>Tasks</h1>'
f"<p>{len(df)} hidden clinical targets. One fixed linear probe reads each "
"one out of your embedding; the cohorts stay anonymous, the biology does "
"not.</p></div></div>"
f'<div class="pm-body">{table}</div>'
)
|