File size: 13,755 Bytes
c9d0ebc
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
4cb9990
c9d0ebc
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
6fbeb8c
 
 
c9d0ebc
 
 
 
 
 
 
 
 
 
 
6fbeb8c
c9d0ebc
 
 
 
 
 
 
 
 
 
65e3518
c9d0ebc
 
 
6fbeb8c
c9d0ebc
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
75e0906
4cb9990
 
 
c9d0ebc
 
 
 
 
 
 
6fbeb8c
c9d0ebc
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
65e3518
6fbeb8c
 
 
 
 
65e3518
 
 
 
 
c9d0ebc
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
4cb9990
c9d0ebc
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
65e3518
c9d0ebc
 
 
 
65e3518
c9d0ebc
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
"""The PRIMO pages, rendered as ``pm-*`` markup for ``gr.HTML`` blocks.

Gradio has no card or rich-table component, so every page here is a string of
HTML styled entirely by ``primo.css`` -- the rail, the board grid, the
leaderboard tables and the tasks table. The functions are pure: they take the
data the app already fetched (``Board`` objects, the results ``DataFrame``, the
registry) and return a string, so they unit-test without Gradio or a network.

Two rules hold everywhere:

* Every string that comes from the registry or from a submission goes through
  ``html.escape`` -- board names, blurbs, disease names and, above all, the
  user-chosen model names that become table cells and column headers.
* The private ``hf_username`` is never rendered. It is a name-ownership lock in
  the results dataset, not a public credit, so no table here carries a
  "submitted by" column.

Navigation is reload-based: the rail is a column of ``<a href="?board=slug">``
and ``<a href="?tab=name">`` links, which costs a page load per click but buys
shareable per-board URLs and needs no JavaScript, matching how the board cards
have always worked.
"""

import math
import numbers
from html import escape

import pandas as pd
from boards import (
    AREA_GROUP,
    CATEGORY_GROUP,
    GROUP_NOTE,
    MODALITY_GROUP,
    Board,
    OpenBoard,
    in_group,
    metric_label,
    modality_label,
    open_in_group,
)
from leaderboard import per_task_table, ranked_table, tasks_table, top_models

SECTIONS = (MODALITY_GROUP, AREA_GROUP, CATEGORY_GROUP)
N_TOP_MODELS = 3

BRAND = (
    '<a class="pm-brand" target="_self" href="?tab=boards">'
    '<img src="/gradio_api/file=assets/primo-mark.webp" alt="">'
    "<b>PRIMO</b><span>benchmark</span></a>"
)

FOOT_LINKS = (
    ("tasks", "Tasks"),
    ("submit", "Submit a model"),
    ("contribute", "Contribute"),
    ("method", "Method"),
)

NAME_COLUMNS = frozenset({"Model", "Best"})
STRONG_COLUMNS = frozenset({"Task"})
METRIC_COLUMNS = frozenset({"Metric"})
RIGHT_COLUMNS = frozenset({"Rank", "Patients"})


# --------------------------------------------------------------------- rail
def rail_html(
    boards: list[Board], active_slug: str | None, active_tab: str | None
) -> str:
    """The left navigation: brand, one link per board grouped by facet, foot links.

    ``active_slug`` highlights the board a visitor is on; ``active_tab`` highlights
    a foot link. Open boards link to Contribute, mirroring the overview cards.
    """
    parts = ['<div class="pm-rail-inner">', BRAND]
    for group in SECTIONS:
        cards, opens = in_group(boards, group), open_in_group(boards, group)
        if not cards and not opens:
            continue
        parts.append(
            f'<div class="pm-rail-group"><p class="pm-rail-label">{escape(group)}</p>'
        )
        for board in cards:
            cls = "pm-link pm-active" if board.slug == active_slug else "pm-link"
            parts.append(
                f'<a class="{cls}" target="_self" href="?board={escape(board.slug)}">'
                f'<span class="pm-label">{escape(board.name)}</span>'
                f'<span class="pm-count">{board.n_tasks}</span></a>'
            )
        for board in opens:
            parts.append(
                '<a class="pm-link pm-link--open" target="_self" href="?tab=contribute">'
                f'<span class="pm-label">{escape(board.name)}</span>'
                '<span class="pm-badge pm-badge--quiet">open</span></a>'
            )
        parts.append("</div>")
    parts.append('<div class="pm-rail-foot">')
    for tab, label in FOOT_LINKS:
        cls = "pm-foot-link pm-active" if tab == active_tab else "pm-foot-link"
        parts.append(
            f'<a class="{cls}" target="_self" href="?tab={tab}">{escape(label)}</a>'
        )
    parts.append("</div></div>")
    return "".join(parts)


# ------------------------------------------------------------------ boards
def _leading(top) -> str:
    """The card's mini-ranking, with the same empty/baseline states as before.

    An empty board says "be the first"; a board held only by our baselines says
    "beat the baseline" instead, because "be the first" misleads once a PCA
    already holds a score.
    """
    parts = ['<p class="pm-over" style="margin-top:14px">Leading</p>']
    if not top:
        parts.append(
            '<p class="pm-leader pm-leader--2">No ranked model yet. Be the first.</p>'
        )
        return "".join(parts)
    for row in top:
        badge = (
            ' <span class="pm-badge pm-badge--marine">baseline</span>'
            if row.is_baseline
            else ""
        )
        score = "n/a" if row.score is None else f"{row.score:.3f}"
        parts.append(
            f'<p class="pm-leader">{escape(row.name)}{badge}'
            f' <span class="pm-num--dim">{score}</span></p>'
        )
    return "".join(parts)


def _live_card(board: Board, df: pd.DataFrame, by_id: dict[str, dict]) -> str:
    top = top_models(df, by_id, board, N_TOP_MODELS)
    return (
        f'<a class="pm-card pm-card--live" target="_self" href="?board={escape(board.slug)}">'
        f"<h3>{escape(board.name)}</h3>"
        f'<p class="pm-meta">{board.n_tasks} tasks · {board.n_cohorts} cohorts<br>'
        f"{board.n_patients:,} patients · {board.n_diseases} diseases</p>"
        f'<p class="pm-asks">{escape(board.blurb)}</p>'
        f"{_leading(top)}</a>"
    )


def _open_card(board: OpenBoard) -> str:
    return (
        '<a class="pm-card pm-card--open" target="_self" href="?tab=contribute">'
        f"<h3>{escape(board.name)}</h3>"
        '<p class="pm-meta"><span class="pm-badge pm-badge--quiet">OPEN · no cohort yet</span></p>'
        f'<p class="pm-asks">{escape(board.blurb)}</p>'
        '<p class="pm-cta">Propose a cohort →</p></a>'
    )


def render_boards(boards: list[Board], df: pd.DataFrame, by_id: dict[str, dict]) -> str:
    """The Boards overview: every board as a card, grouped by facet.

    Live cards link to their board and teaser their leaders; open cards state a
    gap and link to Contribute. A section counts its open slices so the page
    reads as showing its own gaps, not the registry as the whole territory.
    """
    if not boards:
        return (
            '<div class="pm-body"><p class="pm-caption">The task registry is '
            "unavailable right now. Please retry in a moment.</p></div>"
        )
    out = [
        '<div class="pm-head"><div><h1>Leaderboards</h1>',
        "<p>PRIMO evaluates representations of omics samples through "
        "drug-development-related tasks. Benchmarks are organized by data "
        "modality, therapeutic area, or task category.</p></div></div>",
        '<div class="pm-body">',
    ]
    for group in SECTIONS:
        cards, opens = in_group(boards, group), open_in_group(boards, group)
        if not cards and not opens:
            continue
        note = escape(GROUP_NOTE.get(group, ""))
        counter = f" · +{len(opens)} open" if opens else ""
        out.append(
            '<div class="pm-group"><div class="pm-group-head">'
            f'<p class="pm-over">{escape(group)}</p>'
            f'<p class="pm-note">{note}{counter}</p></div><div class="pm-grid">'
        )
        out += [_live_card(b, df, by_id) for b in cards]
        out += [_open_card(b) for b in opens]
        out.append("</div></div>")
    out.append("</div>")
    return "".join(out)


# ------------------------------------------------------------- html tables
def _fmt_value(value, is_score: bool) -> str | None:
    """Display text for one cell; ``None`` marks a blank (``n/a``) cell."""
    try:
        if pd.isna(value):
            return None
    except (TypeError, ValueError):
        pass
    if isinstance(value, str):
        return value
    if isinstance(value, numbers.Integral):
        return str(int(value))
    if isinstance(value, numbers.Real):
        if not math.isfinite(float(value)):
            return None
        return f"{float(value):.3f}" if is_score else str(value)
    return str(value)


def _score_columns(df: pd.DataFrame) -> list[str]:
    """Numeric columns to format and bold -- ``Rank`` is an index, not a score."""
    return [c for c in df.select_dtypes("number").columns if c not in RIGHT_COLUMNS]


def _bold_cells(df: pd.DataFrame, score_cols: list[str], axis: int) -> set:
    """Which ``(row, col)`` cells hold the best value.

    ``axis=0`` bolds the best model per column (the ranked table, read down);
    ``axis=1`` bolds the best model per row (the per-task table, read across).
    """
    bold = set()
    if not score_cols:
        return bold
    if axis == 0:
        for col in score_cols:
            best = df[col].max(skipna=True)
            if pd.notna(best):
                for idx, value in df[col].items():
                    if pd.notna(value) and value == best:
                        bold.add((idx, col))
    else:
        for idx, row in df.iterrows():
            present = {c: row[c] for c in score_cols if pd.notna(row[c])}
            if present:
                best = max(present.values())
                bold.update((idx, c) for c, v in present.items() if v == best)
    return bold


def _cell_class(col: str, is_score: bool, blank: bool) -> str:
    if col in NAME_COLUMNS:
        return "pm-name"
    if col in STRONG_COLUMNS:
        return "pm-strong"
    if col in METRIC_COLUMNS:
        return "pm-metric"
    if is_score or col in RIGHT_COLUMNS:
        return "pm-num pm-num--dim" if blank else "pm-num"
    return ""


def _df_to_table(df: pd.DataFrame, bold_axis: int | None, empty: str) -> str:
    """Render a DataFrame as a ``pm-table``, escaping every header and cell."""
    if df.empty:
        return f'<p class="pm-caption">{escape(empty)}</p>'
    score_cols = _score_columns(df)
    bold = _bold_cells(df, score_cols, bold_axis) if bold_axis is not None else set()

    head = []
    for index, col in enumerate(df.columns):
        align = (
            ' style="text-align:right"'
            if col in score_cols or col in RIGHT_COLUMNS
            else ""
        )
        head.append(
            f'<th class="pm-sort" data-sort-index="{index}" tabindex="0" '
            f'role="button" aria-sort="none" title="Sort by {escape(str(col))}"'
            f"{align}>{escape(str(col))}</th>"
        )
    body = []
    for idx, row in df.iterrows():
        cells = []
        for col in df.columns:
            is_score = col in score_cols
            text = _fmt_value(row[col], is_score)
            cls = _cell_class(col, is_score, text is None)
            weight = ' style="font-weight:700"' if (idx, col) in bold else ""
            shown = "n/a" if text is None else escape(text)
            cells.append(f'<td class="{cls}"{weight}>{shown}</td>')
        body.append(f"<tr>{''.join(cells)}</tr>")
    return (
        '<div class="pm-table-wrap"><table class="pm-table" style="table-layout:auto">'
        f"<thead><tr>{''.join(head)}</tr></thead>"
        f"<tbody>{''.join(body)}</tbody></table></div>"
    )


# ------------------------------------------------------------------- board
def _board_meta(board: Board) -> str:
    metrics = " · ".join(metric_label(m) for m in board.metrics)
    tail = f" · {metrics}" if metrics else ""
    return (
        f"{escape(board.blurb)} · {board.n_tasks} tasks · {board.n_cohorts} cohorts · "
        f"{board.n_patients:,} patients · {escape(modality_label(board.modality))}{tail}"
    )


def render_board(board: Board | None, df: pd.DataFrame, by_id: dict[str, dict]) -> str:
    """One board: title strip, the ranked leaderboard, then the per-task table."""
    if board is None:
        return (
            '<div class="pm-head"><div><h1>No board available</h1>'
            "<p>The task registry could not be loaded. Please retry shortly.</p>"
            "</div></div>"
        )
    ranked = _df_to_table(
        ranked_table(df, by_id, board),
        bold_axis=0,
        empty="No model has covered every task of this board yet. Be the first to submit.",
    )
    per_task = _df_to_table(
        per_task_table(df, by_id, board),
        bold_axis=1,
        empty="No submission has scored on this board yet.",
    )
    return (
        '<div class="pm-head"><div><p class="pm-over">Board</p>'
        f"<h1>{escape(board.name)}</h1><p>{_board_meta(board)}</p></div></div>"
        '<div class="pm-body">'
        '<div class="pm-group"><div class="pm-group-head">'
        '<p class="pm-over pm-over--marine">Ranked</p>'
        f'<p class="pm-note">Only models that covered all {board.n_tasks} tasks are '
        "ranked. Mean averages the family columns and mixes metrics; a tie-break, "
        "not a score.</p>"
        f"</div>{ranked}</div>"
        '<div class="pm-group"><div class="pm-group-head">'
        '<p class="pm-over pm-over--marine">Per task</p>'
        '<p class="pm-note">Every submission, partial ones included. Read across a row.</p>'
        f"</div>{per_task}</div>"
        "</div>"
    )


# ------------------------------------------------------------------- tasks
def render_tasks(by_id: dict[str, dict]) -> str:
    """The Tasks page: one scannable row per hidden target. Provenance is never shown."""
    df = tasks_table(list(by_id.values()))
    table = _df_to_table(
        df, bold_axis=None, empty="The task registry is unavailable right now."
    )
    return (
        '<div class="pm-head"><div><h1>Tasks</h1>'
        f"<p>{len(df)} hidden clinical targets. One fixed linear probe reads each "
        "one out of your embedding; the cohorts stay anonymous, the biology does "
        "not.</p></div></div>"
        f'<div class="pm-body">{table}</div>'
    )