app.py CHANGED
@@ -581,14 +581,19 @@ button.theme-toggle[data-mode="light"] .theme-icon-moon { display: block !import
581
  .prose .ranking-table td {
582
  padding: 8px 10px !important;
583
  }
 
 
 
 
584
  .ranking-table .rank,
585
  .prose .ranking-table .rank,
586
  .ranking-table th.rank {
587
  position: sticky !important;
588
  left: 0 !important;
589
- width: 2.4rem;
590
- min-width: 2.4rem;
591
- box-shadow: 6px 0 8px -6px rgba(0, 0, 0, 0.45);
 
592
  }
593
  .ranking-table .model-cell,
594
  .prose .ranking-table .model-cell {
@@ -598,6 +603,7 @@ button.theme-toggle[data-mode="light"] .theme-icon-moon { display: block !import
598
  min-width: 140px;
599
  max-width: none;
600
  background: transparent !important;
 
601
  }
602
  .ranking-table th.model-cell,
603
  .prose .ranking-table th.model-cell {
@@ -608,6 +614,7 @@ button.theme-toggle[data-mode="light"] .theme-icon-moon { display: block !import
608
  min-width: 140px;
609
  max-width: none;
610
  background: var(--pruna-bg-header) !important;
 
611
  }
612
  .ranking-table tbody tr:hover .model-cell {
613
  background: var(--pruna-table-hover) !important;
@@ -1107,11 +1114,11 @@ button.theme-toggle[data-mode="light"] .theme-icon-moon { display: block !import
1107
  max-height: min(70vh, 720px);
1108
  overflow-x: auto;
1109
  overflow-y: auto;
1110
- -webkit-overflow-scrolling: touch;
1111
- overscroll-behavior-x: contain;
1112
  }
1113
  .ranking-table,
1114
  .prose .ranking-table {
 
1115
  width: 100%;
1116
  margin: 0 !important;
1117
  overflow: visible;
@@ -1187,10 +1194,14 @@ button.theme-toggle[data-mode="light"] .theme-icon-moon { display: block !import
1187
  .prose .ranking-table .rank {
1188
  position: sticky;
1189
  left: 0;
1190
- z-index: 1;
1191
  box-sizing: border-box;
1192
- width: 3.25rem;
1193
- min-width: 3.25rem;
 
 
 
 
1194
  color: var(--pruna-lavender) !important;
1195
  font-weight: 700 !important;
1196
  text-align: center;
@@ -1207,18 +1218,21 @@ button.theme-toggle[data-mode="light"] .theme-icon-moon { display: block !import
1207
  .ranking-table .model-cell,
1208
  .prose .ranking-table .model-cell {
1209
  position: sticky;
1210
- left: 3.25rem;
1211
- z-index: 1;
 
1212
  min-width: 180px;
1213
  max-width: 260px;
1214
  background: var(--pruna-table-sticky) !important;
 
1215
  }
1216
  .ranking-table th.model-cell,
1217
  .prose .ranking-table th.model-cell {
1218
  top: 0;
1219
- left: 3.25rem;
1220
  z-index: 5;
1221
  background: var(--pruna-bg-header) !important;
 
1222
  }
1223
  .ranking-table tbody tr:hover .rank,
1224
  .ranking-table tbody tr:hover .model-cell {
@@ -1769,6 +1783,9 @@ button.theme-toggle[data-mode="light"] .theme-icon-moon { display: block !import
1769
  max-width: 100% !important;
1770
  }
1771
  .compare-cell { min-width: 0; }
 
 
 
1772
  .compare-model-label {
1773
  margin-bottom: 6px;
1774
  color: var(--pruna-lavender);
@@ -1776,15 +1793,23 @@ button.theme-toggle[data-mode="light"] .theme-icon-moon { display: block !import
1776
  font-weight: 700;
1777
  word-break: break-word;
1778
  }
1779
- .compare-cell img {
 
1780
  display: block;
1781
  width: 100%;
1782
- aspect-ratio: 1 / 1;
1783
- object-fit: cover;
1784
  border-radius: 12px;
1785
  border: 1px solid var(--pruna-border);
1786
  background: var(--pruna-bg-elevated);
1787
  }
 
 
 
 
 
 
 
 
 
1788
  .compare-empty,
1789
  .pareto-note-copy {
1790
  margin: 0;
@@ -2143,14 +2168,20 @@ def load_sample_comparison_data(folder):
2143
  return None
2144
 
2145
  prompts = {}
 
2146
  with prompts_path.open() as handle:
2147
  for line in handle:
2148
  if not line.strip():
2149
  continue
2150
  row = json.loads(line)
2151
- prompts[row["prompt_id"]] = row.get("text", "")
 
 
 
 
2152
 
2153
  images = defaultdict(dict)
 
2154
  with generations_path.open() as handle:
2155
  for line in handle:
2156
  if not line.strip():
@@ -2158,20 +2189,33 @@ def load_sample_comparison_data(folder):
2158
  row = json.loads(line)
2159
  model_id = row["model_id"]
2160
  prompt_id = row["prompt_id"]
2161
- image_url = row.get("image")
2162
- if model_id and prompt_id and image_url:
2163
- images[model_id][prompt_id] = image_url
 
 
 
 
 
 
 
 
 
2164
 
2165
  models = sorted(images)
2166
  if not models or not prompts:
2167
  return None
2168
 
2169
- return {
2170
  "prompts": prompts,
2171
  "images": {model: dict(prompt_map) for model, prompt_map in images.items()},
2172
  "models": models,
2173
  "prompt_ids": sorted(prompts),
 
2174
  }
 
 
 
2175
 
2176
 
2177
  def _as_numeric(df, columns):
@@ -2301,6 +2345,123 @@ def load_qwen_combined_dataframe(path):
2301
  return df.reset_index(drop=True)
2302
 
2303
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
2304
  df = load_oneig_dataframe(oneig_path)
2305
 
2306
  oneig_metric_columns = [
@@ -2348,6 +2509,14 @@ qwen_combined_dir = _resolve_data_path(
2348
  data_dir / "qwen_image_bench_combined",
2349
  space_root.parent / "qwen_image_bench_combined",
2350
  )
 
 
 
 
 
 
 
 
2351
  qwen_path = _resolve_data_path(
2352
  data_dir / "qwen_image_bench_model_price_and_median_generation_time_10_august.csv",
2353
  space_root.parent / "qwen_image_bench_model_price_and_median_generation_time_10_august.csv",
@@ -2360,10 +2529,20 @@ arena_path = _resolve_data_path(
2360
  data_dir / "arena_ai_text_to_image_leaderboard.csv",
2361
  space_root.parent / "arena_ai_text_to_image_leaderboard.csv",
2362
  )
 
 
 
 
 
 
 
 
2363
 
2364
  qwen_df = load_qwen_combined_dataframe(qwen_path)
2365
  aa_df = load_artificial_analysis_dataframe(aa_path)
2366
  arena_df = load_arena_ai_dataframe(arena_path)
 
 
2367
  qwen_display_columns = [
2368
  col
2369
  for col in [
@@ -2395,9 +2574,36 @@ arena_display_columns = [
2395
  ]
2396
  if col in arena_df.columns
2397
  ]
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
2398
 
2399
  oneig_samples = load_sample_comparison_data(oneig_combined_dir)
2400
  qwen_samples = load_sample_comparison_data(qwen_combined_dir)
 
 
2401
 
2402
  metrics = [
2403
  {"id": "datapoint_elo", "column": "Datapoint Elo"},
@@ -2456,11 +2662,45 @@ arena_metric_ids = _metric_ids_for(
2456
  "arena_text",
2457
  ],
2458
  )
 
 
 
 
2459
 
2460
  datasets = [
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
2461
  {
2462
  "id": "qwen",
2463
  "name": "Qwen Image Dataset",
 
2464
  "data": qwen_df,
2465
  "columns": qwen_display_columns,
2466
  "metric_ids": qwen_metric_ids,
@@ -2470,6 +2710,7 @@ datasets = [
2470
  {
2471
  "id": "oneig",
2472
  "name": "OneIG Alignment Dataset",
 
2473
  "data": oneig_df,
2474
  "columns": oneig_display_columns,
2475
  "metric_ids": oneig_metric_ids,
@@ -2482,6 +2723,7 @@ datasets = [
2482
  {
2483
  "id": "artificial_analysis",
2484
  "name": "Artificial Analysis Dataset",
 
2485
  "data": aa_df,
2486
  "columns": aa_display_columns,
2487
  "metric_ids": aa_metric_ids,
@@ -2491,6 +2733,7 @@ datasets = [
2491
  {
2492
  "id": "arena_ai",
2493
  "name": "Arena AI Dataset",
 
2494
  "data": arena_df,
2495
  "columns": arena_display_columns,
2496
  "metric_ids": arena_metric_ids,
@@ -2501,8 +2744,14 @@ datasets = [
2501
  datasets = [dataset for dataset in datasets if dataset["metric_ids"]]
2502
 
2503
  DEFAULT_DATASET_ID = next(
2504
- (dataset["id"] for dataset in datasets if dataset["id"] == "qwen"),
2505
- datasets[0]["id"] if datasets else None,
 
 
 
 
 
 
2506
  )
2507
  DEFAULT_METRIC_ID = None
2508
 
 
581
  .prose .ranking-table td {
582
  padding: 8px 10px !important;
583
  }
584
+ .ranking-table,
585
+ .prose .ranking-table {
586
+ --rank-col-width: 3.25rem;
587
+ }
588
  .ranking-table .rank,
589
  .prose .ranking-table .rank,
590
  .ranking-table th.rank {
591
  position: sticky !important;
592
  left: 0 !important;
593
+ width: var(--rank-col-width);
594
+ min-width: var(--rank-col-width);
595
+ max-width: var(--rank-col-width);
596
+ box-shadow: none;
597
  }
598
  .ranking-table .model-cell,
599
  .prose .ranking-table .model-cell {
 
603
  min-width: 140px;
604
  max-width: none;
605
  background: transparent !important;
606
+ box-shadow: none !important;
607
  }
608
  .ranking-table th.model-cell,
609
  .prose .ranking-table th.model-cell {
 
614
  min-width: 140px;
615
  max-width: none;
616
  background: var(--pruna-bg-header) !important;
617
+ box-shadow: 0 1px 0 var(--pruna-hairline) !important;
618
  }
619
  .ranking-table tbody tr:hover .model-cell {
620
  background: var(--pruna-table-hover) !important;
 
1114
  max-height: min(70vh, 720px);
1115
  overflow-x: auto;
1116
  overflow-y: auto;
1117
+ overscroll-behavior: none;
 
1118
  }
1119
  .ranking-table,
1120
  .prose .ranking-table {
1121
+ --rank-col-width: 4.25rem;
1122
  width: 100%;
1123
  margin: 0 !important;
1124
  overflow: visible;
 
1194
  .prose .ranking-table .rank {
1195
  position: sticky;
1196
  left: 0;
1197
+ z-index: 2;
1198
  box-sizing: border-box;
1199
+ width: var(--rank-col-width);
1200
+ min-width: var(--rank-col-width);
1201
+ max-width: var(--rank-col-width);
1202
+ padding-left: 0.5rem !important;
1203
+ padding-right: 0.5rem !important;
1204
+ overflow: hidden;
1205
  color: var(--pruna-lavender) !important;
1206
  font-weight: 700 !important;
1207
  text-align: center;
 
1218
  .ranking-table .model-cell,
1219
  .prose .ranking-table .model-cell {
1220
  position: sticky;
1221
+ left: var(--rank-col-width);
1222
+ z-index: 2;
1223
+ box-sizing: border-box;
1224
  min-width: 180px;
1225
  max-width: 260px;
1226
  background: var(--pruna-table-sticky) !important;
1227
+ box-shadow: 8px 0 10px -8px rgba(0, 0, 0, 0.35) !important;
1228
  }
1229
  .ranking-table th.model-cell,
1230
  .prose .ranking-table th.model-cell {
1231
  top: 0;
1232
+ left: var(--rank-col-width);
1233
  z-index: 5;
1234
  background: var(--pruna-bg-header) !important;
1235
+ box-shadow: 0 1px 0 var(--pruna-hairline), 8px 0 10px -8px rgba(0, 0, 0, 0.35) !important;
1236
  }
1237
  .ranking-table tbody tr:hover .rank,
1238
  .ranking-table tbody tr:hover .model-cell {
 
1783
  max-width: 100% !important;
1784
  }
1785
  .compare-cell { min-width: 0; }
1786
+ .compare-cell.compare-source .compare-model-label {
1787
+ color: var(--pruna-text-muted);
1788
+ }
1789
  .compare-model-label {
1790
  margin-bottom: 6px;
1791
  color: var(--pruna-lavender);
 
1793
  font-weight: 700;
1794
  word-break: break-word;
1795
  }
1796
+ .compare-cell img,
1797
+ .compare-cell video {
1798
  display: block;
1799
  width: 100%;
 
 
1800
  border-radius: 12px;
1801
  border: 1px solid var(--pruna-border);
1802
  background: var(--pruna-bg-elevated);
1803
  }
1804
+ .compare-cell img {
1805
+ aspect-ratio: 1 / 1;
1806
+ object-fit: cover;
1807
+ }
1808
+ .compare-cell video {
1809
+ aspect-ratio: 16 / 9;
1810
+ max-height: 360px;
1811
+ object-fit: contain;
1812
+ }
1813
  .compare-empty,
1814
  .pareto-note-copy {
1815
  margin: 0;
 
2168
  return None
2169
 
2170
  prompts = {}
2171
+ source_videos = {}
2172
  with prompts_path.open() as handle:
2173
  for line in handle:
2174
  if not line.strip():
2175
  continue
2176
  row = json.loads(line)
2177
+ prompt_id = row["prompt_id"]
2178
+ prompts[prompt_id] = row.get("text", "")
2179
+ source_video = row.get("source_video")
2180
+ if source_video:
2181
+ source_videos[prompt_id] = source_video
2182
 
2183
  images = defaultdict(dict)
2184
+ kinds = set()
2185
  with generations_path.open() as handle:
2186
  for line in handle:
2187
  if not line.strip():
 
2189
  row = json.loads(line)
2190
  model_id = row["model_id"]
2191
  prompt_id = row["prompt_id"]
2192
+ media_url = row.get("image") or row.get("video")
2193
+ if row.get("video"):
2194
+ kinds.add("video")
2195
+ elif row.get("image"):
2196
+ kinds.add("image")
2197
+ if model_id and prompt_id and media_url:
2198
+ images[model_id][prompt_id] = media_url
2199
+ if prompt_id and prompt_id not in source_videos:
2200
+ params = row.get("params") or {}
2201
+ input_video = params.get("input_video") or row.get("input_video")
2202
+ if input_video:
2203
+ source_videos[prompt_id] = input_video
2204
 
2205
  models = sorted(images)
2206
  if not models or not prompts:
2207
  return None
2208
 
2209
+ loaded = {
2210
  "prompts": prompts,
2211
  "images": {model: dict(prompt_map) for model, prompt_map in images.items()},
2212
  "models": models,
2213
  "prompt_ids": sorted(prompts),
2214
+ "kind": "video" if "video" in kinds else "image",
2215
  }
2216
+ if source_videos:
2217
+ loaded["source_videos"] = source_videos
2218
+ return loaded
2219
 
2220
 
2221
  def _as_numeric(df, columns):
 
2345
  return df.reset_index(drop=True)
2346
 
2347
 
2348
+ def _is_pruna_video_model(series):
2349
+ models = series.astype(str).str.casefold()
2350
+ return models.str.startswith("p_video") | models.str.startswith("p-video")
2351
+
2352
+
2353
+ def _apply_video_timings(df, *, replace_displayed_time=False):
2354
+ """Mix Fal wall time with Pruna model execution time."""
2355
+ fal = df.get("Time / Output Video Second (s)")
2356
+ execution = df.get("Execution Time / Output Video Second (s)")
2357
+ if fal is None:
2358
+ return df
2359
+
2360
+ if "model_id" in df.columns:
2361
+ is_ours = _is_pruna_video_model(df["model_id"])
2362
+ else:
2363
+ is_ours = _is_pruna_video_model(df["Model"])
2364
+
2365
+ if execution is None:
2366
+ mixed = fal
2367
+ else:
2368
+ ours_time = execution.where(execution.notna(), fal)
2369
+ mixed = fal.where(~is_ours, ours_time)
2370
+ if replace_displayed_time:
2371
+ df["Time / Output Video Second (s)"] = mixed
2372
+ df["Pareto Time / Output Video Second (s)"] = mixed
2373
+ return df
2374
+
2375
+
2376
+ def load_video_editing_dataframe(path):
2377
+ """Load the video-to-video editing leaderboard."""
2378
+ df = pd.read_csv(path, na_values=["N/A", "n/a", ""])
2379
+ df = df.rename(
2380
+ columns={
2381
+ "display_name": "Model",
2382
+ "elo": "Datapoint Elo",
2383
+ "min_generation_s": "Min Generation Time (s)",
2384
+ "median_generation_s": "Median Generation Time (s)",
2385
+ "p20_generation_s": "P20 Generation Time (s)",
2386
+ "generation_s_per_output_video_s": "Time / Output Video Second (s)",
2387
+ "predict_time_s_per_output_video_s": "Predict Time / Output Video Second (s)",
2388
+ "model_execution_time_s_per_output_video_s": (
2389
+ "Execution Time / Output Video Second (s)"
2390
+ ),
2391
+ "price": "Price / Second of Video (USD)",
2392
+ }
2393
+ )
2394
+ df = df.drop(columns=["wandb_run_ids", "n_generations"], errors="ignore")
2395
+ df["Model"] = df["Model"].astype(str).str.strip()
2396
+ df = _as_numeric(
2397
+ df,
2398
+ [
2399
+ "Datapoint Elo",
2400
+ "Min Generation Time (s)",
2401
+ "Median Generation Time (s)",
2402
+ "P20 Generation Time (s)",
2403
+ "Time / Output Video Second (s)",
2404
+ "Predict Time / Output Video Second (s)",
2405
+ "Execution Time / Output Video Second (s)",
2406
+ "Price / Second of Video (USD)",
2407
+ ],
2408
+ )
2409
+ df = _apply_video_timings(df)
2410
+ df = df.drop(columns=["model_id"], errors="ignore")
2411
+ return df.reset_index(drop=True)
2412
+
2413
+
2414
+ def load_text_to_video_dataframe(path):
2415
+ """Load the text-to-video leaderboard (P-Video-2 and Fal models)."""
2416
+ df = pd.read_csv(path, na_values=["N/A", "n/a", ""])
2417
+ model_column = "model_id" if "model_id" in df.columns else "Model"
2418
+ df = df[~df[model_column].astype(str).str.casefold().str.startswith("agnes")].copy()
2419
+ df = df.rename(
2420
+ columns={
2421
+ "model_id": "Model",
2422
+ "datapoint_elo": "Datapoint Elo",
2423
+ "rapidata_elo": "Rapidata Elo",
2424
+ "min_generation_s": "Min Generation Time (s)",
2425
+ "median_generation_s": "Median Generation Time (s)",
2426
+ "p20_generation_s": "P20 Generation Time (s)",
2427
+ "generation_s_per_output_video_s": "Time / Output Video Second (s)",
2428
+ "model_execution_s_per_output_video_s": (
2429
+ "Execution Time / Output Video Second (s)"
2430
+ ),
2431
+ "price_usd_per_second": "Price / Second of Video (USD)",
2432
+ }
2433
+ )
2434
+ df = df.drop(columns=["wandb_run_ids", "n_generations"], errors="ignore")
2435
+ df["Model"] = df["Model"].astype(str).str.strip()
2436
+ df = _as_numeric(
2437
+ df,
2438
+ [
2439
+ "Datapoint Elo",
2440
+ "Rapidata Elo",
2441
+ "Min Generation Time (s)",
2442
+ "Median Generation Time (s)",
2443
+ "P20 Generation Time (s)",
2444
+ "Time / Output Video Second (s)",
2445
+ "Execution Time / Output Video Second (s)",
2446
+ "Price / Second of Video (USD)",
2447
+ ],
2448
+ )
2449
+ # Pruna rows use model execution time; Fal rows keep Fal wall time.
2450
+ df = _apply_video_timings(df, replace_displayed_time=True)
2451
+ df = df.drop(
2452
+ columns=["Execution Time / Output Video Second (s)"],
2453
+ errors="ignore",
2454
+ )
2455
+ elo_columns = [
2456
+ column
2457
+ for column in ("Datapoint Elo", "Rapidata Elo")
2458
+ if column in df.columns
2459
+ ]
2460
+ if elo_columns:
2461
+ df = df.dropna(subset=elo_columns, how="all")
2462
+ return df.reset_index(drop=True)
2463
+
2464
+
2465
  df = load_oneig_dataframe(oneig_path)
2466
 
2467
  oneig_metric_columns = [
 
2509
  data_dir / "qwen_image_bench_combined",
2510
  space_root.parent / "qwen_image_bench_combined",
2511
  )
2512
+ video_combined_dir = _resolve_data_path(
2513
+ data_dir / "video_editing_combined",
2514
+ space_root.parent / "video_editing_combined",
2515
+ )
2516
+ text_to_video_combined_dir = _resolve_data_path(
2517
+ data_dir / "video_generation_combined",
2518
+ space_root.parent / "video_generation_combined",
2519
+ )
2520
  qwen_path = _resolve_data_path(
2521
  data_dir / "qwen_image_bench_model_price_and_median_generation_time_10_august.csv",
2522
  space_root.parent / "qwen_image_bench_model_price_and_median_generation_time_10_august.csv",
 
2529
  data_dir / "arena_ai_text_to_image_leaderboard.csv",
2530
  space_root.parent / "arena_ai_text_to_image_leaderboard.csv",
2531
  )
2532
+ video_path = _resolve_data_path(
2533
+ data_dir / "video-editing-leaderboard.csv",
2534
+ space_root.parent / "video-editing-leaderboard.csv",
2535
+ )
2536
+ text_to_video_path = _resolve_data_path(
2537
+ data_dir / "p-video-2-leaderboard.csv",
2538
+ space_root.parent / "p-video-2-leaderboard.csv",
2539
+ )
2540
 
2541
  qwen_df = load_qwen_combined_dataframe(qwen_path)
2542
  aa_df = load_artificial_analysis_dataframe(aa_path)
2543
  arena_df = load_arena_ai_dataframe(arena_path)
2544
+ video_df = load_video_editing_dataframe(video_path)
2545
+ text_to_video_df = load_text_to_video_dataframe(text_to_video_path)
2546
  qwen_display_columns = [
2547
  col
2548
  for col in [
 
2574
  ]
2575
  if col in arena_df.columns
2576
  ]
2577
+ video_display_columns = [
2578
+ col
2579
+ for col in [
2580
+ "Model",
2581
+ "Datapoint Elo",
2582
+ "Time / Output Video Second (s)",
2583
+ "Median Generation Time (s)",
2584
+ "Min Generation Time (s)",
2585
+ "Price / Second of Video (USD)",
2586
+ ]
2587
+ if col in video_df.columns
2588
+ ]
2589
+ text_to_video_display_columns = [
2590
+ col
2591
+ for col in [
2592
+ "Model",
2593
+ "Datapoint Elo",
2594
+ "Rapidata Elo",
2595
+ "Time / Output Video Second (s)",
2596
+ "Median Generation Time (s)",
2597
+ "Min Generation Time (s)",
2598
+ "Price / Second of Video (USD)",
2599
+ ]
2600
+ if col in text_to_video_df.columns
2601
+ ]
2602
 
2603
  oneig_samples = load_sample_comparison_data(oneig_combined_dir)
2604
  qwen_samples = load_sample_comparison_data(qwen_combined_dir)
2605
+ video_samples = load_sample_comparison_data(video_combined_dir)
2606
+ text_to_video_samples = load_sample_comparison_data(text_to_video_combined_dir)
2607
 
2608
  metrics = [
2609
  {"id": "datapoint_elo", "column": "Datapoint Elo"},
 
2662
  "arena_text",
2663
  ],
2664
  )
2665
+ video_metric_ids = _metric_ids_for(video_df, ["datapoint_elo"])
2666
+ text_to_video_metric_ids = _metric_ids_for(
2667
+ text_to_video_df, ["datapoint_elo", "rapidata_elo"]
2668
+ )
2669
 
2670
  datasets = [
2671
+ {
2672
+ "id": "text_to_video",
2673
+ "name": "VBench-2.0 Dataset",
2674
+ "modality": "text_to_video",
2675
+ "data": text_to_video_df,
2676
+ "columns": text_to_video_display_columns,
2677
+ "metric_ids": text_to_video_metric_ids,
2678
+ "note": (
2679
+ "Datapoint Elo and Rapidata Elo from pairwise text-to-video "
2680
+ "preference. Price is USD per second of output video. Time per "
2681
+ "second of video is Fal wall time, except Pruna models which use "
2682
+ "model execution time."
2683
+ ),
2684
+ "samples": text_to_video_samples,
2685
+ },
2686
+ {
2687
+ "id": "video_editing",
2688
+ "name": "Pruna Internal Video-Edit Benchmark",
2689
+ "modality": "video_to_video",
2690
+ "data": video_df,
2691
+ "columns": video_display_columns,
2692
+ "metric_ids": video_metric_ids,
2693
+ "note": (
2694
+ "Datapoint Elo from pairwise video-edit preference. Price is USD "
2695
+ "per second of output video. Generation time per second of video "
2696
+ "is end-to-end wall time to produce one second of output."
2697
+ ),
2698
+ "samples": video_samples,
2699
+ },
2700
  {
2701
  "id": "qwen",
2702
  "name": "Qwen Image Dataset",
2703
+ "modality": "text_to_image",
2704
  "data": qwen_df,
2705
  "columns": qwen_display_columns,
2706
  "metric_ids": qwen_metric_ids,
 
2710
  {
2711
  "id": "oneig",
2712
  "name": "OneIG Alignment Dataset",
2713
+ "modality": "text_to_image",
2714
  "data": oneig_df,
2715
  "columns": oneig_display_columns,
2716
  "metric_ids": oneig_metric_ids,
 
2723
  {
2724
  "id": "artificial_analysis",
2725
  "name": "Artificial Analysis Dataset",
2726
+ "modality": "text_to_image",
2727
  "data": aa_df,
2728
  "columns": aa_display_columns,
2729
  "metric_ids": aa_metric_ids,
 
2733
  {
2734
  "id": "arena_ai",
2735
  "name": "Arena AI Dataset",
2736
+ "modality": "text_to_image",
2737
  "data": arena_df,
2738
  "columns": arena_display_columns,
2739
  "metric_ids": arena_metric_ids,
 
2744
  datasets = [dataset for dataset in datasets if dataset["metric_ids"]]
2745
 
2746
  DEFAULT_DATASET_ID = next(
2747
+ (dataset["id"] for dataset in datasets if dataset["id"] == "text_to_video"),
2748
+ next(
2749
+ (dataset["id"] for dataset in datasets if dataset["id"] == "video_editing"),
2750
+ next(
2751
+ (dataset["id"] for dataset in datasets if dataset["id"] == "qwen"),
2752
+ datasets[0]["id"] if datasets else None,
2753
+ ),
2754
+ ),
2755
  )
2756
  DEFAULT_METRIC_ID = None
2757
 
data/p-video-2-leaderboard.csv ADDED
@@ -0,0 +1,17 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ model_id,wandb_run_ids,n_generations,min_generation_s,median_generation_s,p20_generation_s,generation_s_per_output_video_s,model_execution_s_per_output_video_s,datapoint_elo,rapidata_elo,price_usd_per_second
2
+ gemini_omni_1_1_flash,mfajagum,89,26.276599962002365,31.884095050001633,29.0771403791965,6.399738597206617,,1037,1197,0.100
3
+ grok_imagine_video,q39r8v9e,90,68.9776645039965,99.17614604350092,78.69151808420138,16.13514411181308,,958,994,0.050
4
+ grok_imagine_video_v1_5,z2fp7hd7,90,38.42492011199647,52.268091876499966,46.47850653079659,11.452002471104521,,959,1016,0.140
5
+ minimax_h3,xpcv2m58,89,97.61419671699696,117.51345164199665,106.57570995900024,24.968780374361877,,1026,1108,0.100
6
+ minimax_h3_max,0d2lkmp6,90,4.173863318999793,4.49869444649994,4.4671880990012145,1.3807388451000264,0.644,1031,1197,0.080
7
+ minimax_h3_max_turbo__prompt_expansion_mode_balanced,kxuwq31c,90,2.891819470001792,3.487746603501364,3.134695611400821,1.35231252479333,0.61,1040,1214,0.040
8
+ p_video_2__draft_false__prompt_upsampling_false__resolution_1080p,9hgkhujw,90,13.33159556199098,16.702878594005597,14.793676785795833,4.033411242884629,2.151682222222222,,,0.050
9
+ p_video_2__draft_false__prompt_upsampling_false__resolution_720p,m2jf4z7h,89,5.859760620005545,8.162274362999597,7.147269867203431,2.1525601170473116,0.9053123595505614,,,0.025
10
+ p_video_2__draft_false__prompt_upsampling_true__resolution_1080p,uoqpv10b,90,14.615217412007041,18.316011337505188,16.441080821005745,4.5638632221758275,2.152566666666667,979,991,0.050
11
+ p_video_2__draft_false__prompt_upsampling_true__resolution_720p,2q5jydr8,90,7.227749306999613,9.880705762494472,9.154919161600992,2.4789854674801206,0.9079755555555558,985,1039,0.025
12
+ p_video_2__draft_true__prompt_upsampling_false__resolution_1080p,qbt0atz4,89,6.250065040003392,9.528199804000906,7.504525098402519,2.431133616615736,0.7517640449438197,,,0.030
13
+ p_video_2__draft_true__prompt_upsampling_false__resolution_720p,9ur57h54,88,3.379928643000312,5.454214365498046,4.208131522199255,1.4960965825225272,0.40954545454545455,,,0.015
14
+ p_video_2__draft_true__prompt_upsampling_true__resolution_1080p,gcbyy4wn,90,8.047415203996934,11.128662197996164,9.22702455239487,2.8557857865532665,0.7514066666666668,,,0.030
15
+ p_video_2__draft_true__prompt_upsampling_true__resolution_720p,6hi994vk,90,4.985804016003385,7.4922584965024726,6.472871531004785,2.4024102996157146,0.40678444444444434,982,1048,0.015
16
+ seedance_2_5_turbo,l9zjnv7n,90,168.997338743,252.07832621250054,211.93920676639829,67.60899374696221,,1037,1080,0.200
17
+ veo_3_1_lite,jjbobes4,90,33.894167409991496,37.83729844300251,37.009501291197374,6.745116482181308,,968,986,0.050
data/video-editing-leaderboard.csv ADDED
@@ -0,0 +1,11 @@
 
 
 
 
 
 
 
 
 
 
 
 
1
+ model_id,display_name,elo,wandb_run_ids,n_generations,min_generation_s,median_generation_s,p20_generation_s,generation_s_per_output_video_s,predict_time_s_per_output_video_s,model_execution_time_s_per_output_video_s,price
2
+ gemini_omni_flash_edit__fal,Gemini Omni Flash Edit,1054,3iuctwj9,56,32.574746752001374,58.2327566820004,50.44402900500063,13.78,,,0.13
3
+ grok_imagine_video__replicate,Grok Imagine Video,974,q0ogvekx,59,31.640928319000523,43.75649321700257,32.88782280539963,11.04,,,0.05
4
+ happyhorse_1_0__wavespeed,HappyHorse 1.0,1048,k4zj5wqk,71,106.5824136010051,201.2431439649954,141.46595527700265,46.73,,,0.14
5
+ lucy_edit_pro__fal,Lucy Edit Pro,879,na6bz9xx,72,118.3459930579993,136.44442766549764,124.13885582720104,27.18,,,0.15
6
+ minimax_h3_reference_to_video__fal,MiniMax H3 Reference-to-Video,1060,639xmakz,63,180.8385945170012,308.29198316100155,248.2394383729996,58.31,,,0.06
7
+ p_video_edit_preview__replicate_final,P-Video-Edit,1000,m8irpsa6,73,31.41135125700021,86.86665409700072,57.21095684959946,23.18,17.99,12.06,0.045
8
+ p_video_edit_preview__replicate_final__draft,P-Video-Edit Draft,994,6f88e6mx,73,23.04809326099712,42.9797806409988,33.19193872759861,11.48,11.28,4.46,0.025
9
+ seedance_2_5_video_edit_turbo__wavespeed,Seedance 2.5 Video Edit Turbo,1063,2nr274vp,68,142.49452170499717,293.7401115540015,223.11569286320045,59.93,,,0.24
10
+ wan_2_7_video_edit__wavespeed,Wan 2.7 Video Edit,1055,9lq803be,66,134.3744560209998,313.957433804002,219.2491260079987,68.04,,,0.2
11
+
data/video_editing_combined/generations.jsonl ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:fb97166ea143a220c9395706627259755d0656534b3162712ecc619020b995db
3
+ size 674713
data/video_editing_combined/prompts.jsonl ADDED
@@ -0,0 +1,78 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {"prompt_id": "video_edit_internal__advertising__p0000", "text": "Replace the white bottle with an orange sunscreen bottle.", "dataset": "video_edit_internal", "category": "advertising", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/advertising/4620329ad65820ce.mp4"}
2
+ {"prompt_id": "video_edit_internal__advertising__p0001", "text": "Replace the woman with an asian woman riding on a donkey through a chinese small town.", "dataset": "video_edit_internal", "category": "advertising", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/advertising/44db5689846be901.mp4"}
3
+ {"prompt_id": "video_edit_internal__advertising__p0002", "text": "Place the couple on a snowy mountain next to a mountain hut.", "dataset": "video_edit_internal", "category": "advertising", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/advertising/481db4a047688bbe.mp4"}
4
+ {"prompt_id": "video_edit_internal__advertising__p0003", "text": "Turn this into a scene in the summer with birds flying in the sky and butterflies in the foreground.", "dataset": "video_edit_internal", "category": "advertising", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/advertising/32b9f56763c56482.mp4"}
5
+ {"prompt_id": "video_edit_internal__advertising__p0004", "text": "Replace the robot arm with a dancing hamster on a small table.", "dataset": "video_edit_internal", "category": "advertising", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/advertising/22d02ae2df7f9045.mp4"}
6
+ {"prompt_id": "video_edit_internal__anonymization__p0000", "text": "Anonymize the womans face", "dataset": "video_edit_internal", "category": "anonymization", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/anonymization/7c410085f2fd939d.mp4"}
7
+ {"prompt_id": "video_edit_internal__anonymization__p0001", "text": "Anonymize the mans face", "dataset": "video_edit_internal", "category": "anonymization", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/anonymization/7b495805c0c270ea.mp4"}
8
+ {"prompt_id": "video_edit_internal__anonymization__p0002", "text": "Anonymize the mans face", "dataset": "video_edit_internal", "category": "anonymization", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/anonymization/736b6fb504bd32be.mp4"}
9
+ {"prompt_id": "video_edit_internal__anonymization__p0003", "text": "Anonymize the girls face", "dataset": "video_edit_internal", "category": "anonymization", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/anonymization/2c82413d39579cd5.mp4"}
10
+ {"prompt_id": "video_edit_internal__artificial_analysis__p0000", "text": "Change the rainforest setting to a neon-lit urban alley at night with steam from vents and wet reflective asphalt, and move into an ariel shot of the character after the initial camera movement to focus on the character", "dataset": "video_edit_internal", "category": "artificial_analysis", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/artificial_analysis/e2f6121269a206d3.mp4"}
11
+ {"prompt_id": "video_edit_internal__artificial_analysis__p0001", "text": "Replace the magenta hover-car with a chrome-blue car, keeping the drift and neon light trails.", "dataset": "video_edit_internal", "category": "artificial_analysis", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/artificial_analysis/55d1baa7f3f71b26.mp4"}
12
+ {"prompt_id": "video_edit_internal__artificial_analysis__p0002", "text": "Make the footage look like it was shot on a 1970s film camera, with grainy film texture, faded warm colors, slight softness, and slightly darker corners around the edges.", "dataset": "video_edit_internal", "category": "artificial_analysis", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/artificial_analysis/fd1ab436de020152.mp4"}
13
+ {"prompt_id": "video_edit_internal__artificial_analysis__p0003", "text": "A person appears at the top of the waterfall, leaps into the lake below, and disappears into the misty water beneath the falls.", "dataset": "video_edit_internal", "category": "artificial_analysis", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/artificial_analysis/7ee1623303a8da68.mp4"}
14
+ {"prompt_id": "video_edit_internal__artificial_analysis__p0004", "text": "A fluffy orange cat leaps gracefully from the floor onto the couch, paws sinking into the soft cushions. It circles once in place, tail swaying gently, before curling up tightly on one of the cushions and settling in comfortably.", "dataset": "video_edit_internal", "category": "artificial_analysis", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/artificial_analysis/9a73c608b343bbbe.mp4"}
15
+ {"prompt_id": "video_edit_internal__camera_editing__p0000", "text": "Zoom in on the man's face to show his focused expression", "dataset": "video_edit_internal", "category": "camera_editing", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/camera_editing/0336e64b0594bd7a.mp4"}
16
+ {"prompt_id": "video_edit_internal__camera_editing__p0001", "text": "Perform an arc shot around the tram as it arrives at the station", "dataset": "video_edit_internal", "category": "camera_editing", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/camera_editing/204aa93703da117b.mp4"}
17
+ {"prompt_id": "video_edit_internal__camera_editing__p0002", "text": "Change the view to a high angle.", "dataset": "video_edit_internal", "category": "camera_editing", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/camera_editing/c2ccb5351be5c718.mp4"}
18
+ {"prompt_id": "video_edit_internal__camera_editing__p0003", "text": "Change the view to a high angle.", "dataset": "video_edit_internal", "category": "camera_editing", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/camera_editing/b725aaded02d65e9.mp4"}
19
+ {"prompt_id": "video_edit_internal__camera_editing__p0004", "text": "Gradually move the camera away from the doctor", "dataset": "video_edit_internal", "category": "camera_editing", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/camera_editing/33f24f4d083df743.mp4"}
20
+ {"prompt_id": "video_edit_internal__design_arena__p0000", "text": "Create a transition of the video of the product whihc is the jeans on the lady model with an atitiude that goes into a zoom out camera that she is in a party .Focus on the vibes", "dataset": "video_edit_internal", "category": "design_arena", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/design_arena/b0b4642272a481b1.mp4"}
21
+ {"prompt_id": "video_edit_internal__design_arena__p0001", "text": "add ethnic urban women dancing at the bottem and at the bar wearing DMI Tshirts", "dataset": "video_edit_internal", "category": "design_arena", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/design_arena/0f730f98f3c51356.mp4"}
22
+ {"prompt_id": "video_edit_internal__design_arena__p0002", "text": "Generate a commercial of this dog drinking beer at an electronic music party in a world where dogs and humans are mixed together.", "dataset": "video_edit_internal", "category": "design_arena", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/design_arena/575fc4b9b6d7b3ab.mp4"}
23
+ {"prompt_id": "video_edit_internal__design_arena__p0003", "text": "everything is the same excpet the winter season, snowing", "dataset": "video_edit_internal", "category": "design_arena", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/design_arena/8674fd022867a736.mp4"}
24
+ {"prompt_id": "video_edit_internal__design_arena__p0004", "text": "A cinematic character introduction of Jason, framed in a medium close-up with subtle camera movement, confident body language, and expressive facial detail. Moody, high-contrast lighting with a cool-toned color palette, shallow depth of field, and a slow dramatic reveal that builds intrigue over a few seconds. I want a video in a 9:16 aspect ratio for tiktok. remove all text and have him in a mid day campus setting", "dataset": "video_edit_internal", "category": "design_arena", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/design_arena/91873bbafe4173da.mp4"}
25
+ {"prompt_id": "video_edit_internal__e_commerce__p0000", "text": "Remove the black blazer.", "dataset": "video_edit_internal", "category": "e_commerce", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/e_commerce/b89faaa5080c31db.mp4"}
26
+ {"prompt_id": "video_edit_internal__e_commerce__p0001", "text": "Replace her skirt with a jeans skirt.", "dataset": "video_edit_internal", "category": "e_commerce", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/e_commerce/c34a33d07e843804.mp4"}
27
+ {"prompt_id": "video_edit_internal__e_commerce__p0002", "text": "Replace the black leggings and the black shirt with a white dress.", "dataset": "video_edit_internal", "category": "e_commerce", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/e_commerce/489d2439cbaeed79.mp4"}
28
+ {"prompt_id": "video_edit_internal__e_commerce__p0003", "text": "Replace the white woman with a lebanese-looking woman.", "dataset": "video_edit_internal", "category": "e_commerce", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/e_commerce/0ad8fb773f303a11.mp4"}
29
+ {"prompt_id": "video_edit_internal__e_commerce__p0004", "text": "Remove all items on the table and place an eyeshadow pallete on the table next to the woman.", "dataset": "video_edit_internal", "category": "e_commerce", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/e_commerce/f28f3aa12b1ec220.mp4"}
30
+ {"prompt_id": "video_edit_internal__lighting__p0000", "text": "After the sun sets behind the mountains, the scene transitions into nighttime.", "dataset": "video_edit_internal", "category": "lighting", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/lighting/229c165e0ac97daf.mp4"}
31
+ {"prompt_id": "video_edit_internal__lighting__p0001", "text": "Change the weather to a dazzling starry night.", "dataset": "video_edit_internal", "category": "lighting", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/lighting/f42c41ab2f3adc26.mp4"}
32
+ {"prompt_id": "video_edit_internal__lighting__p0002", "text": "Change the weather to a thunderstorm with heavy rain.", "dataset": "video_edit_internal", "category": "lighting", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/lighting/daa36567634ca6be.mp4"}
33
+ {"prompt_id": "video_edit_internal__lighting__p0003", "text": "Change the weather to a torrential downpour.", "dataset": "video_edit_internal", "category": "lighting", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/lighting/6817a4142b69d602.mp4"}
34
+ {"prompt_id": "video_edit_internal__lighting__p0004", "text": "Change the weather to a torrential downpour.", "dataset": "video_edit_internal", "category": "lighting", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/lighting/ac01488a685a1096.mp4"}
35
+ {"prompt_id": "video_edit_internal__long__p0000", "text": "Change the weather to a dazzling starry night.", "dataset": "video_edit_internal", "category": "long", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/long/e9af2a7039e43f16.mp4"}
36
+ {"prompt_id": "video_edit_internal__long__p0001", "text": "Adjust the color of book to blue", "dataset": "video_edit_internal", "category": "long", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/long/eab696d1555be316.mp4"}
37
+ {"prompt_id": "video_edit_internal__long__p0002", "text": "Make the young woman turn into sand and blow away.", "dataset": "video_edit_internal", "category": "long", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/long/d18378f1d7e4ca0e.mp4"}
38
+ {"prompt_id": "video_edit_internal__long__p0003", "text": "Make the bird flap its wings", "dataset": "video_edit_internal", "category": "long", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/long/3835b37b43f5e5bf.mp4"}
39
+ {"prompt_id": "video_edit_internal__long__p0004", "text": "Transform the video into a ukiyo-e style", "dataset": "video_edit_internal", "category": "long", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/long/09fe5bb4b5304b66.mp4"}
40
+ {"prompt_id": "video_edit_internal__movie_concept_art__p0000", "text": "Add more blood and wounds to the mans face.", "dataset": "video_edit_internal", "category": "movie_concept_art", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/movie_concept_art/1d1a99167475720b.mp4"}
41
+ {"prompt_id": "video_edit_internal__movie_concept_art__p0001", "text": "Make the scene less dark.", "dataset": "video_edit_internal", "category": "movie_concept_art", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/movie_concept_art/10cb44d8c0bf4116.mp4"}
42
+ {"prompt_id": "video_edit_internal__movie_concept_art__p0002", "text": "Replace the female warrior with a male warrior with ginger hair.", "dataset": "video_edit_internal", "category": "movie_concept_art", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/movie_concept_art/9e45454390c20b8a.mp4"}
43
+ {"prompt_id": "video_edit_internal__movie_concept_art__p0003", "text": "Turn the man's hand into a robotic hand.", "dataset": "video_edit_internal", "category": "movie_concept_art", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/movie_concept_art/99977e7a91a950b4.mp4"}
44
+ {"prompt_id": "video_edit_internal__real_estate__p0000", "text": "Replace the interior with a gold-black interior heavy luxury style.", "dataset": "video_edit_internal", "category": "real_estate", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/real_estate/869064ab06bb5daa.mp4"}
45
+ {"prompt_id": "video_edit_internal__real_estate__p0001", "text": "Replace the grey wallpaper with a beige painted wall and turn the grey curtains a dark brown.", "dataset": "video_edit_internal", "category": "real_estate", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/real_estate/1c2f358bfe9b8299.mp4"}
46
+ {"prompt_id": "video_edit_internal__real_estate__p0002", "text": "Remove all decoration from the walls and keep only the furniture.", "dataset": "video_edit_internal", "category": "real_estate", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/real_estate/2058bcf38389f8fe.mp4"}
47
+ {"prompt_id": "video_edit_internal__real_estate__p0003", "text": "Exchange the wooden floor for a marble floor.", "dataset": "video_edit_internal", "category": "real_estate", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/real_estate/6a4c09a23dda57d9.mp4"}
48
+ {"prompt_id": "video_edit_internal__real_estate__p0004", "text": "Replace the bedframe with a modern wooden bed.", "dataset": "video_edit_internal", "category": "real_estate", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/real_estate/3e0f6dc5df891001.mp4"}
49
+ {"prompt_id": "video_edit_internal__style_transfer__p0000", "text": "Transform the video into a cyberpunk style", "dataset": "video_edit_internal", "category": "style_transfer", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/style_transfer/2253f6ed3674f006.mp4"}
50
+ {"prompt_id": "video_edit_internal__style_transfer__p0001", "text": "Convert to different shades of orange", "dataset": "video_edit_internal", "category": "style_transfer", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/style_transfer/edb6a105be03bae0.mp4"}
51
+ {"prompt_id": "video_edit_internal__style_transfer__p0002", "text": "Convert to black and white", "dataset": "video_edit_internal", "category": "style_transfer", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/style_transfer/c3dc058b3be26cbd.mp4"}
52
+ {"prompt_id": "video_edit_internal__style_transfer__p0003", "text": "Apply Ghibli-style editing to the video", "dataset": "video_edit_internal", "category": "style_transfer", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/style_transfer/83a638223a20911c.mp4"}
53
+ {"prompt_id": "video_edit_internal__style_transfer__p0004", "text": "Relight the scene as if it were shot during golden hour, with warm low-angle sunlight, soft shadows, and natural highlights on faces and surfaces.", "dataset": "video_edit_internal", "category": "style_transfer", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/style_transfer/daa36567634ca6be.mp4"}
54
+ {"prompt_id": "video_edit_internal__subject_editing__p0000", "text": "Replace the rainbow colors of the logo with different shades of purple", "dataset": "video_edit_internal", "category": "subject_editing", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/subject_editing/6358f26b46baadf5.mp4"}
55
+ {"prompt_id": "video_edit_internal__subject_editing__p0001", "text": "Add a small dog running beside the scooter", "dataset": "video_edit_internal", "category": "subject_editing", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/subject_editing/052c0bfd7b68fea7.mp4"}
56
+ {"prompt_id": "video_edit_internal__subject_editing__p0002", "text": "Replace the splashing waves with a calm water surface", "dataset": "video_edit_internal", "category": "subject_editing", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/subject_editing/c3f79578a39f415d.mp4"}
57
+ {"prompt_id": "video_edit_internal__subject_editing__p0003", "text": "Add a violinist in the background", "dataset": "video_edit_internal", "category": "subject_editing", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/subject_editing/a0eebf32a076e2fc.mp4"}
58
+ {"prompt_id": "video_edit_internal__subject_editing__p0004", "text": "Add a group of people walking on the pathway", "dataset": "video_edit_internal", "category": "subject_editing", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/subject_editing/3c2b7e3a5b284e1d.mp4"}
59
+ {"prompt_id": "video_edit_internal__subject_motion_editing__p0000", "text": "Make the bird hop around instead of walking and foraging", "dataset": "video_edit_internal", "category": "subject_motion_editing", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/subject_motion_editing/ac38b2a97d1a5c3a.mp4"}
60
+ {"prompt_id": "video_edit_internal__subject_motion_editing__p0001", "text": "The male colleague is walking around to observe.", "dataset": "video_edit_internal", "category": "subject_motion_editing", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/subject_motion_editing/8aaab7d304c5ad99.mp4"}
61
+ {"prompt_id": "video_edit_internal__subject_motion_editing__p0002", "text": "Make the static spider-man in the mural dynamic and make him swing faster", "dataset": "video_edit_internal", "category": "subject_motion_editing", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/subject_motion_editing/3412b39686fd5879.mp4"}
62
+ {"prompt_id": "video_edit_internal__subject_motion_editing__p0003", "text": "Change the woman's jogging to taking off.", "dataset": "video_edit_internal", "category": "subject_motion_editing", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/subject_motion_editing/17c4dc0514772321.mp4"}
63
+ {"prompt_id": "video_edit_internal__subject_motion_editing__p0004", "text": "Make the knight lunging forward and the creature swiping at the knight", "dataset": "video_edit_internal", "category": "subject_motion_editing", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/subject_motion_editing/980354550011911b.mp4"}
64
+ {"prompt_id": "video_edit_internal__synthetic_data__p0000", "text": "Turn the scene into nighttime.", "dataset": "video_edit_internal", "category": "synthetic_data", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/synthetic_data/e3b1e603bfaa1aec.mp4"}
65
+ {"prompt_id": "video_edit_internal__synthetic_data__p0001", "text": "Remove the crosswalk.", "dataset": "video_edit_internal", "category": "synthetic_data", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/synthetic_data/30df5fd5fe28af65.mp4"}
66
+ {"prompt_id": "video_edit_internal__synthetic_data__p0002", "text": "Add a bicycle riding in front of the car.", "dataset": "video_edit_internal", "category": "synthetic_data", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/synthetic_data/d093491e25682b3a.mp4"}
67
+ {"prompt_id": "video_edit_internal__synthetic_data__p0003", "text": "Turn the scene into a snowstorm scene.", "dataset": "video_edit_internal", "category": "synthetic_data", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/synthetic_data/1da5913f13362f8d.mp4"}
68
+ {"prompt_id": "video_edit_internal__synthetic_data__p0004", "text": "Make it rain in the scene.", "dataset": "video_edit_internal", "category": "synthetic_data", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/synthetic_data/936bdb89558f5a7f.mp4"}
69
+ {"prompt_id": "video_edit_internal__text__p0000", "text": "Add the text \"True Love\" in the foreground in pink romantic font.", "dataset": "video_edit_internal", "category": "text", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/text/c763407793ca9b0e.mp4"}
70
+ {"prompt_id": "video_edit_internal__text__p0001", "text": "Replace any mention of \"Fanta\" with the branding \"Cola\".", "dataset": "video_edit_internal", "category": "text", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/text/7b878ea6d99834b2.mp4"}
71
+ {"prompt_id": "video_edit_internal__text__p0002", "text": "Remove all text from the video.", "dataset": "video_edit_internal", "category": "text", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/text/76763c31b2609109.mp4"}
72
+ {"prompt_id": "video_edit_internal__text__p0003", "text": "Replace the branding \"Royalty\" with the phrasing \"Princess\".", "dataset": "video_edit_internal", "category": "text", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/text/d221d299c9e49064.mp4"}
73
+ {"prompt_id": "video_edit_internal__text__p0004", "text": "Remove all text from the video.", "dataset": "video_edit_internal", "category": "text", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/text/5c169901af8e4ed6.mp4"}
74
+ {"prompt_id": "video_edit_internal__transitions__p0000", "text": "After a black screen transition, the road transforms into lush grass.", "dataset": "video_edit_internal", "category": "transitions", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/transitions/964c35b19ee1fc1f.mp4"}
75
+ {"prompt_id": "video_edit_internal__transitions__p0001", "text": "After a wave-foam transition, the small fishing boat is eaten by a giant whale.", "dataset": "video_edit_internal", "category": "transitions", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/transitions/b8b88f43316363f3.mp4"}
76
+ {"prompt_id": "video_edit_internal__transitions__p0002", "text": "Add a cut transition, then show a bowl of ramen topped with parsley.", "dataset": "video_edit_internal", "category": "transitions", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/transitions/c823d6c77bdc87f4.mp4"}
77
+ {"prompt_id": "video_edit_internal__transitions__p0003", "text": "Add a smoke transition, then show the circuit board burning.", "dataset": "video_edit_internal", "category": "transitions", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/transitions/edb9927ff06c1175.mp4"}
78
+ {"prompt_id": "video_edit_internal__transitions__p0004", "text": "After a smoke transition, the Basilica of the Sacred Heart of Paris catches fire.", "dataset": "video_edit_internal", "category": "transitions", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/transitions/77bb900383d4f370.mp4"}
data/video_generation_combined/generations.jsonl ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:149714ae8c73bf2832fd8b9ac2fbdde069971b9f5cf90e8ac2725f051d805b59
3
+ size 970689
data/video_generation_combined/prompts.jsonl ADDED
@@ -0,0 +1,90 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {"prompt_id": "v-bench2__all__p0000", "text": "Garden, zoom in.", "dataset": "v-bench2", "category": "all"}
2
+ {"prompt_id": "v-bench2__all__p0001", "text": "Garden, zoom out.", "dataset": "v-bench2", "category": "all"}
3
+ {"prompt_id": "v-bench2__all__p0002", "text": "Garden, tilt up.", "dataset": "v-bench2", "category": "all"}
4
+ {"prompt_id": "v-bench2__all__p0003", "text": "The camera starts at the top of the forest, where thick morning mist drifts between the treetops, with sunlight filtering through the leaves and casting dappled spots of light, the air fresh and moist. As the camera slowly moves downward, dewdrops glisten on the leaves, a gentle breeze rustling the leaves, making a soft, soothing sound. The camera moves along a woodland path, occasionally capturing a squirrel leaping between trees, its agile figure darting across the branches. The shot continues, eventually arriving at a tranquil lake. The surface of the water is like a mirror, reflecting the towering trees and distant mountains. In the distance, a heron takes off from the lakeshore, breaking the stillness of the water, ripples forming as the bird skims the surface. The camera follows its flight and gradually pulls back, showing the entire serene beauty of the lake.", "dataset": "v-bench2", "category": "all"}
5
+ {"prompt_id": "v-bench2__all__p0004", "text": "The scene opens in the vast desert, with the camera angled low over the sand, fine grains drifting in the gentle breeze, the undulating dunes in the distance bathed in a faint orange glow. As the camera moves forward, the outlines of the dunes become clearer, and the early morning sky begins to glow with warm hues, transitioning the desert from deep brown to golden yellow. The scene cuts to the top of a dune, where the camera follows a caravan of camels making their way across the endless desert, their bells jingling softly in the wind. The first rays of the sun break through the gaps between the dunes, casting golden light on the sand, as the desert comes to life, glowing in the morning's embrace. Finally, the camera pulls back, revealing the vast expanse of the desert merging with the horizon, the sun climbing higher, illuminating the entire landscape in a mystical, warm glow.", "dataset": "v-bench2", "category": "all"}
6
+ {"prompt_id": "v-bench2__all__p0005", "text": "The camera begins in a vast grassland, where the lush green grass sways gently in the breeze, the air fresh, and the soft rustling of the leaves fills the space. As the camera moves forward, the grass turns from a vibrant green to golden hues, sunlight pouring over the land, the textures of the grass becoming more pronounced in the play of light and shadow. In the distance, a herd of cattle grazes peacefully. Their figures become clearer as the camera moves closer, occasionally lifting their heads and gazing into the distance. The scene shifts to a lakeshore, where the lake reflects the sky, grasslands, and distant mountains, the water clear and calm, with gentle ripples created by the breeze. Finally, the camera pulls back, following the herd as they move off into the distance, the vastness of the grassland merging with the expansive sky, creating a feeling of peaceful openness.", "dataset": "v-bench2", "category": "all"}
7
+ {"prompt_id": "v-bench2__all__p0006", "text": "The camera starts at a tranquil coastline, where the waves gently crash against the rocks, the salty sea breeze fills the air, and the atmosphere feels fresh and alive. As the camera moves forward, the surface of the water glistens with golden light from the setting sun, the sound of the waves becoming more distinct as the tide retreats, revealing a moist sandy shore. The scene shifts to the beach, where a few seagulls peck at the sand, occasionally flying up and breaking the stillness of the sky. The sun slowly sinks below the horizon, painting the sky in vibrant shades of orange and red, as the golden light on the water fades, and the waves begin to intensify. Finally, the camera pulls back, and the entire coastline fades into twilight, with the sea and sky blending together in a serene, solitary embrace.", "dataset": "v-bench2", "category": "all"}
8
+ {"prompt_id": "v-bench2__all__p0007", "text": "The race began, and the first runner quickly took off, leading the other teams. Everyone was focused on his performance. During the baton handoff of the first runner, Team A made a slight mistake, and the baton nearly fell. Despite this, the runner quickly passed it to the second runner. The second runner showed great determination and chased hard, finally overtaking Team B on the bend. However, due to the earlier mistake, Team A's lead was not significant. During the third runner's handoff, the runner from Team A nervously accelerated and managed to maintain the lead, but Team C quickly caught up thanks to their excellent pace control. During the fourth runner's handoff, the final sprint became crucial. The Team A runner started to accelerate in the second-to-last turn and, with a burst of strength, successfully widened the gap, steadily running toward the finish line and winning the race.", "dataset": "v-bench2", "category": "all"}
9
+ {"prompt_id": "v-bench2__all__p0008", "text": "The match began, and Team A quickly organized an attack, breaking through Team B's defense with fast passes. The forward accurately kicked the ball into the goal, taking a 1-0 lead. Team B did not panic; they used a long pass to quickly counterattack. The forward calmly shot past the goalkeeper, and the ball went straight into the net, making it 1-1. At this point, the game entered a stalemate. Both teams' defenses were solid, and the match became a fierce battle. In the second half, Team A had a corner kick. The high ball delivered by the player was headed in by the center-back who jumped high, putting Team A in the lead at 2-1. In the final moments of the game, Team B got a penalty kick. The forward took the shot without hesitation, sending the ball into the net to make it 2-2. Finally, in the last moments of extra time, Team A used a quick counterattack and a long-range shot from outside the box flew straight into the corner, securing the win with a 3-2 score.", "dataset": "v-bench2", "category": "all"}
10
+ {"prompt_id": "v-bench2__all__p0009", "text": "The race began, and all the runners sprinted quickly. The runner from Team A quickly took the lead. After a period of steady running, the runners from Team A and Team B gradually created a gap, almost running shoulder to shoulder. However, at the 25 km mark, the runner from Team A suddenly began to feel unwell, slowing down noticeably. The runner from Team B seized the opportunity and caught up, eventually overtaking Team A to take the lead at the 30 km mark. As the race entered the second half, the Team A runner gave it their all, regained their rhythm, and caught up with Team B at the 35 km mark. The two were neck and neck in the final sprint, and with only 200 meters remaining, the Team A runner pushed through with determination and accelerated to cross the finish line first by a narrow margin, winning the tough race.", "dataset": "v-bench2", "category": "all"}
11
+ {"prompt_id": "v-bench2__all__p0010", "text": "The race began, and the runners quickly started. The Team A runner took the lead initially due to a powerful start. However, the Team B runner did not rush to chase but instead steadily adjusted their pace, ensuring they had more energy for the latter part of the race. By the third lap, the Team A runner began to tire, and the Team B runner gradually reduced the gap, eventually overtaking Team A in the fourth lap. The Team C runner, meanwhile, began to accelerate, and in the last two laps, with strong willpower, they overtook Team B and moved into the lead. In the final sprint, the Team A runner gritted their teeth and tried to close the gap, but with only 50 meters left, the Team C runner exploded with speed, crossing the finish line with a clear lead and winning the race.", "dataset": "v-bench2", "category": "all"}
12
+ {"prompt_id": "v-bench2__all__p0011", "text": "The match began, and Team A quickly found their rhythm. A series of fast attacks put pressure on Team B’s defense. Team A's forward made two three-pointers, quickly increasing the lead. After the first quarter, Team B adjusted their tactics, strengthened their defense, and gradually started to counterattack, narrowing the gap. In the third quarter, Team A’s key player was injured and forced to leave the game. Team B took advantage of this by speeding up their attacks, overtaking the score. In the fourth quarter, Team A's bench players caught up with a series of fast breaks, and in the final moments, a player from Team A calmly made a game-winning three-pointer from beyond the arc, securing a narrow 1-point victory.", "dataset": "v-bench2", "category": "all"}
13
+ {"prompt_id": "v-bench2__all__p0012", "text": "The match began, and Team A quickly gained the upper hand with several precise kills, taking a 11-6 lead. The Team B player stayed calm and gradually adjusted, finding ways to respond. In the second set, Team B improved their serving quality, scoring several powerful smashes to win a set, leveling the score at 1-1. In the deciding set, the Team A player showed signs of fatigue, but with precise net control and quick reflexes, they pulled ahead. In the crucial final point, a lightning-fast smash left Team B with no chance to return, and Team A won the match 21-19.", "dataset": "v-bench2", "category": "all"}
14
+ {"prompt_id": "v-bench2__all__p0013", "text": "A lion with the wings of an eagle, soaring through the sky with majestic ease.", "dataset": "v-bench2", "category": "all"}
15
+ {"prompt_id": "v-bench2__all__p0014", "text": "A giraffe with the scales of a fish, able to glide smoothly through the water.", "dataset": "v-bench2", "category": "all"}
16
+ {"prompt_id": "v-bench2__all__p0015", "text": "A wolf with the body of a horse, galloping across a vast plains with wild abandon.", "dataset": "v-bench2", "category": "all"}
17
+ {"prompt_id": "v-bench2__all__p0016", "text": "A bear with the antlers of a deer, roaming the forest with a regal presence.", "dataset": "v-bench2", "category": "all"}
18
+ {"prompt_id": "v-bench2__all__p0017", "text": "A cheetah with the shell of a tortoise, moving quickly but with a protective outer layer.", "dataset": "v-bench2", "category": "all"}
19
+ {"prompt_id": "v-bench2__all__p0018", "text": "A wooden toy is placed gently on the surface of a small bowl of water.", "dataset": "v-bench2", "category": "all"}
20
+ {"prompt_id": "v-bench2__all__p0019", "text": "A river changes from blue to brown.", "dataset": "v-bench2", "category": "all"}
21
+ {"prompt_id": "v-bench2__all__p0020", "text": "The leaves gradually change from red to green.", "dataset": "v-bench2", "category": "all"}
22
+ {"prompt_id": "v-bench2__all__p0021", "text": "A river changes from brown to blue.", "dataset": "v-bench2", "category": "all"}
23
+ {"prompt_id": "v-bench2__all__p0022", "text": "A car changes from white to red.", "dataset": "v-bench2", "category": "all"}
24
+ {"prompt_id": "v-bench2__all__p0023", "text": "A car changes from red to white.", "dataset": "v-bench2", "category": "all"}
25
+ {"prompt_id": "v-bench2__all__p0024", "text": "A dog is on the left of a table, then the dog runs to the front of the table.", "dataset": "v-bench2", "category": "all"}
26
+ {"prompt_id": "v-bench2__all__p0025", "text": "A dog is on the left of a sofa, then the dog runs to the front of the sofa.", "dataset": "v-bench2", "category": "all"}
27
+ {"prompt_id": "v-bench2__all__p0026", "text": "A dog is on the right of a table, then the dog runs to the left of the table.", "dataset": "v-bench2", "category": "all"}
28
+ {"prompt_id": "v-bench2__all__p0027", "text": "A dog is on the right of a sofa, then the dog runs to the front of the sofa.", "dataset": "v-bench2", "category": "all"}
29
+ {"prompt_id": "v-bench2__all__p0028", "text": "A dog is on the right of a rock, then the dog runs to the left of the rock.", "dataset": "v-bench2", "category": "all"}
30
+ {"prompt_id": "v-bench2__all__p0029", "text": "A dog is behind a chair, then the dog runs to the right of the chair.", "dataset": "v-bench2", "category": "all"}
31
+ {"prompt_id": "v-bench2__all__p0030", "text": "A man is doing yoga.", "dataset": "v-bench2", "category": "all"}
32
+ {"prompt_id": "v-bench2__all__p0031", "text": "A woman is doing yoga.", "dataset": "v-bench2", "category": "all"}
33
+ {"prompt_id": "v-bench2__all__p0032", "text": "people are doing yoga.", "dataset": "v-bench2", "category": "all"}
34
+ {"prompt_id": "v-bench2__all__p0033", "text": "A man is running.", "dataset": "v-bench2", "category": "all"}
35
+ {"prompt_id": "v-bench2__all__p0034", "text": "A woman is running.", "dataset": "v-bench2", "category": "all"}
36
+ {"prompt_id": "v-bench2__all__p0035", "text": "people are running.", "dataset": "v-bench2", "category": "all"}
37
+ {"prompt_id": "v-bench2__all__p0036", "text": "A man is walking.", "dataset": "v-bench2", "category": "all"}
38
+ {"prompt_id": "v-bench2__all__p0037", "text": "A woman is walking.", "dataset": "v-bench2", "category": "all"}
39
+ {"prompt_id": "v-bench2__all__p0038", "text": "people are walking.", "dataset": "v-bench2", "category": "all"}
40
+ {"prompt_id": "v-bench2__all__p0039", "text": "A man is dancing.", "dataset": "v-bench2", "category": "all"}
41
+ {"prompt_id": "v-bench2__all__p0040", "text": "A woman is dancing.", "dataset": "v-bench2", "category": "all"}
42
+ {"prompt_id": "v-bench2__all__p0041", "text": "A man is playing basketball.", "dataset": "v-bench2", "category": "all"}
43
+ {"prompt_id": "v-bench2__all__p0042", "text": "A woman is playing basketball.", "dataset": "v-bench2", "category": "all"}
44
+ {"prompt_id": "v-bench2__all__p0043", "text": "One person hands a cup of water to another.", "dataset": "v-bench2", "category": "all"}
45
+ {"prompt_id": "v-bench2__all__p0044", "text": "One person passes a ball to another.", "dataset": "v-bench2", "category": "all"}
46
+ {"prompt_id": "v-bench2__all__p0045", "text": "Two people shake hands.", "dataset": "v-bench2", "category": "all"}
47
+ {"prompt_id": "v-bench2__all__p0046", "text": "One person ties the shoelaces of another person.", "dataset": "v-bench2", "category": "all"}
48
+ {"prompt_id": "v-bench2__all__p0047", "text": "One person opens the door for another person.", "dataset": "v-bench2", "category": "all"}
49
+ {"prompt_id": "v-bench2__all__p0048", "text": "Two people exchange a book.", "dataset": "v-bench2", "category": "all"}
50
+ {"prompt_id": "v-bench2__all__p0049", "text": "One person puts a coat on another person.", "dataset": "v-bench2", "category": "all"}
51
+ {"prompt_id": "v-bench2__all__p0050", "text": "One person places a chair for another to sit in.", "dataset": "v-bench2", "category": "all"}
52
+ {"prompt_id": "v-bench2__all__p0051", "text": "An orange dog is running.", "dataset": "v-bench2", "category": "all"}
53
+ {"prompt_id": "v-bench2__all__p0052", "text": "Two orange dogs are running.", "dataset": "v-bench2", "category": "all"}
54
+ {"prompt_id": "v-bench2__all__p0053", "text": "A brown dog is on the left of an apple, then the dog moves to the right of the apple.", "dataset": "v-bench2", "category": "all"}
55
+ {"prompt_id": "v-bench2__all__p0054", "text": "An orange cat is running.", "dataset": "v-bench2", "category": "all"}
56
+ {"prompt_id": "v-bench2__all__p0055", "text": "Equal amounts of yellow and blue paint are rapidly combined, with the mixture being vigorously stirred until fully blended.", "dataset": "v-bench2", "category": "all"}
57
+ {"prompt_id": "v-bench2__all__p0056", "text": "Equal amounts of black and white paint are rapidly combined, with the mixture being vigorously stirred until fully blended.", "dataset": "v-bench2", "category": "all"}
58
+ {"prompt_id": "v-bench2__all__p0057", "text": "Equal amounts of white and black paint are rapidly combined, with the mixture being vigorously stirred until fully blended.", "dataset": "v-bench2", "category": "all"}
59
+ {"prompt_id": "v-bench2__all__p0058", "text": "Equal amounts of red and purple paint are rapidly combined, with the mixture being vigorously stirred until fully blended.", "dataset": "v-bench2", "category": "all"}
60
+ {"prompt_id": "v-bench2__all__p0059", "text": "A bowl of soup is tilted in the space station, with the liquid slowly spreading in all directions.", "dataset": "v-bench2", "category": "all"}
61
+ {"prompt_id": "v-bench2__all__p0060", "text": "A jar of peanut butter is opened in the space station, with the viscous liquid slowly dispersing.", "dataset": "v-bench2", "category": "all"}
62
+ {"prompt_id": "v-bench2__all__p0061", "text": "A container of cream is opened in the space station, with the liquid slowly dispersing into the air.", "dataset": "v-bench2", "category": "all"}
63
+ {"prompt_id": "v-bench2__all__p0062", "text": "A cup of tea is carefully tilted in the space station, and the liquid floats in various directions.", "dataset": "v-bench2", "category": "all"}
64
+ {"prompt_id": "v-bench2__all__p0063", "text": "A bottle of ketchup is gently squeezed in the space station, with the thick liquid spreading into the environment.", "dataset": "v-bench2", "category": "all"}
65
+ {"prompt_id": "v-bench2__all__p0064", "text": "A bottle of water is opened in the space station, and the water starts to float out in irregular shapes.", "dataset": "v-bench2", "category": "all"}
66
+ {"prompt_id": "v-bench2__all__p0065", "text": "A person is sitting on the couch, then suddenly they get up and start sweeping the floor.", "dataset": "v-bench2", "category": "all"}
67
+ {"prompt_id": "v-bench2__all__p0066", "text": "A dog is playing with a ball, then it suddenly starts lying down on the carpet.", "dataset": "v-bench2", "category": "all"}
68
+ {"prompt_id": "v-bench2__all__p0067", "text": "A person is cooking dinner, then they suddenly start organizing the pantry.", "dataset": "v-bench2", "category": "all"}
69
+ {"prompt_id": "v-bench2__all__p0068", "text": "A cat is watching birds through the window, then it suddenly starts grooming itself.", "dataset": "v-bench2", "category": "all"}
70
+ {"prompt_id": "v-bench2__all__p0069", "text": "A person is typing on a keyboard, then they suddenly get up and start making the bed.", "dataset": "v-bench2", "category": "all"}
71
+ {"prompt_id": "v-bench2__all__p0070", "text": "A horse is trotting in the field, then it suddenly starts drinking from a stream.", "dataset": "v-bench2", "category": "all"}
72
+ {"prompt_id": "v-bench2__all__p0071", "text": "A person is drinking a glass of water, then they suddenly start cleaning the windows.", "dataset": "v-bench2", "category": "all"}
73
+ {"prompt_id": "v-bench2__all__p0072", "text": "A dog is sitting in the yard, then it suddenly starts running in circles.", "dataset": "v-bench2", "category": "all"}
74
+ {"prompt_id": "v-bench2__all__p0073", "text": "A person is slurping noodles from a steaming bowl.", "dataset": "v-bench2", "category": "all"}
75
+ {"prompt_id": "v-bench2__all__p0074", "text": "A person is eating hamburger.", "dataset": "v-bench2", "category": "all"}
76
+ {"prompt_id": "v-bench2__all__p0075", "text": "A person is eating ice cream.", "dataset": "v-bench2", "category": "all"}
77
+ {"prompt_id": "v-bench2__all__p0076", "text": "A person is drinking coffee from a cup.", "dataset": "v-bench2", "category": "all"}
78
+ {"prompt_id": "v-bench2__all__p0077", "text": "A person is spreading butter on toast.", "dataset": "v-bench2", "category": "all"}
79
+ {"prompt_id": "v-bench2__all__p0078", "text": "A person is eating spaghetti with a fork.", "dataset": "v-bench2", "category": "all"}
80
+ {"prompt_id": "v-bench2__all__p0079", "text": "A person is biting into an apple.", "dataset": "v-bench2", "category": "all"}
81
+ {"prompt_id": "v-bench2__all__p0080", "text": "A person is peeling a banana.", "dataset": "v-bench2", "category": "all"}
82
+ {"prompt_id": "v-bench2__all__p0081", "text": "The camera orbits around. Castle, the camera circles around.", "dataset": "v-bench2", "category": "all"}
83
+ {"prompt_id": "v-bench2__all__p0082", "text": "The camera orbits around. Volcano, the camera circles around.", "dataset": "v-bench2", "category": "all"}
84
+ {"prompt_id": "v-bench2__all__p0083", "text": "The camera orbits around. Statue, the camera circles around.", "dataset": "v-bench2", "category": "all"}
85
+ {"prompt_id": "v-bench2__all__p0084", "text": "The camera orbits around. Clock Tower, the camera circles around.", "dataset": "v-bench2", "category": "all"}
86
+ {"prompt_id": "v-bench2__all__p0085", "text": "The camera orbits around. Playground, the camera circles around.", "dataset": "v-bench2", "category": "all"}
87
+ {"prompt_id": "v-bench2__all__p0086", "text": "A timelapse captures the transformation of water in an untextured bottle as the temperature significantly drops below 0°C.", "dataset": "v-bench2", "category": "all"}
88
+ {"prompt_id": "v-bench2__all__p0087", "text": "A timelapse captures the transformation of a river as the temperature significantly drops below 0°C.", "dataset": "v-bench2", "category": "all"}
89
+ {"prompt_id": "v-bench2__all__p0088", "text": "A timelapse captures the transformation of juice in an untextured bottle as the temperature significantly drops below 0°C.", "dataset": "v-bench2", "category": "all"}
90
+ {"prompt_id": "v-bench2__all__p0089", "text": "A timelapse captures the transformation of milk in an untextured bottle as the temperature significantly drops below 0°C.", "dataset": "v-bench2", "category": "all"}
model_display.py CHANGED
@@ -88,6 +88,31 @@ MODEL_DISPLAY_NAMES = {
88
  "p_image_2_ideogram_high_2k": "P-Image-Ideogram High 2K",
89
  "P-Image-Ideogram (High)": "P-Image-Ideogram High",
90
  "p_image_2_ideogram_very_high_1k": "P-Image-Ideogram Very High 1K",
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
91
  # Others overlapping P-Bench
92
  "z_image": "Z-Image",
93
  "glm_image": "GLM-Image",
@@ -185,6 +210,35 @@ MODEL_DISPLAY_NAMES = {
185
  }
186
 
187
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
188
  def _prettify_snake_case(model_id: str) -> str:
189
  parts = [part for part in str(model_id).split("_") if part]
190
  pretty = []
@@ -209,7 +263,10 @@ def display_model_name(model_id) -> str:
209
  return ""
210
  if raw in MODEL_DISPLAY_NAMES:
211
  return MODEL_DISPLAY_NAMES[raw]
 
 
 
212
  # Already a human label (spaces / punctuation) — keep as-is.
213
- if re.search(r"[\s.\[\]()]", raw):
214
  return raw
215
  return _prettify_snake_case(raw)
 
88
  "p_image_2_ideogram_high_2k": "P-Image-Ideogram High 2K",
89
  "P-Image-Ideogram (High)": "P-Image-Ideogram High",
90
  "p_image_2_ideogram_very_high_1k": "P-Image-Ideogram Very High 1K",
91
+ # P-Video-Edit
92
+ "P-Video-Edit": "P-Video-Edit",
93
+ "P-Video-Edit Draft": "P-Video-Edit Draft",
94
+ "P-Video Edit Final": "P-Video-Edit",
95
+ "P-Video Edit Final (draft)": "P-Video-Edit Draft",
96
+ "p_video_edit_preview__replicate_final": "P-Video-Edit",
97
+ "p_video_edit_preview__replicate_final__draft": "P-Video-Edit Draft",
98
+ # Text-to-video leaderboard
99
+ "gemini_omni_1_1_flash": "Gemini Omni 1.1 Flash",
100
+ "grok_imagine_video": "Grok Imagine Video",
101
+ "grok_imagine_video_v1_5": "Grok Imagine Video 1.5",
102
+ "minimax_h3": "MiniMax H3",
103
+ "minimax_h3_max": "MiniMax H3 Max",
104
+ "minimax_h3_max_turbo": "MiniMax H3 Max Turbo",
105
+ "minimax_h3_max_turbo__prompt_expansion_mode_balanced": "MiniMax H3 Max Turbo",
106
+ "seedance_2_5_turbo": "Seedance 2.5 Turbo",
107
+ "veo_3_1_lite": "Veo 3.1 Lite",
108
+ # Video-to-video leaderboard
109
+ "gemini_omni_flash_edit__fal": "Gemini Omni Flash Edit",
110
+ "grok_imagine_video__replicate": "Grok Imagine Video",
111
+ "happyhorse_1_0__wavespeed": "HappyHorse 1.0",
112
+ "lucy_edit_pro__fal": "Lucy Edit Pro",
113
+ "minimax_h3_reference_to_video__fal": "MiniMax H3 Reference-to-Video",
114
+ "seedance_2_5_video_edit_turbo__wavespeed": "Seedance 2.5 Video Edit Turbo",
115
+ "wan_2_7_video_edit__wavespeed": "Wan 2.7 Video Edit",
116
  # Others overlapping P-Bench
117
  "z_image": "Z-Image",
118
  "glm_image": "GLM-Image",
 
210
  }
211
 
212
 
213
+ def _p_video_2_display_name(model_id: str):
214
+ """Turn p_video_2 variant ids into P-Video-2 Draft 720p labels."""
215
+ raw = str(model_id).strip()
216
+ if raw != "p_video_2" and not raw.startswith("p_video_2__"):
217
+ return None
218
+ if raw == "p_video_2":
219
+ return "P-Video-2"
220
+
221
+ draft = False
222
+ upsample = None
223
+ resolution = None
224
+ for part in raw.split("__")[1:]:
225
+ if part.startswith("draft_"):
226
+ draft = part.endswith("true")
227
+ elif part.startswith("prompt_upsampling_"):
228
+ upsample = part.endswith("true")
229
+ elif part.startswith("resolution_"):
230
+ resolution = part[len("resolution_") :]
231
+
232
+ label = "P-Video-2"
233
+ if draft:
234
+ label += " Draft"
235
+ if resolution:
236
+ label += f" {resolution}"
237
+ if upsample is False:
238
+ label += " (no prompt upsampling)"
239
+ return label
240
+
241
+
242
  def _prettify_snake_case(model_id: str) -> str:
243
  parts = [part for part in str(model_id).split("_") if part]
244
  pretty = []
 
263
  return ""
264
  if raw in MODEL_DISPLAY_NAMES:
265
  return MODEL_DISPLAY_NAMES[raw]
266
+ p_video_2 = _p_video_2_display_name(raw)
267
+ if p_video_2:
268
+ return p_video_2
269
  # Already a human label (spaces / punctuation) — keep as-is.
270
+ if re.search(r"[\s.\[\]()-]", raw):
271
  return raw
272
  return _prettify_snake_case(raw)
ui.py CHANGED
@@ -25,19 +25,43 @@ MAX_COMPARE_PROMPTS = 8
25
  MAX_PARETO_METRICS = 8
26
  _PARETO_SLOT_COUNT = 1 + MAX_PARETO_METRICS * 8
27
  _PARETO_PRICE_COLUMN = "Price / Image (USD)"
 
 
28
  _PARETO_TIME_COLUMN = "Min Generation Time (s)"
 
 
 
 
 
 
 
 
 
 
29
  _PARETO_SCALE_CHOICES = [
30
  ("Log", "Logarithmic"),
31
  ("Linear", "Linear"),
32
  ]
33
  _PARETO_SCALE_VALUES = {value for _, value in _PARETO_SCALE_CHOICES}
34
  _PARETO_SCALE_DEFAULT = "Logarithmic"
 
 
 
35
 
36
  TAB_LEADERBOARDS = "leaderboards"
37
  TAB_PARETO = "pareto"
38
  TAB_SAMPLES = "samples"
39
  TAB_ABOUT = "about"
40
 
 
 
 
 
 
 
 
 
 
41
  _MODEL_CHOICES_CACHE = {}
42
  _VIEW_EVENTS = {
43
  "show_progress": "hidden",
@@ -49,21 +73,24 @@ _VIEW_EVENTS = {
49
  ABOUT_OVERVIEW_CONTENT = """
50
  # About P-Bench
51
 
52
- P-Bench compares **text-to-image models**, including optimized or accelerated
53
- endpoints, on **quality, speed, and price**. Each view is a **dataset** scored
54
- with a **metric**, written as `Dataset | Metric`. There is no single score
55
- across P-Bench.
56
 
57
  ## How to read it
58
 
59
- 1. Pick a **dataset** and a **metric**.
 
60
  2. **Leaderboards**: ranked by that metric. Price and generation time sit in
61
  the same table when the source publishes them.
62
  3. **Pareto plots**: mark models that are not beaten on both higher score
63
  and lower price (or time). Only datasets with price or generation time
64
  can open this tab (not Arena AI).
65
  4. **Samples**: the same prompts, side by side. Only for datasets we
66
- generated (Qwen Image Dataset and OneIG Alignment Dataset).
 
 
67
 
68
  ## How a score is made
69
 
@@ -82,6 +109,24 @@ prompt suites, so samples are not shown.
82
 
83
  ## Current datasets
84
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
85
  ### Qwen Image Dataset
86
  100 prompts from the 1,000-prompt Qwen Image Bench set, sampled for coverage
87
  across its fine-grained (L3) categories. Metrics include Datapoint Elo,
@@ -124,12 +169,17 @@ ABOUT_DETAILS_CONTENT = """
124
  - **Arena Elo**: Elo published by Arena AI on their own dataset, plus
125
  category Elos (branding, 3D, cartoon/anime, photorealistic, art, portraits,
126
  text rendering).
127
- - **Generation time**: median and minimum generation time in seconds, as
128
- reported in the evaluation table. This is not a p95, and we do not state
129
- warm vs cold or concurrent load. Not available for Arena AI.
130
- - **Price**: USD per image in the evaluation table. We do not state list
131
- price vs amount paid, or whether failed generations are included. Not
 
132
  available for Arena AI.
 
 
 
 
133
 
134
  Scores from different datasets or metrics are **not interchangeable**. A high
135
  OneIG alignment score is not the same quantity as a Datapoint Elo. Compare
@@ -143,14 +193,21 @@ models *within* a Dataset | Metric view.
143
  - **Prompt counts:** OneIG Alignment uses 100 anime, 100 human, and 99 object
144
  prompts (299 total). Qwen Image Dataset uses 100 prompts sampled from the
145
  1,000-prompt pool for roughly even coverage of its fine-grained (L3)
146
- categories. Artificial Analysis and Arena AI use their own private prompt
147
- sets.
 
 
 
148
  - **Generation (Qwen and OneIG):** one image per prompt per endpoint when
149
  the run exists. Default resolution is 1024×1024. Exceptions: FLUX 1.1 Pro
150
  Ultra at 2K, FLUX 2 Flex at 1008×1008, and any endpoint labeled 2K. The
151
  seed is derived from the prompt, so every model gets the same seed for the
152
  same prompt. Steps, CFG, prompt rewrite, and safety filters follow each
153
  endpoint's default. This does not describe Artificial Analysis or Arena AI.
 
 
 
 
154
  - **Datapoint (Qwen and OneIG):** every model pair is compared on every
155
  prompt, with 10 votes per battle.
156
  - **Rapidata (Qwen and OneIG):** prompts longer than 400 characters are
@@ -222,7 +279,7 @@ def render_header():
222
  </svg>
223
  </button>
224
  </div>
225
- <p class="app-header-tagline">Compare text-to-image models on quality, speed, and price</p>
226
  </header>
227
  """,
228
  padding=False,
@@ -237,10 +294,42 @@ def _item(items, item_id):
237
  return items[0] if items else None
238
 
239
 
240
- def _dataset_choices(datasets, *, require_samples=False, require_pareto=False):
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
241
  return [
242
  (dataset["name"], dataset["id"])
243
- for dataset in datasets
244
  if (not require_samples or dataset.get("samples"))
245
  and (not require_pareto or _dataset_has_pareto(datasets, dataset["id"]))
246
  ]
@@ -256,20 +345,72 @@ def _sample_model_ids(datasets, dataset_id):
256
  samples = dataset.get("samples") if dataset else None
257
  if not samples:
258
  return set()
259
- return set(samples.get("models") or [])
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
260
 
261
 
262
  def _dataset_has_pareto(datasets, dataset_id):
263
  dataset = _item(datasets, dataset_id)
264
- columns = getattr(dataset.get("data") if dataset else None, "columns", [])
265
- return _PARETO_PRICE_COLUMN in columns or _PARETO_TIME_COLUMN in columns
 
 
 
266
 
267
 
268
- def _dataset_dropdown_update(datasets, tab, dataset_id):
269
  """Limit the dataset list to what the current tab can show."""
 
 
270
  return gr.update(
271
  choices=_dataset_choices(
272
  datasets,
 
273
  require_samples=tab == TAB_SAMPLES
274
  and _dataset_has_samples(datasets, dataset_id),
275
  require_pareto=tab == TAB_PARETO
@@ -375,9 +516,11 @@ _LEADERBOARD_IDENTITY_COLUMNS = [
375
  "Optimized",
376
  ]
377
  _LEADERBOARD_META_COLUMNS = [
 
378
  "Median Generation Time (s)",
379
  "Min Generation Time (s)",
380
  "Price / Image (USD)",
 
381
  "Evaluation Date (UTC)",
382
  "Date",
383
  ]
@@ -604,7 +747,9 @@ def _display_label(column):
604
  "Arena Text Rendering Elo": "Text Rendering",
605
  "Median Generation Time (s)": "Median generation time",
606
  "Min Generation Time (s)": "Min generation time",
 
607
  "Price / Image (USD)": "Price per image",
 
608
  "Evaluation Date (UTC)": "Date",
609
  "Date": "Date",
610
  }
@@ -679,6 +824,23 @@ def _applied_key(view_state):
679
  )
680
 
681
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
682
  def _build_pareto_figure(
683
  data,
684
  score_column,
@@ -703,6 +865,8 @@ def _build_pareto_figure(
703
 
704
  dominated = scatter.loc[[not flag for flag in on_frontier]].copy()
705
  frontier = scatter.loc[on_frontier].sort_values(x_column).copy()
 
 
706
  if not dominated.empty:
707
  dominated["Model"] = dominated["Model"].map(display_model_name)
708
  if not frontier.empty:
@@ -723,10 +887,11 @@ def _build_pareto_figure(
723
  name="Below frontier",
724
  text=dominated["Model"],
725
  hovertemplate=hover,
 
726
  marker={
727
  "size": 9,
728
- "color": "#d8b4fe",
729
- "opacity": 0.8,
730
  "line": {"width": 0},
731
  },
732
  )
@@ -740,14 +905,51 @@ def _build_pareto_figure(
740
  name="On frontier",
741
  text=frontier["Model"],
742
  hovertemplate=hover,
743
- line={"color": "#69a45c", "width": 2.5},
 
744
  marker={
745
  "size": 12,
746
- "color": "#69a45c",
747
- "line": {"width": 1.5, "color": "#86c077"},
748
  },
749
  )
750
  )
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
751
 
752
  score_label = _display_label(score_column)
753
  fig.update_layout(
@@ -887,23 +1089,43 @@ def _pareto_pair(
887
  if data is None or not score_column or score_column not in data.columns:
888
  return None, score_missing, None, score_missing
889
 
 
 
 
 
 
 
 
890
  price_fig, price_message = _pareto_axis(
891
  data,
892
  score_column,
893
- _PARETO_PRICE_COLUMN,
894
- "Price per image (USD)",
895
- "Price per image isn't available for this dataset.",
896
  "No models have both a score and a price for this metric.",
897
  x_hover_prefix="$",
898
  x_axis_type=_pareto_axis_type(price_scale),
899
  )
 
 
 
 
 
 
 
 
 
 
 
 
 
900
  time_fig, time_message = _pareto_axis(
901
  data,
902
  score_column,
903
- _PARETO_TIME_COLUMN,
904
- "Min generation time (s)",
905
- "Min generation time isn't available for this dataset.",
906
- "No models have both a score and a min generation time for this metric.",
907
  x_hover_suffix="s",
908
  x_axis_type=_pareto_axis_type(latency_scale),
909
  )
@@ -911,29 +1133,29 @@ def _pareto_pair(
911
 
912
 
913
  def _pareto_dataset_message(data):
914
- has_price = data is not None and _PARETO_PRICE_COLUMN in data.columns
915
- has_time = data is not None and _PARETO_TIME_COLUMN in data.columns
916
  if has_price or has_time:
917
  return None
918
  return (
919
- "Price per image and min generation time aren't available for "
920
  "this dataset, so these plots can't be drawn."
921
  )
922
 
923
 
924
  def _pareto_slot_note(price_fig, price_message, time_fig, time_message, data):
925
- has_price = data is not None and _PARETO_PRICE_COLUMN in data.columns
926
- has_time = data is not None and _PARETO_TIME_COLUMN in data.columns
927
  notes = []
928
  if has_price and not has_time:
929
  notes.append(
930
- "Min generation time isn't available for this dataset, so only "
931
  "price vs score is shown."
932
  )
933
  elif has_time and not has_price:
934
  notes.append(
935
- "Price per image isn't available for this dataset, so only min "
936
- "generation time vs score is shown."
937
  )
938
  if price_fig is None and has_price:
939
  notes.append(price_message)
@@ -954,8 +1176,8 @@ def _pareto_slot_updates(
954
  score_columns = [column for column in (score_columns or []) if column]
955
  price_scales = _normalize_pareto_scales(price_scales)
956
  time_scales = _normalize_pareto_scales(time_scales)
957
- has_price = data is not None and _PARETO_PRICE_COLUMN in data.columns
958
- has_time = data is not None and _PARETO_TIME_COLUMN in data.columns
959
  dataset_note = _pareto_dataset_message(data)
960
  updates = [_pareto_note_update(dataset_note)]
961
  hide_all_slots = not has_price and not has_time
@@ -1034,15 +1256,48 @@ def _samples_html(samples, selected_models, num_prompts, seed=0):
1034
  return _pareto_unavailable_html(
1035
  "Samples aren't available for this dataset."
1036
  )
1037
- images = samples.get("images", {})
1038
- models = [model for model in (selected_models or []) if model in images]
 
 
 
1039
  if not models:
1040
- models = (samples.get("models") or [])[:2]
1041
  return _build_compare_samples_html(samples, models, num_prompts, seed)
1042
 
1043
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1044
  def _build_compare_samples_html(samples, selected_models, num_prompts, seed=0):
1045
  selected_models = list(selected_models or [])[:MAX_COMPARE_MODELS]
 
 
 
1046
 
1047
  if not selected_models:
1048
  return (
@@ -1053,7 +1308,7 @@ def _build_compare_samples_html(samples, selected_models, num_prompts, seed=0):
1053
 
1054
  shared_prompt_ids = None
1055
  for model in selected_models:
1056
- model_prompt_ids = set(samples["images"][model])
1057
  shared_prompt_ids = (
1058
  model_prompt_ids
1059
  if shared_prompt_ids is None
@@ -1073,29 +1328,31 @@ def _build_compare_samples_html(samples, selected_models, num_prompts, seed=0):
1073
  rng.shuffle(prompt_pool)
1074
  chosen = prompt_pool[: max(1, min(int(num_prompts), len(prompt_pool)))]
1075
 
1076
- columns = len(selected_models)
1077
  blocks = []
1078
  for index, prompt_id in enumerate(chosen, start=1):
1079
  prompt_text = escape(samples["prompts"].get(prompt_id, ""))
1080
  cells = []
 
 
 
 
 
 
 
1081
  for model in selected_models:
1082
- image_url = escape(samples["images"][model][prompt_id], quote=True)
1083
  cells.append(
1084
- f"""
1085
- <div class="compare-cell">
1086
- <div class="compare-model-label">{escape(display_model_name(model))}</div>
1087
- <a href="{image_url}" target="_blank" rel="noopener noreferrer">
1088
- <img src="{image_url}" alt="{escape(display_model_name(model))} sample" loading="lazy" />
1089
- </a>
1090
- </div>
1091
- """
1092
  )
 
1093
  blocks.append(
1094
  f"""
1095
  <div class="compare-prompt-block">
1096
  <div class="compare-prompt-meta">
1097
  <span>Prompt {index}</span>
1098
- <span>{escape(prompt_id)}</span>
1099
  </div>
1100
  <p class="compare-prompt-text">{prompt_text}</p>
1101
  <div class="compare-row" style="grid-template-columns: repeat({columns}, minmax(0, 1fr));">
@@ -1123,9 +1380,19 @@ def _filter_row(datasets, metrics, default_dataset_id, default_metric_id=None):
1123
  metric_id = _coerce_metric(
1124
  datasets, metrics, default_dataset_id, default_metric_id
1125
  )
 
1126
  with gr.Row(elem_classes="view-filters"):
 
 
 
 
 
 
 
 
 
1127
  dataset_dd = gr.Dropdown(
1128
- choices=_dataset_choices(datasets),
1129
  value=default_dataset_id,
1130
  label="Dataset",
1131
  type="value",
@@ -1157,7 +1424,7 @@ def _filter_row(datasets, metrics, default_dataset_id, default_metric_id=None):
1157
  min_width=180,
1158
  elem_classes="filter-chips",
1159
  )
1160
- return dataset_dd, metric_dd, models_dd
1161
 
1162
 
1163
  def render_image_workspace(datasets, metrics, default_dataset_id, default_metric_id):
@@ -1172,15 +1439,17 @@ def render_image_workspace(datasets, metrics, default_dataset_id, default_metric
1172
  with gr.Column(elem_classes="workspace-filters") as filters_host:
1173
  gr.Markdown(
1174
  "<p class='filter-help'>"
1175
- "These filters apply to Leaderboards, Pareto plots, and Samples. "
1176
- "On Samples, only datasets and models we have generations for "
1177
- "are listed. On Pareto plots, only datasets with price or "
1178
- "generation time are listed. Search in Models, or leave it "
1179
- "empty to include every model."
 
 
1180
  "</p>",
1181
  elem_classes="filter-help-host",
1182
  )
1183
- dataset_dd, metric_dd, models_dd = _filter_row(
1184
  datasets, metrics, default_dataset_id, None
1185
  )
1186
  with gr.Tabs(elem_classes="main-tabs") as main_tabs:
@@ -1306,7 +1575,7 @@ def render_image_workspace(datasets, metrics, default_dataset_id, default_metric
1306
  elem_classes="pareto-col",
1307
  ) as slot_time_col:
1308
  slot_time_scale = _pareto_plot_heading(
1309
- "Min generation time vs score"
1310
  )
1311
  slot_time = gr.Plot(
1312
  value=None,
@@ -1342,7 +1611,8 @@ def render_image_workspace(datasets, metrics, default_dataset_id, default_metric
1342
  with gr.Column(visible=bool(initial_samples)) as samples_panel:
1343
  gr.Markdown(
1344
  f"<p class='view-help'>"
1345
- f"The same prompts, side by side. Select up to "
 
1346
  f"<strong>{MAX_COMPARE_MODELS}</strong> models above, or leave "
1347
  f"Models empty for two defaults."
1348
  f"</p>",
@@ -1460,8 +1730,14 @@ def render_image_workspace(datasets, metrics, default_dataset_id, default_metric
1460
  ):
1461
  prev = dict(view_state or {})
1462
  extras = extras or {}
 
 
 
 
1463
  return {
1464
  "dataset_id": dataset_id,
 
 
1465
  "metric_id": metric_id,
1466
  "models": list(models or []),
1467
  "current_tab": tab,
@@ -1599,13 +1875,14 @@ def render_image_workspace(datasets, metrics, default_dataset_id, default_metric
1599
  tab = view_state.get("current_tab") or TAB_LEADERBOARDS
1600
  selected_raw = _normalize_metric_ids(metric_id)
1601
  incoming_models = list(models or [])
1602
- dataset_changed = source == "dataset" and dataset_id != view_state.get(
 
1603
  "dataset_id"
1604
  )
1605
  can_pareto = _dataset_has_pareto(datasets, dataset_id)
1606
  can_samples = _dataset_has_samples(datasets, dataset_id)
1607
  selected_tab = tab
1608
- if source == "dataset":
1609
  if tab == TAB_SAMPLES and not can_samples:
1610
  selected_tab = TAB_LEADERBOARDS
1611
  elif tab == TAB_PARETO and not can_pareto:
@@ -1614,7 +1891,7 @@ def render_image_workspace(datasets, metrics, default_dataset_id, default_metric
1614
  dataset_id,
1615
  metric_id,
1616
  models,
1617
- clear_metric=dataset_changed,
1618
  require_samples=selected_tab == TAB_SAMPLES,
1619
  )
1620
  dataset_id, metric_id, models = synced[:3]
@@ -1659,7 +1936,7 @@ def render_image_workspace(datasets, metrics, default_dataset_id, default_metric
1659
  list(optimized_value or []),
1660
  )
1661
  extra_updates = None
1662
- if source == "dataset":
1663
  view = resolve_view(datasets, metrics, dataset_id, metric_id)
1664
  extra_updates = _leaderboard_extras(
1665
  view["data"] if view else None,
@@ -1712,6 +1989,67 @@ def render_image_workspace(datasets, metrics, default_dataset_id, default_metric
1712
  ),
1713
  }
1714
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1715
  def on_dataset(
1716
  dataset_id,
1717
  metric_id,
@@ -2061,9 +2399,12 @@ def render_image_workspace(datasets, metrics, default_dataset_id, default_metric
2061
  handler.__name__ = f"on_tab_{tab}"
2062
  return handler
2063
 
 
2064
  view_state = gr.State(
2065
  {
2066
  "dataset_id": default_dataset_id,
 
 
2067
  "metric_id": None,
2068
  "models": [],
2069
  "current_tab": TAB_LEADERBOARDS,
@@ -2145,6 +2486,12 @@ def render_image_workspace(datasets, metrics, default_dataset_id, default_metric
2145
  main_tabs,
2146
  view_state,
2147
  ]
 
 
 
 
 
 
2148
  dataset_dd.change(
2149
  on_dataset,
2150
  inputs=filter_inputs,
 
25
  MAX_PARETO_METRICS = 8
26
  _PARETO_SLOT_COUNT = 1 + MAX_PARETO_METRICS * 8
27
  _PARETO_PRICE_COLUMN = "Price / Image (USD)"
28
+ _PARETO_VIDEO_PRICE_COLUMN = "Price / Second of Video (USD)"
29
+ _PARETO_PRICE_COLUMNS = (_PARETO_PRICE_COLUMN, _PARETO_VIDEO_PRICE_COLUMN)
30
  _PARETO_TIME_COLUMN = "Min Generation Time (s)"
31
+ _PARETO_VIDEO_TIME_COLUMN = "Pareto Time / Output Video Second (s)"
32
+ _PARETO_TIME_COLUMNS = (_PARETO_VIDEO_TIME_COLUMN, _PARETO_TIME_COLUMN)
33
+ _PARETO_PRICE_TITLES = {
34
+ _PARETO_PRICE_COLUMN: "Price per image (USD)",
35
+ _PARETO_VIDEO_PRICE_COLUMN: "Price per second of video (USD)",
36
+ }
37
+ _PARETO_TIME_TITLES = {
38
+ _PARETO_TIME_COLUMN: "Min generation time (s)",
39
+ _PARETO_VIDEO_TIME_COLUMN: "Generation time per second of video",
40
+ }
41
  _PARETO_SCALE_CHOICES = [
42
  ("Log", "Logarithmic"),
43
  ("Linear", "Linear"),
44
  ]
45
  _PARETO_SCALE_VALUES = {value for _, value in _PARETO_SCALE_CHOICES}
46
  _PARETO_SCALE_DEFAULT = "Logarithmic"
47
+ _PARETO_PRUNA_COLOR = "#c084fc"
48
+ _PARETO_OTHER_COLOR = "#9aa3b5"
49
+ _PARETO_FRONTIER_OUTLINE = "#3fa87e"
50
 
51
  TAB_LEADERBOARDS = "leaderboards"
52
  TAB_PARETO = "pareto"
53
  TAB_SAMPLES = "samples"
54
  TAB_ABOUT = "about"
55
 
56
+ MODALITY_TEXT_TO_VIDEO = "text_to_video"
57
+ MODALITY_VIDEO_TO_VIDEO = "video_to_video"
58
+ MODALITY_TEXT_TO_IMAGE = "text_to_image"
59
+ MODALITY_CHOICES = [
60
+ ("Text to Video", MODALITY_TEXT_TO_VIDEO),
61
+ ("Video to Video", MODALITY_VIDEO_TO_VIDEO),
62
+ ("Text to Image", MODALITY_TEXT_TO_IMAGE),
63
+ ]
64
+
65
  _MODEL_CHOICES_CACHE = {}
66
  _VIEW_EVENTS = {
67
  "show_progress": "hidden",
 
73
  ABOUT_OVERVIEW_CONTENT = """
74
  # About P-Bench
75
 
76
+ P-Bench compares **text-to-video**, **video-to-video**, and **text-to-image**
77
+ models, including optimized or accelerated endpoints, on **quality, speed,
78
+ and price**. Each view is a **dataset** scored with a **metric**, written as
79
+ `Dataset | Metric`. There is no single score across P-Bench.
80
 
81
  ## How to read it
82
 
83
+ 1. Pick a **model type** (Text to Video, Video to Video, or Text to Image),
84
+ then a **dataset** and a **metric**.
85
  2. **Leaderboards**: ranked by that metric. Price and generation time sit in
86
  the same table when the source publishes them.
87
  3. **Pareto plots**: mark models that are not beaten on both higher score
88
  and lower price (or time). Only datasets with price or generation time
89
  can open this tab (not Arena AI).
90
  4. **Samples**: the same prompts, side by side. Only for datasets we
91
+ generated (VBench-2.0 Dataset, Qwen Image Dataset, OneIG Alignment
92
+ Dataset, and the Pruna Internal Video-Edit Benchmark). Video-edit
93
+ samples show the source clip first, then each model's edit.
94
 
95
  ## How a score is made
96
 
 
109
 
110
  ## Current datasets
111
 
112
+ ### VBench-2.0 Dataset
113
+ VBench-2.0 prompts, comparing P-Video-2 variants with Fal-hosted models.
114
+ Quality is Datapoint Elo and Rapidata Elo from pairwise preference. Price
115
+ is USD per second of output video. Time per second of video is Fal wall
116
+ time, except Pruna models which use model execution time. Samples are
117
+ available.
118
+
119
+ ### Pruna Internal Video-Edit Benchmark
120
+ Pruna's internal video-to-video editing benchmark, collected by our
121
+ research engineers. It combines prompts from public video-editing
122
+ benchmarks with use-case examples we gathered for advertisement,
123
+ e-commerce, real estate, concept art, and similar work. The suite also
124
+ covers camera-angle and movement changes, lighting, and text in video
125
+ (altering, adding, or removing it). Quality is Datapoint Elo from
126
+ pairwise preference. Price is USD per second of output video;
127
+ generation time is wall time per second of output video. Samples show
128
+ the source clip beside each model's edit.
129
+
130
  ### Qwen Image Dataset
131
  100 prompts from the 1,000-prompt Qwen Image Bench set, sampled for coverage
132
  across its fine-grained (L3) categories. Metrics include Datapoint Elo,
 
169
  - **Arena Elo**: Elo published by Arena AI on their own dataset, plus
170
  category Elos (branding, 3D, cartoon/anime, photorealistic, art, portraits,
171
  text rendering).
172
+ - **Generation time**: median and minimum generation time in seconds for
173
+ images, as reported in the evaluation table. For video, generation time
174
+ per second of output video is the more informative figure. On the
175
+ text-to-video benchmark this is Fal wall time, except Pruna models which
176
+ use model execution time. On video-edit it is end-to-end wall time. This
177
+ is not a p95, and we do not state warm vs cold or concurrent load. Not
178
  available for Arena AI.
179
+ - **Price**: USD per image for text-to-image, or USD per second of output
180
+ video for text-to-video and video-to-video. We do not state list price vs
181
+ amount paid, or whether failed generations are included. Not available
182
+ for Arena AI.
183
 
184
  Scores from different datasets or metrics are **not interchangeable**. A high
185
  OneIG alignment score is not the same quantity as a Datapoint Elo. Compare
 
193
  - **Prompt counts:** OneIG Alignment uses 100 anime, 100 human, and 99 object
194
  prompts (299 total). Qwen Image Dataset uses 100 prompts sampled from the
195
  1,000-prompt pool for roughly even coverage of its fine-grained (L3)
196
+ categories. The VBench-2.0 Dataset uses about 90
197
+ generations per model. The Pruna Internal Video-Edit Benchmark uses 78
198
+ prompts across advertising, e-commerce, real estate, camera, lighting,
199
+ text, and related categories. Artificial Analysis and Arena AI use their
200
+ own private prompt sets.
201
  - **Generation (Qwen and OneIG):** one image per prompt per endpoint when
202
  the run exists. Default resolution is 1024×1024. Exceptions: FLUX 1.1 Pro
203
  Ultra at 2K, FLUX 2 Flex at 1008×1008, and any endpoint labeled 2K. The
204
  seed is derived from the prompt, so every model gets the same seed for the
205
  same prompt. Steps, CFG, prompt rewrite, and safety filters follow each
206
  endpoint's default. This does not describe Artificial Analysis or Arena AI.
207
+ - **Generation (Text-to-Video):** one clip per prompt per endpoint when the
208
+ run exists. About 90 generations per model.
209
+ - **Generation (Video-Edit):** one edited clip per prompt per endpoint when
210
+ the run exists. Every model sees the same source video for a prompt.
211
  - **Datapoint (Qwen and OneIG):** every model pair is compared on every
212
  prompt, with 10 votes per battle.
213
  - **Rapidata (Qwen and OneIG):** prompts longer than 400 characters are
 
279
  </svg>
280
  </button>
281
  </div>
282
+ <p class="app-header-tagline">Compare models on quality, speed, and price</p>
283
  </header>
284
  """,
285
  padding=False,
 
294
  return items[0] if items else None
295
 
296
 
297
+ def _dataset_modality(dataset):
298
+ return (dataset or {}).get("modality") or MODALITY_TEXT_TO_IMAGE
299
+
300
+
301
+ def _datasets_for_modality(datasets, modality):
302
+ if not modality:
303
+ return list(datasets)
304
+ scoped = [
305
+ dataset
306
+ for dataset in datasets
307
+ if _dataset_modality(dataset) == modality
308
+ ]
309
+ return scoped or list(datasets)
310
+
311
+
312
+ def _modality_choices(datasets):
313
+ present = {_dataset_modality(dataset) for dataset in datasets}
314
+ return [
315
+ (label, value) for label, value in MODALITY_CHOICES if value in present
316
+ ]
317
+
318
+
319
+ def _default_dataset_id(datasets, modality, preferred=None):
320
+ scoped = _datasets_for_modality(datasets, modality)
321
+ if preferred and any(dataset["id"] == preferred for dataset in scoped):
322
+ return preferred
323
+ return scoped[0]["id"] if scoped else None
324
+
325
+
326
+ def _dataset_choices(
327
+ datasets, *, modality=None, require_samples=False, require_pareto=False
328
+ ):
329
+ scoped = _datasets_for_modality(datasets, modality)
330
  return [
331
  (dataset["name"], dataset["id"])
332
+ for dataset in scoped
333
  if (not require_samples or dataset.get("samples"))
334
  and (not require_pareto or _dataset_has_pareto(datasets, dataset["id"]))
335
  ]
 
345
  samples = dataset.get("samples") if dataset else None
346
  if not samples:
347
  return set()
348
+ models = set(samples.get("models") or [])
349
+ return models | {display_model_name(model) for model in models}
350
+
351
+
352
+ def _sample_media_map(samples):
353
+ return (samples or {}).get("images") or {}
354
+
355
+
356
+ def _resolve_sample_model(samples, model):
357
+ media = _sample_media_map(samples)
358
+ if model in media:
359
+ return model
360
+ wanted = {str(model or "").strip(), display_model_name(model)}
361
+ wanted.discard("")
362
+ for key in media:
363
+ if key in wanted or display_model_name(key) in wanted:
364
+ return key
365
+ return None
366
+
367
+
368
+ def _default_sample_models(samples):
369
+ models = list((samples or {}).get("models") or [])
370
+ preferred = [model for model in models if _is_pruna_model(model)]
371
+ preferred.sort(
372
+ key=lambda model: (
373
+ "draft" in str(model).casefold()
374
+ or "draft" in display_model_name(model).casefold(),
375
+ display_model_name(model).casefold(),
376
+ )
377
+ )
378
+ return (preferred or models)[:2]
379
+
380
+
381
+ def _pareto_price_column(data):
382
+ columns = getattr(data, "columns", []) if data is not None else []
383
+ for column in _PARETO_PRICE_COLUMNS:
384
+ if column in columns:
385
+ return column
386
+ return None
387
+
388
+
389
+ def _pareto_time_column(data):
390
+ columns = getattr(data, "columns", []) if data is not None else []
391
+ for column in _PARETO_TIME_COLUMNS:
392
+ if column in columns:
393
+ return column
394
+ return None
395
 
396
 
397
  def _dataset_has_pareto(datasets, dataset_id):
398
  dataset = _item(datasets, dataset_id)
399
+ data = dataset.get("data") if dataset else None
400
+ return (
401
+ _pareto_price_column(data) is not None
402
+ or _pareto_time_column(data) is not None
403
+ )
404
 
405
 
406
+ def _dataset_dropdown_update(datasets, tab, dataset_id, modality=None):
407
  """Limit the dataset list to what the current tab can show."""
408
+ if modality is None:
409
+ modality = _dataset_modality(_item(datasets, dataset_id))
410
  return gr.update(
411
  choices=_dataset_choices(
412
  datasets,
413
+ modality=modality,
414
  require_samples=tab == TAB_SAMPLES
415
  and _dataset_has_samples(datasets, dataset_id),
416
  require_pareto=tab == TAB_PARETO
 
516
  "Optimized",
517
  ]
518
  _LEADERBOARD_META_COLUMNS = [
519
+ "Time / Output Video Second (s)",
520
  "Median Generation Time (s)",
521
  "Min Generation Time (s)",
522
  "Price / Image (USD)",
523
+ "Price / Second of Video (USD)",
524
  "Evaluation Date (UTC)",
525
  "Date",
526
  ]
 
747
  "Arena Text Rendering Elo": "Text Rendering",
748
  "Median Generation Time (s)": "Median generation time",
749
  "Min Generation Time (s)": "Min generation time",
750
+ "Time / Output Video Second (s)": "Generation time per second of video",
751
  "Price / Image (USD)": "Price per image",
752
+ "Price / Second of Video (USD)": "Price per second of video",
753
  "Evaluation Date (UTC)": "Date",
754
  "Date": "Date",
755
  }
 
824
  )
825
 
826
 
827
+ def _is_pruna_model(model_id) -> bool:
828
+ raw = str(model_id or "").casefold()
829
+ label = display_model_name(model_id).casefold()
830
+ return any(
831
+ value.startswith(prefix)
832
+ for value in (raw, label)
833
+ for prefix in ("p-image", "p_image", "p-video", "p_video")
834
+ )
835
+
836
+
837
+ def _pareto_fill_colors(models):
838
+ return [
839
+ _PARETO_PRUNA_COLOR if _is_pruna_model(model) else _PARETO_OTHER_COLOR
840
+ for model in models
841
+ ]
842
+
843
+
844
  def _build_pareto_figure(
845
  data,
846
  score_column,
 
865
 
866
  dominated = scatter.loc[[not flag for flag in on_frontier]].copy()
867
  frontier = scatter.loc[on_frontier].sort_values(x_column).copy()
868
+ dominated_colors = _pareto_fill_colors(dominated["Model"]) if not dominated.empty else []
869
+ frontier_colors = _pareto_fill_colors(frontier["Model"]) if not frontier.empty else []
870
  if not dominated.empty:
871
  dominated["Model"] = dominated["Model"].map(display_model_name)
872
  if not frontier.empty:
 
887
  name="Below frontier",
888
  text=dominated["Model"],
889
  hovertemplate=hover,
890
+ showlegend=False,
891
  marker={
892
  "size": 9,
893
+ "color": dominated_colors,
894
+ "opacity": 0.85,
895
  "line": {"width": 0},
896
  },
897
  )
 
905
  name="On frontier",
906
  text=frontier["Model"],
907
  hovertemplate=hover,
908
+ showlegend=False,
909
+ line={"color": _PARETO_FRONTIER_OUTLINE, "width": 2.5},
910
  marker={
911
  "size": 12,
912
+ "color": frontier_colors,
913
+ "line": {"width": 2.5, "color": _PARETO_FRONTIER_OUTLINE},
914
  },
915
  )
916
  )
917
+ for name, marker in (
918
+ (
919
+ "Pruna",
920
+ {
921
+ "size": 10,
922
+ "color": _PARETO_PRUNA_COLOR,
923
+ "line": {"width": 0},
924
+ },
925
+ ),
926
+ (
927
+ "Other models",
928
+ {
929
+ "size": 10,
930
+ "color": _PARETO_OTHER_COLOR,
931
+ "line": {"width": 0},
932
+ },
933
+ ),
934
+ (
935
+ "On frontier",
936
+ {
937
+ "size": 12,
938
+ "color": "rgba(0,0,0,0)",
939
+ "line": {"width": 2.5, "color": _PARETO_FRONTIER_OUTLINE},
940
+ },
941
+ ),
942
+ ):
943
+ fig.add_trace(
944
+ go.Scatter(
945
+ x=[None],
946
+ y=[None],
947
+ mode="markers",
948
+ name=name,
949
+ marker=marker,
950
+ hoverinfo="skip",
951
+ )
952
+ )
953
 
954
  score_label = _display_label(score_column)
955
  fig.update_layout(
 
1089
  if data is None or not score_column or score_column not in data.columns:
1090
  return None, score_missing, None, score_missing
1091
 
1092
+ price_column = _pareto_price_column(data) or _PARETO_PRICE_COLUMN
1093
+ price_title = _PARETO_PRICE_TITLES.get(price_column, "Price (USD)")
1094
+ price_missing = (
1095
+ "Price per second of video isn't available for this dataset."
1096
+ if price_column == _PARETO_VIDEO_PRICE_COLUMN
1097
+ else "Price per image isn't available for this dataset."
1098
+ )
1099
  price_fig, price_message = _pareto_axis(
1100
  data,
1101
  score_column,
1102
+ price_column,
1103
+ price_title,
1104
+ price_missing,
1105
  "No models have both a score and a price for this metric.",
1106
  x_hover_prefix="$",
1107
  x_axis_type=_pareto_axis_type(price_scale),
1108
  )
1109
+ time_column = _pareto_time_column(data) or _PARETO_TIME_COLUMN
1110
+ time_title = _PARETO_TIME_TITLES.get(time_column, "Generation time (s)")
1111
+ time_missing = (
1112
+ "Generation time per second of video isn't available for this dataset."
1113
+ if time_column == _PARETO_VIDEO_TIME_COLUMN
1114
+ else "Min generation time isn't available for this dataset."
1115
+ )
1116
+ time_empty = (
1117
+ "No models have both a score and generation time per second of "
1118
+ "video for this metric."
1119
+ if time_column == _PARETO_VIDEO_TIME_COLUMN
1120
+ else "No models have both a score and a min generation time for this metric."
1121
+ )
1122
  time_fig, time_message = _pareto_axis(
1123
  data,
1124
  score_column,
1125
+ time_column,
1126
+ time_title,
1127
+ time_missing,
1128
+ time_empty,
1129
  x_hover_suffix="s",
1130
  x_axis_type=_pareto_axis_type(latency_scale),
1131
  )
 
1133
 
1134
 
1135
  def _pareto_dataset_message(data):
1136
+ has_price = _pareto_price_column(data) is not None
1137
+ has_time = _pareto_time_column(data) is not None
1138
  if has_price or has_time:
1139
  return None
1140
  return (
1141
+ "Price and generation time aren't available for "
1142
  "this dataset, so these plots can't be drawn."
1143
  )
1144
 
1145
 
1146
  def _pareto_slot_note(price_fig, price_message, time_fig, time_message, data):
1147
+ has_price = _pareto_price_column(data) is not None
1148
+ has_time = _pareto_time_column(data) is not None
1149
  notes = []
1150
  if has_price and not has_time:
1151
  notes.append(
1152
+ "Generation time isn't available for this dataset, so only "
1153
  "price vs score is shown."
1154
  )
1155
  elif has_time and not has_price:
1156
  notes.append(
1157
+ "Price isn't available for this dataset, so only "
1158
+ "time vs score is shown."
1159
  )
1160
  if price_fig is None and has_price:
1161
  notes.append(price_message)
 
1176
  score_columns = [column for column in (score_columns or []) if column]
1177
  price_scales = _normalize_pareto_scales(price_scales)
1178
  time_scales = _normalize_pareto_scales(time_scales)
1179
+ has_price = _pareto_price_column(data) is not None
1180
+ has_time = _pareto_time_column(data) is not None
1181
  dataset_note = _pareto_dataset_message(data)
1182
  updates = [_pareto_note_update(dataset_note)]
1183
  hide_all_slots = not has_price and not has_time
 
1256
  return _pareto_unavailable_html(
1257
  "Samples aren't available for this dataset."
1258
  )
1259
+ models = [
1260
+ resolved
1261
+ for model in (selected_models or [])
1262
+ if (resolved := _resolve_sample_model(samples, model))
1263
+ ]
1264
  if not models:
1265
+ models = _default_sample_models(samples)
1266
  return _build_compare_samples_html(samples, models, num_prompts, seed)
1267
 
1268
 
1269
+ def _compare_media_html(url, label, *, kind):
1270
+ safe_url = escape(url, quote=True)
1271
+ safe_label = escape(label)
1272
+ if kind == "video":
1273
+ return (
1274
+ f'<video src="{safe_url}" controls preload="metadata" '
1275
+ f'playsinline></video>'
1276
+ )
1277
+ return (
1278
+ f'<a href="{safe_url}" target="_blank" rel="noopener noreferrer">'
1279
+ f'<img src="{safe_url}" alt="{safe_label} sample" loading="lazy" />'
1280
+ f"</a>"
1281
+ )
1282
+
1283
+
1284
+ def _compare_cell_html(label, url, *, kind, extra_class=""):
1285
+ classes = "compare-cell"
1286
+ if extra_class:
1287
+ classes = f"{classes} {extra_class}"
1288
+ return f"""
1289
+ <div class="{classes}">
1290
+ <div class="compare-model-label">{escape(label)}</div>
1291
+ {_compare_media_html(url, label, kind=kind)}
1292
+ </div>
1293
+ """
1294
+
1295
+
1296
  def _build_compare_samples_html(samples, selected_models, num_prompts, seed=0):
1297
  selected_models = list(selected_models or [])[:MAX_COMPARE_MODELS]
1298
+ media = _sample_media_map(samples)
1299
+ kind = (samples or {}).get("kind") or "image"
1300
+ source_videos = (samples or {}).get("source_videos") or {}
1301
 
1302
  if not selected_models:
1303
  return (
 
1308
 
1309
  shared_prompt_ids = None
1310
  for model in selected_models:
1311
+ model_prompt_ids = set(media.get(model) or [])
1312
  shared_prompt_ids = (
1313
  model_prompt_ids
1314
  if shared_prompt_ids is None
 
1328
  rng.shuffle(prompt_pool)
1329
  chosen = prompt_pool[: max(1, min(int(num_prompts), len(prompt_pool)))]
1330
 
 
1331
  blocks = []
1332
  for index, prompt_id in enumerate(chosen, start=1):
1333
  prompt_text = escape(samples["prompts"].get(prompt_id, ""))
1334
  cells = []
1335
+ source_url = source_videos.get(prompt_id)
1336
+ if source_url:
1337
+ cells.append(
1338
+ _compare_cell_html(
1339
+ "Source", source_url, kind="video", extra_class="compare-source"
1340
+ )
1341
+ )
1342
  for model in selected_models:
 
1343
  cells.append(
1344
+ _compare_cell_html(
1345
+ display_model_name(model),
1346
+ media[model][prompt_id],
1347
+ kind=kind,
1348
+ )
 
 
 
1349
  )
1350
+ columns = len(cells)
1351
  blocks.append(
1352
  f"""
1353
  <div class="compare-prompt-block">
1354
  <div class="compare-prompt-meta">
1355
  <span>Prompt {index}</span>
 
1356
  </div>
1357
  <p class="compare-prompt-text">{prompt_text}</p>
1358
  <div class="compare-row" style="grid-template-columns: repeat({columns}, minmax(0, 1fr));">
 
1380
  metric_id = _coerce_metric(
1381
  datasets, metrics, default_dataset_id, default_metric_id
1382
  )
1383
+ default_modality = _dataset_modality(_item(datasets, default_dataset_id))
1384
  with gr.Row(elem_classes="view-filters"):
1385
+ modality_dd = gr.Dropdown(
1386
+ choices=_modality_choices(datasets),
1387
+ value=default_modality,
1388
+ label="Model Type",
1389
+ type="value",
1390
+ filterable=False,
1391
+ scale=1,
1392
+ min_width=170,
1393
+ )
1394
  dataset_dd = gr.Dropdown(
1395
+ choices=_dataset_choices(datasets, modality=default_modality),
1396
  value=default_dataset_id,
1397
  label="Dataset",
1398
  type="value",
 
1424
  min_width=180,
1425
  elem_classes="filter-chips",
1426
  )
1427
+ return modality_dd, dataset_dd, metric_dd, models_dd
1428
 
1429
 
1430
  def render_image_workspace(datasets, metrics, default_dataset_id, default_metric_id):
 
1439
  with gr.Column(elem_classes="workspace-filters") as filters_host:
1440
  gr.Markdown(
1441
  "<p class='filter-help'>"
1442
+ "Start with Model Type to switch modalities. The rest of "
1443
+ "the filters follow you across Leaderboards, Pareto plots, "
1444
+ "and Samples. "
1445
+ "Samples only lists datasets and models we have generations "
1446
+ "for; Pareto plots only lists datasets with price or "
1447
+ "generation time. Search in Models, or leave it empty to "
1448
+ "include every model."
1449
  "</p>",
1450
  elem_classes="filter-help-host",
1451
  )
1452
+ modality_dd, dataset_dd, metric_dd, models_dd = _filter_row(
1453
  datasets, metrics, default_dataset_id, None
1454
  )
1455
  with gr.Tabs(elem_classes="main-tabs") as main_tabs:
 
1575
  elem_classes="pareto-col",
1576
  ) as slot_time_col:
1577
  slot_time_scale = _pareto_plot_heading(
1578
+ "Time vs score"
1579
  )
1580
  slot_time = gr.Plot(
1581
  value=None,
 
1611
  with gr.Column(visible=bool(initial_samples)) as samples_panel:
1612
  gr.Markdown(
1613
  f"<p class='view-help'>"
1614
+ f"The same prompts, side by side. Video edits show the "
1615
+ f"source clip first. Select up to "
1616
  f"<strong>{MAX_COMPARE_MODELS}</strong> models above, or leave "
1617
  f"Models empty for two defaults."
1618
  f"</p>",
 
1730
  ):
1731
  prev = dict(view_state or {})
1732
  extras = extras or {}
1733
+ modality = _dataset_modality(_item(datasets, dataset_id))
1734
+ last_by_modality = dict(prev.get("dataset_by_modality") or {})
1735
+ if dataset_id:
1736
+ last_by_modality[modality] = dataset_id
1737
  return {
1738
  "dataset_id": dataset_id,
1739
+ "modality": modality,
1740
+ "dataset_by_modality": last_by_modality,
1741
  "metric_id": metric_id,
1742
  "models": list(models or []),
1743
  "current_tab": tab,
 
1875
  tab = view_state.get("current_tab") or TAB_LEADERBOARDS
1876
  selected_raw = _normalize_metric_ids(metric_id)
1877
  incoming_models = list(models or [])
1878
+ filter_changed = source in {"dataset", "modality"}
1879
+ dataset_changed = filter_changed and dataset_id != view_state.get(
1880
  "dataset_id"
1881
  )
1882
  can_pareto = _dataset_has_pareto(datasets, dataset_id)
1883
  can_samples = _dataset_has_samples(datasets, dataset_id)
1884
  selected_tab = tab
1885
+ if filter_changed:
1886
  if tab == TAB_SAMPLES and not can_samples:
1887
  selected_tab = TAB_LEADERBOARDS
1888
  elif tab == TAB_PARETO and not can_pareto:
 
1891
  dataset_id,
1892
  metric_id,
1893
  models,
1894
+ clear_metric=source == "modality" or dataset_changed,
1895
  require_samples=selected_tab == TAB_SAMPLES,
1896
  )
1897
  dataset_id, metric_id, models = synced[:3]
 
1936
  list(optimized_value or []),
1937
  )
1938
  extra_updates = None
1939
+ if filter_changed:
1940
  view = resolve_view(datasets, metrics, dataset_id, metric_id)
1941
  extra_updates = _leaderboard_extras(
1942
  view["data"] if view else None,
 
1989
  ),
1990
  }
1991
 
1992
+ def on_modality(
1993
+ modality,
1994
+ dataset_id,
1995
+ metric_id,
1996
+ models,
1997
+ platform_value,
1998
+ owner_value,
1999
+ optimized_value,
2000
+ num_prompts,
2001
+ seed,
2002
+ view_state,
2003
+ ):
2004
+ view_state = dict(view_state or {})
2005
+ last_by_modality = dict(view_state.get("dataset_by_modality") or {})
2006
+ current_modality = view_state.get("modality") or _dataset_modality(
2007
+ _item(datasets, dataset_id)
2008
+ )
2009
+ if dataset_id:
2010
+ last_by_modality[current_modality] = dataset_id
2011
+ dataset_id = _default_dataset_id(
2012
+ datasets, modality, last_by_modality.get(modality)
2013
+ )
2014
+ view_state["modality"] = modality
2015
+ view_state["dataset_by_modality"] = last_by_modality
2016
+ result = _apply_filter_change(
2017
+ "modality",
2018
+ dataset_id,
2019
+ metric_id,
2020
+ models,
2021
+ platform_value,
2022
+ owner_value,
2023
+ optimized_value,
2024
+ num_prompts,
2025
+ seed,
2026
+ view_state,
2027
+ )
2028
+ if result is None:
2029
+ return _skip_all(len(dataset_outputs))
2030
+ extras = result["extra_updates"]
2031
+ return (
2032
+ _dataset_dropdown_update(
2033
+ datasets,
2034
+ result["selected_tab"],
2035
+ result["dataset_id"],
2036
+ modality=modality,
2037
+ ),
2038
+ result["metric_update"],
2039
+ result["models_update"],
2040
+ extras[6],
2041
+ extras[0],
2042
+ extras[1],
2043
+ extras[2],
2044
+ *result["views"],
2045
+ gr.update(interactive=result["can_pareto"]),
2046
+ gr.update(interactive=result["can_samples"]),
2047
+ gr.update(selected=result["selected_tab"])
2048
+ if result["selected_tab"] != result["tab"]
2049
+ else gr.skip(),
2050
+ result["state"],
2051
+ )
2052
+
2053
  def on_dataset(
2054
  dataset_id,
2055
  metric_id,
 
2399
  handler.__name__ = f"on_tab_{tab}"
2400
  return handler
2401
 
2402
+ default_modality = _dataset_modality(_item(datasets, default_dataset_id))
2403
  view_state = gr.State(
2404
  {
2405
  "dataset_id": default_dataset_id,
2406
+ "modality": default_modality,
2407
+ "dataset_by_modality": {default_modality: default_dataset_id},
2408
  "metric_id": None,
2409
  "models": [],
2410
  "current_tab": TAB_LEADERBOARDS,
 
2486
  main_tabs,
2487
  view_state,
2488
  ]
2489
+ modality_dd.change(
2490
+ on_modality,
2491
+ inputs=[modality_dd, *filter_inputs],
2492
+ outputs=dataset_outputs,
2493
+ **_VIEW_EVENTS,
2494
+ )
2495
  dataset_dd.change(
2496
  on_dataset,
2497
  inputs=filter_inputs,