.gitignore CHANGED
@@ -176,8 +176,4 @@ cython_debug/
176
  evaluation_results/
177
  images/
178
  hf_cache/
179
- *.lock
180
-
181
- # macOS
182
- .DS_Store
183
-
 
176
  evaluation_results/
177
  images/
178
  hf_cache/
179
+ *.lock
 
 
 
 
README.md CHANGED
@@ -30,6 +30,8 @@ pip install "gradio==5.19.0" pandas -r requirements.txt
30
  python app.py
31
  ```
32
 
 
 
33
  `requirements.txt` lists Plotly. Gradio and pandas are required locally;
34
  Hugging Face Spaces installs Gradio from the YAML `sdk_version` above.
35
 
 
30
  python app.py
31
  ```
32
 
33
+ The app is served at `http://127.0.0.1:7860`.
34
+
35
  `requirements.txt` lists Plotly. Gradio and pandas are required locally;
36
  Hugging Face Spaces installs Gradio from the YAML `sdk_version` above.
37
 
app.py CHANGED
@@ -62,9 +62,6 @@ custom_css = """
62
  --pruna-accordion-bg: rgba(255, 255, 255, 0.02);
63
  --pruna-accordion-border: rgba(216, 180, 254, 0.15);
64
  --pruna-dropdown-hover: #2a1844;
65
- --pruna-toggle-track: var(--pruna-bg-header);
66
- --pruna-toggle-thumb: var(--pruna-bg-elevated);
67
- --pruna-toggle-thumb-shadow: 0 1px 2px rgba(0, 0, 0, 0.45), inset 0 1px rgba(255, 255, 255, 0.06);
68
  color-scheme: dark;
69
  }
70
 
@@ -108,27 +105,13 @@ custom_css = """
108
  --pruna-accordion-bg: var(--pruna-bg-card);
109
  --pruna-accordion-border: var(--pruna-border);
110
  --pruna-dropdown-hover: #f3e8ff;
111
- --pruna-toggle-track: var(--pruna-bg-header);
112
- --pruna-toggle-thumb: var(--pruna-bg-card);
113
- --pruna-toggle-thumb-shadow: 0 1px 2px rgba(88, 28, 135, 0.12);
114
  color-scheme: light;
115
  }
116
 
117
- html {
118
  width: 100% !important;
119
  max-width: 100% !important;
120
- min-width: 0 !important;
121
- overflow-x: hidden;
122
- overflow-y: auto;
123
- -webkit-text-size-adjust: 100%;
124
- text-size-adjust: 100%;
125
- -webkit-tap-highlight-color: transparent;
126
- }
127
- body, gradio-app {
128
- width: 100% !important;
129
- max-width: 100% !important;
130
- min-width: 0 !important;
131
- overflow: visible;
132
  -webkit-tap-highlight-color: transparent;
133
  }
134
  html, body, .gradio-container, .main {
@@ -151,25 +134,18 @@ button, a, label, input, select, textarea,
151
  /* Subtle depth — not a marketing-site hero glow */
152
  body, .gradio-container {
153
  background-image: var(--pruna-glow) !important;
154
- background-repeat: no-repeat !important;
155
- background-attachment: scroll !important;
156
- }
157
- @media (min-width: 701px) and (hover: hover) and (pointer: fine) {
158
- body, .gradio-container {
159
- background-attachment: fixed !important;
160
- }
161
  }
162
 
163
  .gradio-container {
164
  width: 100% !important;
165
  max-width: 1200px !important;
166
- min-width: 0 !important;
167
  margin: 0 auto !important;
168
  padding-top: 0 !important;
169
  padding-left: 20px !important;
170
  padding-right: 20px !important;
171
  box-sizing: border-box !important;
172
- overflow-x: hidden;
173
  }
174
  .gradio-container .main,
175
  .gradio-container .wrap,
@@ -190,7 +166,6 @@ body, .gradio-container {
190
  .workspace-filters,
191
  .view-filters {
192
  max-width: 100% !important;
193
- min-width: 0 !important;
194
  }
195
 
196
  /* —— App header (P-Bench only) —— */
@@ -325,9 +300,6 @@ button.theme-toggle[data-mode="light"] .theme-icon-moon { display: block !import
325
  background: transparent !important;
326
  box-shadow: none !important;
327
  }
328
- /* Flatten tabs so the bar sits above shared filters. display:contents is
329
- the fallback; Safari can drop or mis-order those children, so browsers
330
- with subgrid use the grid layout below instead. */
331
  .workspace-shell > .tabs,
332
  .workspace-shell > .main-tabs,
333
  .workspace-shell > .block:not(.workspace-filters),
@@ -351,49 +323,12 @@ button.theme-toggle[data-mode="light"] .theme-icon-moon { display: block !import
351
  margin: 0 0 16px !important;
352
  justify-content: center !important;
353
  width: 100% !important;
354
- overflow: visible !important;
355
- }
356
- @supports (grid-template-rows: subgrid) {
357
- .workspace-shell,
358
- .workspace-shell.block,
359
- .workspace-shell.column,
360
- .workspace-shell.gap {
361
- display: grid !important;
362
- grid-template-columns: minmax(0, 1fr) !important;
363
- grid-template-rows: auto auto auto !important;
364
- align-content: start !important;
365
- }
366
- .workspace-shell > .tabs,
367
- .workspace-shell > .main-tabs,
368
- .workspace-shell .tabs.main-tabs {
369
- display: grid !important;
370
- grid-template-columns: minmax(0, 1fr) !important;
371
- grid-template-rows: subgrid !important;
372
- grid-column: 1 !important;
373
- grid-row: 1 / 4 !important;
374
- position: static !important;
375
- }
376
- .main-tabs > .tab-wrapper {
377
- grid-row: 1 !important;
378
- order: 0 !important;
379
- }
380
- .workspace-filters {
381
- grid-column: 1 !important;
382
- grid-row: 2 !important;
383
- order: 0 !important;
384
- }
385
- .main-tabs .tabitem {
386
- grid-row: 3 !important;
387
- order: 0 !important;
388
- min-width: 0 !important;
389
- }
390
  }
391
  .main-tabs .tab-container {
392
  height: auto !important;
393
- min-height: 0 !important;
394
  justify-content: center !important;
395
  flex-wrap: wrap !important;
396
- overflow: visible !important;
397
  max-width: 100% !important;
398
  gap: 2px;
399
  }
@@ -510,8 +445,6 @@ button.theme-toggle[data-mode="light"] .theme-icon-moon { display: block !import
510
  .app-header-brand h1,
511
  .gradio-container .app-header-brand h1 {
512
  font-size: 1.55rem !important;
513
- width: auto !important;
514
- max-width: 100% !important;
515
  }
516
  .app-header-tagline {
517
  font-size: 0.88rem !important;
@@ -581,43 +514,24 @@ button.theme-toggle[data-mode="light"] .theme-icon-moon { display: block !import
581
  .prose .ranking-table td {
582
  padding: 8px 10px !important;
583
  }
584
- .ranking-table,
585
- .prose .ranking-table {
586
- --rank-col-width: 3.25rem;
587
- }
588
  .ranking-table .rank,
589
  .prose .ranking-table .rank,
590
  .ranking-table th.rank {
591
  position: sticky !important;
592
  left: 0 !important;
593
- width: var(--rank-col-width);
594
- min-width: var(--rank-col-width);
595
- max-width: var(--rank-col-width);
596
- box-shadow: none;
597
  }
598
  .ranking-table .model-cell,
599
- .prose .ranking-table .model-cell {
600
- position: static !important;
601
- left: auto !important;
602
- z-index: auto;
603
- min-width: 140px;
604
- max-width: none;
605
- background: transparent !important;
606
- box-shadow: none !important;
607
- }
608
- .ranking-table th.model-cell,
609
- .prose .ranking-table th.model-cell {
610
  position: sticky !important;
611
- top: 0 !important;
612
- left: auto !important;
613
- z-index: 3;
614
- min-width: 140px;
615
- max-width: none;
616
- background: var(--pruna-bg-header) !important;
617
- box-shadow: 0 1px 0 var(--pruna-hairline) !important;
618
  }
619
- .ranking-table tbody tr:hover .model-cell {
620
- background: var(--pruna-table-hover) !important;
621
  }
622
  .compare-controls,
623
  .compare-controls.row,
@@ -690,7 +604,7 @@ button.theme-toggle[data-mode="light"] .theme-icon-moon { display: block !import
690
  .view-filters {
691
  display: flex !important;
692
  flex-wrap: wrap !important;
693
- align-items: flex-end !important;
694
  gap: 12px !important;
695
  margin: 0;
696
  overflow: visible !important;
@@ -698,7 +612,7 @@ button.theme-toggle[data-mode="light"] .theme-icon-moon { display: block !import
698
  .view-filters > div,
699
  .view-filters > .block,
700
  .view-filters > .form {
701
- flex: 1 1 0% !important;
702
  min-width: 0 !important;
703
  }
704
  .view-filters > .block,
@@ -850,10 +764,6 @@ button.theme-toggle[data-mode="light"] .theme-icon-moon { display: block !import
850
  line-height: 1.45 !important;
851
  font-weight: 400 !important;
852
  }
853
- .view-help + .view-help,
854
- .prose .view-help + .view-help {
855
- margin-top: 0.45rem !important;
856
- }
857
  .view-filters span[data-testid="block-info"],
858
  .view-filters .info,
859
  .view-filters .block-info {
@@ -991,7 +901,7 @@ button.theme-toggle[data-mode="light"] .theme-icon-moon { display: block !import
991
  .leaderboard-controls {
992
  display: flex !important;
993
  flex-wrap: wrap !important;
994
- align-items: flex-end !important;
995
  gap: 10px !important;
996
  margin-bottom: 12px;
997
  overflow: visible !important;
@@ -1114,11 +1024,11 @@ button.theme-toggle[data-mode="light"] .theme-icon-moon { display: block !import
1114
  max-height: min(70vh, 720px);
1115
  overflow-x: auto;
1116
  overflow-y: auto;
1117
- overscroll-behavior: none;
 
1118
  }
1119
  .ranking-table,
1120
  .prose .ranking-table {
1121
- --rank-col-width: 4.25rem;
1122
  width: 100%;
1123
  margin: 0 !important;
1124
  overflow: visible;
@@ -1194,14 +1104,10 @@ button.theme-toggle[data-mode="light"] .theme-icon-moon { display: block !import
1194
  .prose .ranking-table .rank {
1195
  position: sticky;
1196
  left: 0;
1197
- z-index: 2;
1198
  box-sizing: border-box;
1199
- width: var(--rank-col-width);
1200
- min-width: var(--rank-col-width);
1201
- max-width: var(--rank-col-width);
1202
- padding-left: 0.5rem !important;
1203
- padding-right: 0.5rem !important;
1204
- overflow: hidden;
1205
  color: var(--pruna-lavender) !important;
1206
  font-weight: 700 !important;
1207
  text-align: center;
@@ -1218,21 +1124,18 @@ button.theme-toggle[data-mode="light"] .theme-icon-moon { display: block !import
1218
  .ranking-table .model-cell,
1219
  .prose .ranking-table .model-cell {
1220
  position: sticky;
1221
- left: var(--rank-col-width);
1222
- z-index: 2;
1223
- box-sizing: border-box;
1224
  min-width: 180px;
1225
  max-width: 260px;
1226
  background: var(--pruna-table-sticky) !important;
1227
- box-shadow: 8px 0 10px -8px rgba(0, 0, 0, 0.35) !important;
1228
  }
1229
  .ranking-table th.model-cell,
1230
  .prose .ranking-table th.model-cell {
1231
  top: 0;
1232
- left: var(--rank-col-width);
1233
  z-index: 5;
1234
  background: var(--pruna-bg-header) !important;
1235
- box-shadow: 0 1px 0 var(--pruna-hairline), 8px 0 10px -8px rgba(0, 0, 0, 0.35) !important;
1236
  }
1237
  .ranking-table tbody tr:hover .rank,
1238
  .ranking-table tbody tr:hover .model-cell {
@@ -1414,9 +1317,7 @@ button.theme-toggle[data-mode="light"] .theme-icon-moon { display: block !import
1414
  box-shadow: none !important;
1415
  }
1416
  .compare-controls .compare-prompt-count .head {
1417
- display: block !important;
1418
- grid-column: 1 / -1;
1419
- grid-row: 1;
1420
  margin: 0 !important;
1421
  }
1422
  .compare-controls .compare-prompt-count .head label {
@@ -1590,182 +1491,7 @@ button.theme-toggle[data-mode="light"] .theme-icon-moon { display: block !import
1590
  margin: 0 !important;
1591
  }
1592
  .compare-row { display: grid; gap: 12px; min-width: 0; width: 100%; }
1593
- .compare-prompt-text { overflow-wrap: anywhere; word-break: break-word; }
1594
- .pareto-heading-row,
1595
- .pareto-heading-row.row,
1596
- .pareto-heading-row .form {
1597
- display: flex !important;
1598
- flex-wrap: wrap !important;
1599
- align-items: center !important;
1600
- gap: 8px 12px !important;
1601
- width: 100% !important;
1602
- margin-bottom: 0.4rem !important;
1603
- }
1604
- .pareto-heading-row .pareto-subhead,
1605
- .pareto-heading-row > div:first-child,
1606
- .pareto-heading-row .form > div:first-child {
1607
- flex: 1 1 240px !important;
1608
- min-width: 0 !important;
1609
- margin: 0 !important;
1610
- }
1611
- .pareto-heading-row .pareto-scale-control {
1612
- display: flex !important;
1613
- flex-direction: row !important;
1614
- align-items: center !important;
1615
- justify-content: flex-end !important;
1616
- flex: 0 0 auto !important;
1617
- gap: 0 !important;
1618
- margin-left: auto !important;
1619
- max-width: 168px !important;
1620
- padding: 0 !important;
1621
- }
1622
- .pareto-scale-all-row,
1623
- .pareto-scale-all-row.row,
1624
- .pareto-scale-all-row .form {
1625
- display: flex !important;
1626
- flex-direction: row !important;
1627
- flex-wrap: wrap !important;
1628
- align-items: center !important;
1629
- justify-content: flex-start !important;
1630
- gap: 8px 12px !important;
1631
- width: 100% !important;
1632
- margin: 2px 0 14px !important;
1633
- }
1634
- .pareto-scale-all-row .html-container,
1635
- .pareto-scale-all-row .block {
1636
- border: none !important;
1637
- background: transparent !important;
1638
- box-shadow: none !important;
1639
- padding: 0 !important;
1640
- margin: 0 !important;
1641
- width: auto !important;
1642
- flex: 0 0 auto !important;
1643
- }
1644
- .pareto-scale-all-row .html-container {
1645
- flex: 1 1 auto !important;
1646
- min-width: 0 !important;
1647
- }
1648
- .pareto-scale-all-label {
1649
- color: var(--pruna-text-muted);
1650
- font-size: 0.8rem;
1651
- font-weight: 500;
1652
- white-space: nowrap;
1653
- }
1654
- .pareto-scale-all-row .pareto-scale-toggle {
1655
- margin-left: auto !important;
1656
- }
1657
- .pareto-scale-toggle,
1658
- .pareto-scale-toggle.block {
1659
- min-width: 0 !important;
1660
- width: auto !important;
1661
- border: none !important;
1662
- background: transparent !important;
1663
- box-shadow: none !important;
1664
- padding: 0 !important;
1665
- margin: 0 !important;
1666
- }
1667
- .pareto-scale-toggle .wrap,
1668
- .pareto-scale-toggle .form {
1669
- display: block !important;
1670
- width: auto !important;
1671
- margin: 0 !important;
1672
- padding: 0 !important;
1673
- border: none !important;
1674
- background: transparent !important;
1675
- box-shadow: none !important;
1676
- }
1677
- .pareto-scale-toggle legend {
1678
- display: none !important;
1679
- }
1680
- .pareto-scale-toggle fieldset,
1681
- .pareto-scale-toggle .wrap:has(> label),
1682
- .pareto-scale-toggle .form:has(> label) {
1683
- position: relative !important;
1684
- display: grid !important;
1685
- grid-template-columns: 1fr 1fr !important;
1686
- align-items: stretch !important;
1687
- isolation: isolate;
1688
- box-sizing: border-box !important;
1689
- width: max-content !important;
1690
- min-width: 0 !important;
1691
- padding: 4px !important;
1692
- gap: 4px !important;
1693
- border: 1px solid var(--pruna-input-border) !important;
1694
- border-radius: 10px !important;
1695
- background: var(--pruna-toggle-track) !important;
1696
- box-shadow: none !important;
1697
- }
1698
- .pareto-scale-toggle fieldset::before,
1699
- .pareto-scale-toggle .wrap:has(> label)::before,
1700
- .pareto-scale-toggle .form:has(> label)::before {
1701
- content: none !important;
1702
- }
1703
- .pareto-scale-toggle label {
1704
- position: relative !important;
1705
- z-index: 1 !important;
1706
- display: flex !important;
1707
- flex: 1 1 auto !important;
1708
- align-items: center !important;
1709
- justify-content: center !important;
1710
- gap: 0 !important;
1711
- box-sizing: border-box !important;
1712
- min-width: 58px !important;
1713
- min-height: 26px !important;
1714
- margin: 0 !important;
1715
- padding: 5px 12px !important;
1716
- border: none !important;
1717
- border-radius: 6px !important;
1718
- background: transparent !important;
1719
- box-shadow: none !important;
1720
- color: var(--pruna-text-body) !important;
1721
- font-size: 0.72rem !important;
1722
- font-weight: 600 !important;
1723
- line-height: 1.2 !important;
1724
- letter-spacing: 0.01em;
1725
- white-space: nowrap;
1726
- cursor: pointer !important;
1727
- }
1728
- .pareto-scale-toggle-all label {
1729
- min-width: 68px !important;
1730
- min-height: 30px !important;
1731
- padding: 6px 14px !important;
1732
- font-size: 0.85rem !important;
1733
- }
1734
- .pareto-scale-toggle label span {
1735
- margin: 0 !important;
1736
- padding: 0 !important;
1737
- color: inherit !important;
1738
- opacity: 1 !important;
1739
- }
1740
- .pareto-scale-toggle label > * + * {
1741
- margin-left: 0 !important;
1742
- }
1743
- .pareto-scale-toggle label + label::before,
1744
- .pareto-scale-toggle label + label {
1745
- content: none !important;
1746
- border-left: none !important;
1747
- }
1748
- .pareto-scale-toggle input[type="radio"] {
1749
- position: absolute !important;
1750
- appearance: none !important;
1751
- opacity: 0 !important;
1752
- width: 0 !important;
1753
- height: 0 !important;
1754
- margin: 0 !important;
1755
- pointer-events: none !important;
1756
- }
1757
- .pareto-scale-toggle label:hover {
1758
- background: transparent !important;
1759
- color: var(--pruna-text-primary) !important;
1760
- }
1761
- .pareto-scale-toggle label.selected,
1762
- .pareto-scale-toggle label:has(input:checked) {
1763
- background: var(--pruna-toggle-thumb) !important;
1764
- color: var(--pruna-lavender) !important;
1765
- font-weight: 700 !important;
1766
- border-color: transparent !important;
1767
- box-shadow: var(--pruna-toggle-thumb-shadow) !important;
1768
- }
1769
  .pareto-layout,
1770
  .pareto-layout.row,
1771
  .pareto-layout .form {
@@ -1783,9 +1509,6 @@ button.theme-toggle[data-mode="light"] .theme-icon-moon { display: block !import
1783
  max-width: 100% !important;
1784
  }
1785
  .compare-cell { min-width: 0; }
1786
- .compare-cell.compare-source .compare-model-label {
1787
- color: var(--pruna-text-muted);
1788
- }
1789
  .compare-model-label {
1790
  margin-bottom: 6px;
1791
  color: var(--pruna-lavender);
@@ -1793,23 +1516,15 @@ button.theme-toggle[data-mode="light"] .theme-icon-moon { display: block !import
1793
  font-weight: 700;
1794
  word-break: break-word;
1795
  }
1796
- .compare-cell img,
1797
- .compare-cell video {
1798
  display: block;
1799
  width: 100%;
 
 
1800
  border-radius: 12px;
1801
  border: 1px solid var(--pruna-border);
1802
  background: var(--pruna-bg-elevated);
1803
  }
1804
- .compare-cell img {
1805
- aspect-ratio: 1 / 1;
1806
- object-fit: cover;
1807
- }
1808
- .compare-cell video {
1809
- aspect-ratio: 16 / 9;
1810
- max-height: 360px;
1811
- object-fit: contain;
1812
- }
1813
  .compare-empty,
1814
  .pareto-note-copy {
1815
  margin: 0;
@@ -1834,7 +1549,7 @@ button.theme-toggle[data-mode="light"] .theme-icon-moon { display: block !import
1834
  .app-header .app-header-brand h1 {
1835
  display: block !important;
1836
  width: max-content !important;
1837
- max-width: 100% !important;
1838
  flex: 0 0 auto !important;
1839
  margin: 0 !important;
1840
  padding: 0 !important;
@@ -2168,20 +1883,14 @@ def load_sample_comparison_data(folder):
2168
  return None
2169
 
2170
  prompts = {}
2171
- source_videos = {}
2172
  with prompts_path.open() as handle:
2173
  for line in handle:
2174
  if not line.strip():
2175
  continue
2176
  row = json.loads(line)
2177
- prompt_id = row["prompt_id"]
2178
- prompts[prompt_id] = row.get("text", "")
2179
- source_video = row.get("source_video")
2180
- if source_video:
2181
- source_videos[prompt_id] = source_video
2182
 
2183
  images = defaultdict(dict)
2184
- kinds = set()
2185
  with generations_path.open() as handle:
2186
  for line in handle:
2187
  if not line.strip():
@@ -2189,33 +1898,20 @@ def load_sample_comparison_data(folder):
2189
  row = json.loads(line)
2190
  model_id = row["model_id"]
2191
  prompt_id = row["prompt_id"]
2192
- media_url = row.get("image") or row.get("video")
2193
- if row.get("video"):
2194
- kinds.add("video")
2195
- elif row.get("image"):
2196
- kinds.add("image")
2197
- if model_id and prompt_id and media_url:
2198
- images[model_id][prompt_id] = media_url
2199
- if prompt_id and prompt_id not in source_videos:
2200
- params = row.get("params") or {}
2201
- input_video = params.get("input_video") or row.get("input_video")
2202
- if input_video:
2203
- source_videos[prompt_id] = input_video
2204
 
2205
  models = sorted(images)
2206
  if not models or not prompts:
2207
  return None
2208
 
2209
- loaded = {
2210
  "prompts": prompts,
2211
  "images": {model: dict(prompt_map) for model, prompt_map in images.items()},
2212
  "models": models,
2213
  "prompt_ids": sorted(prompts),
2214
- "kind": "video" if "video" in kinds else "image",
2215
  }
2216
- if source_videos:
2217
- loaded["source_videos"] = source_videos
2218
- return loaded
2219
 
2220
 
2221
  def _as_numeric(df, columns):
@@ -2330,7 +2026,7 @@ def load_qwen_combined_dataframe(path):
2330
  df = df[~df["Model"].astype(str).str.startswith("#")].copy()
2331
  df["Model"] = df["Model"].astype(str).str.strip()
2332
 
2333
- df = _as_numeric(
2334
  df,
2335
  [
2336
  "Price / Image (USD)",
@@ -2340,126 +2036,7 @@ def load_qwen_combined_dataframe(path):
2340
  "Rapidata Elo",
2341
  "Datapoint Elo",
2342
  ],
2343
- )
2344
- df = df.drop(columns=["Raw Win Rate"], errors="ignore")
2345
- return df.reset_index(drop=True)
2346
-
2347
-
2348
- def _is_pruna_video_model(series):
2349
- models = series.astype(str).str.casefold()
2350
- return models.str.startswith("p_video") | models.str.startswith("p-video")
2351
-
2352
-
2353
- def _apply_video_timings(df, *, replace_displayed_time=False):
2354
- """Mix Fal wall time with Pruna model execution time."""
2355
- fal = df.get("Time / Output Video Second (s)")
2356
- execution = df.get("Execution Time / Output Video Second (s)")
2357
- if fal is None:
2358
- return df
2359
-
2360
- if "model_id" in df.columns:
2361
- is_ours = _is_pruna_video_model(df["model_id"])
2362
- else:
2363
- is_ours = _is_pruna_video_model(df["Model"])
2364
-
2365
- if execution is None:
2366
- mixed = fal
2367
- else:
2368
- ours_time = execution.where(execution.notna(), fal)
2369
- mixed = fal.where(~is_ours, ours_time)
2370
- if replace_displayed_time:
2371
- df["Time / Output Video Second (s)"] = mixed
2372
- df["Pareto Time / Output Video Second (s)"] = mixed
2373
- return df
2374
-
2375
-
2376
- def load_video_editing_dataframe(path):
2377
- """Load the video-to-video editing leaderboard."""
2378
- df = pd.read_csv(path, na_values=["N/A", "n/a", ""])
2379
- df = df.rename(
2380
- columns={
2381
- "display_name": "Model",
2382
- "elo": "Datapoint Elo",
2383
- "min_generation_s": "Min Generation Time (s)",
2384
- "median_generation_s": "Median Generation Time (s)",
2385
- "p20_generation_s": "P20 Generation Time (s)",
2386
- "generation_s_per_output_video_s": "Time / Output Video Second (s)",
2387
- "predict_time_s_per_output_video_s": "Predict Time / Output Video Second (s)",
2388
- "model_execution_time_s_per_output_video_s": (
2389
- "Execution Time / Output Video Second (s)"
2390
- ),
2391
- "price": "Price / Second of Video (USD)",
2392
- }
2393
- )
2394
- df = df.drop(columns=["wandb_run_ids", "n_generations"], errors="ignore")
2395
- df["Model"] = df["Model"].astype(str).str.strip()
2396
- df = _as_numeric(
2397
- df,
2398
- [
2399
- "Datapoint Elo",
2400
- "Min Generation Time (s)",
2401
- "Median Generation Time (s)",
2402
- "P20 Generation Time (s)",
2403
- "Time / Output Video Second (s)",
2404
- "Predict Time / Output Video Second (s)",
2405
- "Execution Time / Output Video Second (s)",
2406
- "Price / Second of Video (USD)",
2407
- ],
2408
- )
2409
- df = _apply_video_timings(df)
2410
- df = df.drop(columns=["model_id"], errors="ignore")
2411
- return df.reset_index(drop=True)
2412
-
2413
-
2414
- def load_text_to_video_dataframe(path):
2415
- """Load the text-to-video leaderboard (P-Video-2 and Fal models)."""
2416
- df = pd.read_csv(path, na_values=["N/A", "n/a", ""])
2417
- model_column = "model_id" if "model_id" in df.columns else "Model"
2418
- df = df[~df[model_column].astype(str).str.casefold().str.startswith("agnes")].copy()
2419
- df = df.rename(
2420
- columns={
2421
- "model_id": "Model",
2422
- "datapoint_elo": "Datapoint Elo",
2423
- "rapidata_elo": "Rapidata Elo",
2424
- "min_generation_s": "Min Generation Time (s)",
2425
- "median_generation_s": "Median Generation Time (s)",
2426
- "p20_generation_s": "P20 Generation Time (s)",
2427
- "generation_s_per_output_video_s": "Time / Output Video Second (s)",
2428
- "model_execution_s_per_output_video_s": (
2429
- "Execution Time / Output Video Second (s)"
2430
- ),
2431
- "price_usd_per_second": "Price / Second of Video (USD)",
2432
- }
2433
- )
2434
- df = df.drop(columns=["wandb_run_ids", "n_generations"], errors="ignore")
2435
- df["Model"] = df["Model"].astype(str).str.strip()
2436
- df = _as_numeric(
2437
- df,
2438
- [
2439
- "Datapoint Elo",
2440
- "Rapidata Elo",
2441
- "Min Generation Time (s)",
2442
- "Median Generation Time (s)",
2443
- "P20 Generation Time (s)",
2444
- "Time / Output Video Second (s)",
2445
- "Execution Time / Output Video Second (s)",
2446
- "Price / Second of Video (USD)",
2447
- ],
2448
- )
2449
- # Pruna rows use model execution time; Fal rows keep Fal wall time.
2450
- df = _apply_video_timings(df, replace_displayed_time=True)
2451
- df = df.drop(
2452
- columns=["Execution Time / Output Video Second (s)"],
2453
- errors="ignore",
2454
- )
2455
- elo_columns = [
2456
- column
2457
- for column in ("Datapoint Elo", "Rapidata Elo")
2458
- if column in df.columns
2459
- ]
2460
- if elo_columns:
2461
- df = df.dropna(subset=elo_columns, how="all")
2462
- return df.reset_index(drop=True)
2463
 
2464
 
2465
  df = load_oneig_dataframe(oneig_path)
@@ -2509,17 +2086,9 @@ qwen_combined_dir = _resolve_data_path(
2509
  data_dir / "qwen_image_bench_combined",
2510
  space_root.parent / "qwen_image_bench_combined",
2511
  )
2512
- video_combined_dir = _resolve_data_path(
2513
- data_dir / "video_editing_combined",
2514
- space_root.parent / "video_editing_combined",
2515
- )
2516
- text_to_video_combined_dir = _resolve_data_path(
2517
- data_dir / "video_generation_combined",
2518
- space_root.parent / "video_generation_combined",
2519
- )
2520
  qwen_path = _resolve_data_path(
2521
- data_dir / "qwen_image_bench_model_price_and_median_generation_time_10_august.csv",
2522
- space_root.parent / "qwen_image_bench_model_price_and_median_generation_time_10_august.csv",
2523
  )
2524
  aa_path = _resolve_data_path(
2525
  data_dir / "artificial_analysis_text_to_image_leaderboard.csv",
@@ -2529,20 +2098,10 @@ arena_path = _resolve_data_path(
2529
  data_dir / "arena_ai_text_to_image_leaderboard.csv",
2530
  space_root.parent / "arena_ai_text_to_image_leaderboard.csv",
2531
  )
2532
- video_path = _resolve_data_path(
2533
- data_dir / "video-editing-leaderboard.csv",
2534
- space_root.parent / "video-editing-leaderboard.csv",
2535
- )
2536
- text_to_video_path = _resolve_data_path(
2537
- data_dir / "p-video-2-leaderboard.csv",
2538
- space_root.parent / "p-video-2-leaderboard.csv",
2539
- )
2540
 
2541
  qwen_df = load_qwen_combined_dataframe(qwen_path)
2542
  aa_df = load_artificial_analysis_dataframe(aa_path)
2543
  arena_df = load_arena_ai_dataframe(arena_path)
2544
- video_df = load_video_editing_dataframe(video_path)
2545
- text_to_video_df = load_text_to_video_dataframe(text_to_video_path)
2546
  qwen_display_columns = [
2547
  col
2548
  for col in [
@@ -2550,6 +2109,7 @@ qwen_display_columns = [
2550
  "Datapoint Elo",
2551
  "Rapidata Elo",
2552
  "P-Judge Overall",
 
2553
  "Median Generation Time (s)",
2554
  "Min Generation Time (s)",
2555
  "Price / Image (USD)",
@@ -2574,36 +2134,9 @@ arena_display_columns = [
2574
  ]
2575
  if col in arena_df.columns
2576
  ]
2577
- video_display_columns = [
2578
- col
2579
- for col in [
2580
- "Model",
2581
- "Datapoint Elo",
2582
- "Time / Output Video Second (s)",
2583
- "Median Generation Time (s)",
2584
- "Min Generation Time (s)",
2585
- "Price / Second of Video (USD)",
2586
- ]
2587
- if col in video_df.columns
2588
- ]
2589
- text_to_video_display_columns = [
2590
- col
2591
- for col in [
2592
- "Model",
2593
- "Datapoint Elo",
2594
- "Rapidata Elo",
2595
- "Time / Output Video Second (s)",
2596
- "Median Generation Time (s)",
2597
- "Min Generation Time (s)",
2598
- "Price / Second of Video (USD)",
2599
- ]
2600
- if col in text_to_video_df.columns
2601
- ]
2602
 
2603
  oneig_samples = load_sample_comparison_data(oneig_combined_dir)
2604
  qwen_samples = load_sample_comparison_data(qwen_combined_dir)
2605
- video_samples = load_sample_comparison_data(video_combined_dir)
2606
- text_to_video_samples = load_sample_comparison_data(text_to_video_combined_dir)
2607
 
2608
  metrics = [
2609
  {"id": "datapoint_elo", "column": "Datapoint Elo"},
@@ -2662,45 +2195,11 @@ arena_metric_ids = _metric_ids_for(
2662
  "arena_text",
2663
  ],
2664
  )
2665
- video_metric_ids = _metric_ids_for(video_df, ["datapoint_elo"])
2666
- text_to_video_metric_ids = _metric_ids_for(
2667
- text_to_video_df, ["datapoint_elo", "rapidata_elo"]
2668
- )
2669
 
2670
  datasets = [
2671
- {
2672
- "id": "text_to_video",
2673
- "name": "VBench-2.0 Dataset",
2674
- "modality": "text_to_video",
2675
- "data": text_to_video_df,
2676
- "columns": text_to_video_display_columns,
2677
- "metric_ids": text_to_video_metric_ids,
2678
- "note": (
2679
- "Datapoint Elo and Rapidata Elo from pairwise text-to-video "
2680
- "preference. Price is USD per second of output video. Time per "
2681
- "second of video is Fal wall time, except Pruna models which use "
2682
- "model execution time."
2683
- ),
2684
- "samples": text_to_video_samples,
2685
- },
2686
- {
2687
- "id": "video_editing",
2688
- "name": "Pruna Internal Video-Edit Benchmark",
2689
- "modality": "video_to_video",
2690
- "data": video_df,
2691
- "columns": video_display_columns,
2692
- "metric_ids": video_metric_ids,
2693
- "note": (
2694
- "Datapoint Elo from pairwise video-edit preference. Price is USD "
2695
- "per second of output video. Generation time per second of video "
2696
- "is end-to-end wall time to produce one second of output."
2697
- ),
2698
- "samples": video_samples,
2699
- },
2700
  {
2701
  "id": "qwen",
2702
  "name": "Qwen Image Dataset",
2703
- "modality": "text_to_image",
2704
  "data": qwen_df,
2705
  "columns": qwen_display_columns,
2706
  "metric_ids": qwen_metric_ids,
@@ -2710,7 +2209,6 @@ datasets = [
2710
  {
2711
  "id": "oneig",
2712
  "name": "OneIG Alignment Dataset",
2713
- "modality": "text_to_image",
2714
  "data": oneig_df,
2715
  "columns": oneig_display_columns,
2716
  "metric_ids": oneig_metric_ids,
@@ -2723,7 +2221,6 @@ datasets = [
2723
  {
2724
  "id": "artificial_analysis",
2725
  "name": "Artificial Analysis Dataset",
2726
- "modality": "text_to_image",
2727
  "data": aa_df,
2728
  "columns": aa_display_columns,
2729
  "metric_ids": aa_metric_ids,
@@ -2733,7 +2230,6 @@ datasets = [
2733
  {
2734
  "id": "arena_ai",
2735
  "name": "Arena AI Dataset",
2736
- "modality": "text_to_image",
2737
  "data": arena_df,
2738
  "columns": arena_display_columns,
2739
  "metric_ids": arena_metric_ids,
@@ -2744,14 +2240,8 @@ datasets = [
2744
  datasets = [dataset for dataset in datasets if dataset["metric_ids"]]
2745
 
2746
  DEFAULT_DATASET_ID = next(
2747
- (dataset["id"] for dataset in datasets if dataset["id"] == "text_to_video"),
2748
- next(
2749
- (dataset["id"] for dataset in datasets if dataset["id"] == "video_editing"),
2750
- next(
2751
- (dataset["id"] for dataset in datasets if dataset["id"] == "qwen"),
2752
- datasets[0]["id"] if datasets else None,
2753
- ),
2754
- ),
2755
  )
2756
  DEFAULT_METRIC_ID = None
2757
 
@@ -2827,23 +2317,11 @@ custom_head = """
2827
  return found;
2828
  };
2829
 
2830
- const applyPlotTheme = (gd, mode) => {
2831
- const layout = PLOT_LAYOUT[mode];
2832
- if (!layout || typeof Plotly === "undefined" || !gd) return;
2833
- if (gd.layout && gd.layout.paper_bgcolor === layout.paper_bgcolor) return;
2834
- try { Plotly.relayout(gd, layout); } catch (e) {}
2835
- };
2836
-
2837
- const watchPlotTheme = (gd) => {
2838
- if (!gd || gd.__inferbenchThemeBound) return;
2839
- gd.__inferbenchThemeBound = true;
2840
- gd.addEventListener("plotly_afterplot", () => applyPlotTheme(gd, currentMode()));
2841
- };
2842
-
2843
  const restylePlots = (mode) => {
 
 
2844
  queryAll(".js-plotly-plot").forEach((gd) => {
2845
- watchPlotTheme(gd);
2846
- applyPlotTheme(gd, mode);
2847
  });
2848
  };
2849
 
 
62
  --pruna-accordion-bg: rgba(255, 255, 255, 0.02);
63
  --pruna-accordion-border: rgba(216, 180, 254, 0.15);
64
  --pruna-dropdown-hover: #2a1844;
 
 
 
65
  color-scheme: dark;
66
  }
67
 
 
105
  --pruna-accordion-bg: var(--pruna-bg-card);
106
  --pruna-accordion-border: var(--pruna-border);
107
  --pruna-dropdown-hover: #f3e8ff;
 
 
 
108
  color-scheme: light;
109
  }
110
 
111
+ html, body {
112
  width: 100% !important;
113
  max-width: 100% !important;
114
+ overflow-x: clip;
 
 
 
 
 
 
 
 
 
 
 
115
  -webkit-tap-highlight-color: transparent;
116
  }
117
  html, body, .gradio-container, .main {
 
134
  /* Subtle depth — not a marketing-site hero glow */
135
  body, .gradio-container {
136
  background-image: var(--pruna-glow) !important;
137
+ background-attachment: fixed !important;
 
 
 
 
 
 
138
  }
139
 
140
  .gradio-container {
141
  width: 100% !important;
142
  max-width: 1200px !important;
 
143
  margin: 0 auto !important;
144
  padding-top: 0 !important;
145
  padding-left: 20px !important;
146
  padding-right: 20px !important;
147
  box-sizing: border-box !important;
148
+ overflow-x: clip;
149
  }
150
  .gradio-container .main,
151
  .gradio-container .wrap,
 
166
  .workspace-filters,
167
  .view-filters {
168
  max-width: 100% !important;
 
169
  }
170
 
171
  /* —— App header (P-Bench only) —— */
 
300
  background: transparent !important;
301
  box-shadow: none !important;
302
  }
 
 
 
303
  .workspace-shell > .tabs,
304
  .workspace-shell > .main-tabs,
305
  .workspace-shell > .block:not(.workspace-filters),
 
323
  margin: 0 0 16px !important;
324
  justify-content: center !important;
325
  width: 100% !important;
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
326
  }
327
  .main-tabs .tab-container {
328
  height: auto !important;
 
329
  justify-content: center !important;
330
  flex-wrap: wrap !important;
331
+ overflow: hidden !important;
332
  max-width: 100% !important;
333
  gap: 2px;
334
  }
 
445
  .app-header-brand h1,
446
  .gradio-container .app-header-brand h1 {
447
  font-size: 1.55rem !important;
 
 
448
  }
449
  .app-header-tagline {
450
  font-size: 0.88rem !important;
 
514
  .prose .ranking-table td {
515
  padding: 8px 10px !important;
516
  }
 
 
 
 
517
  .ranking-table .rank,
518
  .prose .ranking-table .rank,
519
  .ranking-table th.rank {
520
  position: sticky !important;
521
  left: 0 !important;
522
+ width: 2.4rem;
523
+ min-width: 2.4rem;
 
 
524
  }
525
  .ranking-table .model-cell,
526
+ .prose .ranking-table .model-cell,
527
+ .ranking-table th.model-cell {
 
 
 
 
 
 
 
 
 
528
  position: sticky !important;
529
+ left: 2.4rem !important;
530
+ min-width: 108px;
531
+ max-width: 36vw;
 
 
 
 
532
  }
533
+ .ranking-table .model-cell strong {
534
+ white-space: nowrap;
535
  }
536
  .compare-controls,
537
  .compare-controls.row,
 
604
  .view-filters {
605
  display: flex !important;
606
  flex-wrap: wrap !important;
607
+ align-items: end !important;
608
  gap: 12px !important;
609
  margin: 0;
610
  overflow: visible !important;
 
612
  .view-filters > div,
613
  .view-filters > .block,
614
  .view-filters > .form {
615
+ flex: 1 1 0 !important;
616
  min-width: 0 !important;
617
  }
618
  .view-filters > .block,
 
764
  line-height: 1.45 !important;
765
  font-weight: 400 !important;
766
  }
 
 
 
 
767
  .view-filters span[data-testid="block-info"],
768
  .view-filters .info,
769
  .view-filters .block-info {
 
901
  .leaderboard-controls {
902
  display: flex !important;
903
  flex-wrap: wrap !important;
904
+ align-items: end !important;
905
  gap: 10px !important;
906
  margin-bottom: 12px;
907
  overflow: visible !important;
 
1024
  max-height: min(70vh, 720px);
1025
  overflow-x: auto;
1026
  overflow-y: auto;
1027
+ -webkit-overflow-scrolling: touch;
1028
+ overscroll-behavior-x: contain;
1029
  }
1030
  .ranking-table,
1031
  .prose .ranking-table {
 
1032
  width: 100%;
1033
  margin: 0 !important;
1034
  overflow: visible;
 
1104
  .prose .ranking-table .rank {
1105
  position: sticky;
1106
  left: 0;
1107
+ z-index: 1;
1108
  box-sizing: border-box;
1109
+ width: 3.25rem;
1110
+ min-width: 3.25rem;
 
 
 
 
1111
  color: var(--pruna-lavender) !important;
1112
  font-weight: 700 !important;
1113
  text-align: center;
 
1124
  .ranking-table .model-cell,
1125
  .prose .ranking-table .model-cell {
1126
  position: sticky;
1127
+ left: 3.25rem;
1128
+ z-index: 1;
 
1129
  min-width: 180px;
1130
  max-width: 260px;
1131
  background: var(--pruna-table-sticky) !important;
 
1132
  }
1133
  .ranking-table th.model-cell,
1134
  .prose .ranking-table th.model-cell {
1135
  top: 0;
1136
+ left: 3.25rem;
1137
  z-index: 5;
1138
  background: var(--pruna-bg-header) !important;
 
1139
  }
1140
  .ranking-table tbody tr:hover .rank,
1141
  .ranking-table tbody tr:hover .model-cell {
 
1317
  box-shadow: none !important;
1318
  }
1319
  .compare-controls .compare-prompt-count .head {
1320
+ display: contents;
 
 
1321
  margin: 0 !important;
1322
  }
1323
  .compare-controls .compare-prompt-count .head label {
 
1491
  margin: 0 !important;
1492
  }
1493
  .compare-row { display: grid; gap: 12px; min-width: 0; width: 100%; }
1494
+ .compare-prompt-text { overflow-wrap: anywhere; }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1495
  .pareto-layout,
1496
  .pareto-layout.row,
1497
  .pareto-layout .form {
 
1509
  max-width: 100% !important;
1510
  }
1511
  .compare-cell { min-width: 0; }
 
 
 
1512
  .compare-model-label {
1513
  margin-bottom: 6px;
1514
  color: var(--pruna-lavender);
 
1516
  font-weight: 700;
1517
  word-break: break-word;
1518
  }
1519
+ .compare-cell img {
 
1520
  display: block;
1521
  width: 100%;
1522
+ aspect-ratio: 1 / 1;
1523
+ object-fit: cover;
1524
  border-radius: 12px;
1525
  border: 1px solid var(--pruna-border);
1526
  background: var(--pruna-bg-elevated);
1527
  }
 
 
 
 
 
 
 
 
 
1528
  .compare-empty,
1529
  .pareto-note-copy {
1530
  margin: 0;
 
1549
  .app-header .app-header-brand h1 {
1550
  display: block !important;
1551
  width: max-content !important;
1552
+ max-width: none !important;
1553
  flex: 0 0 auto !important;
1554
  margin: 0 !important;
1555
  padding: 0 !important;
 
1883
  return None
1884
 
1885
  prompts = {}
 
1886
  with prompts_path.open() as handle:
1887
  for line in handle:
1888
  if not line.strip():
1889
  continue
1890
  row = json.loads(line)
1891
+ prompts[row["prompt_id"]] = row.get("text", "")
 
 
 
 
1892
 
1893
  images = defaultdict(dict)
 
1894
  with generations_path.open() as handle:
1895
  for line in handle:
1896
  if not line.strip():
 
1898
  row = json.loads(line)
1899
  model_id = row["model_id"]
1900
  prompt_id = row["prompt_id"]
1901
+ image_url = row.get("image")
1902
+ if model_id and prompt_id and image_url:
1903
+ images[model_id][prompt_id] = image_url
 
 
 
 
 
 
 
 
 
1904
 
1905
  models = sorted(images)
1906
  if not models or not prompts:
1907
  return None
1908
 
1909
+ return {
1910
  "prompts": prompts,
1911
  "images": {model: dict(prompt_map) for model, prompt_map in images.items()},
1912
  "models": models,
1913
  "prompt_ids": sorted(prompts),
 
1914
  }
 
 
 
1915
 
1916
 
1917
  def _as_numeric(df, columns):
 
2026
  df = df[~df["Model"].astype(str).str.startswith("#")].copy()
2027
  df["Model"] = df["Model"].astype(str).str.strip()
2028
 
2029
+ return _as_numeric(
2030
  df,
2031
  [
2032
  "Price / Image (USD)",
 
2036
  "Rapidata Elo",
2037
  "Datapoint Elo",
2038
  ],
2039
+ ).reset_index(drop=True)
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
2040
 
2041
 
2042
  df = load_oneig_dataframe(oneig_path)
 
2086
  data_dir / "qwen_image_bench_combined",
2087
  space_root.parent / "qwen_image_bench_combined",
2088
  )
 
 
 
 
 
 
 
 
2089
  qwen_path = _resolve_data_path(
2090
+ data_dir / "qwen_image_bench_model_price_and_median_generation_time.csv",
2091
+ space_root.parent / "qwen_image_bench_model_price_and_median_generation_time.csv",
2092
  )
2093
  aa_path = _resolve_data_path(
2094
  data_dir / "artificial_analysis_text_to_image_leaderboard.csv",
 
2098
  data_dir / "arena_ai_text_to_image_leaderboard.csv",
2099
  space_root.parent / "arena_ai_text_to_image_leaderboard.csv",
2100
  )
 
 
 
 
 
 
 
 
2101
 
2102
  qwen_df = load_qwen_combined_dataframe(qwen_path)
2103
  aa_df = load_artificial_analysis_dataframe(aa_path)
2104
  arena_df = load_arena_ai_dataframe(arena_path)
 
 
2105
  qwen_display_columns = [
2106
  col
2107
  for col in [
 
2109
  "Datapoint Elo",
2110
  "Rapidata Elo",
2111
  "P-Judge Overall",
2112
+ "Raw Win Rate",
2113
  "Median Generation Time (s)",
2114
  "Min Generation Time (s)",
2115
  "Price / Image (USD)",
 
2134
  ]
2135
  if col in arena_df.columns
2136
  ]
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
2137
 
2138
  oneig_samples = load_sample_comparison_data(oneig_combined_dir)
2139
  qwen_samples = load_sample_comparison_data(qwen_combined_dir)
 
 
2140
 
2141
  metrics = [
2142
  {"id": "datapoint_elo", "column": "Datapoint Elo"},
 
2195
  "arena_text",
2196
  ],
2197
  )
 
 
 
 
2198
 
2199
  datasets = [
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
2200
  {
2201
  "id": "qwen",
2202
  "name": "Qwen Image Dataset",
 
2203
  "data": qwen_df,
2204
  "columns": qwen_display_columns,
2205
  "metric_ids": qwen_metric_ids,
 
2209
  {
2210
  "id": "oneig",
2211
  "name": "OneIG Alignment Dataset",
 
2212
  "data": oneig_df,
2213
  "columns": oneig_display_columns,
2214
  "metric_ids": oneig_metric_ids,
 
2221
  {
2222
  "id": "artificial_analysis",
2223
  "name": "Artificial Analysis Dataset",
 
2224
  "data": aa_df,
2225
  "columns": aa_display_columns,
2226
  "metric_ids": aa_metric_ids,
 
2230
  {
2231
  "id": "arena_ai",
2232
  "name": "Arena AI Dataset",
 
2233
  "data": arena_df,
2234
  "columns": arena_display_columns,
2235
  "metric_ids": arena_metric_ids,
 
2240
  datasets = [dataset for dataset in datasets if dataset["metric_ids"]]
2241
 
2242
  DEFAULT_DATASET_ID = next(
2243
+ (dataset["id"] for dataset in datasets if dataset["id"] == "qwen"),
2244
+ datasets[0]["id"] if datasets else None,
 
 
 
 
 
 
2245
  )
2246
  DEFAULT_METRIC_ID = None
2247
 
 
2317
  return found;
2318
  };
2319
 
 
 
 
 
 
 
 
 
 
 
 
 
 
2320
  const restylePlots = (mode) => {
2321
+ const layout = PLOT_LAYOUT[mode];
2322
+ if (!layout || typeof Plotly === "undefined") return;
2323
  queryAll(".js-plotly-plot").forEach((gd) => {
2324
+ try { Plotly.relayout(gd, layout); } catch (e) {}
 
2325
  });
2326
  };
2327
 
data/p-video-2-leaderboard.csv DELETED
@@ -1,19 +0,0 @@
1
- model_id,wandb_run_ids,n_generations,min_generation_s,median_generation_s,p20_generation_s,generation_s_per_output_video_s,model_execution_s_per_output_video_s,datapoint_elo,rapidata_elo,price_usd_per_second
2
- gemini_omni_1_1_flash,mfajagum,89,26.276599962002365,31.884095050001633,29.0771403791965,6.399738597206617,,1037,1197,0.100
3
- grok_imagine_video,q39r8v9e,90,68.9776645039965,99.17614604350092,78.69151808420138,16.13514411181308,,958,994,0.050
4
- grok_imagine_video_v1_5,z2fp7hd7,90,38.42492011199647,52.268091876499966,46.47850653079659,11.452002471104521,,959,1016,0.140
5
- ltx_2_5_fast,h6se1hfi,90,21.695961473000352,26.684429597007693,24.118977216200438,5.270079915707363,,1002,1035,0.090
6
- ltx_2_5_pro,kjbgxbab,90,23.997383733993047,27.897931821993552,25.263995286799037,4.936325968288616,,1007,978,0.120
7
- minimax_h3,xpcv2m58,89,97.61419671699696,117.51345164199665,106.57570995900024,24.968780374361877,,1026,1108,0.100
8
- minimax_h3_max,0d2lkmp6,90,4.173863318999793,4.49869444649994,4.4671880990012145,1.3807388451000264,0.644,1031,1197,0.080
9
- minimax_h3_max_turbo__prompt_expansion_mode_balanced,kxuwq31c,90,2.891819470001792,3.487746603501364,3.134695611400821,1.35231252479333,0.61,1040,1214,0.040
10
- p_video_2__draft_false__prompt_upsampling_false__resolution_1080p,9hgkhujw,90,13.33159556199098,16.702878594005597,14.793676785795833,4.033411242884629,2.151682222222222,,,0.050
11
- p_video_2__draft_false__prompt_upsampling_false__resolution_720p,m2jf4z7h,89,5.859760620005545,8.162274362999597,7.147269867203431,2.1525601170473116,0.9053123595505614,,,0.025
12
- p_video_2__draft_false__prompt_upsampling_true__resolution_1080p,uoqpv10b,90,14.615217412007041,18.316011337505188,16.441080821005745,4.5638632221758275,2.152566666666667,979,991,0.050
13
- p_video_2__draft_false__prompt_upsampling_true__resolution_720p,2q5jydr8,90,7.227749306999613,9.880705762494472,9.154919161600992,2.4789854674801206,0.9079755555555558,985,1039,0.025
14
- p_video_2__draft_true__prompt_upsampling_false__resolution_1080p,qbt0atz4,89,6.250065040003392,9.528199804000906,7.504525098402519,2.431133616615736,0.7517640449438197,,,0.030
15
- p_video_2__draft_true__prompt_upsampling_false__resolution_720p,9ur57h54,88,3.379928643000312,5.454214365498046,4.208131522199255,1.4960965825225272,0.40954545454545455,,,0.015
16
- p_video_2__draft_true__prompt_upsampling_true__resolution_1080p,gcbyy4wn,90,8.047415203996934,11.128662197996164,9.22702455239487,2.8557857865532665,0.7514066666666668,,,0.030
17
- p_video_2__draft_true__prompt_upsampling_true__resolution_720p,6hi994vk,90,4.985804016003385,7.4922584965024726,6.472871531004785,2.4024102996157146,0.40678444444444434,982,1048,0.015
18
- seedance_2_5_turbo,l9zjnv7n,90,168.997338743,252.07832621250054,211.93920676639829,67.60899374696221,,1037,1080,0.200
19
- veo_3_1_lite,jjbobes4,90,33.894167409991496,37.83729844300251,37.009501291197374,6.745116482181308,,968,986,0.050
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
data/qwen_image_bench_model_price_and_median_generation_time.csv ADDED
@@ -0,0 +1,60 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Model,Price / Image (USD),Median Generation Time (s),Min Generation Time (s),P-Judge Overall,Raw Win Rate,Rapidata Elo,Datapoint Elo,Benchmark.ai Elo
2
+ reve_2_1,N/A,N/A,28.3,,,,,1173.2
3
+ ideogram_4_0_quality,N/A,N/A,66.6,,,,,1131.0
4
+ gpt_image_2,0.21,77.8,77.8,59.02083099999998,,1172.58,1116,1124.5
5
+ nano_banana_2_0,N/A,N/A,N/A,56.362956,59.4%,1071.79,1067,1056.3
6
+ gpt_image_1_5,0.135,38.0,38.0,57.864434999999986,64.5%,1102.09,1064,910.3
7
+ hidream_i1_dev,0.0086,2.823569217998738,2.06,49.17682099999999,51.7%,999.17,984,
8
+ gpt_image_1,0.167,38.8,38.8,54.975741000000006,,1095.39,987,
9
+ imagen_4_0_ultra,0.06,11.4,11.4,53.385603999999965,59.7%,1075.22,1023,
10
+ flux_2_flex,0.06,10.856533817990567,8.07,53.9245676767677,57.8%,1054.34,1018,921.3
11
+ qwen_image,0.025,4.8,4.8,51.561746,51.5%,1073.72,1005,975.6
12
+ seedream_5_0,N/A,N/A,N/A,55.715153,,1070.35,1012,
13
+ nano_banana_pro,0.134,17.2,17.2,56.452189898989914,58.0%,1029.51,1045,1102.4
14
+ hidream_i1_fast,0.0051,9.920469530501578,1.40,48.89298600000002,50.1%,1024.67,984,
15
+ imagen_4_fast,0.02,3.7531301500021073,2.71,50.155055208333344,,972.6,981,
16
+ seedream_4_5,0.04,16.6,16.6,55.66328800000001,,1048.96,1035,964.2
17
+ seedream_4_0,0.03,12.2,12.2,55.241443999999994,,1050.41,1035,
18
+ p_image_2_ideogram_low_1k,0.0075,2.59,1.55,54.827397,53.2%,1000.07,1009,1103.9
19
+ #p_image_2_ideogram_low_2k,0.016,5.17,4.25,53.971723,48.4%,1074.91,1003,
20
+ qwen_image_2_0_pro,0.035,35.5,35.5,56.181776,,1033.25,1017,959.9
21
+ juggernaut_base_flux,0.035,4.115834823496698,3.84,49.474676,46.4%,1038.22,972,
22
+ flux_2_pro,N/A,N/A,N/A,54.43423900000002,,993.54,1019,1011.6
23
+ z_image,0.005,1.5122045120006078,1.22,49.946227999999984,48.1%,1028.1,1001,
24
+ p_image_2_ideogram_high_1k,0.015,4.28,3.07,55.754507999999994,52.1%,972.92,1022,1104.0
25
+ #p_image_2_ideogram_high_2k,0.03,8.69,7.12,54.757842000000004,49.4%,1074.91,1007,
26
+ qwen_image_2512,0.02,19.1,19.1,51.677326,,1029.6,1009,
27
+ flux_2_max,0.07,26.4,26.4,54.047976,60.9%,955.19,1028,1001.6
28
+ flux_1_1_pro_ultra,0.06,9.026992494000297,6.20,50.358445,52.0%,979.56,995,
29
+ juggernaut_pro_flux,0.055,3.699636150500737,3.36,49.741183,46.4%,972.14,964,
30
+ flux_dev,0.025,1.6931055715031107,1.49,48.241479,42.1%,925.04,940,
31
+ p_image_2_ideogram_very_low_1k,0.003,2.56,1.49,53.51894200000001,48.1%,959.86,995,1071.7
32
+ #p_image_2_ideogram_very_low_2k,0.006,3.94,3.05,53.68248699999998,47.2%,962.38,980,
33
+ #p_image_2_ideogram_very_low_1k_no_upsampling,0.005,0.82,,46.33,,,,1071.7
34
+ #p_image_2_ideogram_very_low_2k_no_upsampling,0.005,2.18,,46.39,,,,1071.7
35
+ #p_image_2_ideogram_low_1k_no_upsampling,0.01,3.33,,45.32,,,,1103.9
36
+ #p_image_2_ideogram_low_2k_no_upsampling,0.01,3.34,,46.25,,,,1103.9
37
+ #p_image_2_ideogram_medium_1k_no_upsampling,0.015,2.13,,46.88,,,,1115.1
38
+ #p_image_2_ideogram_medium_2k_no_upsampling,0.015,5.55,,47.11,,,,1115.1
39
+ #p_image_2_ideogram_high_1k_no_upsampling,0.03,3.11,,46.88,,,,1104.0
40
+ #p_image_2_ideogram_high_2k_no_upsampling,0.03,5.55,,47.06,,,,1104.0
41
+ hidream_i1_full,0.014,6.008430051002506,5.69,46.82559300000002,36.9%,955.46,944,
42
+ flux_2_dev,0.025,4.310278721997747,4.03,52.71644489795918,53.1%,1007.6,1021,942.0
43
+ imagen_4_0,0.04,14.1,14.1,52.08996199999999,53.5%,979.62,1005,
44
+ wan_2_2_image,0.02,3.005390542501118,2.96,48.19959399999999,,944.87,960,
45
+ flux_krea,0.025,1.7150160090022837,1.7150160090022837,50.35734,50.0%,919.73,975,
46
+ p_image_2_ideogram_medium_1k,0.01,3.06,2.05,54.719193000000004,51.7%,941.46,1002,1115.1
47
+ #p_image_2_ideogram_medium_2k,0.02,7.44,6.47,54.23124444444446,48.1%,949.35,1000,
48
+ glm_image,0.05,188.2,188.2,51.42623399999999,,923.35,953,
49
+ p_image,0.005,1.0640762715011078,0.95,48.75217099999999,44.8%,924.37,961,1098.7
50
+ hunyuanimage_3_0,0.09,41.0,41.0,52.32440099999998,52.4%,1009.61,979,765.3
51
+ juggernaut_lightning_flux,0.006,1.1787893719956628,0.93,48.30471699999998,40.5%,916.68,929,
52
+ flux_1_1_pro,0.04,3.0104645500032348,2.34,49.92882700000001,50.6%,925.04,984,
53
+ flux_schnell,0.003,0.8411653029907029,0.80,46.685981818181816,34.8%,892.32,915,
54
+ kling_v2_1,N/A,N/A,N/A,51.044512,,870.04,981,
55
+ #p_image_2_ideogram_very_high_high_1k,0.075,9.34,7.25,58.45,,,1025,
56
+ #p_image_2_ideogram_very_high_low_1k,0.0375,26.17,10.44,57.59,,,1024,
57
+ #p_image_2_ideogram_very_high_medium_1k,0.05,9.28,5.96,57.88,,,1020,
58
+ #p_image_2_ideogram_very_high_very_low_1k,0.015,26.52,10.60,56.92,,,1002,
59
+ #p_image_2_ideogram_final_1k,0.0375,11.17,5.55,58.28,,,,
60
+ #p_image_2_ideogram_final_2k,0.075,14.34,9.96,57.68,,,,
data/qwen_image_bench_model_price_and_median_generation_time_10_august.csv DELETED
@@ -1,64 +0,0 @@
1
- Rapidata Model,Price / Image (USD),Median Generation Time (s),Min Generation Time (s),P-Judge Overall,Raw Win Rate,Rapidata Elo,Datapoint Elo,Benchmark.ai Elo
2
- reve_2_1,N/A,N/A,28.3,,,,,1173.2
3
- ideogram_4_0_quality,N/A,N/A,66.6,,,,,1131.0
4
- gpt_image_2,0.21,77.8,77.8,59.02083099999998,65.6%,1172.58,1110,1124.5
5
- nano_banana_2_0,N/A,N/A,N/A,56.362956,59.2%,1071.79,1063,1056.3
6
- gpt_image_1_5,0.135,38.0,38.0,57.864434999999986,58.9%,1102.09,1060,910.3
7
- hidream_i1_dev,0.0086,2.823569217998738,2.06,49.17682099999999,47.7%,999.17,983,
8
- gpt_image_1,0.167,38.8,38.8,54.975741000000006,47.4%,1095.39,984,
9
- imagen_4_0_ultra,0.06,11.4,11.4,53.385603999999965,52.9%,1075.22,1019,
10
- flux_2_flex,0.06,10.856533817990567,8.07,53.9245676767677,52.4%,1054.34,1014,921.3
11
- #flux_2_turbo,0.008,2.14,1.81,53.02,,,1003,
12
- #flux_2_flash,0.005,1.43,1.03,52.03,,,1001,
13
- qwen_image,0.025,4.8,4.8,51.561746,50.3%,1073.72,1002,975.6
14
- seedream_5_0,N/A,N/A,N/A,55.715153,51.6%,1070.35,1010,
15
- nano_banana_pro,0.134,17.2,17.2,56.452189898989914,56.4%,1029.51,1041,1102.4
16
- hidream_i1_fast,0.0051,9.920469530501578,1.40,48.89298600000002,47.2%,1024.67,980,
17
- imagen_4_fast,0.02,3.7531301500021073,2.71,50.155055208333344,47.2%,972.6,980,
18
- seedream_4_5,0.04,16.6,16.6,55.66328800000001,54.7%,1048.96,1032,964.2
19
- seedream_4_0,0.03,12.2,12.2,55.241443999999994,54.9%,1050.41,1032,
20
- p_image_2_ideogram_low_1k,0.0075,2.59,1.55,54.827397,50.8%,1000.07,1006,1103.9
21
- #p_image_2_ideogram_low_2k,0.016,5.17,4.25,53.971723,50.0%,1074.91,1000,1103.9
22
- qwen_image_2_0_pro,0.035,35.5,35.5,56.181776,52.2%,1033.25,1014,959.9
23
- juggernaut_base_flux,0.035,4.115834823496698,3.84,49.474676,46.1%,1038.22,971,
24
- flux_2_pro,N/A,N/A,N/A,54.43423900000002,52.2%,993.54,1014,1011.6
25
- z_image,0.005,1.5122045120006078,1.22,49.946227999999984,49.9%,1028.1,999,
26
- p_image_2_ideogram_high_1k,0.015,4.28,3.07,55.754507999999994,52.5%,972.92,1017,1104.0
27
- #p_image_2_ideogram_high_2k,0.06,8.69,7.12,54.757842000000004,50.7%,1074.91,1006,1104.0
28
- qwen_image_2512,0.02,19.1,19.1,51.677326,50.9%,1029.6,1005,
29
- flux_2_max,0.07,26.4,26.4,54.047976,53.9%,955.19,1026,1001.6
30
- flux_1_1_pro_ultra,0.06,9.026992494000297,6.20,50.358445,48.6%,979.56,990,
31
- juggernaut_pro_flux,0.055,3.699636150500737,3.36,49.741183,44.7%,972.14,963,
32
- flux_dev,0.025,1.6931055715031107,1.49,48.241479,41.5%,925.04,941,
33
- p_image_2_ideogram_very_low_1k,0.003,2.56,1.49,53.51894200000001,49.1%,959.86,994,1071.7
34
- #p_image_2_ideogram_very_low_2k,0.006,3.94,3.05,53.68248699999998,46.3%,962.38,976,1071.7
35
- #p_image_2_ideogram_very_low_1k_no_upsampling,0.005,0.82,,46.33,,,,1071.7
36
- #p_image_2_ideogram_very_low_2k_no_upsampling,0.005,2.18,,46.39,,,,1071.7
37
- #p_image_2_ideogram_low_1k_no_upsampling,0.01,3.33,,45.32,,,,1103.9
38
- #p_image_2_ideogram_low_2k_no_upsampling,0.01,3.34,,46.25,,,,1103.9
39
- #p_image_2_ideogram_medium_1k_no_upsampling,0.015,2.13,,46.88,,,,1115.1
40
- #p_image_2_ideogram_medium_2k_no_upsampling,0.015,5.55,,47.11,,,,1115.1
41
- #p_image_2_ideogram_high_1k_no_upsampling,0.03,3.11,,46.88,,,,1104.0
42
- #p_image_2_ideogram_high_2k_no_upsampling,0.03,5.55,,47.06,,,,1104.0
43
- hidream_i1_full,0.014,6.008430051002506,5.69,46.82559300000002,41.4%,955.46,941,
44
- flux_2_dev,0.025,4.310278721997747,4.03,52.71644489795918,52.5%,1007.6,1016,942.0
45
- imagen_4_0,0.04,14.1,14.1,52.08996199999999,50.4%,979.62,1002,
46
- wan_2_2_image,0.02,3.005390542501118,2.96,48.19959399999999,44.1%,944.87,958,
47
- flux_krea,0.025,1.7150160090022837,1.7150160090022837,50.35734,46.2%,919.73,973,
48
- p_image_2_ideogram_medium_1k,0.01,3.06,2.05,54.719193000000004,49.8%,941.46,999,1115.1
49
- #p_image_2_ideogram_medium_2k,0.02,7.44,6.47,54.23124444444446,49.3%,949.35,996,1115.1
50
- glm_image,0.05,188.2,188.2,51.42623399999999,42.9%,923.35,950,
51
- p_image,0.005,1.0640762715011078,0.95,48.75217099999999,44.3%,924.37,960,1098.7
52
- hunyuanimage_3_0,0.09,41.0,41.0,52.32440099999998,46.6%,1009.61,977,765.3
53
- juggernaut_lightning_flux,0.006,1.1787893719956628,0.93,48.30471699999998,40.0%,916.68,930,
54
- flux_1_1_pro,0.04,3.0104645500032348,2.34,49.92882700000001,47.7%,925.04,984,
55
- flux_schnell,0.003,0.8411653029907029,0.80,46.685981818181816,37.6%,892.32,914,
56
- kling_v2_1,N/A,N/A,N/A,51.044512,47.2%,870.04,980,
57
- #p_image_2_ideogram_very_high_high_1k,0.075,9.34,7.25,58.45,52.9%,,1024,
58
- #p_image_2_ideogram_very_high_low_1k,0.0375,26.17,10.44,57.59,52.9%,,1023,
59
- #p_image_2_ideogram_very_high_medium_1k,0.05,9.28,5.96,57.88,52.3%,,1019,
60
- #p_image_2_ideogram_very_high_very_low_1k,0.015,26.52,10.60,56.92,49.4%,,1001,
61
- p_image_2_ideogram_very_high_1k,0.033,11.17,5.55,58.28,55.5%,,1034,
62
- #p_image_2_ideogram_very_high_2k,0.066,14.34,9.96,57.68,53.4%,,1020,
63
- #p_image_2_ideogram_very_high_john_1k,,8.53,6.43,58.19,,,1015,
64
- #p_image_2_ideogram_very_high_john_2k,,13.86,10.64,57.75,,,1002,
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
data/video-editing-leaderboard.csv DELETED
@@ -1,12 +0,0 @@
1
- model_id,display_name,elo,wandb_run_ids,n_generations,min_generation_s,median_generation_s,p20_generation_s,generation_s_per_output_video_s,predict_time_s_per_output_video_s,model_execution_time_s_per_output_video_s,price
2
- gemini_omni_flash_edit__fal,Gemini Omni Flash Edit,1054,3iuctwj9,56,32.574746752001374,58.2327566820004,50.44402900500063,13.78,,,0.13
3
- grok_imagine_video__replicate,Grok Imagine Video,974,q0ogvekx,59,31.640928319000523,43.75649321700257,32.88782280539963,11.04,,,0.05
4
- happyhorse_1_0__wavespeed,HappyHorse 1.0,1048,k4zj5wqk,71,106.5824136010051,201.2431439649954,141.46595527700265,46.73,,,0.14
5
- ltx_2_3_quality_reference_video_to_video__fal,LTX 2.3 Video Edit,911,p19ks8ad,73,60.768441981999786,73.62109239100027,70.20137865180223,15.83,,,0.054
6
- lucy_edit_pro__fal,Lucy Edit Pro,879,na6bz9xx,72,118.3459930579993,136.44442766549764,124.13885582720104,27.18,,,0.15
7
- minimax_h3_reference_to_video__fal,MiniMax H3 Reference-to-Video,1060,639xmakz,63,180.8385945170012,308.29198316100155,248.2394383729996,58.31,,,0.06
8
- p_video_edit_preview__replicate_final,P-Video-Edit,1000,m8irpsa6,73,31.41135125700021,86.86665409700072,57.21095684959946,23.18,17.99,12.06,0.045
9
- p_video_edit_preview__replicate_final__draft,P-Video-Edit Draft,994,6f88e6mx,73,23.04809326099712,42.9797806409988,33.19193872759861,11.48,11.28,4.46,0.025
10
- seedance_2_5_video_edit_turbo__wavespeed,Seedance 2.5 Video Edit Turbo,1063,2nr274vp,68,142.49452170499717,293.7401115540015,223.11569286320045,59.93,,,0.24
11
- wan_2_7_video_edit__wavespeed,Wan 2.7 Video Edit,1055,9lq803be,66,134.3744560209998,313.957433804002,219.2491260079987,68.04,,,0.2
12
-
 
 
 
 
 
 
 
 
 
 
 
 
 
data/video_editing_combined/generations.jsonl DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:3ec4ea0a431484e65f30990c8cc4292b00c665be96db587775426da1c84a4ab8
3
- size 762227
 
 
 
 
data/video_editing_combined/prompts.jsonl DELETED
@@ -1,78 +0,0 @@
1
- {"prompt_id": "video_edit_internal__advertising__p0000", "text": "Replace the white bottle with an orange sunscreen bottle.", "dataset": "video_edit_internal", "category": "advertising", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/advertising/4620329ad65820ce.mp4"}
2
- {"prompt_id": "video_edit_internal__advertising__p0001", "text": "Replace the woman with an asian woman riding on a donkey through a chinese small town.", "dataset": "video_edit_internal", "category": "advertising", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/advertising/44db5689846be901.mp4"}
3
- {"prompt_id": "video_edit_internal__advertising__p0002", "text": "Place the couple on a snowy mountain next to a mountain hut.", "dataset": "video_edit_internal", "category": "advertising", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/advertising/481db4a047688bbe.mp4"}
4
- {"prompt_id": "video_edit_internal__advertising__p0003", "text": "Turn this into a scene in the summer with birds flying in the sky and butterflies in the foreground.", "dataset": "video_edit_internal", "category": "advertising", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/advertising/32b9f56763c56482.mp4"}
5
- {"prompt_id": "video_edit_internal__advertising__p0004", "text": "Replace the robot arm with a dancing hamster on a small table.", "dataset": "video_edit_internal", "category": "advertising", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/advertising/22d02ae2df7f9045.mp4"}
6
- {"prompt_id": "video_edit_internal__anonymization__p0000", "text": "Anonymize the womans face", "dataset": "video_edit_internal", "category": "anonymization", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/anonymization/7c410085f2fd939d.mp4"}
7
- {"prompt_id": "video_edit_internal__anonymization__p0001", "text": "Anonymize the mans face", "dataset": "video_edit_internal", "category": "anonymization", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/anonymization/7b495805c0c270ea.mp4"}
8
- {"prompt_id": "video_edit_internal__anonymization__p0002", "text": "Anonymize the mans face", "dataset": "video_edit_internal", "category": "anonymization", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/anonymization/736b6fb504bd32be.mp4"}
9
- {"prompt_id": "video_edit_internal__anonymization__p0003", "text": "Anonymize the girls face", "dataset": "video_edit_internal", "category": "anonymization", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/anonymization/2c82413d39579cd5.mp4"}
10
- {"prompt_id": "video_edit_internal__artificial_analysis__p0000", "text": "Change the rainforest setting to a neon-lit urban alley at night with steam from vents and wet reflective asphalt, and move into an ariel shot of the character after the initial camera movement to focus on the character", "dataset": "video_edit_internal", "category": "artificial_analysis", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/artificial_analysis/e2f6121269a206d3.mp4"}
11
- {"prompt_id": "video_edit_internal__artificial_analysis__p0001", "text": "Replace the magenta hover-car with a chrome-blue car, keeping the drift and neon light trails.", "dataset": "video_edit_internal", "category": "artificial_analysis", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/artificial_analysis/55d1baa7f3f71b26.mp4"}
12
- {"prompt_id": "video_edit_internal__artificial_analysis__p0002", "text": "Make the footage look like it was shot on a 1970s film camera, with grainy film texture, faded warm colors, slight softness, and slightly darker corners around the edges.", "dataset": "video_edit_internal", "category": "artificial_analysis", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/artificial_analysis/fd1ab436de020152.mp4"}
13
- {"prompt_id": "video_edit_internal__artificial_analysis__p0003", "text": "A person appears at the top of the waterfall, leaps into the lake below, and disappears into the misty water beneath the falls.", "dataset": "video_edit_internal", "category": "artificial_analysis", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/artificial_analysis/7ee1623303a8da68.mp4"}
14
- {"prompt_id": "video_edit_internal__artificial_analysis__p0004", "text": "A fluffy orange cat leaps gracefully from the floor onto the couch, paws sinking into the soft cushions. It circles once in place, tail swaying gently, before curling up tightly on one of the cushions and settling in comfortably.", "dataset": "video_edit_internal", "category": "artificial_analysis", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/artificial_analysis/9a73c608b343bbbe.mp4"}
15
- {"prompt_id": "video_edit_internal__camera_editing__p0000", "text": "Zoom in on the man's face to show his focused expression", "dataset": "video_edit_internal", "category": "camera_editing", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/camera_editing/0336e64b0594bd7a.mp4"}
16
- {"prompt_id": "video_edit_internal__camera_editing__p0001", "text": "Perform an arc shot around the tram as it arrives at the station", "dataset": "video_edit_internal", "category": "camera_editing", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/camera_editing/204aa93703da117b.mp4"}
17
- {"prompt_id": "video_edit_internal__camera_editing__p0002", "text": "Change the view to a high angle.", "dataset": "video_edit_internal", "category": "camera_editing", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/camera_editing/c2ccb5351be5c718.mp4"}
18
- {"prompt_id": "video_edit_internal__camera_editing__p0003", "text": "Change the view to a high angle.", "dataset": "video_edit_internal", "category": "camera_editing", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/camera_editing/b725aaded02d65e9.mp4"}
19
- {"prompt_id": "video_edit_internal__camera_editing__p0004", "text": "Gradually move the camera away from the doctor", "dataset": "video_edit_internal", "category": "camera_editing", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/camera_editing/33f24f4d083df743.mp4"}
20
- {"prompt_id": "video_edit_internal__design_arena__p0000", "text": "Create a transition of the video of the product whihc is the jeans on the lady model with an atitiude that goes into a zoom out camera that she is in a party .Focus on the vibes", "dataset": "video_edit_internal", "category": "design_arena", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/design_arena/b0b4642272a481b1.mp4"}
21
- {"prompt_id": "video_edit_internal__design_arena__p0001", "text": "add ethnic urban women dancing at the bottem and at the bar wearing DMI Tshirts", "dataset": "video_edit_internal", "category": "design_arena", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/design_arena/0f730f98f3c51356.mp4"}
22
- {"prompt_id": "video_edit_internal__design_arena__p0002", "text": "Generate a commercial of this dog drinking beer at an electronic music party in a world where dogs and humans are mixed together.", "dataset": "video_edit_internal", "category": "design_arena", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/design_arena/575fc4b9b6d7b3ab.mp4"}
23
- {"prompt_id": "video_edit_internal__design_arena__p0003", "text": "everything is the same excpet the winter season, snowing", "dataset": "video_edit_internal", "category": "design_arena", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/design_arena/8674fd022867a736.mp4"}
24
- {"prompt_id": "video_edit_internal__design_arena__p0004", "text": "A cinematic character introduction of Jason, framed in a medium close-up with subtle camera movement, confident body language, and expressive facial detail. Moody, high-contrast lighting with a cool-toned color palette, shallow depth of field, and a slow dramatic reveal that builds intrigue over a few seconds. I want a video in a 9:16 aspect ratio for tiktok. remove all text and have him in a mid day campus setting", "dataset": "video_edit_internal", "category": "design_arena", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/design_arena/91873bbafe4173da.mp4"}
25
- {"prompt_id": "video_edit_internal__e_commerce__p0000", "text": "Remove the black blazer.", "dataset": "video_edit_internal", "category": "e_commerce", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/e_commerce/b89faaa5080c31db.mp4"}
26
- {"prompt_id": "video_edit_internal__e_commerce__p0001", "text": "Replace her skirt with a jeans skirt.", "dataset": "video_edit_internal", "category": "e_commerce", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/e_commerce/c34a33d07e843804.mp4"}
27
- {"prompt_id": "video_edit_internal__e_commerce__p0002", "text": "Replace the black leggings and the black shirt with a white dress.", "dataset": "video_edit_internal", "category": "e_commerce", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/e_commerce/489d2439cbaeed79.mp4"}
28
- {"prompt_id": "video_edit_internal__e_commerce__p0003", "text": "Replace the white woman with a lebanese-looking woman.", "dataset": "video_edit_internal", "category": "e_commerce", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/e_commerce/0ad8fb773f303a11.mp4"}
29
- {"prompt_id": "video_edit_internal__e_commerce__p0004", "text": "Remove all items on the table and place an eyeshadow pallete on the table next to the woman.", "dataset": "video_edit_internal", "category": "e_commerce", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/e_commerce/f28f3aa12b1ec220.mp4"}
30
- {"prompt_id": "video_edit_internal__lighting__p0000", "text": "After the sun sets behind the mountains, the scene transitions into nighttime.", "dataset": "video_edit_internal", "category": "lighting", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/lighting/229c165e0ac97daf.mp4"}
31
- {"prompt_id": "video_edit_internal__lighting__p0001", "text": "Change the weather to a dazzling starry night.", "dataset": "video_edit_internal", "category": "lighting", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/lighting/f42c41ab2f3adc26.mp4"}
32
- {"prompt_id": "video_edit_internal__lighting__p0002", "text": "Change the weather to a thunderstorm with heavy rain.", "dataset": "video_edit_internal", "category": "lighting", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/lighting/daa36567634ca6be.mp4"}
33
- {"prompt_id": "video_edit_internal__lighting__p0003", "text": "Change the weather to a torrential downpour.", "dataset": "video_edit_internal", "category": "lighting", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/lighting/6817a4142b69d602.mp4"}
34
- {"prompt_id": "video_edit_internal__lighting__p0004", "text": "Change the weather to a torrential downpour.", "dataset": "video_edit_internal", "category": "lighting", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/lighting/ac01488a685a1096.mp4"}
35
- {"prompt_id": "video_edit_internal__long__p0000", "text": "Change the weather to a dazzling starry night.", "dataset": "video_edit_internal", "category": "long", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/long/e9af2a7039e43f16.mp4"}
36
- {"prompt_id": "video_edit_internal__long__p0001", "text": "Adjust the color of book to blue", "dataset": "video_edit_internal", "category": "long", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/long/eab696d1555be316.mp4"}
37
- {"prompt_id": "video_edit_internal__long__p0002", "text": "Make the young woman turn into sand and blow away.", "dataset": "video_edit_internal", "category": "long", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/long/d18378f1d7e4ca0e.mp4"}
38
- {"prompt_id": "video_edit_internal__long__p0003", "text": "Make the bird flap its wings", "dataset": "video_edit_internal", "category": "long", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/long/3835b37b43f5e5bf.mp4"}
39
- {"prompt_id": "video_edit_internal__long__p0004", "text": "Transform the video into a ukiyo-e style", "dataset": "video_edit_internal", "category": "long", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/long/09fe5bb4b5304b66.mp4"}
40
- {"prompt_id": "video_edit_internal__movie_concept_art__p0000", "text": "Add more blood and wounds to the mans face.", "dataset": "video_edit_internal", "category": "movie_concept_art", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/movie_concept_art/1d1a99167475720b.mp4"}
41
- {"prompt_id": "video_edit_internal__movie_concept_art__p0001", "text": "Make the scene less dark.", "dataset": "video_edit_internal", "category": "movie_concept_art", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/movie_concept_art/10cb44d8c0bf4116.mp4"}
42
- {"prompt_id": "video_edit_internal__movie_concept_art__p0002", "text": "Replace the female warrior with a male warrior with ginger hair.", "dataset": "video_edit_internal", "category": "movie_concept_art", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/movie_concept_art/9e45454390c20b8a.mp4"}
43
- {"prompt_id": "video_edit_internal__movie_concept_art__p0003", "text": "Turn the man's hand into a robotic hand.", "dataset": "video_edit_internal", "category": "movie_concept_art", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/movie_concept_art/99977e7a91a950b4.mp4"}
44
- {"prompt_id": "video_edit_internal__real_estate__p0000", "text": "Replace the interior with a gold-black interior heavy luxury style.", "dataset": "video_edit_internal", "category": "real_estate", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/real_estate/869064ab06bb5daa.mp4"}
45
- {"prompt_id": "video_edit_internal__real_estate__p0001", "text": "Replace the grey wallpaper with a beige painted wall and turn the grey curtains a dark brown.", "dataset": "video_edit_internal", "category": "real_estate", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/real_estate/1c2f358bfe9b8299.mp4"}
46
- {"prompt_id": "video_edit_internal__real_estate__p0002", "text": "Remove all decoration from the walls and keep only the furniture.", "dataset": "video_edit_internal", "category": "real_estate", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/real_estate/2058bcf38389f8fe.mp4"}
47
- {"prompt_id": "video_edit_internal__real_estate__p0003", "text": "Exchange the wooden floor for a marble floor.", "dataset": "video_edit_internal", "category": "real_estate", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/real_estate/6a4c09a23dda57d9.mp4"}
48
- {"prompt_id": "video_edit_internal__real_estate__p0004", "text": "Replace the bedframe with a modern wooden bed.", "dataset": "video_edit_internal", "category": "real_estate", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/real_estate/3e0f6dc5df891001.mp4"}
49
- {"prompt_id": "video_edit_internal__style_transfer__p0000", "text": "Transform the video into a cyberpunk style", "dataset": "video_edit_internal", "category": "style_transfer", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/style_transfer/2253f6ed3674f006.mp4"}
50
- {"prompt_id": "video_edit_internal__style_transfer__p0001", "text": "Convert to different shades of orange", "dataset": "video_edit_internal", "category": "style_transfer", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/style_transfer/edb6a105be03bae0.mp4"}
51
- {"prompt_id": "video_edit_internal__style_transfer__p0002", "text": "Convert to black and white", "dataset": "video_edit_internal", "category": "style_transfer", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/style_transfer/c3dc058b3be26cbd.mp4"}
52
- {"prompt_id": "video_edit_internal__style_transfer__p0003", "text": "Apply Ghibli-style editing to the video", "dataset": "video_edit_internal", "category": "style_transfer", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/style_transfer/83a638223a20911c.mp4"}
53
- {"prompt_id": "video_edit_internal__style_transfer__p0004", "text": "Relight the scene as if it were shot during golden hour, with warm low-angle sunlight, soft shadows, and natural highlights on faces and surfaces.", "dataset": "video_edit_internal", "category": "style_transfer", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/style_transfer/daa36567634ca6be.mp4"}
54
- {"prompt_id": "video_edit_internal__subject_editing__p0000", "text": "Replace the rainbow colors of the logo with different shades of purple", "dataset": "video_edit_internal", "category": "subject_editing", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/subject_editing/6358f26b46baadf5.mp4"}
55
- {"prompt_id": "video_edit_internal__subject_editing__p0001", "text": "Add a small dog running beside the scooter", "dataset": "video_edit_internal", "category": "subject_editing", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/subject_editing/052c0bfd7b68fea7.mp4"}
56
- {"prompt_id": "video_edit_internal__subject_editing__p0002", "text": "Replace the splashing waves with a calm water surface", "dataset": "video_edit_internal", "category": "subject_editing", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/subject_editing/c3f79578a39f415d.mp4"}
57
- {"prompt_id": "video_edit_internal__subject_editing__p0003", "text": "Add a violinist in the background", "dataset": "video_edit_internal", "category": "subject_editing", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/subject_editing/a0eebf32a076e2fc.mp4"}
58
- {"prompt_id": "video_edit_internal__subject_editing__p0004", "text": "Add a group of people walking on the pathway", "dataset": "video_edit_internal", "category": "subject_editing", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/subject_editing/3c2b7e3a5b284e1d.mp4"}
59
- {"prompt_id": "video_edit_internal__subject_motion_editing__p0000", "text": "Make the bird hop around instead of walking and foraging", "dataset": "video_edit_internal", "category": "subject_motion_editing", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/subject_motion_editing/ac38b2a97d1a5c3a.mp4"}
60
- {"prompt_id": "video_edit_internal__subject_motion_editing__p0001", "text": "The male colleague is walking around to observe.", "dataset": "video_edit_internal", "category": "subject_motion_editing", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/subject_motion_editing/8aaab7d304c5ad99.mp4"}
61
- {"prompt_id": "video_edit_internal__subject_motion_editing__p0002", "text": "Make the static spider-man in the mural dynamic and make him swing faster", "dataset": "video_edit_internal", "category": "subject_motion_editing", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/subject_motion_editing/3412b39686fd5879.mp4"}
62
- {"prompt_id": "video_edit_internal__subject_motion_editing__p0003", "text": "Change the woman's jogging to taking off.", "dataset": "video_edit_internal", "category": "subject_motion_editing", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/subject_motion_editing/17c4dc0514772321.mp4"}
63
- {"prompt_id": "video_edit_internal__subject_motion_editing__p0004", "text": "Make the knight lunging forward and the creature swiping at the knight", "dataset": "video_edit_internal", "category": "subject_motion_editing", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/subject_motion_editing/980354550011911b.mp4"}
64
- {"prompt_id": "video_edit_internal__synthetic_data__p0000", "text": "Turn the scene into nighttime.", "dataset": "video_edit_internal", "category": "synthetic_data", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/synthetic_data/e3b1e603bfaa1aec.mp4"}
65
- {"prompt_id": "video_edit_internal__synthetic_data__p0001", "text": "Remove the crosswalk.", "dataset": "video_edit_internal", "category": "synthetic_data", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/synthetic_data/30df5fd5fe28af65.mp4"}
66
- {"prompt_id": "video_edit_internal__synthetic_data__p0002", "text": "Add a bicycle riding in front of the car.", "dataset": "video_edit_internal", "category": "synthetic_data", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/synthetic_data/d093491e25682b3a.mp4"}
67
- {"prompt_id": "video_edit_internal__synthetic_data__p0003", "text": "Turn the scene into a snowstorm scene.", "dataset": "video_edit_internal", "category": "synthetic_data", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/synthetic_data/1da5913f13362f8d.mp4"}
68
- {"prompt_id": "video_edit_internal__synthetic_data__p0004", "text": "Make it rain in the scene.", "dataset": "video_edit_internal", "category": "synthetic_data", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/synthetic_data/936bdb89558f5a7f.mp4"}
69
- {"prompt_id": "video_edit_internal__text__p0000", "text": "Add the text \"True Love\" in the foreground in pink romantic font.", "dataset": "video_edit_internal", "category": "text", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/text/c763407793ca9b0e.mp4"}
70
- {"prompt_id": "video_edit_internal__text__p0001", "text": "Replace any mention of \"Fanta\" with the branding \"Cola\".", "dataset": "video_edit_internal", "category": "text", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/text/7b878ea6d99834b2.mp4"}
71
- {"prompt_id": "video_edit_internal__text__p0002", "text": "Remove all text from the video.", "dataset": "video_edit_internal", "category": "text", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/text/76763c31b2609109.mp4"}
72
- {"prompt_id": "video_edit_internal__text__p0003", "text": "Replace the branding \"Royalty\" with the phrasing \"Princess\".", "dataset": "video_edit_internal", "category": "text", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/text/d221d299c9e49064.mp4"}
73
- {"prompt_id": "video_edit_internal__text__p0004", "text": "Remove all text from the video.", "dataset": "video_edit_internal", "category": "text", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/text/5c169901af8e4ed6.mp4"}
74
- {"prompt_id": "video_edit_internal__transitions__p0000", "text": "After a black screen transition, the road transforms into lush grass.", "dataset": "video_edit_internal", "category": "transitions", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/transitions/964c35b19ee1fc1f.mp4"}
75
- {"prompt_id": "video_edit_internal__transitions__p0001", "text": "After a wave-foam transition, the small fishing boat is eaten by a giant whale.", "dataset": "video_edit_internal", "category": "transitions", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/transitions/b8b88f43316363f3.mp4"}
76
- {"prompt_id": "video_edit_internal__transitions__p0002", "text": "Add a cut transition, then show a bowl of ramen topped with parsley.", "dataset": "video_edit_internal", "category": "transitions", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/transitions/c823d6c77bdc87f4.mp4"}
77
- {"prompt_id": "video_edit_internal__transitions__p0003", "text": "Add a smoke transition, then show the circuit board burning.", "dataset": "video_edit_internal", "category": "transitions", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/transitions/edb9927ff06c1175.mp4"}
78
- {"prompt_id": "video_edit_internal__transitions__p0004", "text": "After a smoke transition, the Basilica of the Sacred Heart of Paris catches fire.", "dataset": "video_edit_internal", "category": "transitions", "source_video": "https://d2j1a65dna040x.cloudfront.net/benchmark_inputs/video_edit_internal/transitions/77bb900383d4f370.mp4"}
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
data/video_generation_combined/generations.jsonl DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:8395b068f818174bbbbd0af450c6cedcc76142352f35ccf565f55c4837cb43b4
3
- size 1135969
 
 
 
 
data/video_generation_combined/prompts.jsonl DELETED
@@ -1,90 +0,0 @@
1
- {"prompt_id": "v-bench2__all__p0000", "text": "Garden, zoom in.", "dataset": "v-bench2", "category": "all"}
2
- {"prompt_id": "v-bench2__all__p0001", "text": "Garden, zoom out.", "dataset": "v-bench2", "category": "all"}
3
- {"prompt_id": "v-bench2__all__p0002", "text": "Garden, tilt up.", "dataset": "v-bench2", "category": "all"}
4
- {"prompt_id": "v-bench2__all__p0003", "text": "The camera starts at the top of the forest, where thick morning mist drifts between the treetops, with sunlight filtering through the leaves and casting dappled spots of light, the air fresh and moist. As the camera slowly moves downward, dewdrops glisten on the leaves, a gentle breeze rustling the leaves, making a soft, soothing sound. The camera moves along a woodland path, occasionally capturing a squirrel leaping between trees, its agile figure darting across the branches. The shot continues, eventually arriving at a tranquil lake. The surface of the water is like a mirror, reflecting the towering trees and distant mountains. In the distance, a heron takes off from the lakeshore, breaking the stillness of the water, ripples forming as the bird skims the surface. The camera follows its flight and gradually pulls back, showing the entire serene beauty of the lake.", "dataset": "v-bench2", "category": "all"}
5
- {"prompt_id": "v-bench2__all__p0004", "text": "The scene opens in the vast desert, with the camera angled low over the sand, fine grains drifting in the gentle breeze, the undulating dunes in the distance bathed in a faint orange glow. As the camera moves forward, the outlines of the dunes become clearer, and the early morning sky begins to glow with warm hues, transitioning the desert from deep brown to golden yellow. The scene cuts to the top of a dune, where the camera follows a caravan of camels making their way across the endless desert, their bells jingling softly in the wind. The first rays of the sun break through the gaps between the dunes, casting golden light on the sand, as the desert comes to life, glowing in the morning's embrace. Finally, the camera pulls back, revealing the vast expanse of the desert merging with the horizon, the sun climbing higher, illuminating the entire landscape in a mystical, warm glow.", "dataset": "v-bench2", "category": "all"}
6
- {"prompt_id": "v-bench2__all__p0005", "text": "The camera begins in a vast grassland, where the lush green grass sways gently in the breeze, the air fresh, and the soft rustling of the leaves fills the space. As the camera moves forward, the grass turns from a vibrant green to golden hues, sunlight pouring over the land, the textures of the grass becoming more pronounced in the play of light and shadow. In the distance, a herd of cattle grazes peacefully. Their figures become clearer as the camera moves closer, occasionally lifting their heads and gazing into the distance. The scene shifts to a lakeshore, where the lake reflects the sky, grasslands, and distant mountains, the water clear and calm, with gentle ripples created by the breeze. Finally, the camera pulls back, following the herd as they move off into the distance, the vastness of the grassland merging with the expansive sky, creating a feeling of peaceful openness.", "dataset": "v-bench2", "category": "all"}
7
- {"prompt_id": "v-bench2__all__p0006", "text": "The camera starts at a tranquil coastline, where the waves gently crash against the rocks, the salty sea breeze fills the air, and the atmosphere feels fresh and alive. As the camera moves forward, the surface of the water glistens with golden light from the setting sun, the sound of the waves becoming more distinct as the tide retreats, revealing a moist sandy shore. The scene shifts to the beach, where a few seagulls peck at the sand, occasionally flying up and breaking the stillness of the sky. The sun slowly sinks below the horizon, painting the sky in vibrant shades of orange and red, as the golden light on the water fades, and the waves begin to intensify. Finally, the camera pulls back, and the entire coastline fades into twilight, with the sea and sky blending together in a serene, solitary embrace.", "dataset": "v-bench2", "category": "all"}
8
- {"prompt_id": "v-bench2__all__p0007", "text": "The race began, and the first runner quickly took off, leading the other teams. Everyone was focused on his performance. During the baton handoff of the first runner, Team A made a slight mistake, and the baton nearly fell. Despite this, the runner quickly passed it to the second runner. The second runner showed great determination and chased hard, finally overtaking Team B on the bend. However, due to the earlier mistake, Team A's lead was not significant. During the third runner's handoff, the runner from Team A nervously accelerated and managed to maintain the lead, but Team C quickly caught up thanks to their excellent pace control. During the fourth runner's handoff, the final sprint became crucial. The Team A runner started to accelerate in the second-to-last turn and, with a burst of strength, successfully widened the gap, steadily running toward the finish line and winning the race.", "dataset": "v-bench2", "category": "all"}
9
- {"prompt_id": "v-bench2__all__p0008", "text": "The match began, and Team A quickly organized an attack, breaking through Team B's defense with fast passes. The forward accurately kicked the ball into the goal, taking a 1-0 lead. Team B did not panic; they used a long pass to quickly counterattack. The forward calmly shot past the goalkeeper, and the ball went straight into the net, making it 1-1. At this point, the game entered a stalemate. Both teams' defenses were solid, and the match became a fierce battle. In the second half, Team A had a corner kick. The high ball delivered by the player was headed in by the center-back who jumped high, putting Team A in the lead at 2-1. In the final moments of the game, Team B got a penalty kick. The forward took the shot without hesitation, sending the ball into the net to make it 2-2. Finally, in the last moments of extra time, Team A used a quick counterattack and a long-range shot from outside the box flew straight into the corner, securing the win with a 3-2 score.", "dataset": "v-bench2", "category": "all"}
10
- {"prompt_id": "v-bench2__all__p0009", "text": "The race began, and all the runners sprinted quickly. The runner from Team A quickly took the lead. After a period of steady running, the runners from Team A and Team B gradually created a gap, almost running shoulder to shoulder. However, at the 25 km mark, the runner from Team A suddenly began to feel unwell, slowing down noticeably. The runner from Team B seized the opportunity and caught up, eventually overtaking Team A to take the lead at the 30 km mark. As the race entered the second half, the Team A runner gave it their all, regained their rhythm, and caught up with Team B at the 35 km mark. The two were neck and neck in the final sprint, and with only 200 meters remaining, the Team A runner pushed through with determination and accelerated to cross the finish line first by a narrow margin, winning the tough race.", "dataset": "v-bench2", "category": "all"}
11
- {"prompt_id": "v-bench2__all__p0010", "text": "The race began, and the runners quickly started. The Team A runner took the lead initially due to a powerful start. However, the Team B runner did not rush to chase but instead steadily adjusted their pace, ensuring they had more energy for the latter part of the race. By the third lap, the Team A runner began to tire, and the Team B runner gradually reduced the gap, eventually overtaking Team A in the fourth lap. The Team C runner, meanwhile, began to accelerate, and in the last two laps, with strong willpower, they overtook Team B and moved into the lead. In the final sprint, the Team A runner gritted their teeth and tried to close the gap, but with only 50 meters left, the Team C runner exploded with speed, crossing the finish line with a clear lead and winning the race.", "dataset": "v-bench2", "category": "all"}
12
- {"prompt_id": "v-bench2__all__p0011", "text": "The match began, and Team A quickly found their rhythm. A series of fast attacks put pressure on Team B’s defense. Team A's forward made two three-pointers, quickly increasing the lead. After the first quarter, Team B adjusted their tactics, strengthened their defense, and gradually started to counterattack, narrowing the gap. In the third quarter, Team A’s key player was injured and forced to leave the game. Team B took advantage of this by speeding up their attacks, overtaking the score. In the fourth quarter, Team A's bench players caught up with a series of fast breaks, and in the final moments, a player from Team A calmly made a game-winning three-pointer from beyond the arc, securing a narrow 1-point victory.", "dataset": "v-bench2", "category": "all"}
13
- {"prompt_id": "v-bench2__all__p0012", "text": "The match began, and Team A quickly gained the upper hand with several precise kills, taking a 11-6 lead. The Team B player stayed calm and gradually adjusted, finding ways to respond. In the second set, Team B improved their serving quality, scoring several powerful smashes to win a set, leveling the score at 1-1. In the deciding set, the Team A player showed signs of fatigue, but with precise net control and quick reflexes, they pulled ahead. In the crucial final point, a lightning-fast smash left Team B with no chance to return, and Team A won the match 21-19.", "dataset": "v-bench2", "category": "all"}
14
- {"prompt_id": "v-bench2__all__p0013", "text": "A lion with the wings of an eagle, soaring through the sky with majestic ease.", "dataset": "v-bench2", "category": "all"}
15
- {"prompt_id": "v-bench2__all__p0014", "text": "A giraffe with the scales of a fish, able to glide smoothly through the water.", "dataset": "v-bench2", "category": "all"}
16
- {"prompt_id": "v-bench2__all__p0015", "text": "A wolf with the body of a horse, galloping across a vast plains with wild abandon.", "dataset": "v-bench2", "category": "all"}
17
- {"prompt_id": "v-bench2__all__p0016", "text": "A bear with the antlers of a deer, roaming the forest with a regal presence.", "dataset": "v-bench2", "category": "all"}
18
- {"prompt_id": "v-bench2__all__p0017", "text": "A cheetah with the shell of a tortoise, moving quickly but with a protective outer layer.", "dataset": "v-bench2", "category": "all"}
19
- {"prompt_id": "v-bench2__all__p0018", "text": "A wooden toy is placed gently on the surface of a small bowl of water.", "dataset": "v-bench2", "category": "all"}
20
- {"prompt_id": "v-bench2__all__p0019", "text": "A river changes from blue to brown.", "dataset": "v-bench2", "category": "all"}
21
- {"prompt_id": "v-bench2__all__p0020", "text": "The leaves gradually change from red to green.", "dataset": "v-bench2", "category": "all"}
22
- {"prompt_id": "v-bench2__all__p0021", "text": "A river changes from brown to blue.", "dataset": "v-bench2", "category": "all"}
23
- {"prompt_id": "v-bench2__all__p0022", "text": "A car changes from white to red.", "dataset": "v-bench2", "category": "all"}
24
- {"prompt_id": "v-bench2__all__p0023", "text": "A car changes from red to white.", "dataset": "v-bench2", "category": "all"}
25
- {"prompt_id": "v-bench2__all__p0024", "text": "A dog is on the left of a table, then the dog runs to the front of the table.", "dataset": "v-bench2", "category": "all"}
26
- {"prompt_id": "v-bench2__all__p0025", "text": "A dog is on the left of a sofa, then the dog runs to the front of the sofa.", "dataset": "v-bench2", "category": "all"}
27
- {"prompt_id": "v-bench2__all__p0026", "text": "A dog is on the right of a table, then the dog runs to the left of the table.", "dataset": "v-bench2", "category": "all"}
28
- {"prompt_id": "v-bench2__all__p0027", "text": "A dog is on the right of a sofa, then the dog runs to the front of the sofa.", "dataset": "v-bench2", "category": "all"}
29
- {"prompt_id": "v-bench2__all__p0028", "text": "A dog is on the right of a rock, then the dog runs to the left of the rock.", "dataset": "v-bench2", "category": "all"}
30
- {"prompt_id": "v-bench2__all__p0029", "text": "A dog is behind a chair, then the dog runs to the right of the chair.", "dataset": "v-bench2", "category": "all"}
31
- {"prompt_id": "v-bench2__all__p0030", "text": "A man is doing yoga.", "dataset": "v-bench2", "category": "all"}
32
- {"prompt_id": "v-bench2__all__p0031", "text": "A woman is doing yoga.", "dataset": "v-bench2", "category": "all"}
33
- {"prompt_id": "v-bench2__all__p0032", "text": "people are doing yoga.", "dataset": "v-bench2", "category": "all"}
34
- {"prompt_id": "v-bench2__all__p0033", "text": "A man is running.", "dataset": "v-bench2", "category": "all"}
35
- {"prompt_id": "v-bench2__all__p0034", "text": "A woman is running.", "dataset": "v-bench2", "category": "all"}
36
- {"prompt_id": "v-bench2__all__p0035", "text": "people are running.", "dataset": "v-bench2", "category": "all"}
37
- {"prompt_id": "v-bench2__all__p0036", "text": "A man is walking.", "dataset": "v-bench2", "category": "all"}
38
- {"prompt_id": "v-bench2__all__p0037", "text": "A woman is walking.", "dataset": "v-bench2", "category": "all"}
39
- {"prompt_id": "v-bench2__all__p0038", "text": "people are walking.", "dataset": "v-bench2", "category": "all"}
40
- {"prompt_id": "v-bench2__all__p0039", "text": "A man is dancing.", "dataset": "v-bench2", "category": "all"}
41
- {"prompt_id": "v-bench2__all__p0040", "text": "A woman is dancing.", "dataset": "v-bench2", "category": "all"}
42
- {"prompt_id": "v-bench2__all__p0041", "text": "A man is playing basketball.", "dataset": "v-bench2", "category": "all"}
43
- {"prompt_id": "v-bench2__all__p0042", "text": "A woman is playing basketball.", "dataset": "v-bench2", "category": "all"}
44
- {"prompt_id": "v-bench2__all__p0043", "text": "One person hands a cup of water to another.", "dataset": "v-bench2", "category": "all"}
45
- {"prompt_id": "v-bench2__all__p0044", "text": "One person passes a ball to another.", "dataset": "v-bench2", "category": "all"}
46
- {"prompt_id": "v-bench2__all__p0045", "text": "Two people shake hands.", "dataset": "v-bench2", "category": "all"}
47
- {"prompt_id": "v-bench2__all__p0046", "text": "One person ties the shoelaces of another person.", "dataset": "v-bench2", "category": "all"}
48
- {"prompt_id": "v-bench2__all__p0047", "text": "One person opens the door for another person.", "dataset": "v-bench2", "category": "all"}
49
- {"prompt_id": "v-bench2__all__p0048", "text": "Two people exchange a book.", "dataset": "v-bench2", "category": "all"}
50
- {"prompt_id": "v-bench2__all__p0049", "text": "One person puts a coat on another person.", "dataset": "v-bench2", "category": "all"}
51
- {"prompt_id": "v-bench2__all__p0050", "text": "One person places a chair for another to sit in.", "dataset": "v-bench2", "category": "all"}
52
- {"prompt_id": "v-bench2__all__p0051", "text": "An orange dog is running.", "dataset": "v-bench2", "category": "all"}
53
- {"prompt_id": "v-bench2__all__p0052", "text": "Two orange dogs are running.", "dataset": "v-bench2", "category": "all"}
54
- {"prompt_id": "v-bench2__all__p0053", "text": "A brown dog is on the left of an apple, then the dog moves to the right of the apple.", "dataset": "v-bench2", "category": "all"}
55
- {"prompt_id": "v-bench2__all__p0054", "text": "An orange cat is running.", "dataset": "v-bench2", "category": "all"}
56
- {"prompt_id": "v-bench2__all__p0055", "text": "Equal amounts of yellow and blue paint are rapidly combined, with the mixture being vigorously stirred until fully blended.", "dataset": "v-bench2", "category": "all"}
57
- {"prompt_id": "v-bench2__all__p0056", "text": "Equal amounts of black and white paint are rapidly combined, with the mixture being vigorously stirred until fully blended.", "dataset": "v-bench2", "category": "all"}
58
- {"prompt_id": "v-bench2__all__p0057", "text": "Equal amounts of white and black paint are rapidly combined, with the mixture being vigorously stirred until fully blended.", "dataset": "v-bench2", "category": "all"}
59
- {"prompt_id": "v-bench2__all__p0058", "text": "Equal amounts of red and purple paint are rapidly combined, with the mixture being vigorously stirred until fully blended.", "dataset": "v-bench2", "category": "all"}
60
- {"prompt_id": "v-bench2__all__p0059", "text": "A bowl of soup is tilted in the space station, with the liquid slowly spreading in all directions.", "dataset": "v-bench2", "category": "all"}
61
- {"prompt_id": "v-bench2__all__p0060", "text": "A jar of peanut butter is opened in the space station, with the viscous liquid slowly dispersing.", "dataset": "v-bench2", "category": "all"}
62
- {"prompt_id": "v-bench2__all__p0061", "text": "A container of cream is opened in the space station, with the liquid slowly dispersing into the air.", "dataset": "v-bench2", "category": "all"}
63
- {"prompt_id": "v-bench2__all__p0062", "text": "A cup of tea is carefully tilted in the space station, and the liquid floats in various directions.", "dataset": "v-bench2", "category": "all"}
64
- {"prompt_id": "v-bench2__all__p0063", "text": "A bottle of ketchup is gently squeezed in the space station, with the thick liquid spreading into the environment.", "dataset": "v-bench2", "category": "all"}
65
- {"prompt_id": "v-bench2__all__p0064", "text": "A bottle of water is opened in the space station, and the water starts to float out in irregular shapes.", "dataset": "v-bench2", "category": "all"}
66
- {"prompt_id": "v-bench2__all__p0065", "text": "A person is sitting on the couch, then suddenly they get up and start sweeping the floor.", "dataset": "v-bench2", "category": "all"}
67
- {"prompt_id": "v-bench2__all__p0066", "text": "A dog is playing with a ball, then it suddenly starts lying down on the carpet.", "dataset": "v-bench2", "category": "all"}
68
- {"prompt_id": "v-bench2__all__p0067", "text": "A person is cooking dinner, then they suddenly start organizing the pantry.", "dataset": "v-bench2", "category": "all"}
69
- {"prompt_id": "v-bench2__all__p0068", "text": "A cat is watching birds through the window, then it suddenly starts grooming itself.", "dataset": "v-bench2", "category": "all"}
70
- {"prompt_id": "v-bench2__all__p0069", "text": "A person is typing on a keyboard, then they suddenly get up and start making the bed.", "dataset": "v-bench2", "category": "all"}
71
- {"prompt_id": "v-bench2__all__p0070", "text": "A horse is trotting in the field, then it suddenly starts drinking from a stream.", "dataset": "v-bench2", "category": "all"}
72
- {"prompt_id": "v-bench2__all__p0071", "text": "A person is drinking a glass of water, then they suddenly start cleaning the windows.", "dataset": "v-bench2", "category": "all"}
73
- {"prompt_id": "v-bench2__all__p0072", "text": "A dog is sitting in the yard, then it suddenly starts running in circles.", "dataset": "v-bench2", "category": "all"}
74
- {"prompt_id": "v-bench2__all__p0073", "text": "A person is slurping noodles from a steaming bowl.", "dataset": "v-bench2", "category": "all"}
75
- {"prompt_id": "v-bench2__all__p0074", "text": "A person is eating hamburger.", "dataset": "v-bench2", "category": "all"}
76
- {"prompt_id": "v-bench2__all__p0075", "text": "A person is eating ice cream.", "dataset": "v-bench2", "category": "all"}
77
- {"prompt_id": "v-bench2__all__p0076", "text": "A person is drinking coffee from a cup.", "dataset": "v-bench2", "category": "all"}
78
- {"prompt_id": "v-bench2__all__p0077", "text": "A person is spreading butter on toast.", "dataset": "v-bench2", "category": "all"}
79
- {"prompt_id": "v-bench2__all__p0078", "text": "A person is eating spaghetti with a fork.", "dataset": "v-bench2", "category": "all"}
80
- {"prompt_id": "v-bench2__all__p0079", "text": "A person is biting into an apple.", "dataset": "v-bench2", "category": "all"}
81
- {"prompt_id": "v-bench2__all__p0080", "text": "A person is peeling a banana.", "dataset": "v-bench2", "category": "all"}
82
- {"prompt_id": "v-bench2__all__p0081", "text": "The camera orbits around. Castle, the camera circles around.", "dataset": "v-bench2", "category": "all"}
83
- {"prompt_id": "v-bench2__all__p0082", "text": "The camera orbits around. Volcano, the camera circles around.", "dataset": "v-bench2", "category": "all"}
84
- {"prompt_id": "v-bench2__all__p0083", "text": "The camera orbits around. Statue, the camera circles around.", "dataset": "v-bench2", "category": "all"}
85
- {"prompt_id": "v-bench2__all__p0084", "text": "The camera orbits around. Clock Tower, the camera circles around.", "dataset": "v-bench2", "category": "all"}
86
- {"prompt_id": "v-bench2__all__p0085", "text": "The camera orbits around. Playground, the camera circles around.", "dataset": "v-bench2", "category": "all"}
87
- {"prompt_id": "v-bench2__all__p0086", "text": "A timelapse captures the transformation of water in an untextured bottle as the temperature significantly drops below 0°C.", "dataset": "v-bench2", "category": "all"}
88
- {"prompt_id": "v-bench2__all__p0087", "text": "A timelapse captures the transformation of a river as the temperature significantly drops below 0°C.", "dataset": "v-bench2", "category": "all"}
89
- {"prompt_id": "v-bench2__all__p0088", "text": "A timelapse captures the transformation of juice in an untextured bottle as the temperature significantly drops below 0°C.", "dataset": "v-bench2", "category": "all"}
90
- {"prompt_id": "v-bench2__all__p0089", "text": "A timelapse captures the transformation of milk in an untextured bottle as the temperature significantly drops below 0°C.", "dataset": "v-bench2", "category": "all"}
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
model_display.py CHANGED
@@ -87,35 +87,6 @@ MODEL_DISPLAY_NAMES = {
87
  "p_image_2_ideogram_high_1k": "P-Image-Ideogram High 1K",
88
  "p_image_2_ideogram_high_2k": "P-Image-Ideogram High 2K",
89
  "P-Image-Ideogram (High)": "P-Image-Ideogram High",
90
- "p_image_2_ideogram_very_high_1k": "P-Image-Ideogram Very High 1K",
91
- # P-Video-Edit
92
- "P-Video-Edit": "P-Video-Edit",
93
- "P-Video-Edit Draft": "P-Video-Edit Draft",
94
- "P-Video Edit Final": "P-Video-Edit",
95
- "P-Video Edit Final (draft)": "P-Video-Edit Draft",
96
- "p_video_edit_preview__replicate_final": "P-Video-Edit",
97
- "p_video_edit_preview__replicate_final__draft": "P-Video-Edit Draft",
98
- # Text-to-video leaderboard
99
- "gemini_omni_1_1_flash": "Gemini Omni 1.1 Flash",
100
- "grok_imagine_video": "Grok Imagine Video",
101
- "grok_imagine_video_v1_5": "Grok Imagine Video 1.5",
102
- "ltx_2_5_fast": "LTX 2.5 Fast",
103
- "ltx_2_5_pro": "LTX 2.5 Pro",
104
- "minimax_h3": "MiniMax H3",
105
- "minimax_h3_max": "MiniMax H3 Max",
106
- "minimax_h3_max_turbo": "MiniMax H3 Max Turbo",
107
- "minimax_h3_max_turbo__prompt_expansion_mode_balanced": "MiniMax H3 Max Turbo",
108
- "seedance_2_5_turbo": "Seedance 2.5 Turbo",
109
- "veo_3_1_lite": "Veo 3.1 Lite",
110
- # Video-to-video leaderboard
111
- "gemini_omni_flash_edit__fal": "Gemini Omni Flash Edit",
112
- "grok_imagine_video__replicate": "Grok Imagine Video",
113
- "happyhorse_1_0__wavespeed": "HappyHorse 1.0",
114
- "ltx_2_3_quality_reference_video_to_video__fal": "LTX 2.3 Video Edit",
115
- "lucy_edit_pro__fal": "Lucy Edit Pro",
116
- "minimax_h3_reference_to_video__fal": "MiniMax H3 Reference-to-Video",
117
- "seedance_2_5_video_edit_turbo__wavespeed": "Seedance 2.5 Video Edit Turbo",
118
- "wan_2_7_video_edit__wavespeed": "Wan 2.7 Video Edit",
119
  # Others overlapping P-Bench
120
  "z_image": "Z-Image",
121
  "glm_image": "GLM-Image",
@@ -213,35 +184,6 @@ MODEL_DISPLAY_NAMES = {
213
  }
214
 
215
 
216
- def _p_video_2_display_name(model_id: str):
217
- """Turn p_video_2 variant ids into P-Video-2 Draft 720p labels."""
218
- raw = str(model_id).strip()
219
- if raw != "p_video_2" and not raw.startswith("p_video_2__"):
220
- return None
221
- if raw == "p_video_2":
222
- return "P-Video-2"
223
-
224
- draft = False
225
- upsample = None
226
- resolution = None
227
- for part in raw.split("__")[1:]:
228
- if part.startswith("draft_"):
229
- draft = part.endswith("true")
230
- elif part.startswith("prompt_upsampling_"):
231
- upsample = part.endswith("true")
232
- elif part.startswith("resolution_"):
233
- resolution = part[len("resolution_") :]
234
-
235
- label = "P-Video-2"
236
- if draft:
237
- label += " Draft"
238
- if resolution:
239
- label += f" {resolution}"
240
- if upsample is False:
241
- label += " (no prompt upsampling)"
242
- return label
243
-
244
-
245
  def _prettify_snake_case(model_id: str) -> str:
246
  parts = [part for part in str(model_id).split("_") if part]
247
  pretty = []
@@ -266,10 +208,7 @@ def display_model_name(model_id) -> str:
266
  return ""
267
  if raw in MODEL_DISPLAY_NAMES:
268
  return MODEL_DISPLAY_NAMES[raw]
269
- p_video_2 = _p_video_2_display_name(raw)
270
- if p_video_2:
271
- return p_video_2
272
  # Already a human label (spaces / punctuation) — keep as-is.
273
- if re.search(r"[\s.\[\]()-]", raw):
274
  return raw
275
  return _prettify_snake_case(raw)
 
87
  "p_image_2_ideogram_high_1k": "P-Image-Ideogram High 1K",
88
  "p_image_2_ideogram_high_2k": "P-Image-Ideogram High 2K",
89
  "P-Image-Ideogram (High)": "P-Image-Ideogram High",
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
90
  # Others overlapping P-Bench
91
  "z_image": "Z-Image",
92
  "glm_image": "GLM-Image",
 
184
  }
185
 
186
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
187
  def _prettify_snake_case(model_id: str) -> str:
188
  parts = [part for part in str(model_id).split("_") if part]
189
  pretty = []
 
208
  return ""
209
  if raw in MODEL_DISPLAY_NAMES:
210
  return MODEL_DISPLAY_NAMES[raw]
 
 
 
211
  # Already a human label (spaces / punctuation) — keep as-is.
212
+ if re.search(r"[\s.\[\]()]", raw):
213
  return raw
214
  return _prettify_snake_case(raw)
ui.py CHANGED
@@ -1,5 +1,4 @@
1
  from html import escape
2
- from math import ceil, floor, log10
3
  from pathlib import Path
4
  import base64
5
  import random
@@ -25,43 +24,13 @@ MAX_COMPARE_PROMPTS = 8
25
  MAX_PARETO_METRICS = 8
26
  _PARETO_SLOT_COUNT = 1 + MAX_PARETO_METRICS * 8
27
  _PARETO_PRICE_COLUMN = "Price / Image (USD)"
28
- _PARETO_VIDEO_PRICE_COLUMN = "Price / Second of Video (USD)"
29
- _PARETO_PRICE_COLUMNS = (_PARETO_PRICE_COLUMN, _PARETO_VIDEO_PRICE_COLUMN)
30
  _PARETO_TIME_COLUMN = "Min Generation Time (s)"
31
- _PARETO_VIDEO_TIME_COLUMN = "Pareto Time / Output Video Second (s)"
32
- _PARETO_TIME_COLUMNS = (_PARETO_VIDEO_TIME_COLUMN, _PARETO_TIME_COLUMN)
33
- _PARETO_PRICE_TITLES = {
34
- _PARETO_PRICE_COLUMN: "Price per image (USD)",
35
- _PARETO_VIDEO_PRICE_COLUMN: "Price per second of video (USD)",
36
- }
37
- _PARETO_TIME_TITLES = {
38
- _PARETO_TIME_COLUMN: "Min generation time (s)",
39
- _PARETO_VIDEO_TIME_COLUMN: "Generation time per second of video",
40
- }
41
- _PARETO_SCALE_CHOICES = [
42
- ("Log", "Logarithmic"),
43
- ("Linear", "Linear"),
44
- ]
45
- _PARETO_SCALE_VALUES = {value for _, value in _PARETO_SCALE_CHOICES}
46
- _PARETO_SCALE_DEFAULT = "Logarithmic"
47
- _PARETO_PRUNA_COLOR = "#c084fc"
48
- _PARETO_OTHER_COLOR = "#9aa3b5"
49
- _PARETO_FRONTIER_OUTLINE = "#3fa87e"
50
 
51
  TAB_LEADERBOARDS = "leaderboards"
52
  TAB_PARETO = "pareto"
53
  TAB_SAMPLES = "samples"
54
  TAB_ABOUT = "about"
55
 
56
- MODALITY_TEXT_TO_VIDEO = "text_to_video"
57
- MODALITY_VIDEO_TO_VIDEO = "video_to_video"
58
- MODALITY_TEXT_TO_IMAGE = "text_to_image"
59
- MODALITY_CHOICES = [
60
- ("Text to Video", MODALITY_TEXT_TO_VIDEO),
61
- ("Video to Video", MODALITY_VIDEO_TO_VIDEO),
62
- ("Text to Image", MODALITY_TEXT_TO_IMAGE),
63
- ]
64
-
65
  _MODEL_CHOICES_CACHE = {}
66
  _VIEW_EVENTS = {
67
  "show_progress": "hidden",
@@ -73,24 +42,21 @@ _VIEW_EVENTS = {
73
  ABOUT_OVERVIEW_CONTENT = """
74
  # About P-Bench
75
 
76
- P-Bench compares **text-to-video**, **video-to-video**, and **text-to-image**
77
- models, including optimized or accelerated endpoints, on **quality, speed,
78
- and price**. Each view is a **dataset** scored with a **metric**, written as
79
- `Dataset | Metric`. There is no single score across P-Bench.
80
 
81
  ## How to read it
82
 
83
- 1. Pick a **model type** (Text to Video, Video to Video, or Text to Image),
84
- then a **dataset** and a **metric**.
85
  2. **Leaderboards**: ranked by that metric. Price and generation time sit in
86
  the same table when the source publishes them.
87
  3. **Pareto plots**: mark models that are not beaten on both higher score
88
  and lower price (or time). Only datasets with price or generation time
89
  can open this tab (not Arena AI).
90
  4. **Samples**: the same prompts, side by side. Only for datasets we
91
- generated (VBench-2.0 Dataset, Qwen Image Dataset, OneIG Alignment
92
- Dataset, and the Pruna Internal Video-Edit Benchmark). Video-edit
93
- samples show the source clip first, then each model's edit.
94
 
95
  ## How a score is made
96
 
@@ -109,24 +75,6 @@ prompt suites, so samples are not shown.
109
 
110
  ## Current datasets
111
 
112
- ### VBench-2.0 Dataset
113
- VBench-2.0 prompts, comparing P-Video-2 variants with Fal-hosted models.
114
- Quality is Datapoint Elo and Rapidata Elo from pairwise preference. Price
115
- is USD per second of output video. Time per second of video is Fal wall
116
- time, except Pruna models which use model execution time. Samples are
117
- available.
118
-
119
- ### Pruna Internal Video-Edit Benchmark
120
- Pruna's internal video-to-video editing benchmark, collected by our
121
- research engineers. It combines prompts from public video-editing
122
- benchmarks with use-case examples we gathered for advertisement,
123
- e-commerce, real estate, concept art, and similar work. The suite also
124
- covers camera-angle and movement changes, lighting, and text in video
125
- (altering, adding, or removing it). Quality is Datapoint Elo from
126
- pairwise preference. Price is USD per second of output video;
127
- generation time is wall time per second of output video. Samples show
128
- the source clip beside each model's edit.
129
-
130
  ### Qwen Image Dataset
131
  100 prompts from the 1,000-prompt Qwen Image Bench set, sampled for coverage
132
  across its fine-grained (L3) categories. Metrics include Datapoint Elo,
@@ -169,17 +117,12 @@ ABOUT_DETAILS_CONTENT = """
169
  - **Arena Elo**: Elo published by Arena AI on their own dataset, plus
170
  category Elos (branding, 3D, cartoon/anime, photorealistic, art, portraits,
171
  text rendering).
172
- - **Generation time**: median and minimum generation time in seconds for
173
- images, as reported in the evaluation table. For video, generation time
174
- per second of output video is the more informative figure. On the
175
- text-to-video benchmark this is Fal wall time, except Pruna models which
176
- use model execution time. On video-edit it is end-to-end wall time. This
177
- is not a p95, and we do not state warm vs cold or concurrent load. Not
178
  available for Arena AI.
179
- - **Price**: USD per image for text-to-image, or USD per second of output
180
- video for text-to-video and video-to-video. We do not state list price vs
181
- amount paid, or whether failed generations are included. Not available
182
- for Arena AI.
183
 
184
  Scores from different datasets or metrics are **not interchangeable**. A high
185
  OneIG alignment score is not the same quantity as a Datapoint Elo. Compare
@@ -193,21 +136,14 @@ models *within* a Dataset | Metric view.
193
  - **Prompt counts:** OneIG Alignment uses 100 anime, 100 human, and 99 object
194
  prompts (299 total). Qwen Image Dataset uses 100 prompts sampled from the
195
  1,000-prompt pool for roughly even coverage of its fine-grained (L3)
196
- categories. The VBench-2.0 Dataset uses about 90
197
- generations per model. The Pruna Internal Video-Edit Benchmark uses 78
198
- prompts across advertising, e-commerce, real estate, camera, lighting,
199
- text, and related categories. Artificial Analysis and Arena AI use their
200
- own private prompt sets.
201
  - **Generation (Qwen and OneIG):** one image per prompt per endpoint when
202
  the run exists. Default resolution is 1024×1024. Exceptions: FLUX 1.1 Pro
203
  Ultra at 2K, FLUX 2 Flex at 1008×1008, and any endpoint labeled 2K. The
204
  seed is derived from the prompt, so every model gets the same seed for the
205
  same prompt. Steps, CFG, prompt rewrite, and safety filters follow each
206
  endpoint's default. This does not describe Artificial Analysis or Arena AI.
207
- - **Generation (Text-to-Video):** one clip per prompt per endpoint when the
208
- run exists. About 90 generations per model.
209
- - **Generation (Video-Edit):** one edited clip per prompt per endpoint when
210
- the run exists. Every model sees the same source video for a prompt.
211
  - **Datapoint (Qwen and OneIG):** every model pair is compared on every
212
  prompt, with 10 votes per battle.
213
  - **Rapidata (Qwen and OneIG):** prompts longer than 400 characters are
@@ -279,7 +215,7 @@ def render_header():
279
  </svg>
280
  </button>
281
  </div>
282
- <p class="app-header-tagline">Compare models on quality, speed, and price</p>
283
  </header>
284
  """,
285
  padding=False,
@@ -294,42 +230,10 @@ def _item(items, item_id):
294
  return items[0] if items else None
295
 
296
 
297
- def _dataset_modality(dataset):
298
- return (dataset or {}).get("modality") or MODALITY_TEXT_TO_IMAGE
299
-
300
-
301
- def _datasets_for_modality(datasets, modality):
302
- if not modality:
303
- return list(datasets)
304
- scoped = [
305
- dataset
306
- for dataset in datasets
307
- if _dataset_modality(dataset) == modality
308
- ]
309
- return scoped or list(datasets)
310
-
311
-
312
- def _modality_choices(datasets):
313
- present = {_dataset_modality(dataset) for dataset in datasets}
314
- return [
315
- (label, value) for label, value in MODALITY_CHOICES if value in present
316
- ]
317
-
318
-
319
- def _default_dataset_id(datasets, modality, preferred=None):
320
- scoped = _datasets_for_modality(datasets, modality)
321
- if preferred and any(dataset["id"] == preferred for dataset in scoped):
322
- return preferred
323
- return scoped[0]["id"] if scoped else None
324
-
325
-
326
- def _dataset_choices(
327
- datasets, *, modality=None, require_samples=False, require_pareto=False
328
- ):
329
- scoped = _datasets_for_modality(datasets, modality)
330
  return [
331
  (dataset["name"], dataset["id"])
332
- for dataset in scoped
333
  if (not require_samples or dataset.get("samples"))
334
  and (not require_pareto or _dataset_has_pareto(datasets, dataset["id"]))
335
  ]
@@ -340,77 +244,17 @@ def _dataset_has_samples(datasets, dataset_id):
340
  return bool(dataset and dataset.get("samples"))
341
 
342
 
343
- def _sample_model_ids(datasets, dataset_id):
344
- dataset = _item(datasets, dataset_id)
345
- samples = dataset.get("samples") if dataset else None
346
- if not samples:
347
- return set()
348
- models = set(samples.get("models") or [])
349
- return models | {display_model_name(model) for model in models}
350
-
351
-
352
- def _sample_media_map(samples):
353
- return (samples or {}).get("images") or {}
354
-
355
-
356
- def _resolve_sample_model(samples, model):
357
- media = _sample_media_map(samples)
358
- if model in media:
359
- return model
360
- wanted = {str(model or "").strip(), display_model_name(model)}
361
- wanted.discard("")
362
- for key in media:
363
- if key in wanted or display_model_name(key) in wanted:
364
- return key
365
- return None
366
-
367
-
368
- def _default_sample_models(samples):
369
- models = list((samples or {}).get("models") or [])
370
- preferred = [model for model in models if _is_pruna_model(model)]
371
- preferred.sort(
372
- key=lambda model: (
373
- "draft" in str(model).casefold()
374
- or "draft" in display_model_name(model).casefold(),
375
- display_model_name(model).casefold(),
376
- )
377
- )
378
- return (preferred or models)[:2]
379
-
380
-
381
- def _pareto_price_column(data):
382
- columns = getattr(data, "columns", []) if data is not None else []
383
- for column in _PARETO_PRICE_COLUMNS:
384
- if column in columns:
385
- return column
386
- return None
387
-
388
-
389
- def _pareto_time_column(data):
390
- columns = getattr(data, "columns", []) if data is not None else []
391
- for column in _PARETO_TIME_COLUMNS:
392
- if column in columns:
393
- return column
394
- return None
395
-
396
-
397
  def _dataset_has_pareto(datasets, dataset_id):
398
  dataset = _item(datasets, dataset_id)
399
- data = dataset.get("data") if dataset else None
400
- return (
401
- _pareto_price_column(data) is not None
402
- or _pareto_time_column(data) is not None
403
- )
404
 
405
 
406
- def _dataset_dropdown_update(datasets, tab, dataset_id, modality=None):
407
  """Limit the dataset list to what the current tab can show."""
408
- if modality is None:
409
- modality = _dataset_modality(_item(datasets, dataset_id))
410
  return gr.update(
411
  choices=_dataset_choices(
412
  datasets,
413
- modality=modality,
414
  require_samples=tab == TAB_SAMPLES
415
  and _dataset_has_samples(datasets, dataset_id),
416
  require_pareto=tab == TAB_PARETO
@@ -470,25 +314,23 @@ def _metric_dropdown_value(metric_id):
470
  ]
471
 
472
 
473
- def _model_choices(datasets, dataset_id, *, require_samples=False):
474
  cached = _MODEL_CHOICES_CACHE.get(dataset_id)
475
- if cached is None:
476
- dataset = _item(datasets, dataset_id)
477
- data = dataset.get("data") if dataset else None
478
- if data is None or "Model" not in getattr(data, "columns", []):
479
- cached = []
480
- else:
481
- models = data["Model"].dropna().astype(str).unique().tolist()
482
- # (label, value) so the UI shows the shared name but filters on the raw id.
483
- cached = sorted(
484
- ((display_model_name(model), model) for model in models),
485
- key=lambda item: item[0].casefold(),
486
- )
487
- _MODEL_CHOICES_CACHE[dataset_id] = cached
488
- if not require_samples:
489
  return cached
490
- allowed = _sample_model_ids(datasets, dataset_id)
491
- return [choice for choice in cached if choice[1] in allowed]
 
 
 
 
 
 
 
 
 
 
 
492
 
493
 
494
  def _model_choice_values(choices):
@@ -516,11 +358,9 @@ _LEADERBOARD_IDENTITY_COLUMNS = [
516
  "Optimized",
517
  ]
518
  _LEADERBOARD_META_COLUMNS = [
519
- "Time / Output Video Second (s)",
520
  "Median Generation Time (s)",
521
  "Min Generation Time (s)",
522
  "Price / Image (USD)",
523
- "Price / Second of Video (USD)",
524
  "Evaluation Date (UTC)",
525
  "Date",
526
  ]
@@ -745,11 +585,10 @@ def _display_label(column):
745
  "Arena Art Elo": "Art",
746
  "Arena Portraits Elo": "Portraits",
747
  "Arena Text Rendering Elo": "Text Rendering",
 
748
  "Median Generation Time (s)": "Median generation time",
749
  "Min Generation Time (s)": "Min generation time",
750
- "Time / Output Video Second (s)": "Generation time per second of video",
751
  "Price / Image (USD)": "Price per image",
752
- "Price / Second of Video (USD)": "Price per second of video",
753
  "Evaluation Date (UTC)": "Date",
754
  "Date": "Date",
755
  }
@@ -824,23 +663,6 @@ def _applied_key(view_state):
824
  )
825
 
826
 
827
- def _is_pruna_model(model_id) -> bool:
828
- raw = str(model_id or "").casefold()
829
- label = display_model_name(model_id).casefold()
830
- return any(
831
- value.startswith(prefix)
832
- for value in (raw, label)
833
- for prefix in ("p-image", "p_image", "p-video", "p_video")
834
- )
835
-
836
-
837
- def _pareto_fill_colors(models):
838
- return [
839
- _PARETO_PRUNA_COLOR if _is_pruna_model(model) else _PARETO_OTHER_COLOR
840
- for model in models
841
- ]
842
-
843
-
844
  def _build_pareto_figure(
845
  data,
846
  score_column,
@@ -848,7 +670,6 @@ def _build_pareto_figure(
848
  x_title,
849
  x_hover_prefix="",
850
  x_hover_suffix="",
851
- x_axis_type="linear",
852
  ):
853
  scatter = (
854
  data[["Model", score_column, x_column]]
@@ -865,8 +686,6 @@ def _build_pareto_figure(
865
 
866
  dominated = scatter.loc[[not flag for flag in on_frontier]].copy()
867
  frontier = scatter.loc[on_frontier].sort_values(x_column).copy()
868
- dominated_colors = _pareto_fill_colors(dominated["Model"]) if not dominated.empty else []
869
- frontier_colors = _pareto_fill_colors(frontier["Model"]) if not frontier.empty else []
870
  if not dominated.empty:
871
  dominated["Model"] = dominated["Model"].map(display_model_name)
872
  if not frontier.empty:
@@ -887,11 +706,10 @@ def _build_pareto_figure(
887
  name="Below frontier",
888
  text=dominated["Model"],
889
  hovertemplate=hover,
890
- showlegend=False,
891
  marker={
892
  "size": 9,
893
- "color": dominated_colors,
894
- "opacity": 0.85,
895
  "line": {"width": 0},
896
  },
897
  )
@@ -905,51 +723,14 @@ def _build_pareto_figure(
905
  name="On frontier",
906
  text=frontier["Model"],
907
  hovertemplate=hover,
908
- showlegend=False,
909
- line={"color": _PARETO_FRONTIER_OUTLINE, "width": 2.5},
910
  marker={
911
  "size": 12,
912
- "color": frontier_colors,
913
- "line": {"width": 2.5, "color": _PARETO_FRONTIER_OUTLINE},
914
  },
915
  )
916
  )
917
- for name, marker in (
918
- (
919
- "Pruna",
920
- {
921
- "size": 10,
922
- "color": _PARETO_PRUNA_COLOR,
923
- "line": {"width": 0},
924
- },
925
- ),
926
- (
927
- "Other models",
928
- {
929
- "size": 10,
930
- "color": _PARETO_OTHER_COLOR,
931
- "line": {"width": 0},
932
- },
933
- ),
934
- (
935
- "On frontier",
936
- {
937
- "size": 12,
938
- "color": "rgba(0,0,0,0)",
939
- "line": {"width": 2.5, "color": _PARETO_FRONTIER_OUTLINE},
940
- },
941
- ),
942
- ):
943
- fig.add_trace(
944
- go.Scatter(
945
- x=[None],
946
- y=[None],
947
- mode="markers",
948
- name=name,
949
- marker=marker,
950
- hoverinfo="skip",
951
- )
952
- )
953
 
954
  score_label = _display_label(score_column)
955
  fig.update_layout(
@@ -974,29 +755,7 @@ def _build_pareto_figure(
974
  )
975
  axis_font = {"color": "#fafafa", "size": 13}
976
  tick_font = {"color": "#a3a3a3", "size": 12}
977
- x_axis_ticks = {}
978
- if x_axis_type == "log":
979
- positive_x = scatter.loc[scatter[x_column] > 0, x_column].astype(float)
980
- if not positive_x.empty:
981
- minimum = positive_x.min()
982
- maximum = positive_x.max()
983
- tick_values = [
984
- factor * (10**exponent)
985
- for exponent in range(
986
- floor(log10(minimum)),
987
- ceil(log10(maximum)) + 1,
988
- )
989
- for factor in (1, 2, 5)
990
- if minimum * 0.8 <= factor * (10**exponent) <= maximum * 1.2
991
- ]
992
- x_axis_ticks = {
993
- "tickmode": "array",
994
- "tickvals": tick_values,
995
- "ticktext": [f"{value:g}" for value in tick_values],
996
- }
997
  fig.update_xaxes(
998
- type=x_axis_type,
999
- **x_axis_ticks,
1000
  showgrid=True,
1001
  gridcolor="rgba(74, 57, 98, 0.55)",
1002
  zeroline=False,
@@ -1015,55 +774,6 @@ def _build_pareto_figure(
1015
  return fig
1016
 
1017
 
1018
- def _is_log_scale(scale):
1019
- return scale == "Logarithmic"
1020
-
1021
-
1022
- def _pareto_axis_type(scale):
1023
- return "log" if _is_log_scale(scale) else "linear"
1024
-
1025
-
1026
- def _pareto_scale_radio(*extra_classes):
1027
- return gr.Radio(
1028
- choices=_PARETO_SCALE_CHOICES,
1029
- value=_PARETO_SCALE_DEFAULT,
1030
- show_label=False,
1031
- container=False,
1032
- elem_classes=["pareto-scale-toggle", *extra_classes],
1033
- )
1034
-
1035
-
1036
- def _pareto_plot_heading(title):
1037
- with gr.Row(equal_height=False, elem_classes="pareto-heading-row"):
1038
- gr.Markdown(f"#### {title}", elem_classes="pareto-subhead")
1039
- with gr.Column(min_width=140, elem_classes="pareto-scale-control"):
1040
- return _pareto_scale_radio()
1041
-
1042
-
1043
- def _default_pareto_scales():
1044
- return [_PARETO_SCALE_DEFAULT] * MAX_PARETO_METRICS
1045
-
1046
-
1047
- def _normalize_pareto_scales(scales):
1048
- values = list(scales or [])
1049
- if len(values) < MAX_PARETO_METRICS:
1050
- values.extend(
1051
- [_PARETO_SCALE_DEFAULT] * (MAX_PARETO_METRICS - len(values))
1052
- )
1053
- return values[:MAX_PARETO_METRICS]
1054
-
1055
-
1056
- def _uniform_pareto_scales(scale):
1057
- return [scale] * MAX_PARETO_METRICS
1058
-
1059
-
1060
- def _pareto_master_scale_update(price_scales, time_scales):
1061
- values = list(price_scales) + list(time_scales)
1062
- if values and all(value == values[0] for value in values):
1063
- return gr.update(value=values[0])
1064
- return gr.update(value=None)
1065
-
1066
-
1067
  def _pareto_axis(data, score_column, x_column, x_title, missing_message, empty_message, **hover):
1068
  if x_column not in data.columns:
1069
  return None, missing_message
@@ -1079,83 +789,56 @@ def _pareto_axis(data, score_column, x_column, x_title, missing_message, empty_m
1079
  return fig, None
1080
 
1081
 
1082
- def _pareto_pair(
1083
- data,
1084
- score_column,
1085
- latency_scale=_PARETO_SCALE_DEFAULT,
1086
- price_scale=_PARETO_SCALE_DEFAULT,
1087
- ):
1088
  score_missing = "No score data is available for this metric."
1089
  if data is None or not score_column or score_column not in data.columns:
1090
  return None, score_missing, None, score_missing
1091
 
1092
- price_column = _pareto_price_column(data) or _PARETO_PRICE_COLUMN
1093
- price_title = _PARETO_PRICE_TITLES.get(price_column, "Price (USD)")
1094
- price_missing = (
1095
- "Price per second of video isn't available for this dataset."
1096
- if price_column == _PARETO_VIDEO_PRICE_COLUMN
1097
- else "Price per image isn't available for this dataset."
1098
- )
1099
  price_fig, price_message = _pareto_axis(
1100
  data,
1101
  score_column,
1102
- price_column,
1103
- price_title,
1104
- price_missing,
1105
  "No models have both a score and a price for this metric.",
1106
  x_hover_prefix="$",
1107
- x_axis_type=_pareto_axis_type(price_scale),
1108
- )
1109
- time_column = _pareto_time_column(data) or _PARETO_TIME_COLUMN
1110
- time_title = _PARETO_TIME_TITLES.get(time_column, "Generation time (s)")
1111
- time_missing = (
1112
- "Generation time per second of video isn't available for this dataset."
1113
- if time_column == _PARETO_VIDEO_TIME_COLUMN
1114
- else "Min generation time isn't available for this dataset."
1115
- )
1116
- time_empty = (
1117
- "No models have both a score and generation time per second of "
1118
- "video for this metric."
1119
- if time_column == _PARETO_VIDEO_TIME_COLUMN
1120
- else "No models have both a score and a min generation time for this metric."
1121
  )
1122
  time_fig, time_message = _pareto_axis(
1123
  data,
1124
  score_column,
1125
- time_column,
1126
- time_title,
1127
- time_missing,
1128
- time_empty,
1129
  x_hover_suffix="s",
1130
- x_axis_type=_pareto_axis_type(latency_scale),
1131
  )
1132
  return price_fig, price_message, time_fig, time_message
1133
 
1134
 
1135
  def _pareto_dataset_message(data):
1136
- has_price = _pareto_price_column(data) is not None
1137
- has_time = _pareto_time_column(data) is not None
1138
  if has_price or has_time:
1139
  return None
1140
  return (
1141
- "Price and generation time aren't available for "
1142
  "this dataset, so these plots can't be drawn."
1143
  )
1144
 
1145
 
1146
  def _pareto_slot_note(price_fig, price_message, time_fig, time_message, data):
1147
- has_price = _pareto_price_column(data) is not None
1148
- has_time = _pareto_time_column(data) is not None
1149
  notes = []
1150
  if has_price and not has_time:
1151
  notes.append(
1152
- "Generation time isn't available for this dataset, so only "
1153
  "price vs score is shown."
1154
  )
1155
  elif has_time and not has_price:
1156
  notes.append(
1157
- "Price isn't available for this dataset, so only "
1158
- "time vs score is shown."
1159
  )
1160
  if price_fig is None and has_price:
1161
  notes.append(price_message)
@@ -1166,18 +849,11 @@ def _pareto_slot_note(price_fig, price_message, time_fig, time_message, data):
1166
  return " ".join(notes)
1167
 
1168
 
1169
- def _pareto_slot_updates(
1170
- data,
1171
- score_columns,
1172
- price_scales=None,
1173
- time_scales=None,
1174
- ):
1175
  """Updates for a fixed bank of Gradio Plot slots (visible/hidden)."""
1176
  score_columns = [column for column in (score_columns or []) if column]
1177
- price_scales = _normalize_pareto_scales(price_scales)
1178
- time_scales = _normalize_pareto_scales(time_scales)
1179
- has_price = _pareto_price_column(data) is not None
1180
- has_time = _pareto_time_column(data) is not None
1181
  dataset_note = _pareto_dataset_message(data)
1182
  updates = [_pareto_note_update(dataset_note)]
1183
  hide_all_slots = not has_price and not has_time
@@ -1197,10 +873,7 @@ def _pareto_slot_updates(
1197
  continue
1198
  score_column = score_columns[index]
1199
  price_fig, price_message, time_fig, time_message = _pareto_pair(
1200
- data,
1201
- score_column,
1202
- latency_scale=time_scales[index],
1203
- price_scale=price_scales[index],
1204
  )
1205
  show_price = price_fig is not None
1206
  show_time = time_fig is not None
@@ -1227,77 +900,20 @@ def _pareto_slot_updates(
1227
  return updates
1228
 
1229
 
1230
- def _pareto_all_scale_updates(data, score_columns, scale):
1231
- """Apply one scale to every Pareto plot and radio."""
1232
- score_columns = [column for column in (score_columns or []) if column]
1233
- price_updates = []
1234
- time_updates = []
1235
- for index in range(MAX_PARETO_METRICS):
1236
- if index >= len(score_columns):
1237
- price_updates.append(gr.skip())
1238
- time_updates.append(gr.skip())
1239
- continue
1240
- price_fig, _, time_fig, _ = _pareto_pair(
1241
- data,
1242
- score_columns[index],
1243
- latency_scale=scale,
1244
- price_scale=scale,
1245
- )
1246
- price_updates.append(_pareto_plot_update(price_fig))
1247
- time_updates.append(_pareto_plot_update(time_fig))
1248
- radio_updates = [
1249
- gr.update(value=scale) for _ in range(MAX_PARETO_METRICS * 2)
1250
- ]
1251
- return price_updates + time_updates + radio_updates
1252
-
1253
-
1254
  def _samples_html(samples, selected_models, num_prompts, seed=0):
1255
  if not samples:
1256
  return _pareto_unavailable_html(
1257
  "Samples aren't available for this dataset."
1258
  )
1259
- models = [
1260
- resolved
1261
- for model in (selected_models or [])
1262
- if (resolved := _resolve_sample_model(samples, model))
1263
- ]
1264
  if not models:
1265
- models = _default_sample_models(samples)
1266
  return _build_compare_samples_html(samples, models, num_prompts, seed)
1267
 
1268
 
1269
- def _compare_media_html(url, label, *, kind):
1270
- safe_url = escape(url, quote=True)
1271
- safe_label = escape(label)
1272
- if kind == "video":
1273
- return (
1274
- f'<video src="{safe_url}" controls preload="metadata" '
1275
- f'playsinline></video>'
1276
- )
1277
- return (
1278
- f'<a href="{safe_url}" target="_blank" rel="noopener noreferrer">'
1279
- f'<img src="{safe_url}" alt="{safe_label} sample" loading="lazy" />'
1280
- f"</a>"
1281
- )
1282
-
1283
-
1284
- def _compare_cell_html(label, url, *, kind, extra_class=""):
1285
- classes = "compare-cell"
1286
- if extra_class:
1287
- classes = f"{classes} {extra_class}"
1288
- return f"""
1289
- <div class="{classes}">
1290
- <div class="compare-model-label">{escape(label)}</div>
1291
- {_compare_media_html(url, label, kind=kind)}
1292
- </div>
1293
- """
1294
-
1295
-
1296
  def _build_compare_samples_html(samples, selected_models, num_prompts, seed=0):
1297
  selected_models = list(selected_models or [])[:MAX_COMPARE_MODELS]
1298
- media = _sample_media_map(samples)
1299
- kind = (samples or {}).get("kind") or "image"
1300
- source_videos = (samples or {}).get("source_videos") or {}
1301
 
1302
  if not selected_models:
1303
  return (
@@ -1308,7 +924,7 @@ def _build_compare_samples_html(samples, selected_models, num_prompts, seed=0):
1308
 
1309
  shared_prompt_ids = None
1310
  for model in selected_models:
1311
- model_prompt_ids = set(media.get(model) or [])
1312
  shared_prompt_ids = (
1313
  model_prompt_ids
1314
  if shared_prompt_ids is None
@@ -1328,31 +944,29 @@ def _build_compare_samples_html(samples, selected_models, num_prompts, seed=0):
1328
  rng.shuffle(prompt_pool)
1329
  chosen = prompt_pool[: max(1, min(int(num_prompts), len(prompt_pool)))]
1330
 
 
1331
  blocks = []
1332
  for index, prompt_id in enumerate(chosen, start=1):
1333
  prompt_text = escape(samples["prompts"].get(prompt_id, ""))
1334
  cells = []
1335
- source_url = source_videos.get(prompt_id)
1336
- if source_url:
1337
- cells.append(
1338
- _compare_cell_html(
1339
- "Source", source_url, kind="video", extra_class="compare-source"
1340
- )
1341
- )
1342
  for model in selected_models:
 
1343
  cells.append(
1344
- _compare_cell_html(
1345
- display_model_name(model),
1346
- media[model][prompt_id],
1347
- kind=kind,
1348
- )
 
 
 
1349
  )
1350
- columns = len(cells)
1351
  blocks.append(
1352
  f"""
1353
  <div class="compare-prompt-block">
1354
  <div class="compare-prompt-meta">
1355
  <span>Prompt {index}</span>
 
1356
  </div>
1357
  <p class="compare-prompt-text">{prompt_text}</p>
1358
  <div class="compare-row" style="grid-template-columns: repeat({columns}, minmax(0, 1fr));">
@@ -1380,19 +994,9 @@ def _filter_row(datasets, metrics, default_dataset_id, default_metric_id=None):
1380
  metric_id = _coerce_metric(
1381
  datasets, metrics, default_dataset_id, default_metric_id
1382
  )
1383
- default_modality = _dataset_modality(_item(datasets, default_dataset_id))
1384
  with gr.Row(elem_classes="view-filters"):
1385
- modality_dd = gr.Dropdown(
1386
- choices=_modality_choices(datasets),
1387
- value=default_modality,
1388
- label="Model Type",
1389
- type="value",
1390
- filterable=False,
1391
- scale=1,
1392
- min_width=170,
1393
- )
1394
  dataset_dd = gr.Dropdown(
1395
- choices=_dataset_choices(datasets, modality=default_modality),
1396
  value=default_dataset_id,
1397
  label="Dataset",
1398
  type="value",
@@ -1424,7 +1028,7 @@ def _filter_row(datasets, metrics, default_dataset_id, default_metric_id=None):
1424
  min_width=180,
1425
  elem_classes="filter-chips",
1426
  )
1427
- return modality_dd, dataset_dd, metric_dd, models_dd
1428
 
1429
 
1430
  def render_image_workspace(datasets, metrics, default_dataset_id, default_metric_id):
@@ -1439,17 +1043,15 @@ def render_image_workspace(datasets, metrics, default_dataset_id, default_metric
1439
  with gr.Column(elem_classes="workspace-filters") as filters_host:
1440
  gr.Markdown(
1441
  "<p class='filter-help'>"
1442
- "Start with Model Type to switch modalities. The rest of "
1443
- "the filters follow you across Leaderboards, Pareto plots, "
1444
- "and Samples. "
1445
- "Samples only lists datasets and models we have generations "
1446
- "for; Pareto plots only lists datasets with price or "
1447
- "generation time. Search in Models, or leave it empty to "
1448
- "include every model."
1449
  "</p>",
1450
  elem_classes="filter-help-host",
1451
  )
1452
- modality_dd, dataset_dd, metric_dd, models_dd = _filter_row(
1453
  datasets, metrics, default_dataset_id, None
1454
  )
1455
  with gr.Tabs(elem_classes="main-tabs") as main_tabs:
@@ -1515,27 +1117,12 @@ def render_image_workspace(datasets, metrics, default_dataset_id, default_metric
1515
  ) as pp_tab:
1516
  gr.Markdown(
1517
  "<p class='view-help'>"
1518
- "Score against price and generation time. Green points are on "
1519
- "the frontier; lavender points sit below it. Hover a point to "
1520
- "see which model it is."
1521
- "</p>"
1522
- "<p class='view-help'>"
1523
- "Plots use a logarithmic scale by default. You can switch "
1524
- "to linear for all plots, or individually for each plot."
1525
  "</p>",
1526
  elem_classes="view-help-host",
1527
  )
1528
- with gr.Row(
1529
- equal_height=False,
1530
- elem_classes="pareto-scale-all-row",
1531
- ):
1532
- gr.HTML(
1533
- "<span class='pareto-scale-all-label'>All plots</span>",
1534
- padding=False,
1535
- )
1536
- pareto_all_scale = _pareto_scale_radio(
1537
- "pareto-scale-toggle-all",
1538
- )
1539
  pareto_dataset_note = gr.HTML(
1540
  "",
1541
  padding=False,
@@ -1561,8 +1148,9 @@ def render_image_workspace(datasets, metrics, default_dataset_id, default_metric
1561
  min_width=320,
1562
  elem_classes="pareto-col",
1563
  ) as slot_price_col:
1564
- slot_price_scale = _pareto_plot_heading(
1565
- "Price vs score"
 
1566
  )
1567
  slot_price = gr.Plot(
1568
  value=None,
@@ -1574,8 +1162,9 @@ def render_image_workspace(datasets, metrics, default_dataset_id, default_metric
1574
  min_width=320,
1575
  elem_classes="pareto-col",
1576
  ) as slot_time_col:
1577
- slot_time_scale = _pareto_plot_heading(
1578
- "Time vs score"
 
1579
  )
1580
  slot_time = gr.Plot(
1581
  value=None,
@@ -1596,10 +1185,8 @@ def render_image_workspace(datasets, metrics, default_dataset_id, default_metric
1596
  slot_layout,
1597
  slot_price_col,
1598
  slot_price,
1599
- slot_price_scale,
1600
  slot_time_col,
1601
  slot_time,
1602
- slot_time_scale,
1603
  )
1604
  )
1605
 
@@ -1611,8 +1198,7 @@ def render_image_workspace(datasets, metrics, default_dataset_id, default_metric
1611
  with gr.Column(visible=bool(initial_samples)) as samples_panel:
1612
  gr.Markdown(
1613
  f"<p class='view-help'>"
1614
- f"The same prompts, side by side. Video edits show the "
1615
- f"source clip first. Select up to "
1616
  f"<strong>{MAX_COMPARE_MODELS}</strong> models above, or leave "
1617
  f"Models empty for two defaults."
1618
  f"</p>",
@@ -1649,16 +1235,12 @@ def render_image_workspace(datasets, metrics, default_dataset_id, default_metric
1649
  with gr.TabItem("About", id=TAB_ABOUT) as about_tab:
1650
  render_about()
1651
 
1652
- def _synced_filters(
1653
- dataset_id, metric_id, models, *, clear_metric=False, require_samples=False
1654
- ):
1655
  if clear_metric:
1656
  metric_id = []
1657
  else:
1658
  metric_id = _coerce_metric(datasets, metrics, dataset_id, metric_id)
1659
- model_choices = _model_choices(
1660
- datasets, dataset_id, require_samples=require_samples
1661
- )
1662
  model_values = set(_model_choice_values(model_choices))
1663
  models = [model for model in (models or []) if model in model_values]
1664
  metric_choices = _metric_dropdown_choices(datasets, metrics, dataset_id)
@@ -1730,14 +1312,8 @@ def render_image_workspace(datasets, metrics, default_dataset_id, default_metric
1730
  ):
1731
  prev = dict(view_state or {})
1732
  extras = extras or {}
1733
- modality = _dataset_modality(_item(datasets, dataset_id))
1734
- last_by_modality = dict(prev.get("dataset_by_modality") or {})
1735
- if dataset_id:
1736
- last_by_modality[modality] = dataset_id
1737
  return {
1738
  "dataset_id": dataset_id,
1739
- "modality": modality,
1740
- "dataset_by_modality": last_by_modality,
1741
  "metric_id": metric_id,
1742
  "models": list(models or []),
1743
  "current_tab": tab,
@@ -1748,8 +1324,6 @@ def render_image_workspace(datasets, metrics, default_dataset_id, default_metric
1748
  "optimized": list(
1749
  extras.get("optimized", prev.get("optimized") or [])
1750
  ),
1751
- "price_scales": _normalize_pareto_scales(prev.get("price_scales")),
1752
- "time_scales": _normalize_pareto_scales(prev.get("time_scales")),
1753
  "stale": {
1754
  TAB_LEADERBOARDS: not flags["include_leaderboard"],
1755
  TAB_PARETO: not flags["include_pareto"],
@@ -1809,8 +1383,6 @@ def render_image_workspace(datasets, metrics, default_dataset_id, default_metric
1809
  include_leaderboard=True,
1810
  include_pareto=False,
1811
  include_samples=False,
1812
- price_scales=None,
1813
- time_scales=None,
1814
  ):
1815
  view = resolve_view(datasets, metrics, dataset_id, metric_id)
1816
  data = view["data"]
@@ -1831,12 +1403,7 @@ def render_image_workspace(datasets, metrics, default_dataset_id, default_metric
1831
  ranking_html = gr.skip()
1832
  if include_pareto:
1833
  pareto_data = _filter_leaderboard(data, [], [], [], models=models)
1834
- pareto_updates = _pareto_slot_updates(
1835
- pareto_data,
1836
- view["score_columns"],
1837
- price_scales=price_scales,
1838
- time_scales=time_scales,
1839
- )
1840
  else:
1841
  pareto_updates = _pareto_skip_updates()
1842
  if include_samples:
@@ -1875,24 +1442,13 @@ def render_image_workspace(datasets, metrics, default_dataset_id, default_metric
1875
  tab = view_state.get("current_tab") or TAB_LEADERBOARDS
1876
  selected_raw = _normalize_metric_ids(metric_id)
1877
  incoming_models = list(models or [])
1878
- filter_changed = source in {"dataset", "modality"}
1879
- dataset_changed = filter_changed and dataset_id != view_state.get(
1880
  "dataset_id"
1881
  )
1882
- can_pareto = _dataset_has_pareto(datasets, dataset_id)
1883
- can_samples = _dataset_has_samples(datasets, dataset_id)
1884
- selected_tab = tab
1885
- if filter_changed:
1886
- if tab == TAB_SAMPLES and not can_samples:
1887
- selected_tab = TAB_LEADERBOARDS
1888
- elif tab == TAB_PARETO and not can_pareto:
1889
- selected_tab = TAB_LEADERBOARDS
1890
  synced = _synced_filters(
1891
- dataset_id,
1892
- metric_id,
1893
- models,
1894
- clear_metric=source == "modality" or dataset_changed,
1895
- require_samples=selected_tab == TAB_SAMPLES,
1896
  )
1897
  dataset_id, metric_id, models = synced[:3]
1898
  metric_update, models_update = synced[3], synced[4]
@@ -1930,13 +1486,20 @@ def render_image_workspace(datasets, metrics, default_dataset_id, default_metric
1930
  ):
1931
  return None
1932
 
 
1933
  extras = (
1934
  list(platform_value or []),
1935
  list(owner_value or []),
1936
  list(optimized_value or []),
1937
  )
1938
  extra_updates = None
1939
- if filter_changed:
 
 
 
 
 
 
1940
  view = resolve_view(datasets, metrics, dataset_id, metric_id)
1941
  extra_updates = _leaderboard_extras(
1942
  view["data"] if view else None,
@@ -1974,8 +1537,6 @@ def render_image_workspace(datasets, metrics, default_dataset_id, default_metric
1974
  extras[2],
1975
  num_prompts,
1976
  seed,
1977
- price_scales=view_state.get("price_scales"),
1978
- time_scales=view_state.get("time_scales"),
1979
  **flags,
1980
  ),
1981
  "state": _commit_state(
@@ -1989,67 +1550,6 @@ def render_image_workspace(datasets, metrics, default_dataset_id, default_metric
1989
  ),
1990
  }
1991
 
1992
- def on_modality(
1993
- modality,
1994
- dataset_id,
1995
- metric_id,
1996
- models,
1997
- platform_value,
1998
- owner_value,
1999
- optimized_value,
2000
- num_prompts,
2001
- seed,
2002
- view_state,
2003
- ):
2004
- view_state = dict(view_state or {})
2005
- last_by_modality = dict(view_state.get("dataset_by_modality") or {})
2006
- current_modality = view_state.get("modality") or _dataset_modality(
2007
- _item(datasets, dataset_id)
2008
- )
2009
- if dataset_id:
2010
- last_by_modality[current_modality] = dataset_id
2011
- dataset_id = _default_dataset_id(
2012
- datasets, modality, last_by_modality.get(modality)
2013
- )
2014
- view_state["modality"] = modality
2015
- view_state["dataset_by_modality"] = last_by_modality
2016
- result = _apply_filter_change(
2017
- "modality",
2018
- dataset_id,
2019
- metric_id,
2020
- models,
2021
- platform_value,
2022
- owner_value,
2023
- optimized_value,
2024
- num_prompts,
2025
- seed,
2026
- view_state,
2027
- )
2028
- if result is None:
2029
- return _skip_all(len(dataset_outputs))
2030
- extras = result["extra_updates"]
2031
- return (
2032
- _dataset_dropdown_update(
2033
- datasets,
2034
- result["selected_tab"],
2035
- result["dataset_id"],
2036
- modality=modality,
2037
- ),
2038
- result["metric_update"],
2039
- result["models_update"],
2040
- extras[6],
2041
- extras[0],
2042
- extras[1],
2043
- extras[2],
2044
- *result["views"],
2045
- gr.update(interactive=result["can_pareto"]),
2046
- gr.update(interactive=result["can_samples"]),
2047
- gr.update(selected=result["selected_tab"])
2048
- if result["selected_tab"] != result["tab"]
2049
- else gr.skip(),
2050
- result["state"],
2051
- )
2052
-
2053
  def on_dataset(
2054
  dataset_id,
2055
  metric_id,
@@ -2177,17 +1677,7 @@ def render_image_workspace(datasets, metrics, default_dataset_id, default_metric
2177
  )
2178
  dataset_update = _dataset_dropdown_update(datasets, tab, dataset_id)
2179
  metric_id = _coerce_metric(datasets, metrics, dataset_id, metric_id)
2180
- require_samples = tab == TAB_SAMPLES
2181
- model_choices = _model_choices(
2182
- datasets, dataset_id, require_samples=require_samples
2183
- )
2184
- allowed_models = set(_model_choice_values(model_choices))
2185
- models = [model for model in (models or []) if model in allowed_models]
2186
- models_update = (
2187
- gr.update(choices=model_choices, value=models)
2188
- if (prev_tab == TAB_SAMPLES) != require_samples
2189
- else gr.skip()
2190
- )
2191
  view_state["current_tab"] = tab
2192
  view_state["dataset_id"] = dataset_id
2193
  view_state["metric_id"] = metric_id
@@ -2224,7 +1714,7 @@ def render_image_workspace(datasets, metrics, default_dataset_id, default_metric
2224
  filters_vis,
2225
  dataset_update,
2226
  metric_vis,
2227
- models_update,
2228
  *lb_filters,
2229
  )
2230
  tab_select = (
@@ -2249,8 +1739,6 @@ def render_image_workspace(datasets, metrics, default_dataset_id, default_metric
2249
  optimized_value,
2250
  num_prompts,
2251
  seed,
2252
- price_scales=view_state.get("price_scales"),
2253
- time_scales=view_state.get("time_scales"),
2254
  **flags,
2255
  )
2256
  stale[tab] = False
@@ -2307,70 +1795,6 @@ def render_image_workspace(datasets, metrics, default_dataset_id, default_metric
2307
  next_seed,
2308
  )
2309
 
2310
- def _on_pareto_plot_scale(slot_index, axis):
2311
- def handler(dataset_id, metric_id, models, scale, view_state):
2312
- view_state = dict(view_state or {})
2313
- price_scales = _normalize_pareto_scales(
2314
- view_state.get("price_scales")
2315
- )
2316
- time_scales = _normalize_pareto_scales(
2317
- view_state.get("time_scales")
2318
- )
2319
- if axis == "price":
2320
- if price_scales[slot_index] == scale:
2321
- return gr.skip(), gr.skip(), gr.skip()
2322
- price_scales[slot_index] = scale
2323
- else:
2324
- if time_scales[slot_index] == scale:
2325
- return gr.skip(), gr.skip(), gr.skip()
2326
- time_scales[slot_index] = scale
2327
- view_state["price_scales"] = price_scales
2328
- view_state["time_scales"] = time_scales
2329
- master_scale = _pareto_master_scale_update(
2330
- price_scales, time_scales
2331
- )
2332
- view = resolve_view(datasets, metrics, dataset_id, metric_id)
2333
- score_columns = [
2334
- column for column in (view["score_columns"] or []) if column
2335
- ]
2336
- if slot_index >= len(score_columns):
2337
- return gr.skip(), master_scale, view_state
2338
- data = _filter_leaderboard(
2339
- view["data"], [], [], [], models=list(models or [])
2340
- )
2341
- price_fig, _, time_fig, _ = _pareto_pair(
2342
- data,
2343
- score_columns[slot_index],
2344
- latency_scale=time_scales[slot_index],
2345
- price_scale=price_scales[slot_index],
2346
- )
2347
- fig = price_fig if axis == "price" else time_fig
2348
- return _pareto_plot_update(fig), master_scale, view_state
2349
-
2350
- handler.__name__ = f"on_pareto_{axis}_scale_{slot_index}"
2351
- return handler
2352
-
2353
- def on_pareto_all_scale(dataset_id, metric_id, models, scale, view_state):
2354
- if scale not in _PARETO_SCALE_VALUES:
2355
- return (*_skip_all(MAX_PARETO_METRICS * 4), gr.skip())
2356
- view_state = dict(view_state or {})
2357
- scales = _uniform_pareto_scales(scale)
2358
- if (
2359
- _normalize_pareto_scales(view_state.get("price_scales")) == scales
2360
- and _normalize_pareto_scales(view_state.get("time_scales")) == scales
2361
- ):
2362
- return (*_skip_all(MAX_PARETO_METRICS * 4), gr.skip())
2363
- view_state["price_scales"] = scales
2364
- view_state["time_scales"] = scales
2365
- view = resolve_view(datasets, metrics, dataset_id, metric_id)
2366
- data = _filter_leaderboard(
2367
- view["data"], [], [], [], models=list(models or [])
2368
- )
2369
- return (
2370
- *_pareto_all_scale_updates(data, view["score_columns"], scale),
2371
- view_state,
2372
- )
2373
-
2374
  def _on_tab(tab):
2375
  def handler(
2376
  dataset_id,
@@ -2399,20 +1823,15 @@ def render_image_workspace(datasets, metrics, default_dataset_id, default_metric
2399
  handler.__name__ = f"on_tab_{tab}"
2400
  return handler
2401
 
2402
- default_modality = _dataset_modality(_item(datasets, default_dataset_id))
2403
  view_state = gr.State(
2404
  {
2405
  "dataset_id": default_dataset_id,
2406
- "modality": default_modality,
2407
- "dataset_by_modality": {default_modality: default_dataset_id},
2408
  "metric_id": None,
2409
  "models": [],
2410
  "current_tab": TAB_LEADERBOARDS,
2411
  "platform": [],
2412
  "owner": [],
2413
  "optimized": [],
2414
- "price_scales": _default_pareto_scales(),
2415
- "time_scales": _default_pareto_scales(),
2416
  "stale": {
2417
  TAB_LEADERBOARDS: False,
2418
  TAB_PARETO: True,
@@ -2424,7 +1843,7 @@ def render_image_workspace(datasets, metrics, default_dataset_id, default_metric
2424
  pareto_dataset_note,
2425
  *[
2426
  component
2427
- for slot_group, slot_title, slot_note, slot_layout, slot_price_col, slot_price, slot_price_scale, slot_time_col, slot_time, slot_time_scale in pareto_slots
2428
  for component in (
2429
  slot_group,
2430
  slot_title,
@@ -2437,24 +1856,6 @@ def render_image_workspace(datasets, metrics, default_dataset_id, default_metric
2437
  )
2438
  ],
2439
  ]
2440
- pareto_all_scale_outputs = [
2441
- *[
2442
- slot_price
2443
- for _, _, _, _, _, slot_price, _, _, _, _ in pareto_slots
2444
- ],
2445
- *[
2446
- slot_time
2447
- for _, _, _, _, _, _, _, _, slot_time, _ in pareto_slots
2448
- ],
2449
- *[
2450
- slot_price_scale
2451
- for _, _, _, _, _, _, slot_price_scale, _, _, _ in pareto_slots
2452
- ],
2453
- *[
2454
- slot_time_scale
2455
- for _, _, _, _, _, _, _, _, _, slot_time_scale in pareto_slots
2456
- ],
2457
- ]
2458
  view_inputs = [
2459
  platform,
2460
  owner,
@@ -2486,12 +1887,6 @@ def render_image_workspace(datasets, metrics, default_dataset_id, default_metric
2486
  main_tabs,
2487
  view_state,
2488
  ]
2489
- modality_dd.change(
2490
- on_modality,
2491
- inputs=[modality_dd, *filter_inputs],
2492
- outputs=dataset_outputs,
2493
- **_VIEW_EVENTS,
2494
- )
2495
  dataset_dd.change(
2496
  on_dataset,
2497
  inputs=filter_inputs,
@@ -2565,56 +1960,6 @@ def render_image_workspace(datasets, metrics, default_dataset_id, default_metric
2565
  show_progress="hidden",
2566
  )
2567
 
2568
- pareto_all_scale.change(
2569
- on_pareto_all_scale,
2570
- inputs=[
2571
- dataset_dd,
2572
- metric_dd,
2573
- models_dd,
2574
- pareto_all_scale,
2575
- view_state,
2576
- ],
2577
- outputs=[*pareto_all_scale_outputs, view_state],
2578
- **_VIEW_EVENTS,
2579
- )
2580
-
2581
- for slot_index, (
2582
- _,
2583
- _,
2584
- _,
2585
- _,
2586
- _,
2587
- slot_price,
2588
- slot_price_scale,
2589
- _,
2590
- slot_time,
2591
- slot_time_scale,
2592
- ) in enumerate(pareto_slots):
2593
- slot_price_scale.change(
2594
- _on_pareto_plot_scale(slot_index, "price"),
2595
- inputs=[
2596
- dataset_dd,
2597
- metric_dd,
2598
- models_dd,
2599
- slot_price_scale,
2600
- view_state,
2601
- ],
2602
- outputs=[slot_price, pareto_all_scale, view_state],
2603
- **_VIEW_EVENTS,
2604
- )
2605
- slot_time_scale.change(
2606
- _on_pareto_plot_scale(slot_index, "time"),
2607
- inputs=[
2608
- dataset_dd,
2609
- metric_dd,
2610
- models_dd,
2611
- slot_time_scale,
2612
- view_state,
2613
- ],
2614
- outputs=[slot_time, pareto_all_scale, view_state],
2615
- **_VIEW_EVENTS,
2616
- )
2617
-
2618
  prompt_count.change(
2619
  on_samples_controls,
2620
  inputs=[dataset_dd, models_dd, prompt_count, seed_state],
 
1
  from html import escape
 
2
  from pathlib import Path
3
  import base64
4
  import random
 
24
  MAX_PARETO_METRICS = 8
25
  _PARETO_SLOT_COUNT = 1 + MAX_PARETO_METRICS * 8
26
  _PARETO_PRICE_COLUMN = "Price / Image (USD)"
 
 
27
  _PARETO_TIME_COLUMN = "Min Generation Time (s)"
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
28
 
29
  TAB_LEADERBOARDS = "leaderboards"
30
  TAB_PARETO = "pareto"
31
  TAB_SAMPLES = "samples"
32
  TAB_ABOUT = "about"
33
 
 
 
 
 
 
 
 
 
 
34
  _MODEL_CHOICES_CACHE = {}
35
  _VIEW_EVENTS = {
36
  "show_progress": "hidden",
 
42
  ABOUT_OVERVIEW_CONTENT = """
43
  # About P-Bench
44
 
45
+ P-Bench compares **text-to-image models**, including optimized or accelerated
46
+ endpoints, on **quality, speed, and price**. Each view is a **dataset** scored
47
+ with a **metric**, written as `Dataset | Metric`. There is no single score
48
+ across P-Bench.
49
 
50
  ## How to read it
51
 
52
+ 1. Pick a **dataset** and a **metric**.
 
53
  2. **Leaderboards**: ranked by that metric. Price and generation time sit in
54
  the same table when the source publishes them.
55
  3. **Pareto plots**: mark models that are not beaten on both higher score
56
  and lower price (or time). Only datasets with price or generation time
57
  can open this tab (not Arena AI).
58
  4. **Samples**: the same prompts, side by side. Only for datasets we
59
+ generated (Qwen Image Dataset and OneIG Alignment Dataset).
 
 
60
 
61
  ## How a score is made
62
 
 
75
 
76
  ## Current datasets
77
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
78
  ### Qwen Image Dataset
79
  100 prompts from the 1,000-prompt Qwen Image Bench set, sampled for coverage
80
  across its fine-grained (L3) categories. Metrics include Datapoint Elo,
 
117
  - **Arena Elo**: Elo published by Arena AI on their own dataset, plus
118
  category Elos (branding, 3D, cartoon/anime, photorealistic, art, portraits,
119
  text rendering).
120
+ - **Generation time**: median and minimum generation time in seconds, as
121
+ reported in the evaluation table. This is not a p95, and we do not state
122
+ warm vs cold or concurrent load. Not available for Arena AI.
123
+ - **Price**: USD per image in the evaluation table. We do not state list
124
+ price vs amount paid, or whether failed generations are included. Not
 
125
  available for Arena AI.
 
 
 
 
126
 
127
  Scores from different datasets or metrics are **not interchangeable**. A high
128
  OneIG alignment score is not the same quantity as a Datapoint Elo. Compare
 
136
  - **Prompt counts:** OneIG Alignment uses 100 anime, 100 human, and 99 object
137
  prompts (299 total). Qwen Image Dataset uses 100 prompts sampled from the
138
  1,000-prompt pool for roughly even coverage of its fine-grained (L3)
139
+ categories. Artificial Analysis and Arena AI use their own private prompt
140
+ sets.
 
 
 
141
  - **Generation (Qwen and OneIG):** one image per prompt per endpoint when
142
  the run exists. Default resolution is 1024×1024. Exceptions: FLUX 1.1 Pro
143
  Ultra at 2K, FLUX 2 Flex at 1008×1008, and any endpoint labeled 2K. The
144
  seed is derived from the prompt, so every model gets the same seed for the
145
  same prompt. Steps, CFG, prompt rewrite, and safety filters follow each
146
  endpoint's default. This does not describe Artificial Analysis or Arena AI.
 
 
 
 
147
  - **Datapoint (Qwen and OneIG):** every model pair is compared on every
148
  prompt, with 10 votes per battle.
149
  - **Rapidata (Qwen and OneIG):** prompts longer than 400 characters are
 
215
  </svg>
216
  </button>
217
  </div>
218
+ <p class="app-header-tagline">Compare text-to-image models on quality, speed, and price</p>
219
  </header>
220
  """,
221
  padding=False,
 
230
  return items[0] if items else None
231
 
232
 
233
+ def _dataset_choices(datasets, *, require_samples=False, require_pareto=False):
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
234
  return [
235
  (dataset["name"], dataset["id"])
236
+ for dataset in datasets
237
  if (not require_samples or dataset.get("samples"))
238
  and (not require_pareto or _dataset_has_pareto(datasets, dataset["id"]))
239
  ]
 
244
  return bool(dataset and dataset.get("samples"))
245
 
246
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
247
  def _dataset_has_pareto(datasets, dataset_id):
248
  dataset = _item(datasets, dataset_id)
249
+ columns = getattr(dataset.get("data") if dataset else None, "columns", [])
250
+ return _PARETO_PRICE_COLUMN in columns or _PARETO_TIME_COLUMN in columns
 
 
 
251
 
252
 
253
+ def _dataset_dropdown_update(datasets, tab, dataset_id):
254
  """Limit the dataset list to what the current tab can show."""
 
 
255
  return gr.update(
256
  choices=_dataset_choices(
257
  datasets,
 
258
  require_samples=tab == TAB_SAMPLES
259
  and _dataset_has_samples(datasets, dataset_id),
260
  require_pareto=tab == TAB_PARETO
 
314
  ]
315
 
316
 
317
+ def _model_choices(datasets, dataset_id):
318
  cached = _MODEL_CHOICES_CACHE.get(dataset_id)
319
+ if cached is not None:
 
 
 
 
 
 
 
 
 
 
 
 
 
320
  return cached
321
+ dataset = _item(datasets, dataset_id)
322
+ data = dataset.get("data") if dataset else None
323
+ if data is None or "Model" not in getattr(data, "columns", []):
324
+ _MODEL_CHOICES_CACHE[dataset_id] = []
325
+ return []
326
+ models = data["Model"].dropna().astype(str).unique().tolist()
327
+ # (label, value) so the UI shows the shared name but filters on the raw id.
328
+ choices = sorted(
329
+ ((display_model_name(model), model) for model in models),
330
+ key=lambda item: item[0].casefold(),
331
+ )
332
+ _MODEL_CHOICES_CACHE[dataset_id] = choices
333
+ return choices
334
 
335
 
336
  def _model_choice_values(choices):
 
358
  "Optimized",
359
  ]
360
  _LEADERBOARD_META_COLUMNS = [
 
361
  "Median Generation Time (s)",
362
  "Min Generation Time (s)",
363
  "Price / Image (USD)",
 
364
  "Evaluation Date (UTC)",
365
  "Date",
366
  ]
 
585
  "Arena Art Elo": "Art",
586
  "Arena Portraits Elo": "Portraits",
587
  "Arena Text Rendering Elo": "Text Rendering",
588
+ "Raw Win Rate": "Raw win rate",
589
  "Median Generation Time (s)": "Median generation time",
590
  "Min Generation Time (s)": "Min generation time",
 
591
  "Price / Image (USD)": "Price per image",
 
592
  "Evaluation Date (UTC)": "Date",
593
  "Date": "Date",
594
  }
 
663
  )
664
 
665
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
666
  def _build_pareto_figure(
667
  data,
668
  score_column,
 
670
  x_title,
671
  x_hover_prefix="",
672
  x_hover_suffix="",
 
673
  ):
674
  scatter = (
675
  data[["Model", score_column, x_column]]
 
686
 
687
  dominated = scatter.loc[[not flag for flag in on_frontier]].copy()
688
  frontier = scatter.loc[on_frontier].sort_values(x_column).copy()
 
 
689
  if not dominated.empty:
690
  dominated["Model"] = dominated["Model"].map(display_model_name)
691
  if not frontier.empty:
 
706
  name="Below frontier",
707
  text=dominated["Model"],
708
  hovertemplate=hover,
 
709
  marker={
710
  "size": 9,
711
+ "color": "#d8b4fe",
712
+ "opacity": 0.8,
713
  "line": {"width": 0},
714
  },
715
  )
 
723
  name="On frontier",
724
  text=frontier["Model"],
725
  hovertemplate=hover,
726
+ line={"color": "#69a45c", "width": 2.5},
 
727
  marker={
728
  "size": 12,
729
+ "color": "#69a45c",
730
+ "line": {"width": 1.5, "color": "#86c077"},
731
  },
732
  )
733
  )
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
734
 
735
  score_label = _display_label(score_column)
736
  fig.update_layout(
 
755
  )
756
  axis_font = {"color": "#fafafa", "size": 13}
757
  tick_font = {"color": "#a3a3a3", "size": 12}
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
758
  fig.update_xaxes(
 
 
759
  showgrid=True,
760
  gridcolor="rgba(74, 57, 98, 0.55)",
761
  zeroline=False,
 
774
  return fig
775
 
776
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
777
  def _pareto_axis(data, score_column, x_column, x_title, missing_message, empty_message, **hover):
778
  if x_column not in data.columns:
779
  return None, missing_message
 
789
  return fig, None
790
 
791
 
792
+ def _pareto_pair(data, score_column):
 
 
 
 
 
793
  score_missing = "No score data is available for this metric."
794
  if data is None or not score_column or score_column not in data.columns:
795
  return None, score_missing, None, score_missing
796
 
 
 
 
 
 
 
 
797
  price_fig, price_message = _pareto_axis(
798
  data,
799
  score_column,
800
+ _PARETO_PRICE_COLUMN,
801
+ "Price per image (USD)",
802
+ "Price per image isn't available for this dataset.",
803
  "No models have both a score and a price for this metric.",
804
  x_hover_prefix="$",
 
 
 
 
 
 
 
 
 
 
 
 
 
 
805
  )
806
  time_fig, time_message = _pareto_axis(
807
  data,
808
  score_column,
809
+ _PARETO_TIME_COLUMN,
810
+ "Min generation time (s)",
811
+ "Min generation time isn't available for this dataset.",
812
+ "No models have both a score and a min generation time for this metric.",
813
  x_hover_suffix="s",
 
814
  )
815
  return price_fig, price_message, time_fig, time_message
816
 
817
 
818
  def _pareto_dataset_message(data):
819
+ has_price = data is not None and _PARETO_PRICE_COLUMN in data.columns
820
+ has_time = data is not None and _PARETO_TIME_COLUMN in data.columns
821
  if has_price or has_time:
822
  return None
823
  return (
824
+ "Price per image and min generation time aren't available for "
825
  "this dataset, so these plots can't be drawn."
826
  )
827
 
828
 
829
  def _pareto_slot_note(price_fig, price_message, time_fig, time_message, data):
830
+ has_price = data is not None and _PARETO_PRICE_COLUMN in data.columns
831
+ has_time = data is not None and _PARETO_TIME_COLUMN in data.columns
832
  notes = []
833
  if has_price and not has_time:
834
  notes.append(
835
+ "Min generation time isn't available for this dataset, so only "
836
  "price vs score is shown."
837
  )
838
  elif has_time and not has_price:
839
  notes.append(
840
+ "Price per image isn't available for this dataset, so only min "
841
+ "generation time vs score is shown."
842
  )
843
  if price_fig is None and has_price:
844
  notes.append(price_message)
 
849
  return " ".join(notes)
850
 
851
 
852
+ def _pareto_slot_updates(data, score_columns):
 
 
 
 
 
853
  """Updates for a fixed bank of Gradio Plot slots (visible/hidden)."""
854
  score_columns = [column for column in (score_columns or []) if column]
855
+ has_price = data is not None and _PARETO_PRICE_COLUMN in data.columns
856
+ has_time = data is not None and _PARETO_TIME_COLUMN in data.columns
 
 
857
  dataset_note = _pareto_dataset_message(data)
858
  updates = [_pareto_note_update(dataset_note)]
859
  hide_all_slots = not has_price and not has_time
 
873
  continue
874
  score_column = score_columns[index]
875
  price_fig, price_message, time_fig, time_message = _pareto_pair(
876
+ data, score_column
 
 
 
877
  )
878
  show_price = price_fig is not None
879
  show_time = time_fig is not None
 
900
  return updates
901
 
902
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
903
  def _samples_html(samples, selected_models, num_prompts, seed=0):
904
  if not samples:
905
  return _pareto_unavailable_html(
906
  "Samples aren't available for this dataset."
907
  )
908
+ images = samples.get("images", {})
909
+ models = [model for model in (selected_models or []) if model in images]
 
 
 
910
  if not models:
911
+ models = (samples.get("models") or [])[:2]
912
  return _build_compare_samples_html(samples, models, num_prompts, seed)
913
 
914
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
915
  def _build_compare_samples_html(samples, selected_models, num_prompts, seed=0):
916
  selected_models = list(selected_models or [])[:MAX_COMPARE_MODELS]
 
 
 
917
 
918
  if not selected_models:
919
  return (
 
924
 
925
  shared_prompt_ids = None
926
  for model in selected_models:
927
+ model_prompt_ids = set(samples["images"][model])
928
  shared_prompt_ids = (
929
  model_prompt_ids
930
  if shared_prompt_ids is None
 
944
  rng.shuffle(prompt_pool)
945
  chosen = prompt_pool[: max(1, min(int(num_prompts), len(prompt_pool)))]
946
 
947
+ columns = len(selected_models)
948
  blocks = []
949
  for index, prompt_id in enumerate(chosen, start=1):
950
  prompt_text = escape(samples["prompts"].get(prompt_id, ""))
951
  cells = []
 
 
 
 
 
 
 
952
  for model in selected_models:
953
+ image_url = escape(samples["images"][model][prompt_id], quote=True)
954
  cells.append(
955
+ f"""
956
+ <div class="compare-cell">
957
+ <div class="compare-model-label">{escape(display_model_name(model))}</div>
958
+ <a href="{image_url}" target="_blank" rel="noopener noreferrer">
959
+ <img src="{image_url}" alt="{escape(display_model_name(model))} sample" loading="lazy" />
960
+ </a>
961
+ </div>
962
+ """
963
  )
 
964
  blocks.append(
965
  f"""
966
  <div class="compare-prompt-block">
967
  <div class="compare-prompt-meta">
968
  <span>Prompt {index}</span>
969
+ <span>{escape(prompt_id)}</span>
970
  </div>
971
  <p class="compare-prompt-text">{prompt_text}</p>
972
  <div class="compare-row" style="grid-template-columns: repeat({columns}, minmax(0, 1fr));">
 
994
  metric_id = _coerce_metric(
995
  datasets, metrics, default_dataset_id, default_metric_id
996
  )
 
997
  with gr.Row(elem_classes="view-filters"):
 
 
 
 
 
 
 
 
 
998
  dataset_dd = gr.Dropdown(
999
+ choices=_dataset_choices(datasets),
1000
  value=default_dataset_id,
1001
  label="Dataset",
1002
  type="value",
 
1028
  min_width=180,
1029
  elem_classes="filter-chips",
1030
  )
1031
+ return dataset_dd, metric_dd, models_dd
1032
 
1033
 
1034
  def render_image_workspace(datasets, metrics, default_dataset_id, default_metric_id):
 
1043
  with gr.Column(elem_classes="workspace-filters") as filters_host:
1044
  gr.Markdown(
1045
  "<p class='filter-help'>"
1046
+ "These filters apply to Leaderboards, Pareto plots, and Samples. "
1047
+ "On Samples, only datasets we have generations for are listed. "
1048
+ "On Pareto plots, only datasets with price or generation time "
1049
+ "are listed. Search in Models, or leave it empty to include "
1050
+ "every model."
 
 
1051
  "</p>",
1052
  elem_classes="filter-help-host",
1053
  )
1054
+ dataset_dd, metric_dd, models_dd = _filter_row(
1055
  datasets, metrics, default_dataset_id, None
1056
  )
1057
  with gr.Tabs(elem_classes="main-tabs") as main_tabs:
 
1117
  ) as pp_tab:
1118
  gr.Markdown(
1119
  "<p class='view-help'>"
1120
+ "Score against price and generation time. Green points are on the "
1121
+ "frontier; lavender points sit below it. Hover a point to see "
1122
+ "which model it is."
 
 
 
 
1123
  "</p>",
1124
  elem_classes="view-help-host",
1125
  )
 
 
 
 
 
 
 
 
 
 
 
1126
  pareto_dataset_note = gr.HTML(
1127
  "",
1128
  padding=False,
 
1148
  min_width=320,
1149
  elem_classes="pareto-col",
1150
  ) as slot_price_col:
1151
+ gr.Markdown(
1152
+ "#### Price vs score",
1153
+ elem_classes="pareto-subhead",
1154
  )
1155
  slot_price = gr.Plot(
1156
  value=None,
 
1162
  min_width=320,
1163
  elem_classes="pareto-col",
1164
  ) as slot_time_col:
1165
+ gr.Markdown(
1166
+ "#### Min generation time vs score",
1167
+ elem_classes="pareto-subhead",
1168
  )
1169
  slot_time = gr.Plot(
1170
  value=None,
 
1185
  slot_layout,
1186
  slot_price_col,
1187
  slot_price,
 
1188
  slot_time_col,
1189
  slot_time,
 
1190
  )
1191
  )
1192
 
 
1198
  with gr.Column(visible=bool(initial_samples)) as samples_panel:
1199
  gr.Markdown(
1200
  f"<p class='view-help'>"
1201
+ f"The same prompts, side by side. Select up to "
 
1202
  f"<strong>{MAX_COMPARE_MODELS}</strong> models above, or leave "
1203
  f"Models empty for two defaults."
1204
  f"</p>",
 
1235
  with gr.TabItem("About", id=TAB_ABOUT) as about_tab:
1236
  render_about()
1237
 
1238
+ def _synced_filters(dataset_id, metric_id, models, *, clear_metric=False):
 
 
1239
  if clear_metric:
1240
  metric_id = []
1241
  else:
1242
  metric_id = _coerce_metric(datasets, metrics, dataset_id, metric_id)
1243
+ model_choices = _model_choices(datasets, dataset_id)
 
 
1244
  model_values = set(_model_choice_values(model_choices))
1245
  models = [model for model in (models or []) if model in model_values]
1246
  metric_choices = _metric_dropdown_choices(datasets, metrics, dataset_id)
 
1312
  ):
1313
  prev = dict(view_state or {})
1314
  extras = extras or {}
 
 
 
 
1315
  return {
1316
  "dataset_id": dataset_id,
 
 
1317
  "metric_id": metric_id,
1318
  "models": list(models or []),
1319
  "current_tab": tab,
 
1324
  "optimized": list(
1325
  extras.get("optimized", prev.get("optimized") or [])
1326
  ),
 
 
1327
  "stale": {
1328
  TAB_LEADERBOARDS: not flags["include_leaderboard"],
1329
  TAB_PARETO: not flags["include_pareto"],
 
1383
  include_leaderboard=True,
1384
  include_pareto=False,
1385
  include_samples=False,
 
 
1386
  ):
1387
  view = resolve_view(datasets, metrics, dataset_id, metric_id)
1388
  data = view["data"]
 
1403
  ranking_html = gr.skip()
1404
  if include_pareto:
1405
  pareto_data = _filter_leaderboard(data, [], [], [], models=models)
1406
+ pareto_updates = _pareto_slot_updates(pareto_data, view["score_columns"])
 
 
 
 
 
1407
  else:
1408
  pareto_updates = _pareto_skip_updates()
1409
  if include_samples:
 
1442
  tab = view_state.get("current_tab") or TAB_LEADERBOARDS
1443
  selected_raw = _normalize_metric_ids(metric_id)
1444
  incoming_models = list(models or [])
1445
+ dataset_changed = source == "dataset" and dataset_id != view_state.get(
 
1446
  "dataset_id"
1447
  )
1448
+
1449
+ if source == "dataset":
 
 
 
 
 
 
1450
  synced = _synced_filters(
1451
+ dataset_id, metric_id, models, clear_metric=dataset_changed
 
 
 
 
1452
  )
1453
  dataset_id, metric_id, models = synced[:3]
1454
  metric_update, models_update = synced[3], synced[4]
 
1486
  ):
1487
  return None
1488
 
1489
+ selected_tab = tab
1490
  extras = (
1491
  list(platform_value or []),
1492
  list(owner_value or []),
1493
  list(optimized_value or []),
1494
  )
1495
  extra_updates = None
1496
+ can_pareto = _dataset_has_pareto(datasets, dataset_id)
1497
+ can_samples = _dataset_has_samples(datasets, dataset_id)
1498
+ if source == "dataset":
1499
+ if tab == TAB_SAMPLES and not can_samples:
1500
+ selected_tab = TAB_LEADERBOARDS
1501
+ elif tab == TAB_PARETO and not can_pareto:
1502
+ selected_tab = TAB_LEADERBOARDS
1503
  view = resolve_view(datasets, metrics, dataset_id, metric_id)
1504
  extra_updates = _leaderboard_extras(
1505
  view["data"] if view else None,
 
1537
  extras[2],
1538
  num_prompts,
1539
  seed,
 
 
1540
  **flags,
1541
  ),
1542
  "state": _commit_state(
 
1550
  ),
1551
  }
1552
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1553
  def on_dataset(
1554
  dataset_id,
1555
  metric_id,
 
1677
  )
1678
  dataset_update = _dataset_dropdown_update(datasets, tab, dataset_id)
1679
  metric_id = _coerce_metric(datasets, metrics, dataset_id, metric_id)
1680
+ models = list(models or [])
 
 
 
 
 
 
 
 
 
 
1681
  view_state["current_tab"] = tab
1682
  view_state["dataset_id"] = dataset_id
1683
  view_state["metric_id"] = metric_id
 
1714
  filters_vis,
1715
  dataset_update,
1716
  metric_vis,
1717
+ gr.skip(),
1718
  *lb_filters,
1719
  )
1720
  tab_select = (
 
1739
  optimized_value,
1740
  num_prompts,
1741
  seed,
 
 
1742
  **flags,
1743
  )
1744
  stale[tab] = False
 
1795
  next_seed,
1796
  )
1797
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1798
  def _on_tab(tab):
1799
  def handler(
1800
  dataset_id,
 
1823
  handler.__name__ = f"on_tab_{tab}"
1824
  return handler
1825
 
 
1826
  view_state = gr.State(
1827
  {
1828
  "dataset_id": default_dataset_id,
 
 
1829
  "metric_id": None,
1830
  "models": [],
1831
  "current_tab": TAB_LEADERBOARDS,
1832
  "platform": [],
1833
  "owner": [],
1834
  "optimized": [],
 
 
1835
  "stale": {
1836
  TAB_LEADERBOARDS: False,
1837
  TAB_PARETO: True,
 
1843
  pareto_dataset_note,
1844
  *[
1845
  component
1846
+ for slot_group, slot_title, slot_note, slot_layout, slot_price_col, slot_price, slot_time_col, slot_time in pareto_slots
1847
  for component in (
1848
  slot_group,
1849
  slot_title,
 
1856
  )
1857
  ],
1858
  ]
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1859
  view_inputs = [
1860
  platform,
1861
  owner,
 
1887
  main_tabs,
1888
  view_state,
1889
  ]
 
 
 
 
 
 
1890
  dataset_dd.change(
1891
  on_dataset,
1892
  inputs=filter_inputs,
 
1960
  show_progress="hidden",
1961
  )
1962
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1963
  prompt_count.change(
1964
  on_samples_controls,
1965
  inputs=[dataset_dd, models_dd, prompt_count, seed_state],