SIMIT demo: BAGEL, Lance, Qwen3.8-27B-FP8 + FLUX.2-klein on ZeroGPU
Browse filesThis view is limited to 50 files because it contains too many changes. See raw diff
- .gitattributes +21 -0
- README.md +70 -6
- app.py +297 -0
- assets/architect.webp +0 -0
- assets/artist.webp +0 -0
- assets/critic.webp +0 -0
- assets/improved.webp +0 -0
- assets/initial.webp +0 -0
- assets/router.webp +0 -0
- assets/solver.webp +0 -0
- assets/style.css +32 -0
- assets/synthesizer.webp +0 -0
- demo_models.py +251 -0
- examples/bagel_amazon/demo_0.png +3 -0
- examples/bagel_amazon/demo_1.png +3 -0
- examples/bagel_amazon/query.png +3 -0
- examples/bagel_infovqa_val_449/demo_0.png +0 -0
- examples/bagel_infovqa_val_449/query.jpg +3 -0
- examples/bagel_ok_vqa_val2014_492_always/demo_0.png +3 -0
- examples/bagel_ok_vqa_val2014_492_always/demo_1.png +3 -0
- examples/bagel_ok_vqa_val2014_492_always/query.jpg +0 -0
- examples/bagel_vizwiz_vqa_val_345_always/demo_0.png +3 -0
- examples/bagel_vizwiz_vqa_val_345_always/demo_1.png +3 -0
- examples/bagel_vizwiz_vqa_val_345_always/query.jpg +3 -0
- examples/bagel_vizwiz_vqa_val_47_always/demo_0.png +3 -0
- examples/bagel_vizwiz_vqa_val_47_always/demo_1.png +3 -0
- examples/bagel_vizwiz_vqa_val_47_always/query.jpg +3 -0
- examples/bagel_vizwiz_vqa_val_87_always/demo_0.png +3 -0
- examples/bagel_vizwiz_vqa_val_87_always/demo_1.png +3 -0
- examples/bagel_vizwiz_vqa_val_87_always/query.jpg +3 -0
- examples/index.json +274 -0
- examples/lance_ok_vqa_val2014_253/demo_0.png +3 -0
- examples/lance_ok_vqa_val2014_253/query.jpg +3 -0
- examples/lance_ok_vqa_val2014_8_always/demo_0.png +3 -0
- examples/lance_ok_vqa_val2014_8_always/demo_1.png +3 -0
- examples/lance_ok_vqa_val2014_8_always/demo_2.png +3 -0
- examples/lance_ok_vqa_val2014_8_always/demo_3.png +3 -0
- examples/lance_ok_vqa_val2014_8_always/query.jpg +0 -0
- packages.txt +18 -0
- presets.json +233 -0
- requirements.txt +4 -0
- tools/build_presets.py +91 -0
- tools/fork_test.py +55 -0
- tools/general_qa.py +70 -0
- tools/latency.py +74 -0
- tools/make_examples.py +121 -0
- tools/results/latency_bagel.json +882 -0
- tools/results/latency_lance.json +446 -0
- tools/results/latency_qwen.json +347 -0
- tools/results/presets_bagel.json +0 -0
.gitattributes
CHANGED
|
@@ -33,3 +33,24 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
|
| 33 |
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 33 |
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
| 36 |
+
examples/bagel_amazon/demo_0.png filter=lfs diff=lfs merge=lfs -text
|
| 37 |
+
examples/bagel_amazon/demo_1.png filter=lfs diff=lfs merge=lfs -text
|
| 38 |
+
examples/bagel_amazon/query.png filter=lfs diff=lfs merge=lfs -text
|
| 39 |
+
examples/bagel_infovqa_val_449/query.jpg filter=lfs diff=lfs merge=lfs -text
|
| 40 |
+
examples/bagel_ok_vqa_val2014_492_always/demo_0.png filter=lfs diff=lfs merge=lfs -text
|
| 41 |
+
examples/bagel_ok_vqa_val2014_492_always/demo_1.png filter=lfs diff=lfs merge=lfs -text
|
| 42 |
+
examples/bagel_vizwiz_vqa_val_345_always/demo_0.png filter=lfs diff=lfs merge=lfs -text
|
| 43 |
+
examples/bagel_vizwiz_vqa_val_345_always/demo_1.png filter=lfs diff=lfs merge=lfs -text
|
| 44 |
+
examples/bagel_vizwiz_vqa_val_345_always/query.jpg filter=lfs diff=lfs merge=lfs -text
|
| 45 |
+
examples/bagel_vizwiz_vqa_val_47_always/demo_0.png filter=lfs diff=lfs merge=lfs -text
|
| 46 |
+
examples/bagel_vizwiz_vqa_val_47_always/demo_1.png filter=lfs diff=lfs merge=lfs -text
|
| 47 |
+
examples/bagel_vizwiz_vqa_val_47_always/query.jpg filter=lfs diff=lfs merge=lfs -text
|
| 48 |
+
examples/bagel_vizwiz_vqa_val_87_always/demo_0.png filter=lfs diff=lfs merge=lfs -text
|
| 49 |
+
examples/bagel_vizwiz_vqa_val_87_always/demo_1.png filter=lfs diff=lfs merge=lfs -text
|
| 50 |
+
examples/bagel_vizwiz_vqa_val_87_always/query.jpg filter=lfs diff=lfs merge=lfs -text
|
| 51 |
+
examples/lance_ok_vqa_val2014_253/demo_0.png filter=lfs diff=lfs merge=lfs -text
|
| 52 |
+
examples/lance_ok_vqa_val2014_253/query.jpg filter=lfs diff=lfs merge=lfs -text
|
| 53 |
+
examples/lance_ok_vqa_val2014_8_always/demo_0.png filter=lfs diff=lfs merge=lfs -text
|
| 54 |
+
examples/lance_ok_vqa_val2014_8_always/demo_1.png filter=lfs diff=lfs merge=lfs -text
|
| 55 |
+
examples/lance_ok_vqa_val2014_8_always/demo_2.png filter=lfs diff=lfs merge=lfs -text
|
| 56 |
+
examples/lance_ok_vqa_val2014_8_always/demo_3.png filter=lfs diff=lfs merge=lfs -text
|
README.md
CHANGED
|
@@ -1,13 +1,77 @@
|
|
| 1 |
---
|
| 2 |
-
title:
|
| 3 |
-
emoji:
|
| 4 |
-
colorFrom:
|
| 5 |
-
colorTo:
|
| 6 |
sdk: gradio
|
| 7 |
sdk_version: 6.29.1
|
| 8 |
-
python_version:
|
| 9 |
app_file: app.py
|
| 10 |
pinned: false
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 11 |
---
|
| 12 |
|
| 13 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
---
|
| 2 |
+
title: SIMIT - models that imagine their own practice examples
|
| 3 |
+
emoji: 🥯
|
| 4 |
+
colorFrom: yellow
|
| 5 |
+
colorTo: red
|
| 6 |
sdk: gradio
|
| 7 |
sdk_version: 6.29.1
|
| 8 |
+
python_version: "3.12"
|
| 9 |
app_file: app.py
|
| 10 |
pinned: false
|
| 11 |
+
license: apache-2.0
|
| 12 |
+
startup_duration_timeout: 1h
|
| 13 |
+
short_description: A VLM imagines practice examples for your question
|
| 14 |
+
models:
|
| 15 |
+
- ByteDance-Seed/BAGEL-7B-MoT
|
| 16 |
+
- bytedance-research/Lance
|
| 17 |
+
- Qwen/Qwen3.8-27B-FP8
|
| 18 |
+
- black-forest-labs/FLUX.2-klein-4B
|
| 19 |
---
|
| 20 |
|
| 21 |
+
# SIMIT demo
|
| 22 |
+
|
| 23 |
+
Upload an image, ask a question, and compare the model's plain answer with its
|
| 24 |
+
SIMIT-ICL answer: before answering again, the model imagines a few practice
|
| 25 |
+
examples (question, answer decided first, an image made to fit), checks them,
|
| 26 |
+
and keeps them in context.
|
| 27 |
+
|
| 28 |
+
Built on the [`simit`](https://github.com/monurcan/simit) library. See `tools/` for how the per-budget
|
| 29 |
+
hyperparameters were chosen and how the cached examples were produced.
|
| 30 |
+
|
| 31 |
+
## Hardware notes (ZeroGPU)
|
| 32 |
+
|
| 33 |
+
* Each request runs in a fresh `@spaces.GPU` process and uses the visitor's own
|
| 34 |
+
ZeroGPU quota for the selected time budget (20-180 s).
|
| 35 |
+
* BAGEL-7B-MoT and Lance run on the default 48 GB size; Qwen3.8-27B-FP8 +
|
| 36 |
+
FLUX.2-klein needs the 96 GB size (2x quota).
|
| 37 |
+
* All weights stay on CPU in the main process (~91 GB of RAM, ~92 GB of disk for
|
| 38 |
+
the model files). If the Space's disk is too small, attach persistent storage
|
| 39 |
+
and set `HF_HOME=/data/.huggingface`, or limit the models with
|
| 40 |
+
`SIMIT_DEMO_MODELS=bagel,lance`.
|
| 41 |
+
* The first request after start-up compiles and autotunes GPU kernels (Qwen:
|
| 42 |
+
~25 s); later requests reuse the on-disk cache.
|
| 43 |
+
|
| 44 |
+
## How the time budgets map to settings
|
| 45 |
+
|
| 46 |
+
Each budget (20 / 40 / 60 / 90 / 180 s of GPU time per request) uses its own settings per model
|
| 47 |
+
(`presets.json`, built by `tools/build_presets.py`):
|
| 48 |
+
|
| 49 |
+
1. **Accuracy per K_max** (`tools/tune_presets.py`): one synthesis cache with 4 candidates per query on a
|
| 50 |
+
mixed 64-question general-QA validation set: 8 each from VizWiz, VQAv2, OK-VQA, TextVQA, ChartQA,
|
| 51 |
+
DocVQA, MMBench and AI2D (LMMs-Eval-Lite rows 300-307), each scored with its own metric. Then 800
|
| 52 |
+
Optuna trials over the ABA/DF thresholds and K_max ∈ {1,2,3,4}. Per K_max, the pick is regularized:
|
| 53 |
+
within one validation query of the best score, the widest difficulty-filter band.
|
| 54 |
+
2. **Latency** (`tools/latency.py`): per-request timelines on the demo's own code path in fresh forked
|
| 55 |
+
processes (as on ZeroGPU, weight transfer included). Measured on an H100 and scaled by spaces'
|
| 56 |
+
own factor for the ZeroGPU GPU sizes (1.5× on `large`, 1× on `xlarge`).
|
| 57 |
+
3. **Choice**: the tuned (best-quality) speed setting whenever some K fits the budget for ≥80% of
|
| 58 |
+
requests; among those, the K within one validation query of the best score with the widest band.
|
| 59 |
+
More budget also buys more attempts and verification rounds. If no K beats zero-shot on
|
| 60 |
+
validation, SIMIT keeps the base answer by default; "Imagine even when confident" still runs it.
|
| 61 |
+
|
| 62 |
+
Validation scores (64 general questions):
|
| 63 |
+
|
| 64 |
+
| model | zero-shot | K=1 | K=2 | K=3 | K=4 | used by default |
|
| 65 |
+
|---|---|---|---|---|---|---|
|
| 66 |
+
| BAGEL-7B-MoT | 0.703 | 0.708 | **0.750** | 0.760 | 0.755 | K=2 from 60 s (20-40 s: zero-shot) |
|
| 67 |
+
| Lance | 0.432 | 0.442 | 0.442 | 0.458 | **0.468** | K=1 at 20 s, K=4 from 40 s |
|
| 68 |
+
| Qwen3.8-27B-FP8 + FLUX | **0.854** | 0.839 | 0.823 | 0.839 | 0.844 | zero-shot (no gain found) |
|
| 69 |
+
|
| 70 |
+
Lance's answers are sampled (greedy decoding degenerates on this checkpoint), so its outputs vary
|
| 71 |
+
between runs and SIMIT can also hurt a correct answer. Qwen3.8-27B is strong enough that imagined
|
| 72 |
+
examples did not improve it on these questions.
|
| 73 |
+
|
| 74 |
+
The cached examples are real runs of this app (BAGEL-7B-MoT and Lance) on queries from the paper's
|
| 75 |
+
figures, plus one product-page screenshot (`examples/bagel_amazon`). Several of them used "Imagine even when confident" (shown in the example), because the base
|
| 76 |
+
model was confidently wrong (e.g. "Unanswerable"). Qwen3.8-27B already answered those queries
|
| 77 |
+
correctly, so it has no cached example.
|
app.py
ADDED
|
@@ -0,0 +1,297 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""SIMIT demo: a vision-language model imagines its own practice examples for
|
| 2 |
+
your question, then answers again with them in context.
|
| 3 |
+
|
| 4 |
+
Runs on Hugging Face ZeroGPU (each request uses the visitor's own GPU quota)
|
| 5 |
+
and on any machine with a CUDA GPU (``python app.py``).
|
| 6 |
+
"""
|
| 7 |
+
|
| 8 |
+
import os
|
| 9 |
+
|
| 10 |
+
# ZeroGPU starts a fresh process per request: keep Triton's autotuning results on disk
|
| 11 |
+
# (otherwise every request re-tunes the FP8 / linear-attention kernels, ~20 s).
|
| 12 |
+
os.environ.setdefault("TRITON_CACHE_AUTOTUNING", "1")
|
| 13 |
+
os.environ.setdefault("TRANSFORMERS_DISABLE_DEEPGEMM_LINEAR", "1")
|
| 14 |
+
|
| 15 |
+
import spaces # before anything CUDA-related: ZeroGPU patches torch # noqa: E402 # isort: skip
|
| 16 |
+
|
| 17 |
+
import base64 # noqa: E402
|
| 18 |
+
import html # noqa: E402
|
| 19 |
+
import json # noqa: E402
|
| 20 |
+
from pathlib import Path # noqa: E402
|
| 21 |
+
|
| 22 |
+
import gradio as gr # noqa: E402
|
| 23 |
+
from PIL import Image # noqa: E402
|
| 24 |
+
|
| 25 |
+
from demo_models import BUDGETS, ENABLED, MODELS, code_snippet, run # noqa: E402
|
| 26 |
+
|
| 27 |
+
HERE = Path(__file__).parent
|
| 28 |
+
ASSETS = HERE / "assets"
|
| 29 |
+
EXAMPLES_DIR = HERE / "examples"
|
| 30 |
+
|
| 31 |
+
|
| 32 |
+
def _data_uri(path: Path) -> str:
|
| 33 |
+
return "data:image/webp;base64," + base64.b64encode(path.read_bytes()).decode()
|
| 34 |
+
|
| 35 |
+
|
| 36 |
+
MASCOT = {p.stem: _data_uri(p) for p in ASSETS.glob("*.webp")}
|
| 37 |
+
STAGES = { # stage -> (mascot, message)
|
| 38 |
+
"idle": ("initial", "Ask me anything about an image"),
|
| 39 |
+
"loading": ("initial", "Loading the model weights (first request only)"),
|
| 40 |
+
"warmup": ("initial", "Waking up the model on the GPU"),
|
| 41 |
+
"solving": ("initial", "Solving it on its own first"),
|
| 42 |
+
"synthesizing": ("synthesizer", "Writing practice questions it already knows the answers to"),
|
| 43 |
+
"routing": ("router", "Choosing how to draw each practice image"),
|
| 44 |
+
"drawing": ("artist", "Painting a practice image"),
|
| 45 |
+
"coding": ("architect", "Writing code for a chart, diagram or document"),
|
| 46 |
+
"checking": ("critic", "Checking that each picture matches its question"),
|
| 47 |
+
"improving": ("improved", "Answering again, with its own practice examples in context"),
|
| 48 |
+
}
|
| 49 |
+
LABELS = {MODELS[k].label: k for k in ENABLED}
|
| 50 |
+
BUDGET_LABELS = [f"{b} s" for b in BUDGETS]
|
| 51 |
+
GPU_OVERHEAD = 20 # seconds requested on top of the budget (worker start-up, final answer); weight transfer counts inside it
|
| 52 |
+
|
| 53 |
+
|
| 54 |
+
def _seconds(label: str) -> int:
|
| 55 |
+
return int(str(label).split()[0])
|
| 56 |
+
|
| 57 |
+
|
| 58 |
+
# --------------------------------------------------------------------- HTML
|
| 59 |
+
def status_html(stage: str, elapsed: float = 0.0, limit: float = 0.0, n_demos: int = 0, note: str = "") -> str:
|
| 60 |
+
mascot, message = STAGES.get(stage, STAGES["solving"])
|
| 61 |
+
dots = "" if stage == "idle" else '<span class="simit-dots"></span>'
|
| 62 |
+
bar = ""
|
| 63 |
+
if limit:
|
| 64 |
+
pct = max(2.0, min(100.0, 100.0 * elapsed / limit))
|
| 65 |
+
bar = (f'<div class="simit-bar"><div style="width:{pct:.0f}%"></div></div>'
|
| 66 |
+
f'<div class="simit-sub">{elapsed:.0f} s of {limit:.0f} s'
|
| 67 |
+
+ (f' · {n_demos} practice example{"s" if n_demos != 1 else ""} ready' if n_demos else "")
|
| 68 |
+
+ "</div>")
|
| 69 |
+
return (f'<div class="simit-status"><img class="simit-mascot" src="{MASCOT[mascot]}" alt="">'
|
| 70 |
+
f'<div><div class="simit-msg">{html.escape(message)}{dots}</div>'
|
| 71 |
+
f'{bar}{f"<div class=simit-sub>{html.escape(note)}</div>" if note else ""}</div></div>')
|
| 72 |
+
|
| 73 |
+
|
| 74 |
+
def done_html(note: str) -> str:
|
| 75 |
+
return (f'<div class="simit-status done"><img class="simit-mascot still" src="{MASCOT["improved"]}" alt="">'
|
| 76 |
+
f'<div><div class="simit-msg">Done</div><div class="simit-sub">{html.escape(note)}</div></div></div>')
|
| 77 |
+
|
| 78 |
+
|
| 79 |
+
def error_html(note: str) -> str:
|
| 80 |
+
return (f'<div class="simit-status error"><img class="simit-mascot still" src="{MASCOT["critic"]}" alt="">'
|
| 81 |
+
f'<div><div class="simit-msg">Something went wrong</div><div class="simit-sub">{html.escape(note)}'
|
| 82 |
+
'</div></div></div>')
|
| 83 |
+
|
| 84 |
+
|
| 85 |
+
def answer_card(kind: str, answer: str = "", sub: str = "", mark: str = "") -> str:
|
| 86 |
+
title = "Base model (greedy)" if kind == "base" else "SIMIT-ICL (self-improved)"
|
| 87 |
+
body = html.escape(answer) if answer else '<span class="simit-wait">…</span>'
|
| 88 |
+
size = " long" if len(answer) > 60 else ""
|
| 89 |
+
badge = {"right": '<span class="simit-ok">✓</span>', "wrong": '<span class="simit-no">✗</span>'}.get(mark, "")
|
| 90 |
+
return (f'<div class="simit-card {kind}"><div class="simit-card-title">{title}{badge}</div>'
|
| 91 |
+
f'<div class="simit-answer{size}">{body}</div><div class="simit-card-sub">{html.escape(sub)}</div></div>')
|
| 92 |
+
|
| 93 |
+
|
| 94 |
+
def demo_caption(d) -> str:
|
| 95 |
+
bits = [f"Q: {d.question}", f"A: {d.answer}"]
|
| 96 |
+
meta = [d.skill] if d.skill else []
|
| 97 |
+
if d.verify_score is not None:
|
| 98 |
+
meta.append(f"critic {d.verify_score}/100")
|
| 99 |
+
if d.confidence is not None:
|
| 100 |
+
meta.append(f"confidence {d.confidence:.2f}")
|
| 101 |
+
return "\n".join(bits) + (f"\n({', '.join(meta)})" if meta else "")
|
| 102 |
+
|
| 103 |
+
|
| 104 |
+
# ------------------------------------------------------------- GPU requests
|
| 105 |
+
def _duration(image, question, key, budget_label, always, *args, **kwargs):
|
| 106 |
+
return _seconds(budget_label) + GPU_OVERHEAD
|
| 107 |
+
|
| 108 |
+
|
| 109 |
+
def _session(image, question, key, budget_label, always):
|
| 110 |
+
"""Runs inside the GPU call: turns pipeline events into UI updates."""
|
| 111 |
+
spec, budget = MODELS[key], _seconds(budget_label)
|
| 112 |
+
try:
|
| 113 |
+
yield from _events(spec, image, question, budget, always)
|
| 114 |
+
except Exception as e: # show it in the page instead of a bare error toast
|
| 115 |
+
print(f"[demo] request failed: {type(e).__name__}: {e}", flush=True)
|
| 116 |
+
yield (error_html(f"{type(e).__name__}: {str(e)[:300]}"), answer_card("base"), answer_card("simit"), [], "")
|
| 117 |
+
|
| 118 |
+
|
| 119 |
+
def _events(spec, image, question, budget, always):
|
| 120 |
+
greedy, k_budget, demos, gallery = None, 0, [], []
|
| 121 |
+
base, improved = answer_card("base"), answer_card("simit")
|
| 122 |
+
status = status_html("warmup")
|
| 123 |
+
for event in run(spec, image, question, budget, always):
|
| 124 |
+
kind = event[0]
|
| 125 |
+
if kind == "stage":
|
| 126 |
+
_, stage, info = event
|
| 127 |
+
status = status_html(stage, info.get("elapsed", 0), info.get("limit", 0), len(demos))
|
| 128 |
+
elif kind == "greedy":
|
| 129 |
+
_, zs, k_budget = event
|
| 130 |
+
greedy = zs
|
| 131 |
+
sub = f"confidence p0 = {zs.confidence:.2f}" if zs.confidence is not None else ""
|
| 132 |
+
base = answer_card("base", zs.answer, sub)
|
| 133 |
+
elif kind == "demo":
|
| 134 |
+
demos.append(event[1])
|
| 135 |
+
gallery = [(d.image, demo_caption(d)) for d in demos]
|
| 136 |
+
else: # final
|
| 137 |
+
_, answer, info = event
|
| 138 |
+
reason = info.get("skipped")
|
| 139 |
+
if reason == "confident":
|
| 140 |
+
sub = "The model was already confident, so SIMIT kept its answer (adaptive budget: 0 examples)."
|
| 141 |
+
elif reason == "policy":
|
| 142 |
+
sub = ("On general questions, our tuning found no gain from imagination for this model at this "
|
| 143 |
+
"budget, so SIMIT keeps the base answer. Tick 'Imagine even when confident' to try it anyway.")
|
| 144 |
+
elif reason == "no_time":
|
| 145 |
+
sub = "Not enough time budget left to imagine examples; kept the base answer."
|
| 146 |
+
elif reason == "none_passed":
|
| 147 |
+
sub = "No imagined example passed the checks in time; kept the base answer."
|
| 148 |
+
else:
|
| 149 |
+
changed = "changed" if answer.strip() != greedy.answer.strip() else "kept"
|
| 150 |
+
sub = f"{changed} the answer after {info['demos']} imagined example{'s' if info['demos'] > 1 else ''}"
|
| 151 |
+
improved = answer_card("simit", answer, sub)
|
| 152 |
+
status = done_html(f"{info['elapsed']:.0f} s on the GPU")
|
| 153 |
+
yield status, base, improved, gallery, _details(greedy, k_budget, demos, spec, budget)
|
| 154 |
+
|
| 155 |
+
|
| 156 |
+
@spaces.GPU(duration=_duration, size="large")
|
| 157 |
+
def gpu_large(image, question, key, budget_label, always):
|
| 158 |
+
yield from _session(image, question, key, budget_label, always)
|
| 159 |
+
|
| 160 |
+
|
| 161 |
+
@spaces.GPU(duration=_duration, size="xlarge")
|
| 162 |
+
def gpu_xlarge(image, question, key, budget_label, always):
|
| 163 |
+
yield from _session(image, question, key, budget_label, always)
|
| 164 |
+
|
| 165 |
+
|
| 166 |
+
def _details(greedy, k_budget, demos, spec, budget) -> str:
|
| 167 |
+
if greedy is None:
|
| 168 |
+
return ""
|
| 169 |
+
p0 = f"{greedy.confidence:.2f}" if greedy.confidence is not None else "n/a"
|
| 170 |
+
return (f"**{spec.label}**, {budget} s budget. Zero-shot confidence p0 = {p0}, so the adaptive budget asked "
|
| 171 |
+
f"for **{k_budget}** example{'s' if k_budget != 1 else ''}; **{len(demos)}** passed the checks. "
|
| 172 |
+
"Each example is a question the model wrote, whose answer it decided first, with an image made "
|
| 173 |
+
"to fit that answer.")
|
| 174 |
+
|
| 175 |
+
|
| 176 |
+
def submit(image, question, model_label, budget_label, always):
|
| 177 |
+
if image is None:
|
| 178 |
+
raise gr.Error("Please upload an image.")
|
| 179 |
+
if not question or not question.strip():
|
| 180 |
+
raise gr.Error("Please type a question about the image.")
|
| 181 |
+
key = LABELS[model_label]
|
| 182 |
+
spec = MODELS[key]
|
| 183 |
+
image = image.convert("RGB")
|
| 184 |
+
if spec._weights is None: # main process, CPU: no GPU time is spent here
|
| 185 |
+
yield (status_html("loading"), answer_card("base"), answer_card("simit"), [], "")
|
| 186 |
+
spec.load()
|
| 187 |
+
fn = gpu_xlarge if spec.gpu_size == "xlarge" else gpu_large
|
| 188 |
+
yield from fn(image, question.strip(), key, budget_label, always)
|
| 189 |
+
|
| 190 |
+
|
| 191 |
+
# ------------------------------------------------------------------ cached
|
| 192 |
+
def _load_examples():
|
| 193 |
+
index = EXAMPLES_DIR / "index.json"
|
| 194 |
+
return json.loads(index.read_text()) if index.exists() else []
|
| 195 |
+
|
| 196 |
+
|
| 197 |
+
EXAMPLES = [e for e in _load_examples() if e["model"] in ENABLED]
|
| 198 |
+
|
| 199 |
+
|
| 200 |
+
def show_cached(image, question, model_label, budget_label, always=False):
|
| 201 |
+
"""A cached example: the stored outputs of a real run, shown without using the GPU."""
|
| 202 |
+
entry = next((e for e in EXAMPLES if e["question"] == question and MODELS[e["model"]].label == model_label
|
| 203 |
+
and f"{e['budget']} s" == budget_label and bool(e.get("always")) == bool(always)), None)
|
| 204 |
+
if entry is None:
|
| 205 |
+
return status_html("idle", note="Press Submit to run this example."), answer_card("base"), \
|
| 206 |
+
answer_card("simit"), [], ""
|
| 207 |
+
d = EXAMPLES_DIR / entry["id"]
|
| 208 |
+
mark = lambda ok: "right" if ok else "wrong" # noqa: E731
|
| 209 |
+
gt = f"reference answer: {entry['reference']}"
|
| 210 |
+
base = answer_card("base", entry["greedy"], f"confidence p0 = {entry['p0']:.2f} · {gt}", mark(entry["greedy_ok"]))
|
| 211 |
+
n = len(entry["demos"])
|
| 212 |
+
improved = answer_card("simit", entry["simit"], f"after {n} imagined example{'s' if n != 1 else ''} · {gt}",
|
| 213 |
+
mark(entry["simit_ok"]))
|
| 214 |
+
gallery = [(str(d / x["image"]), "\n".join([f"Q: {x['question']}", f"A: {x['answer']}",
|
| 215 |
+
f"({x['skill']}" + (f", critic {x['verify_score']}/100" if
|
| 216 |
+
x.get("verify_score") is not None else "") + ")"]))
|
| 217 |
+
for x in entry["demos"]]
|
| 218 |
+
note = (f"Cached result from a real run ({entry['elapsed']:.0f} s on an H100, {entry['budget']} s budget"
|
| 219 |
+
+ (", imagining even when confident" if entry.get("always") else "") + f"). Source: {entry['source']}.")
|
| 220 |
+
details = (f"**{MODELS[entry['model']].label}**, {entry['budget']} s budget. Zero-shot confidence "
|
| 221 |
+
f"p0 = {entry['p0']:.2f}; {n} imagined example{'s' if n != 1 else ''} passed the checks.")
|
| 222 |
+
return done_html(note), base, improved, gallery, details
|
| 223 |
+
|
| 224 |
+
|
| 225 |
+
# ---------------------------------------------------------------------- UI
|
| 226 |
+
CSS = (ASSETS / "style.css").read_text()
|
| 227 |
+
INTRO = """
|
| 228 |
+
<div class="simit-hero">
|
| 229 |
+
<h1>SIMIT: models that imagine their own practice examples</h1>
|
| 230 |
+
<p>Ask any question about an image. The model first answers on its own. Then, without any labels, it writes
|
| 231 |
+
similar practice questions whose answers it decides first, makes an image for each one, keeps only those it
|
| 232 |
+
can verify, and answers your question again with them in context.</p>
|
| 233 |
+
</div>
|
| 234 |
+
"""
|
| 235 |
+
|
| 236 |
+
with gr.Blocks(title="SIMIT demo") as demo:
|
| 237 |
+
gr.HTML(INTRO)
|
| 238 |
+
with gr.Row(equal_height=False):
|
| 239 |
+
with gr.Column(scale=5):
|
| 240 |
+
image = gr.Image(type="pil", label="Image", height=360)
|
| 241 |
+
question = gr.Textbox(label="Question", placeholder="e.g. In what country would you find this hat?",
|
| 242 |
+
lines=2)
|
| 243 |
+
model = gr.Dropdown(list(LABELS), value=next(iter(LABELS)), label="Model")
|
| 244 |
+
budget = gr.Radio(BUDGET_LABELS, value="60 s", label="Max GPU time per request",
|
| 245 |
+
info="More time lets the model imagine and check more practice examples.")
|
| 246 |
+
with gr.Accordion("Advanced", open=False):
|
| 247 |
+
always = gr.Checkbox(False, label="Imagine even when the model is already confident",
|
| 248 |
+
info="By default SIMIT skips imagination for confident answers.")
|
| 249 |
+
run_btn = gr.Button("Submit", variant="primary")
|
| 250 |
+
with gr.Column(scale=6):
|
| 251 |
+
status = gr.HTML(status_html("idle", note="Upload an image and type a question, or pick an example below."))
|
| 252 |
+
with gr.Row():
|
| 253 |
+
base_out = gr.HTML(answer_card("base"), min_width=260)
|
| 254 |
+
simit_out = gr.HTML(answer_card("simit"), min_width=260)
|
| 255 |
+
with gr.Accordion("See the imagined practice examples", open=False):
|
| 256 |
+
details = gr.Markdown()
|
| 257 |
+
gallery = gr.Gallery(columns=4, height=300, object_fit="contain", show_label=False)
|
| 258 |
+
with gr.Accordion("Use SIMIT in your own code", open=False):
|
| 259 |
+
code = gr.Code(code_snippet(MODELS[next(iter(LABELS.values()))], 60, ""), language="python",
|
| 260 |
+
interactive=False)
|
| 261 |
+
|
| 262 |
+
outputs = [status, base_out, simit_out, gallery, details]
|
| 263 |
+
# On ZeroGPU each request gets its own GPU process; run locally, requests share this process and GPU.
|
| 264 |
+
from spaces.config import Config as _SpacesConfig
|
| 265 |
+
run_btn.click(submit, [image, question, model, budget, always], outputs,
|
| 266 |
+
concurrency_limit=4 if _SpacesConfig.zero_gpu else 1)
|
| 267 |
+
|
| 268 |
+
def _code(model_label, budget_label, q, force):
|
| 269 |
+
return code_snippet(MODELS[LABELS[model_label]], _seconds(budget_label), q or "", force)
|
| 270 |
+
for comp in (model, budget, question, always):
|
| 271 |
+
comp.change(_code, [model, budget, question, always], code, queue=False)
|
| 272 |
+
|
| 273 |
+
if EXAMPLES:
|
| 274 |
+
gr.Markdown("### Examples\nCached results from real runs of this demo (click one; press Submit to "
|
| 275 |
+
"run it again live).")
|
| 276 |
+
examples = gr.Gallery(
|
| 277 |
+
[(str(EXAMPLES_DIR / e["id"] / e["query_file"]),
|
| 278 |
+
f"{e['question'].splitlines()[0]} ({MODELS[e['model']].label})") for e in EXAMPLES],
|
| 279 |
+
columns=4, height=300 * ((len(EXAMPLES) + 3) // 4) + 20, object_fit="cover", allow_preview=False,
|
| 280 |
+
show_label=False)
|
| 281 |
+
|
| 282 |
+
def pick(evt: gr.SelectData):
|
| 283 |
+
e = EXAMPLES[evt.index]
|
| 284 |
+
inputs = [Image.open(EXAMPLES_DIR / e["id"] / e["query_file"]), e["question"],
|
| 285 |
+
MODELS[e["model"]].label, f"{e['budget']} s", bool(e.get("always"))]
|
| 286 |
+
return (*inputs, *show_cached(*inputs))
|
| 287 |
+
examples.select(pick, None, [image, question, model, budget, always] + outputs)
|
| 288 |
+
gr.Markdown("Models: " + ", ".join(f"[{MODELS[k].label}](https://huggingface.co/{MODELS[k].repo})"
|
| 289 |
+
for k in ENABLED)
|
| 290 |
+
+ ". Each request uses your own ZeroGPU quota for the selected time budget.")
|
| 291 |
+
|
| 292 |
+
if os.environ.get("SIMIT_DEMO_PRELOAD", "1") == "1":
|
| 293 |
+
for key in ENABLED: # load every model on CPU before serving (main process, no CUDA)
|
| 294 |
+
MODELS[key].load()
|
| 295 |
+
|
| 296 |
+
if __name__ == "__main__":
|
| 297 |
+
demo.queue(max_size=32).launch(css=CSS, theme=gr.themes.Soft(primary_hue="orange", secondary_hue="amber"))
|
assets/architect.webp
ADDED
|
assets/artist.webp
ADDED
|
assets/critic.webp
ADDED
|
assets/improved.webp
ADDED
|
assets/initial.webp
ADDED
|
assets/router.webp
ADDED
|
assets/solver.webp
ADDED
|
assets/style.css
ADDED
|
@@ -0,0 +1,32 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
.simit-hero h1 { font-size: 1.7rem; margin: 0.2rem 0 0.4rem; }
|
| 2 |
+
.simit-hero p { margin: 0 0 0.6rem; opacity: 0.85; max-width: 60rem; }
|
| 3 |
+
|
| 4 |
+
.simit-status { display: flex; align-items: center; gap: 1rem; min-height: 132px; padding: 0.6rem 0.8rem;
|
| 5 |
+
border-radius: 14px; background: var(--block-background-fill); border: 1px solid var(--border-color-primary); }
|
| 6 |
+
.simit-mascot { height: 112px; width: auto; animation: simit-bob 1.6s ease-in-out infinite; transform-origin: 50% 90%; }
|
| 7 |
+
.simit-mascot.still { animation: simit-pop 0.5s ease-out 1; }
|
| 8 |
+
.simit-msg { font-size: 1.05rem; font-weight: 600; }
|
| 9 |
+
.simit-sub { font-size: 0.85rem; opacity: 0.75; margin-top: 0.3rem; }
|
| 10 |
+
.simit-dots::after { content: ""; animation: simit-dots 1.4s steps(4, end) infinite; }
|
| 11 |
+
.simit-bar { height: 8px; width: 16rem; max-width: 100%; border-radius: 4px; margin-top: 0.5rem;
|
| 12 |
+
background: var(--border-color-primary); overflow: hidden; }
|
| 13 |
+
.simit-bar > div { height: 100%; border-radius: 4px; transition: width 0.4s linear;
|
| 14 |
+
background: linear-gradient(90deg, #f2a65a, #e07a3f); }
|
| 15 |
+
|
| 16 |
+
.simit-card { border-radius: 14px; padding: 0.8rem 1rem; min-height: 120px;
|
| 17 |
+
border: 1px solid var(--border-color-primary); background: var(--block-background-fill); }
|
| 18 |
+
.simit-card.base { border-top: 5px solid #c9c4bd; }
|
| 19 |
+
.simit-card.simit { border-top: 5px solid #e07a3f; }
|
| 20 |
+
.simit-card-title { font-size: 0.85rem; text-transform: uppercase; letter-spacing: 0.04em; opacity: 0.7;
|
| 21 |
+
display: flex; justify-content: space-between; }
|
| 22 |
+
.simit-answer { font-size: 1.35rem; font-weight: 650; margin: 0.4rem 0; white-space: pre-wrap; }
|
| 23 |
+
.simit-card-sub { font-size: 0.8rem; opacity: 0.7; }
|
| 24 |
+
.simit-wait { opacity: 0.4; }
|
| 25 |
+
.simit-ok { color: #2e9b5a; font-size: 1.1rem; }
|
| 26 |
+
.simit-no { color: #c94a3c; font-size: 1.1rem; }
|
| 27 |
+
|
| 28 |
+
@keyframes simit-bob { 0%, 100% { transform: translateY(0) rotate(-2deg); } 50% { transform: translateY(-8px) rotate(2deg); } }
|
| 29 |
+
@keyframes simit-pop { 0% { transform: scale(0.8); } 70% { transform: scale(1.06); } 100% { transform: scale(1); } }
|
| 30 |
+
@keyframes simit-dots { 0% { content: ""; } 25% { content: "."; } 50% { content: ".."; } 75% { content: "..."; } }
|
| 31 |
+
@media (prefers-reduced-motion: reduce) { .simit-mascot, .simit-dots::after { animation: none; } }
|
| 32 |
+
.simit-answer.long { font-size: 1.0rem; font-weight: 550; line-height: 1.4; }
|
assets/synthesizer.webp
ADDED
|
demo_models.py
ADDED
|
@@ -0,0 +1,251 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Models of the demo and how one request runs on ZeroGPU.
|
| 2 |
+
|
| 3 |
+
ZeroGPU forks the Space process for every ``@spaces.GPU`` call and kills the
|
| 4 |
+
fork afterwards. So the weights are loaded once, on CPU, in the main process
|
| 5 |
+
(no CUDA there), and each call moves only the chosen model to the GPU and
|
| 6 |
+
builds a fresh SIMIT engine around it.
|
| 7 |
+
"""
|
| 8 |
+
|
| 9 |
+
from __future__ import annotations
|
| 10 |
+
|
| 11 |
+
import contextlib
|
| 12 |
+
import json
|
| 13 |
+
import os
|
| 14 |
+
import queue
|
| 15 |
+
import threading
|
| 16 |
+
import time
|
| 17 |
+
from dataclasses import dataclass, field
|
| 18 |
+
from pathlib import Path
|
| 19 |
+
from typing import Optional
|
| 20 |
+
|
| 21 |
+
import torch
|
| 22 |
+
|
| 23 |
+
from simit import SIMIT, SIMITConfig
|
| 24 |
+
|
| 25 |
+
HERE = Path(__file__).parent
|
| 26 |
+
PRESETS = json.loads((HERE / "presets.json").read_text())
|
| 27 |
+
BUDGETS = [20, 40, 60, 90, 180] # seconds of GPU time per request
|
| 28 |
+
ANSWER_TOKENS = 128
|
| 29 |
+
ANSWER_RESERVE = 5.0 # seconds kept for the final SIMIT-ICL answer
|
| 30 |
+
|
| 31 |
+
|
| 32 |
+
@dataclass
|
| 33 |
+
class ModelSpec:
|
| 34 |
+
key: str
|
| 35 |
+
label: str
|
| 36 |
+
repo: str
|
| 37 |
+
kind: str # "bagel" | "lance" | "hf"
|
| 38 |
+
gpu_size: str # ZeroGPU size: "large" (48 GB) or "xlarge" (96 GB)
|
| 39 |
+
image_generator: Optional[str] = None
|
| 40 |
+
blurb: str = ""
|
| 41 |
+
_weights: object = None
|
| 42 |
+
_lock: threading.Lock = field(default_factory=threading.Lock)
|
| 43 |
+
|
| 44 |
+
# ----------------------------------------------------- main process (CPU)
|
| 45 |
+
def load(self):
|
| 46 |
+
"""Load the weights on CPU (main process; nothing touches CUDA)."""
|
| 47 |
+
with self._lock:
|
| 48 |
+
if self._weights is None:
|
| 49 |
+
t = time.time()
|
| 50 |
+
self._weights = _LOADERS[self.kind](self)
|
| 51 |
+
print(f"[demo] loaded {self.label} on CPU in {time.time() - t:.0f}s", flush=True)
|
| 52 |
+
return self._weights
|
| 53 |
+
|
| 54 |
+
# ------------------------------------------------- inside @spaces.GPU call
|
| 55 |
+
def build(self, preset: dict) -> SIMIT:
|
| 56 |
+
"""Move the weights to the GPU and wrap them in a SIMIT engine."""
|
| 57 |
+
w = self.load()
|
| 58 |
+
# Without ZeroGPU every request runs in this process: keep one model on the GPU at a time.
|
| 59 |
+
# (On ZeroGPU each request is a fresh fork in which nothing is on the GPU yet.)
|
| 60 |
+
if _ON_GPU[0] not in (None, self.key):
|
| 61 |
+
MODELS[_ON_GPU[0]]._to("cpu")
|
| 62 |
+
torch.cuda.empty_cache()
|
| 63 |
+
_ON_GPU[0] = self.key
|
| 64 |
+
cfg = SIMITConfig(**{k: v for k, v in preset.items() if k in SIMITConfig.__dataclass_fields__})
|
| 65 |
+
if self.kind == "bagel":
|
| 66 |
+
from simit.backends.bagel import BagelBackend
|
| 67 |
+
backend = BagelBackend(w, device="cuda", image_steps=preset.get("image_steps"), use_cuda_graphs=False)
|
| 68 |
+
return SIMIT(backend, config=cfg, model_id=self.repo)
|
| 69 |
+
if self.kind == "lance":
|
| 70 |
+
from simit.backends.lance import LanceBackend
|
| 71 |
+
backend = LanceBackend(w, device="cuda", image_steps=preset.get("image_steps"), use_cuda_graphs=False)
|
| 72 |
+
return SIMIT(backend, config=cfg, model_id=self.repo)
|
| 73 |
+
from simit.backends.hf import HFBackend
|
| 74 |
+
from simit.backends.imagegen import DiffusersImageGenerator
|
| 75 |
+
model, processor, pipe = w
|
| 76 |
+
model.to("cuda")
|
| 77 |
+
pipe.to("cuda")
|
| 78 |
+
backend = HFBackend(model, processor=processor, max_batch_size=8, max_batch_tokens=32768)
|
| 79 |
+
return SIMIT(backend, image_generator=DiffusersImageGenerator(pipe), config=cfg, model_id=self.repo)
|
| 80 |
+
|
| 81 |
+
def _to(self, device: str):
|
| 82 |
+
if self.kind in ("bagel", "lance"):
|
| 83 |
+
self._weights.to(device)
|
| 84 |
+
else:
|
| 85 |
+
model, _, pipe = self._weights
|
| 86 |
+
model.to(device)
|
| 87 |
+
pipe.to(device)
|
| 88 |
+
|
| 89 |
+
def preset(self, budget: int) -> dict:
|
| 90 |
+
return dict(PRESETS[self.key][str(budget)])
|
| 91 |
+
|
| 92 |
+
|
| 93 |
+
_ON_GPU = [None] # key of the model whose weights are on the GPU in this process
|
| 94 |
+
|
| 95 |
+
|
| 96 |
+
@contextlib.contextmanager
|
| 97 |
+
def _fp8_without_gpu():
|
| 98 |
+
"""transformers dequantizes FP8 checkpoints to bf16 when it sees no GPU at load
|
| 99 |
+
time. FP8 weights only need a GPU to *run*, so let the load keep them in FP8
|
| 100 |
+
on CPU; they are moved to the GPU inside each request."""
|
| 101 |
+
saved = torch.cuda.is_available, torch.cuda.get_device_capability
|
| 102 |
+
torch.cuda.is_available = lambda: True
|
| 103 |
+
torch.cuda.get_device_capability = lambda *a, **k: (9, 0)
|
| 104 |
+
try:
|
| 105 |
+
yield
|
| 106 |
+
finally:
|
| 107 |
+
torch.cuda.is_available, torch.cuda.get_device_capability = saved
|
| 108 |
+
|
| 109 |
+
|
| 110 |
+
def _load_bagel(spec):
|
| 111 |
+
from simit.backends.bagel import BagelModel
|
| 112 |
+
return BagelModel.load(spec.repo, device="cpu")
|
| 113 |
+
|
| 114 |
+
|
| 115 |
+
def _load_lance(spec):
|
| 116 |
+
from simit.backends.lance import LanceModel
|
| 117 |
+
return LanceModel.load(spec.repo, device="cpu")
|
| 118 |
+
|
| 119 |
+
|
| 120 |
+
def _load_hf(spec):
|
| 121 |
+
from diffusers import DiffusionPipeline
|
| 122 |
+
from transformers import AutoModelForMultimodalLM, AutoProcessor
|
| 123 |
+
|
| 124 |
+
from simit.backends.hf import _dense_fp8_fix
|
| 125 |
+
os.environ.setdefault("TRANSFORMERS_DISABLE_DEEPGEMM_LINEAR", "1") # Triton FP8 kernels: no nvcc JIT at runtime
|
| 126 |
+
processor = AutoProcessor.from_pretrained(spec.repo)
|
| 127 |
+
kw = dict(device_map="cpu", dtype=torch.bfloat16)
|
| 128 |
+
_dense_fp8_fix(spec.repo, kw) # transformers would skip FP8 conversion of gate_proj for this checkpoint
|
| 129 |
+
with _fp8_without_gpu():
|
| 130 |
+
model = AutoModelForMultimodalLM.from_pretrained(spec.repo, **kw)
|
| 131 |
+
model.eval()
|
| 132 |
+
pipe = DiffusionPipeline.from_pretrained(spec.image_generator, dtype=torch.bfloat16)
|
| 133 |
+
return model, processor, pipe
|
| 134 |
+
|
| 135 |
+
|
| 136 |
+
_LOADERS = {"bagel": _load_bagel, "lance": _load_lance, "hf": _load_hf}
|
| 137 |
+
|
| 138 |
+
MODELS = {m.key: m for m in [
|
| 139 |
+
ModelSpec("bagel", "BAGEL-7B-MoT", "ByteDance-Seed/BAGEL-7B-MoT", "bagel", "large",
|
| 140 |
+
blurb="Unified model: draws its own photos and writes code for charts and diagrams."),
|
| 141 |
+
ModelSpec("lance", "Lance", "bytedance-research/Lance", "lance", "large",
|
| 142 |
+
blurb="Small unified model: fast native image generation."),
|
| 143 |
+
ModelSpec("qwen", "Qwen3.8-27B (FP8) + FLUX.2-klein", "Qwen/Qwen3.8-27B-FP8", "hf", "xlarge",
|
| 144 |
+
image_generator="black-forest-labs/FLUX.2-klein-4B",
|
| 145 |
+
blurb="Standard VLM: FLUX.2-klein draws the photos; it writes code for charts and diagrams."),
|
| 146 |
+
]}
|
| 147 |
+
ENABLED = [k for k in os.environ.get("SIMIT_DEMO_MODELS", "bagel,lance,qwen").split(",") if k in MODELS]
|
| 148 |
+
|
| 149 |
+
|
| 150 |
+
# ---------------------------------------------------------------- a request
|
| 151 |
+
# Pipeline call tags (``on_progress``) -> what the loading screen shows.
|
| 152 |
+
STAGE_OF_TAG = {"synthesis": "synthesizing", "synthesis_fresh": "synthesizing", "revise": "synthesizing",
|
| 153 |
+
"route": "routing", "skill_spec": "coding", "render": "coding", "image": "drawing",
|
| 154 |
+
"critic": "checking", "critic_score": "checking"}
|
| 155 |
+
|
| 156 |
+
|
| 157 |
+
def run(spec: ModelSpec, image, question: str, budget: int, always: bool):
|
| 158 |
+
"""Generator of progress events for one request (runs inside the GPU call):
|
| 159 |
+
("stage", name, info) / ("greedy", ZeroShot, k) / ("demo", Demo) / ("final", answer, info)."""
|
| 160 |
+
t_start = time.time()
|
| 161 |
+
preset = spec.preset(budget)
|
| 162 |
+
yield ("stage", "warmup", {"elapsed": 0.0})
|
| 163 |
+
sim = spec.build(preset)
|
| 164 |
+
try:
|
| 165 |
+
yield ("stage", "solving", {"elapsed": time.time() - t_start})
|
| 166 |
+
zs = sim.zero_shot(image, question, max_new_tokens=ANSWER_TOKENS) # the greedy answer + p0 (ABA input)
|
| 167 |
+
k_budget = sim.config.budget(zs.confidence)
|
| 168 |
+
yield ("greedy", zs, k_budget)
|
| 169 |
+
|
| 170 |
+
# what is left of the budget, keeping room for the final answer (measured: ~1-1.5 s after the last demo)
|
| 171 |
+
time_left = budget - (time.time() - t_start) - ANSWER_RESERVE
|
| 172 |
+
if (k_budget == 0 and not always) or time_left < 3:
|
| 173 |
+
if time_left < 3:
|
| 174 |
+
reason = "no_time"
|
| 175 |
+
else:
|
| 176 |
+
reason = "confident" if preset.get("imagine_by_default", True) else "policy"
|
| 177 |
+
yield ("final", zs.answer, {"demos": 0, "skipped": reason, "elapsed": time.time() - t_start})
|
| 178 |
+
return
|
| 179 |
+
|
| 180 |
+
events: "queue.Queue" = queue.Queue()
|
| 181 |
+
current = ["synthesis"]
|
| 182 |
+
result = {}
|
| 183 |
+
|
| 184 |
+
def work():
|
| 185 |
+
try:
|
| 186 |
+
result["det"] = sim.imagine(image, question, k=preset["k_max"] if always else None,
|
| 187 |
+
return_details=True, max_new_tokens=ANSWER_TOKENS, time_limit=time_left,
|
| 188 |
+
on_demo=events.put, on_progress=lambda tag: current.__setitem__(0, tag))
|
| 189 |
+
except Exception as e: # surfaced below
|
| 190 |
+
result["error"] = e
|
| 191 |
+
th = threading.Thread(target=work, daemon=True)
|
| 192 |
+
th.start()
|
| 193 |
+
while th.is_alive() or not events.empty():
|
| 194 |
+
try:
|
| 195 |
+
yield ("demo", events.get(timeout=0.4))
|
| 196 |
+
continue
|
| 197 |
+
except queue.Empty:
|
| 198 |
+
pass
|
| 199 |
+
stage = STAGE_OF_TAG.get(current[0].split(":")[0], "synthesizing")
|
| 200 |
+
yield ("stage", stage, {"elapsed": time.time() - t_start, "limit": budget})
|
| 201 |
+
if "error" in result:
|
| 202 |
+
raise result["error"]
|
| 203 |
+
det = result["det"]
|
| 204 |
+
if not det.demos:
|
| 205 |
+
yield ("final", zs.answer, {"demos": 0, "skipped": "none_passed", "elapsed": time.time() - t_start,
|
| 206 |
+
"stats": det.stats})
|
| 207 |
+
return
|
| 208 |
+
yield ("stage", "improving", {"elapsed": time.time() - t_start})
|
| 209 |
+
answer = sim.answer(image, question, det.demos, max_new_tokens=ANSWER_TOKENS)
|
| 210 |
+
yield ("final", answer, {"demos": len(det.demos), "elapsed": time.time() - t_start, "stats": det.stats})
|
| 211 |
+
finally:
|
| 212 |
+
sim.close()
|
| 213 |
+
|
| 214 |
+
|
| 215 |
+
def code_snippet(spec: ModelSpec, budget: int, question: str, always: bool = False) -> str:
|
| 216 |
+
"""Copy-paste code reproducing this request with the library."""
|
| 217 |
+
p = spec.preset(budget)
|
| 218 |
+
force = always or not p.get("imagine_by_default", True)
|
| 219 |
+
keys = ["k_max", "epsilon", "A0", "A1", "B", "t_low", "t_high", "verify", "verify_think", "use_skills",
|
| 220 |
+
"attempts_per_slot", "verify_rounds"]
|
| 221 |
+
args = [f"{k}={round(p[k], 3) if isinstance(p[k], float) else p[k]!r}" for k in keys if k in p]
|
| 222 |
+
lines, line = [], ""
|
| 223 |
+
for a in args: # wrap the config arguments, a few per line
|
| 224 |
+
if line and len(line) + len(a) > 62:
|
| 225 |
+
lines.append(line.rstrip())
|
| 226 |
+
line = ""
|
| 227 |
+
line += a + ", "
|
| 228 |
+
lines.append(line.rstrip(", "))
|
| 229 |
+
cfg = "\n " + "\n ".join(lines) + ",\n "
|
| 230 |
+
gen = f',\n image_generator="{spec.image_generator}",' if spec.image_generator else ","
|
| 231 |
+
q = question.strip().replace('"""', "'''") or "What is shown in this image?"
|
| 232 |
+
steps = f"\n image_steps={p['image_steps']}," if p.get("image_steps") else ""
|
| 233 |
+
return f'''# pip install simit
|
| 234 |
+
from simit import SIMIT, SIMITConfig
|
| 235 |
+
|
| 236 |
+
model = SIMIT.from_pretrained(
|
| 237 |
+
"{spec.repo}"{gen}
|
| 238 |
+
config=SIMITConfig({cfg}),{steps}
|
| 239 |
+
)
|
| 240 |
+
|
| 241 |
+
image = "my_image.jpg"
|
| 242 |
+
question = """{q}"""
|
| 243 |
+
|
| 244 |
+
greedy = model.greedy(image, question) # base model, zero-shot
|
| 245 |
+
demos = model.imagine(image, question{f", k={p['k_max']}" if force else ""}, time_limit={max(5, budget - 15)}) # imagined examples
|
| 246 |
+
improved = model.answer(image, question, demos) # SIMIT-ICL
|
| 247 |
+
print(greedy, "->", improved)
|
| 248 |
+
|
| 249 |
+
from simit import save_demos
|
| 250 |
+
save_demos(demos, "imagined/") # PNGs + jsonl, e.g. to fine-tune on
|
| 251 |
+
'''
|
examples/bagel_amazon/demo_0.png
ADDED
|
Git LFS Details
|
examples/bagel_amazon/demo_1.png
ADDED
|
Git LFS Details
|
examples/bagel_amazon/query.png
ADDED
|
Git LFS Details
|
examples/bagel_infovqa_val_449/demo_0.png
ADDED
|
examples/bagel_infovqa_val_449/query.jpg
ADDED
|
Git LFS Details
|
examples/bagel_ok_vqa_val2014_492_always/demo_0.png
ADDED
|
Git LFS Details
|
examples/bagel_ok_vqa_val2014_492_always/demo_1.png
ADDED
|
Git LFS Details
|
examples/bagel_ok_vqa_val2014_492_always/query.jpg
ADDED
|
examples/bagel_vizwiz_vqa_val_345_always/demo_0.png
ADDED
|
Git LFS Details
|
examples/bagel_vizwiz_vqa_val_345_always/demo_1.png
ADDED
|
Git LFS Details
|
examples/bagel_vizwiz_vqa_val_345_always/query.jpg
ADDED
|
Git LFS Details
|
examples/bagel_vizwiz_vqa_val_47_always/demo_0.png
ADDED
|
Git LFS Details
|
examples/bagel_vizwiz_vqa_val_47_always/demo_1.png
ADDED
|
Git LFS Details
|
examples/bagel_vizwiz_vqa_val_47_always/query.jpg
ADDED
|
Git LFS Details
|
examples/bagel_vizwiz_vqa_val_87_always/demo_0.png
ADDED
|
Git LFS Details
|
examples/bagel_vizwiz_vqa_val_87_always/demo_1.png
ADDED
|
Git LFS Details
|
examples/bagel_vizwiz_vqa_val_87_always/query.jpg
ADDED
|
Git LFS Details
|
examples/index.json
ADDED
|
@@ -0,0 +1,274 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
[
|
| 2 |
+
{
|
| 3 |
+
"id": "bagel_ok_vqa_val2014_492_always",
|
| 4 |
+
"query_file": "query.jpg",
|
| 5 |
+
"model": "bagel",
|
| 6 |
+
"budget": 60,
|
| 7 |
+
"always": true,
|
| 8 |
+
"question": "In what country would you find this hat?\nWhen the provided information is insufficient, respond with 'Unanswerable'.\nAnswer the question using a single word or phrase.",
|
| 9 |
+
"reference": "vietnam",
|
| 10 |
+
"greedy": "Unanswerable",
|
| 11 |
+
"p0": 0.8883122759482616,
|
| 12 |
+
"simit": "Vietnam",
|
| 13 |
+
"greedy_ok": false,
|
| 14 |
+
"simit_ok": true,
|
| 15 |
+
"demos": [
|
| 16 |
+
{
|
| 17 |
+
"image": "demo_0.png",
|
| 18 |
+
"question": "How many wheels does the bicycle have?",
|
| 19 |
+
"answer": "Two",
|
| 20 |
+
"skill": "natural",
|
| 21 |
+
"verify_score": 100,
|
| 22 |
+
"confidence": null
|
| 23 |
+
},
|
| 24 |
+
{
|
| 25 |
+
"image": "demo_1.png",
|
| 26 |
+
"question": "What color is the hat the person is wearing?",
|
| 27 |
+
"answer": "Brown",
|
| 28 |
+
"skill": "natural",
|
| 29 |
+
"verify_score": 70,
|
| 30 |
+
"confidence": null
|
| 31 |
+
}
|
| 32 |
+
],
|
| 33 |
+
"elapsed": 16.20292091369629,
|
| 34 |
+
"source": "OK-VQA row 492, paper Fig. (qualitative samples)"
|
| 35 |
+
},
|
| 36 |
+
{
|
| 37 |
+
"id": "bagel_infovqa_val_449",
|
| 38 |
+
"query_file": "query.jpg",
|
| 39 |
+
"model": "bagel",
|
| 40 |
+
"budget": 60,
|
| 41 |
+
"always": false,
|
| 42 |
+
"question": "How many points are listed under when to wash your hands?\nAnswer the question using a single word or phrase.",
|
| 43 |
+
"reference": "6",
|
| 44 |
+
"greedy": "7",
|
| 45 |
+
"p0": 0.6461464658834921,
|
| 46 |
+
"simit": "6",
|
| 47 |
+
"greedy_ok": false,
|
| 48 |
+
"simit_ok": true,
|
| 49 |
+
"demos": [
|
| 50 |
+
{
|
| 51 |
+
"image": "demo_0.png",
|
| 52 |
+
"question": "How many steps are there in the handwashing process?",
|
| 53 |
+
"answer": "11",
|
| 54 |
+
"skill": "figure",
|
| 55 |
+
"verify_score": 70,
|
| 56 |
+
"confidence": 0.38085715656328817
|
| 57 |
+
}
|
| 58 |
+
],
|
| 59 |
+
"elapsed": 55.53464651107788,
|
| 60 |
+
"source": "InfographicVQA row 449, paper Fig. (qualitative samples)"
|
| 61 |
+
},
|
| 62 |
+
{
|
| 63 |
+
"id": "bagel_amazon",
|
| 64 |
+
"query_file": "query.png",
|
| 65 |
+
"model": "bagel",
|
| 66 |
+
"budget": 60,
|
| 67 |
+
"always": false,
|
| 68 |
+
"question": "Does this contain any trans fat?",
|
| 69 |
+
"reference": "No",
|
| 70 |
+
"greedy": "Yes, the bagel contains trans fat. According to the nutrition facts label on the packaging, the bagel has 0.5 grams of trans fat per serving.",
|
| 71 |
+
"p0": 0.7576221457584613,
|
| 72 |
+
"simit": "No",
|
| 73 |
+
"greedy_ok": false,
|
| 74 |
+
"simit_ok": true,
|
| 75 |
+
"demos": [
|
| 76 |
+
{
|
| 77 |
+
"image": "demo_0.png",
|
| 78 |
+
"question": "How many calories are in one serving of the Turkish Sesame Bagel?",
|
| 79 |
+
"answer": "290",
|
| 80 |
+
"skill": "natural",
|
| 81 |
+
"verify_score": 75,
|
| 82 |
+
"confidence": 0.3624771327791413
|
| 83 |
+
},
|
| 84 |
+
{
|
| 85 |
+
"image": "demo_1.png",
|
| 86 |
+
"question": "How many calories are listed for each serving of Turkish Sessame Bagel?",
|
| 87 |
+
"answer": "290",
|
| 88 |
+
"skill": "natural",
|
| 89 |
+
"verify_score": 100,
|
| 90 |
+
"confidence": 0.147100984163098
|
| 91 |
+
}
|
| 92 |
+
],
|
| 93 |
+
"elapsed": 27.87025022506714,
|
| 94 |
+
"source": "Amazon product page (screenshot)"
|
| 95 |
+
},
|
| 96 |
+
{
|
| 97 |
+
"id": "bagel_vizwiz_vqa_val_345_always",
|
| 98 |
+
"query_file": "query.jpg",
|
| 99 |
+
"model": "bagel",
|
| 100 |
+
"budget": 60,
|
| 101 |
+
"always": true,
|
| 102 |
+
"question": "What color is this cat?\nWhen the provided information is insufficient, respond with 'Unanswerable'.\nAnswer the question using a single word or phrase.",
|
| 103 |
+
"reference": "brown",
|
| 104 |
+
"greedy": "Unanswerable",
|
| 105 |
+
"p0": 0.6977114977650781,
|
| 106 |
+
"simit": "Brown",
|
| 107 |
+
"greedy_ok": false,
|
| 108 |
+
"simit_ok": true,
|
| 109 |
+
"demos": [
|
| 110 |
+
{
|
| 111 |
+
"image": "demo_0.png",
|
| 112 |
+
"question": "How many cats are in the image?",
|
| 113 |
+
"answer": "One",
|
| 114 |
+
"skill": "natural",
|
| 115 |
+
"verify_score": 70,
|
| 116 |
+
"confidence": null
|
| 117 |
+
},
|
| 118 |
+
{
|
| 119 |
+
"image": "demo_1.png",
|
| 120 |
+
"question": "What type of furniture is visible in the background?",
|
| 121 |
+
"answer": "A chair",
|
| 122 |
+
"skill": "natural",
|
| 123 |
+
"verify_score": 70,
|
| 124 |
+
"confidence": null
|
| 125 |
+
}
|
| 126 |
+
],
|
| 127 |
+
"elapsed": 10.399970293045044,
|
| 128 |
+
"source": "VizWiz row 345, paper teaser"
|
| 129 |
+
},
|
| 130 |
+
{
|
| 131 |
+
"id": "bagel_vizwiz_vqa_val_47_always",
|
| 132 |
+
"query_file": "query.jpg",
|
| 133 |
+
"model": "bagel",
|
| 134 |
+
"budget": 60,
|
| 135 |
+
"always": true,
|
| 136 |
+
"question": "What is this item?\nWhen the provided information is insufficient, respond with 'Unanswerable'.\nAnswer the question using a single word or phrase.",
|
| 137 |
+
"reference": "angel",
|
| 138 |
+
"greedy": "Unanswerable.",
|
| 139 |
+
"p0": 0.8626286828795229,
|
| 140 |
+
"simit": "Angel",
|
| 141 |
+
"greedy_ok": false,
|
| 142 |
+
"simit_ok": true,
|
| 143 |
+
"demos": [
|
| 144 |
+
{
|
| 145 |
+
"image": "demo_0.png",
|
| 146 |
+
"question": "How many cats are depicted in the image?",
|
| 147 |
+
"answer": "One",
|
| 148 |
+
"skill": "natural",
|
| 149 |
+
"verify_score": 80,
|
| 150 |
+
"confidence": null
|
| 151 |
+
},
|
| 152 |
+
{
|
| 153 |
+
"image": "demo_1.png",
|
| 154 |
+
"question": "How many cats are featured in the image?",
|
| 155 |
+
"answer": "One",
|
| 156 |
+
"skill": "natural",
|
| 157 |
+
"verify_score": 70,
|
| 158 |
+
"confidence": null
|
| 159 |
+
}
|
| 160 |
+
],
|
| 161 |
+
"elapsed": 15.055564641952515,
|
| 162 |
+
"source": "VizWiz row 47, paper appendix"
|
| 163 |
+
},
|
| 164 |
+
{
|
| 165 |
+
"id": "bagel_vizwiz_vqa_val_87_always",
|
| 166 |
+
"query_file": "query.jpg",
|
| 167 |
+
"model": "bagel",
|
| 168 |
+
"budget": 60,
|
| 169 |
+
"always": true,
|
| 170 |
+
"question": "What is this a picture of?\nWhen the provided information is insufficient, respond with 'Unanswerable'.\nAnswer the question using a single word or phrase.",
|
| 171 |
+
"reference": "dog in lace dress",
|
| 172 |
+
"greedy": "Unanswerable.",
|
| 173 |
+
"p0": 0.8959448319074929,
|
| 174 |
+
"simit": "Dog",
|
| 175 |
+
"greedy_ok": false,
|
| 176 |
+
"simit_ok": true,
|
| 177 |
+
"demos": [
|
| 178 |
+
{
|
| 179 |
+
"image": "demo_0.png",
|
| 180 |
+
"question": "How many legs does the dog have?",
|
| 181 |
+
"answer": "Four",
|
| 182 |
+
"skill": "natural",
|
| 183 |
+
"verify_score": 100,
|
| 184 |
+
"confidence": null
|
| 185 |
+
},
|
| 186 |
+
{
|
| 187 |
+
"image": "demo_1.png",
|
| 188 |
+
"question": "What color is the dog's belly?",
|
| 189 |
+
"answer": "Pink",
|
| 190 |
+
"skill": "natural",
|
| 191 |
+
"verify_score": 100,
|
| 192 |
+
"confidence": null
|
| 193 |
+
}
|
| 194 |
+
],
|
| 195 |
+
"elapsed": 9.160282850265503,
|
| 196 |
+
"source": "VizWiz row 87, paper appendix"
|
| 197 |
+
},
|
| 198 |
+
{
|
| 199 |
+
"id": "lance_ok_vqa_val2014_253",
|
| 200 |
+
"query_file": "query.jpg",
|
| 201 |
+
"model": "lance",
|
| 202 |
+
"budget": 60,
|
| 203 |
+
"always": false,
|
| 204 |
+
"question": "What time period is the action pictured here based off of?\nAnswer the question using a single word or phrase.",
|
| 205 |
+
"reference": "medieval",
|
| 206 |
+
"greedy": "Polo",
|
| 207 |
+
"p0": 0.6023501950034147,
|
| 208 |
+
"simit": "Medieval",
|
| 209 |
+
"greedy_ok": false,
|
| 210 |
+
"simit_ok": true,
|
| 211 |
+
"demos": [
|
| 212 |
+
{
|
| 213 |
+
"image": "demo_0.png",
|
| 214 |
+
"question": "What color does the umbrella look like?",
|
| 215 |
+
"answer": "Red",
|
| 216 |
+
"skill": "natural",
|
| 217 |
+
"verify_score": null,
|
| 218 |
+
"confidence": 0.36976529385784346
|
| 219 |
+
}
|
| 220 |
+
],
|
| 221 |
+
"elapsed": 16.72637391090393,
|
| 222 |
+
"source": "OK-VQA row 253, paper appendix"
|
| 223 |
+
},
|
| 224 |
+
{
|
| 225 |
+
"id": "lance_ok_vqa_val2014_8_always",
|
| 226 |
+
"query_file": "query.jpg",
|
| 227 |
+
"model": "lance",
|
| 228 |
+
"budget": 60,
|
| 229 |
+
"always": true,
|
| 230 |
+
"question": "What kind of house is next to the stop sign?\nAnswer the question using a single word or phrase.",
|
| 231 |
+
"reference": "apartment",
|
| 232 |
+
"greedy": "Single story",
|
| 233 |
+
"p0": 0.5895590491401231,
|
| 234 |
+
"simit": "Apartment",
|
| 235 |
+
"greedy_ok": false,
|
| 236 |
+
"simit_ok": true,
|
| 237 |
+
"demos": [
|
| 238 |
+
{
|
| 239 |
+
"image": "demo_0.png",
|
| 240 |
+
"question": "What are the colors of the umbrellas in the painting?",
|
| 241 |
+
"answer": "No",
|
| 242 |
+
"skill": "natural",
|
| 243 |
+
"verify_score": null,
|
| 244 |
+
"confidence": null
|
| 245 |
+
},
|
| 246 |
+
{
|
| 247 |
+
"image": "demo_1.png",
|
| 248 |
+
"question": "What color does the door have?",
|
| 249 |
+
"answer": "Brown",
|
| 250 |
+
"skill": "natural",
|
| 251 |
+
"verify_score": null,
|
| 252 |
+
"confidence": null
|
| 253 |
+
},
|
| 254 |
+
{
|
| 255 |
+
"image": "demo_2.png",
|
| 256 |
+
"question": "What color is the umbrella?",
|
| 257 |
+
"answer": "No",
|
| 258 |
+
"skill": "natural",
|
| 259 |
+
"verify_score": null,
|
| 260 |
+
"confidence": null
|
| 261 |
+
},
|
| 262 |
+
{
|
| 263 |
+
"image": "demo_3.png",
|
| 264 |
+
"question": "What is the color of the stop sign?",
|
| 265 |
+
"answer": "Red",
|
| 266 |
+
"skill": "natural",
|
| 267 |
+
"verify_score": null,
|
| 268 |
+
"confidence": null
|
| 269 |
+
}
|
| 270 |
+
],
|
| 271 |
+
"elapsed": 7.836957216262817,
|
| 272 |
+
"source": "OK-VQA row 8, paper appendix"
|
| 273 |
+
}
|
| 274 |
+
]
|
examples/lance_ok_vqa_val2014_253/demo_0.png
ADDED
|
Git LFS Details
|
examples/lance_ok_vqa_val2014_253/query.jpg
ADDED
|
Git LFS Details
|
examples/lance_ok_vqa_val2014_8_always/demo_0.png
ADDED
|
Git LFS Details
|
examples/lance_ok_vqa_val2014_8_always/demo_1.png
ADDED
|
Git LFS Details
|
examples/lance_ok_vqa_val2014_8_always/demo_2.png
ADDED
|
Git LFS Details
|
examples/lance_ok_vqa_val2014_8_always/demo_3.png
ADDED
|
Git LFS Details
|
examples/lance_ok_vqa_val2014_8_always/query.jpg
ADDED
|
packages.txt
ADDED
|
@@ -0,0 +1,18 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
libnss3
|
| 2 |
+
libnspr4
|
| 3 |
+
libatk1.0-0
|
| 4 |
+
libatk-bridge2.0-0
|
| 5 |
+
libcups2
|
| 6 |
+
libdrm2
|
| 7 |
+
libxkbcommon0
|
| 8 |
+
libxcomposite1
|
| 9 |
+
libxdamage1
|
| 10 |
+
libxfixes3
|
| 11 |
+
libxrandr2
|
| 12 |
+
libgbm1
|
| 13 |
+
libasound2
|
| 14 |
+
libpango-1.0-0
|
| 15 |
+
libcairo2
|
| 16 |
+
libxrender1
|
| 17 |
+
libxext6
|
| 18 |
+
fonts-dejavu-core
|
presets.json
ADDED
|
@@ -0,0 +1,233 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_note": "built by tools/build_presets.py from tune_presets.py + latency.py runs",
|
| 3 |
+
"bagel": {
|
| 4 |
+
"20": {
|
| 5 |
+
"epsilon": 1.0,
|
| 6 |
+
"A0": 0.5631057319035666,
|
| 7 |
+
"A1": 0.6871778307441961,
|
| 8 |
+
"B": 0.3818684686169197,
|
| 9 |
+
"t_low": 0.01568469261109281,
|
| 10 |
+
"t_high": 0.9941888593223634,
|
| 11 |
+
"k_max": 1,
|
| 12 |
+
"attempts_per_slot": 2,
|
| 13 |
+
"repair_retries": 0,
|
| 14 |
+
"verify_rounds": 1,
|
| 15 |
+
"imagine_by_default": false,
|
| 16 |
+
"verify": false,
|
| 17 |
+
"image_steps": 25
|
| 18 |
+
},
|
| 19 |
+
"40": {
|
| 20 |
+
"epsilon": 1.0,
|
| 21 |
+
"A0": 0.5631057319035666,
|
| 22 |
+
"A1": 0.6871778307441961,
|
| 23 |
+
"B": 0.3818684686169197,
|
| 24 |
+
"t_low": 0.01568469261109281,
|
| 25 |
+
"t_high": 0.9941888593223634,
|
| 26 |
+
"k_max": 1,
|
| 27 |
+
"attempts_per_slot": 2,
|
| 28 |
+
"repair_retries": 1,
|
| 29 |
+
"verify_rounds": 1,
|
| 30 |
+
"imagine_by_default": false,
|
| 31 |
+
"verify": true,
|
| 32 |
+
"verify_think": true,
|
| 33 |
+
"image_steps": 50
|
| 34 |
+
},
|
| 35 |
+
"60": {
|
| 36 |
+
"epsilon": 0.13150801401314569,
|
| 37 |
+
"A0": 0.5697034469957714,
|
| 38 |
+
"A1": 0.6928285171478893,
|
| 39 |
+
"B": 0.4731007761788684,
|
| 40 |
+
"t_low": 0.012450289537920928,
|
| 41 |
+
"t_high": 0.702696037286609,
|
| 42 |
+
"k_max": 2,
|
| 43 |
+
"attempts_per_slot": 3,
|
| 44 |
+
"repair_retries": 1,
|
| 45 |
+
"verify_rounds": 2,
|
| 46 |
+
"imagine_by_default": true,
|
| 47 |
+
"verify": true,
|
| 48 |
+
"verify_think": true,
|
| 49 |
+
"image_steps": 50
|
| 50 |
+
},
|
| 51 |
+
"90": {
|
| 52 |
+
"epsilon": 0.13150801401314569,
|
| 53 |
+
"A0": 0.5697034469957714,
|
| 54 |
+
"A1": 0.6928285171478893,
|
| 55 |
+
"B": 0.4731007761788684,
|
| 56 |
+
"t_low": 0.012450289537920928,
|
| 57 |
+
"t_high": 0.702696037286609,
|
| 58 |
+
"k_max": 2,
|
| 59 |
+
"attempts_per_slot": 4,
|
| 60 |
+
"repair_retries": 2,
|
| 61 |
+
"verify_rounds": 3,
|
| 62 |
+
"imagine_by_default": true,
|
| 63 |
+
"verify": true,
|
| 64 |
+
"verify_think": true,
|
| 65 |
+
"image_steps": 50
|
| 66 |
+
},
|
| 67 |
+
"180": {
|
| 68 |
+
"epsilon": 0.13150801401314569,
|
| 69 |
+
"A0": 0.5697034469957714,
|
| 70 |
+
"A1": 0.6928285171478893,
|
| 71 |
+
"B": 0.4731007761788684,
|
| 72 |
+
"t_low": 0.012450289537920928,
|
| 73 |
+
"t_high": 0.702696037286609,
|
| 74 |
+
"k_max": 2,
|
| 75 |
+
"attempts_per_slot": 4,
|
| 76 |
+
"repair_retries": 2,
|
| 77 |
+
"verify_rounds": 5,
|
| 78 |
+
"imagine_by_default": true,
|
| 79 |
+
"verify": true,
|
| 80 |
+
"verify_think": true,
|
| 81 |
+
"image_steps": 50
|
| 82 |
+
}
|
| 83 |
+
},
|
| 84 |
+
"lance": {
|
| 85 |
+
"20": {
|
| 86 |
+
"epsilon": 0.10540046109011536,
|
| 87 |
+
"A0": 0.5284418351778891,
|
| 88 |
+
"A1": 0.7185259059153423,
|
| 89 |
+
"B": 0.20052864010313778,
|
| 90 |
+
"t_low": 0.24157862499461874,
|
| 91 |
+
"t_high": 0.8741991398657891,
|
| 92 |
+
"k_max": 1,
|
| 93 |
+
"attempts_per_slot": 2,
|
| 94 |
+
"repair_retries": 0,
|
| 95 |
+
"verify_rounds": 1,
|
| 96 |
+
"imagine_by_default": true,
|
| 97 |
+
"image_steps": 30
|
| 98 |
+
},
|
| 99 |
+
"40": {
|
| 100 |
+
"epsilon": 0.11989932190804062,
|
| 101 |
+
"A0": 0.35382385806255806,
|
| 102 |
+
"A1": 0.4638199725338076,
|
| 103 |
+
"B": 0.2584592356693373,
|
| 104 |
+
"t_low": 0.25642426481228886,
|
| 105 |
+
"t_high": 0.9450464322610527,
|
| 106 |
+
"k_max": 4,
|
| 107 |
+
"attempts_per_slot": 2,
|
| 108 |
+
"repair_retries": 1,
|
| 109 |
+
"verify_rounds": 1,
|
| 110 |
+
"imagine_by_default": true,
|
| 111 |
+
"image_steps": 30
|
| 112 |
+
},
|
| 113 |
+
"60": {
|
| 114 |
+
"epsilon": 0.11989932190804062,
|
| 115 |
+
"A0": 0.35382385806255806,
|
| 116 |
+
"A1": 0.4638199725338076,
|
| 117 |
+
"B": 0.2584592356693373,
|
| 118 |
+
"t_low": 0.25642426481228886,
|
| 119 |
+
"t_high": 0.9450464322610527,
|
| 120 |
+
"k_max": 4,
|
| 121 |
+
"attempts_per_slot": 3,
|
| 122 |
+
"repair_retries": 1,
|
| 123 |
+
"verify_rounds": 2,
|
| 124 |
+
"imagine_by_default": true,
|
| 125 |
+
"image_steps": 30
|
| 126 |
+
},
|
| 127 |
+
"90": {
|
| 128 |
+
"epsilon": 0.11989932190804062,
|
| 129 |
+
"A0": 0.35382385806255806,
|
| 130 |
+
"A1": 0.4638199725338076,
|
| 131 |
+
"B": 0.2584592356693373,
|
| 132 |
+
"t_low": 0.25642426481228886,
|
| 133 |
+
"t_high": 0.9450464322610527,
|
| 134 |
+
"k_max": 4,
|
| 135 |
+
"attempts_per_slot": 4,
|
| 136 |
+
"repair_retries": 2,
|
| 137 |
+
"verify_rounds": 3,
|
| 138 |
+
"imagine_by_default": true,
|
| 139 |
+
"image_steps": 30
|
| 140 |
+
},
|
| 141 |
+
"180": {
|
| 142 |
+
"epsilon": 0.11989932190804062,
|
| 143 |
+
"A0": 0.35382385806255806,
|
| 144 |
+
"A1": 0.4638199725338076,
|
| 145 |
+
"B": 0.2584592356693373,
|
| 146 |
+
"t_low": 0.25642426481228886,
|
| 147 |
+
"t_high": 0.9450464322610527,
|
| 148 |
+
"k_max": 4,
|
| 149 |
+
"attempts_per_slot": 4,
|
| 150 |
+
"repair_retries": 2,
|
| 151 |
+
"verify_rounds": 5,
|
| 152 |
+
"imagine_by_default": true,
|
| 153 |
+
"image_steps": 30
|
| 154 |
+
}
|
| 155 |
+
},
|
| 156 |
+
"qwen": {
|
| 157 |
+
"20": {
|
| 158 |
+
"epsilon": 1.0,
|
| 159 |
+
"A0": 0.29123083240427616,
|
| 160 |
+
"A1": 0.892922351072982,
|
| 161 |
+
"B": 0.051022405374014035,
|
| 162 |
+
"t_low": 0.10443837804741735,
|
| 163 |
+
"t_high": 0.24890099152089373,
|
| 164 |
+
"k_max": 1,
|
| 165 |
+
"attempts_per_slot": 2,
|
| 166 |
+
"repair_retries": 0,
|
| 167 |
+
"verify_rounds": 1,
|
| 168 |
+
"imagine_by_default": false,
|
| 169 |
+
"verify": false,
|
| 170 |
+
"use_skills": false
|
| 171 |
+
},
|
| 172 |
+
"40": {
|
| 173 |
+
"epsilon": 1.0,
|
| 174 |
+
"A0": 0.29123083240427616,
|
| 175 |
+
"A1": 0.892922351072982,
|
| 176 |
+
"B": 0.051022405374014035,
|
| 177 |
+
"t_low": 0.10443837804741735,
|
| 178 |
+
"t_high": 0.24890099152089373,
|
| 179 |
+
"k_max": 1,
|
| 180 |
+
"attempts_per_slot": 2,
|
| 181 |
+
"repair_retries": 1,
|
| 182 |
+
"verify_rounds": 1,
|
| 183 |
+
"imagine_by_default": false,
|
| 184 |
+
"verify": false,
|
| 185 |
+
"use_skills": false
|
| 186 |
+
},
|
| 187 |
+
"60": {
|
| 188 |
+
"epsilon": 1.0,
|
| 189 |
+
"A0": 0.4895372103191979,
|
| 190 |
+
"A1": 0.6789573167296538,
|
| 191 |
+
"B": 0.39026458814322773,
|
| 192 |
+
"t_low": 0.05913721293446661,
|
| 193 |
+
"t_high": 0.6612150885625033,
|
| 194 |
+
"k_max": 2,
|
| 195 |
+
"attempts_per_slot": 3,
|
| 196 |
+
"repair_retries": 1,
|
| 197 |
+
"verify_rounds": 2,
|
| 198 |
+
"imagine_by_default": false,
|
| 199 |
+
"verify": false,
|
| 200 |
+
"use_skills": true
|
| 201 |
+
},
|
| 202 |
+
"90": {
|
| 203 |
+
"epsilon": 1.0,
|
| 204 |
+
"A0": 0.4895372103191979,
|
| 205 |
+
"A1": 0.6789573167296538,
|
| 206 |
+
"B": 0.39026458814322773,
|
| 207 |
+
"t_low": 0.05913721293446661,
|
| 208 |
+
"t_high": 0.6612150885625033,
|
| 209 |
+
"k_max": 2,
|
| 210 |
+
"attempts_per_slot": 4,
|
| 211 |
+
"repair_retries": 2,
|
| 212 |
+
"verify_rounds": 3,
|
| 213 |
+
"imagine_by_default": false,
|
| 214 |
+
"verify": false,
|
| 215 |
+
"use_skills": true
|
| 216 |
+
},
|
| 217 |
+
"180": {
|
| 218 |
+
"epsilon": 1.0,
|
| 219 |
+
"A0": 0.29123083240427616,
|
| 220 |
+
"A1": 0.892922351072982,
|
| 221 |
+
"B": 0.051022405374014035,
|
| 222 |
+
"t_low": 0.10443837804741735,
|
| 223 |
+
"t_high": 0.24890099152089373,
|
| 224 |
+
"k_max": 1,
|
| 225 |
+
"attempts_per_slot": 4,
|
| 226 |
+
"repair_retries": 2,
|
| 227 |
+
"verify_rounds": 5,
|
| 228 |
+
"imagine_by_default": false,
|
| 229 |
+
"verify": true,
|
| 230 |
+
"use_skills": true
|
| 231 |
+
}
|
| 232 |
+
}
|
| 233 |
+
}
|
requirements.txt
ADDED
|
@@ -0,0 +1,4 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# gradio, spaces and huggingface_hub are preinstalled and managed by the Spaces platform: don't list them.
|
| 2 |
+
simit>=0.1.1
|
| 3 |
+
# simit needs torch >= 2.12 (torch.nn.attention.varlen); 2.13.0 is on ZeroGPU's supported list
|
| 4 |
+
torch==2.13.0
|
tools/build_presets.py
ADDED
|
@@ -0,0 +1,91 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Combine accuracy (tune_presets.py) and latency (latency.py) into presets.json:
|
| 2 |
+
for every model and time budget, the largest K_max whose demos arrive in time
|
| 3 |
+
for most requests, at the best-quality speed setting that achieves it, with the
|
| 4 |
+
ABA/DF thresholds tuned for that K_max.
|
| 5 |
+
|
| 6 |
+
python tools/build_presets.py RUN_DIR
|
| 7 |
+
"""
|
| 8 |
+
import json
|
| 9 |
+
import sys
|
| 10 |
+
from pathlib import Path
|
| 11 |
+
|
| 12 |
+
HERE = Path(__file__).parent.parent
|
| 13 |
+
sys.path.insert(0, str(HERE))
|
| 14 |
+
from demo_models import ANSWER_RESERVE, BUDGETS, MODELS # noqa: E402
|
| 15 |
+
|
| 16 |
+
run = Path(sys.argv[1])
|
| 17 |
+
# ZeroGPU vs. our H100 (spaces' own duration_factor: 1.5 on the half GPU, none on the full one)
|
| 18 |
+
SLOWDOWN = {"large": 1.5, "xlarge": 1.0}
|
| 19 |
+
QUALITY = { # speed settings, best quality first (as timed by latency.py)
|
| 20 |
+
"bagel": ["think50", "nothink50", "nothink25", "nocritic25"],
|
| 21 |
+
"lance": ["steps30", "steps20"],
|
| 22 |
+
"qwen": ["critic", "skills", "natural"],
|
| 23 |
+
}
|
| 24 |
+
SETTING = {
|
| 25 |
+
"think50": dict(verify=True, verify_think=True, image_steps=50),
|
| 26 |
+
"nothink50": dict(verify=True, verify_think=False, image_steps=50),
|
| 27 |
+
"nothink25": dict(verify=True, verify_think=False, image_steps=25),
|
| 28 |
+
"nocritic25": dict(verify=False, image_steps=25),
|
| 29 |
+
"steps30": dict(image_steps=30), "steps20": dict(image_steps=20),
|
| 30 |
+
"critic": dict(verify=True, use_skills=True), "skills": dict(verify=False, use_skills=True),
|
| 31 |
+
"natural": dict(verify=False, use_skills=False),
|
| 32 |
+
}
|
| 33 |
+
SHARE = 0.8 # fraction of requests that must get all K demos in time
|
| 34 |
+
|
| 35 |
+
out_file = HERE / "presets.json"
|
| 36 |
+
presets = json.loads(out_file.read_text()) if out_file.exists() else {} # models without data keep theirs
|
| 37 |
+
presets["_note"] = "built by tools/build_presets.py from tune_presets.py + latency.py runs"
|
| 38 |
+
report = []
|
| 39 |
+
for key in MODELS:
|
| 40 |
+
tune_file, lat_file = run / f"presets_{key}.json", run / f"latency_{key}.json"
|
| 41 |
+
if not (tune_file.exists() and lat_file.exists()):
|
| 42 |
+
print(f"skip {key}: missing {tune_file.name if not tune_file.exists() else lat_file.name}")
|
| 43 |
+
continue
|
| 44 |
+
tuning = json.loads(tune_file.read_text())
|
| 45 |
+
tuned, zero_shot = tuning["robust_per_k_max"], tuning["zero_shot"]
|
| 46 |
+
timelines = json.loads(lat_file.read_text())
|
| 47 |
+
f = SLOWDOWN[MODELS[key].gpu_size]
|
| 48 |
+
scores = {int(k): v["score"] for k, v in tuned.items()}
|
| 49 |
+
presets[key] = {}
|
| 50 |
+
n_val = tuning["n_val"]
|
| 51 |
+
band = {int(k): v["params"]["t_high"] - v["params"]["t_low"] for k, v in tuned.items()}
|
| 52 |
+
|
| 53 |
+
def fits(setting, k, deadline):
|
| 54 |
+
"""Most requests have k demos before the deadline. Pooled over every run that asked for at least
|
| 55 |
+
k demos (when its k-th demo arrived): 6 requests per cell alone are too noisy."""
|
| 56 |
+
runs = [r for r in timelines if r["setting"] == setting and r["k"] >= k]
|
| 57 |
+
if not runs:
|
| 58 |
+
return False
|
| 59 |
+
return sum(len(r.get("demos", [])) >= k and r["demos"][k - 1] <= deadline for r in runs) >= SHARE * len(runs)
|
| 60 |
+
|
| 61 |
+
for budget in BUDGETS:
|
| 62 |
+
deadline = (budget - ANSWER_RESERVE) / f # the demo's imagination deadline, in H100 seconds
|
| 63 |
+
choice = None
|
| 64 |
+
# The tuned accuracies hold for the setting the tuning cache was made with (the first, best-quality
|
| 65 |
+
# one), so use it whenever any K fits; faster settings only when it cannot fit at all.
|
| 66 |
+
for setting in QUALITY[key]:
|
| 67 |
+
ks = [k for k in sorted(scores) if fits(setting, k, deadline)]
|
| 68 |
+
if not ks:
|
| 69 |
+
continue
|
| 70 |
+
top = max(scores[k] for k in ks)
|
| 71 |
+
near = [k for k in ks if scores[k] >= top - 1.0 / n_val] # within one validation query
|
| 72 |
+
k = min(near, key=lambda k: (-band[k], k)) # widest band, then fewest demos
|
| 73 |
+
choice = (k, setting)
|
| 74 |
+
break
|
| 75 |
+
if choice is None: # nothing fits: the fastest setting, one demo; the deadline decides
|
| 76 |
+
choice = (1, QUALITY[key][-1])
|
| 77 |
+
k, setting = choice
|
| 78 |
+
# more time: more attempts and checking rounds (the paper's 4 / 2 / 5 at the largest budget)
|
| 79 |
+
effort = {20: (2, 0, 1), 40: (2, 1, 1), 60: (3, 1, 2), 90: (4, 2, 3), 180: (4, 2, 5)}[budget]
|
| 80 |
+
params = {p: v for p, v in tuned[str(k)]["params"].items() if p != "k_max"}
|
| 81 |
+
helps = scores[k] > zero_shot + 0.5 / n_val
|
| 82 |
+
if not helps: # imagination does not beat zero-shot here: ABA never imagines (the UI's checkbox still can)
|
| 83 |
+
params["epsilon"] = 1.0
|
| 84 |
+
presets[key][str(budget)] = dict(params, k_max=k, attempts_per_slot=effort[0], repair_retries=effort[1],
|
| 85 |
+
verify_rounds=effort[2], imagine_by_default=helps, **SETTING[setting])
|
| 86 |
+
report.append((key, budget, k, setting, round(scores[k], 3), round(zero_shot, 3), round(band[k], 2),
|
| 87 |
+
"" if helps else " -> no gain: zero-shot by default"))
|
| 88 |
+
|
| 89 |
+
out_file.write_text(json.dumps(presets, indent=2))
|
| 90 |
+
for row in report:
|
| 91 |
+
print("%-6s %4ss K_max=%d %-11s val %.3f (zero-shot %.3f) DF band %.2f%s" % row)
|
tools/fork_test.py
ADDED
|
@@ -0,0 +1,55 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Simulate ZeroGPU locally: load weights on CPU in the parent (CUDA must stay
|
| 2 |
+
uninitialized there), then run whole requests in forked children, which is
|
| 3 |
+
what ``@spaces.GPU`` does on ZeroGPU.
|
| 4 |
+
|
| 5 |
+
CUDA_VISIBLE_DEVICES=0 python tools/fork_test.py bagel [n_requests]
|
| 6 |
+
"""
|
| 7 |
+
import multiprocessing as mp
|
| 8 |
+
import sys
|
| 9 |
+
import time
|
| 10 |
+
from pathlib import Path
|
| 11 |
+
|
| 12 |
+
sys.path.insert(0, str(Path(__file__).parent.parent))
|
| 13 |
+
sys.path.insert(0, str(Path(__file__).parent))
|
| 14 |
+
|
| 15 |
+
# ZeroGPU's own main-process emulation (fake CUDA queries, CUDA init forbidden); children undo it,
|
| 16 |
+
# as spaces' worker does before attaching the GPU.
|
| 17 |
+
from spaces.zero.torch import patching # noqa: E402
|
| 18 |
+
import torch # noqa: E402
|
| 19 |
+
|
| 20 |
+
patching.patch()
|
| 21 |
+
from demo_models import MODELS, run # noqa: E402
|
| 22 |
+
from general_qa import general_qa # noqa: E402
|
| 23 |
+
|
| 24 |
+
key = sys.argv[1]
|
| 25 |
+
n = int(sys.argv[2]) if len(sys.argv) > 2 else 2
|
| 26 |
+
spec = MODELS[key]
|
| 27 |
+
t = time.time()
|
| 28 |
+
spec.load()
|
| 29 |
+
print(f"parent: loaded on CPU in {time.time() - t:.0f}s; CUDA initialized in parent: {torch.cuda.is_initialized()}",
|
| 30 |
+
flush=True)
|
| 31 |
+
assert not torch.cuda.is_initialized(), "the main process must not initialize CUDA (ZeroGPU forks it)"
|
| 32 |
+
queries = general_qa(per_subset=1, offset=420)[:n]
|
| 33 |
+
|
| 34 |
+
|
| 35 |
+
def child(i, q):
|
| 36 |
+
patching.unpatch()
|
| 37 |
+
t0 = time.time()
|
| 38 |
+
for ev in run(spec, q["image"], q["question"], 60, False):
|
| 39 |
+
if ev[0] == "greedy":
|
| 40 |
+
print(f" [{i}] greedy={ev[1].answer!r} p0={ev[1].confidence:.2f} k={ev[2]} at {time.time() - t0:.1f}s",
|
| 41 |
+
flush=True)
|
| 42 |
+
elif ev[0] == "demo":
|
| 43 |
+
print(f" [{i}] demo [{ev[1].skill}] {ev[1].question!r} -> {ev[1].answer!r} at {time.time() - t0:.1f}s",
|
| 44 |
+
flush=True)
|
| 45 |
+
elif ev[0] == "final":
|
| 46 |
+
print(f" [{i}] final={ev[1]!r} {ev[2]} refs={q['answers'][:3]} total {time.time() - t0:.1f}s", flush=True)
|
| 47 |
+
|
| 48 |
+
|
| 49 |
+
ctx = mp.get_context("fork")
|
| 50 |
+
for i, q in enumerate(queries):
|
| 51 |
+
p = ctx.Process(target=child, args=(i, q))
|
| 52 |
+
t0 = time.time()
|
| 53 |
+
p.start()
|
| 54 |
+
p.join()
|
| 55 |
+
print(f"request {i} ({q['subset']}): exit={p.exitcode} in {time.time() - t0:.1f}s", flush=True)
|
tools/general_qa.py
ADDED
|
@@ -0,0 +1,70 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""A mixed "general visual QA" set from LMMs-Eval-Lite, close to what demo
|
| 2 |
+
visitors ask: everyday photos (VizWiz, VQAv2), knowledge (OK-VQA), reading text
|
| 3 |
+
(TextVQA), charts and documents (ChartQA, DocVQA) and multiple-choice reasoning
|
| 4 |
+
(MMBench, AI2D). Each example carries its benchmark's own metric.
|
| 5 |
+
"""
|
| 6 |
+
|
| 7 |
+
import ast
|
| 8 |
+
|
| 9 |
+
from datasets import load_dataset
|
| 10 |
+
|
| 11 |
+
VQA = "Answer the question using a single word or phrase."
|
| 12 |
+
|
| 13 |
+
|
| 14 |
+
def _list(v):
|
| 15 |
+
if isinstance(v, list):
|
| 16 |
+
return v
|
| 17 |
+
try:
|
| 18 |
+
v = ast.literal_eval(v)
|
| 19 |
+
except Exception:
|
| 20 |
+
pass
|
| 21 |
+
return v if isinstance(v, list) else [v]
|
| 22 |
+
|
| 23 |
+
|
| 24 |
+
def _options(row, keys):
|
| 25 |
+
opts = [(k, str(row[k])) for k in keys if str(row.get(k, "nan")) not in ("nan", "None", "")]
|
| 26 |
+
return "\n".join(f"{k}. {v}" for k, v in opts)
|
| 27 |
+
|
| 28 |
+
|
| 29 |
+
def _mmbench(r):
|
| 30 |
+
hint = str(r.get("hint", "nan"))
|
| 31 |
+
q = (hint + "\n" if hint not in ("nan", "None", "") else "") + r["question"]
|
| 32 |
+
return (q + "\n" + _options(r, "ABCD") + "\nAnswer with the option's letter from the given choices directly.",
|
| 33 |
+
[str(r["answer"])], "multiple_choice")
|
| 34 |
+
|
| 35 |
+
|
| 36 |
+
def _ai2d(r):
|
| 37 |
+
opts = _list(r["options"])
|
| 38 |
+
q = r["question"] + "\n" + "\n".join(f"{chr(65 + i)}. {o}" for i, o in enumerate(opts))
|
| 39 |
+
return q + "\nAnswer with the option's letter from the given choices directly.", [chr(65 + int(r["answer"]))], \
|
| 40 |
+
"multiple_choice"
|
| 41 |
+
|
| 42 |
+
|
| 43 |
+
SUBSETS = {
|
| 44 |
+
"vizwiz_vqa_val": lambda r: (f"{r['question']}\nWhen the provided information is insufficient, respond with "
|
| 45 |
+
f"'Unanswerable'.\n{VQA}", [str(a) for a in _list(r["answers"])], "vqa_accuracy"),
|
| 46 |
+
"vqav2_val": lambda r: (f"{r['question']}\n{VQA}",
|
| 47 |
+
[a["answer"] if isinstance(a, dict) else str(a) for a in _list(r["answers"])],
|
| 48 |
+
"vqa_accuracy"),
|
| 49 |
+
"ok_vqa_val2014": lambda r: (f"{r['question']}\n{VQA}", [str(a) for a in _list(r["answers"])], "vqa_accuracy"),
|
| 50 |
+
"textvqa_val": lambda r: (f"{r['question']}\n{VQA}", [str(a) for a in _list(r["answers"])], "vqa_accuracy"),
|
| 51 |
+
"chartqa": lambda r: (f"{r['question']}\nAnswer the question with a single word.",
|
| 52 |
+
[str(a) for a in _list(r["answer"])], "relaxed_accuracy"),
|
| 53 |
+
"docvqa_val": lambda r: (f"{r['question']}\n{VQA}", [str(a) for a in _list(r["answers"])], "anls"),
|
| 54 |
+
"mmbench_en_dev": _mmbench,
|
| 55 |
+
"ai2d": _ai2d,
|
| 56 |
+
}
|
| 57 |
+
|
| 58 |
+
|
| 59 |
+
def general_qa(per_subset: int = 8, offset: int = 300) -> list[dict]:
|
| 60 |
+
"""``per_subset`` examples from each subset, starting at row ``offset``
|
| 61 |
+
(rows below 300 were used for the package benchmarks)."""
|
| 62 |
+
out = []
|
| 63 |
+
for sub, fn in SUBSETS.items():
|
| 64 |
+
ds = load_dataset("lmms-lab/LMMs-Eval-Lite", sub)["lite"]
|
| 65 |
+
for i in range(offset, offset + per_subset):
|
| 66 |
+
r = ds[i]
|
| 67 |
+
question, answers, metric = fn(r)
|
| 68 |
+
out.append({"image": r["image"].convert("RGB"), "question": question, "answers": answers,
|
| 69 |
+
"metric": metric, "subset": sub})
|
| 70 |
+
return out
|
tools/latency.py
ADDED
|
@@ -0,0 +1,74 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Per-request timelines on the demo's own code path, as ZeroGPU runs it (a fresh
|
| 2 |
+
forked process per request, weights moved to the GPU each time): when the greedy
|
| 3 |
+
answer and each imagined demo arrive, for forced K and a given speed setting.
|
| 4 |
+
|
| 5 |
+
CUDA_VISIBLE_DEVICES=1 python tools/latency.py lance OUT.json
|
| 6 |
+
"""
|
| 7 |
+
import json
|
| 8 |
+
import multiprocessing as mp
|
| 9 |
+
import os
|
| 10 |
+
import sys
|
| 11 |
+
import time
|
| 12 |
+
from pathlib import Path
|
| 13 |
+
|
| 14 |
+
os.environ.setdefault("TRITON_CACHE_AUTOTUNING", "1")
|
| 15 |
+
os.environ.setdefault("TRANSFORMERS_DISABLE_DEEPGEMM_LINEAR", "1")
|
| 16 |
+
sys.path.insert(0, str(Path(__file__).parent.parent))
|
| 17 |
+
sys.path.insert(0, str(Path(__file__).parent))
|
| 18 |
+
from spaces.zero.torch import patching # noqa: E402
|
| 19 |
+
import torch # noqa: E402
|
| 20 |
+
|
| 21 |
+
patching.patch()
|
| 22 |
+
from demo_models import MODELS, run # noqa: E402
|
| 23 |
+
from general_qa import general_qa # noqa: E402
|
| 24 |
+
|
| 25 |
+
ABA = dict(epsilon=0.06, A0=0.5, A1=0.85, B=0.4, t_low=0.0, t_high=1.0)
|
| 26 |
+
SETTINGS = { # speed settings to time (accuracy comes from tune_presets.py)
|
| 27 |
+
"bagel": {"think50": dict(verify=True, verify_think=True, image_steps=50),
|
| 28 |
+
"nothink50": dict(verify=True, verify_think=False, image_steps=50),
|
| 29 |
+
"nothink25": dict(verify=True, verify_think=False, image_steps=25),
|
| 30 |
+
"nocritic25": dict(verify=False, image_steps=25)},
|
| 31 |
+
"lance": {"steps30": dict(image_steps=30), "steps20": dict(image_steps=20)},
|
| 32 |
+
"qwen": {"natural": dict(verify=False, use_skills=False), "skills": dict(verify=False, use_skills=True),
|
| 33 |
+
"critic": dict(verify=True, use_skills=True)},
|
| 34 |
+
}
|
| 35 |
+
KS = {"bagel": [1, 2, 4], "lance": [1, 2, 4], "qwen": [1, 2]}
|
| 36 |
+
|
| 37 |
+
key, out = sys.argv[1], Path(sys.argv[2])
|
| 38 |
+
n_q = int(sys.argv[3]) if len(sys.argv) > 3 else 6
|
| 39 |
+
spec = MODELS[key]
|
| 40 |
+
spec.load()
|
| 41 |
+
assert not torch.cuda.is_initialized()
|
| 42 |
+
queries = general_qa(per_subset=1, offset=440)[:n_q]
|
| 43 |
+
|
| 44 |
+
|
| 45 |
+
def child(conn, q, preset):
|
| 46 |
+
patching.unpatch()
|
| 47 |
+
t0 = time.time()
|
| 48 |
+
rec = {"demos": []}
|
| 49 |
+
spec.preset = lambda budget: dict(preset) # this process only
|
| 50 |
+
for ev in run(spec, q["image"], q["question"], 1000, True):
|
| 51 |
+
t = time.time() - t0
|
| 52 |
+
if ev[0] == "greedy":
|
| 53 |
+
rec["greedy"] = t
|
| 54 |
+
elif ev[0] == "demo":
|
| 55 |
+
rec["demos"].append(t)
|
| 56 |
+
elif ev[0] == "final":
|
| 57 |
+
rec["total"] = t
|
| 58 |
+
conn.send(rec)
|
| 59 |
+
|
| 60 |
+
|
| 61 |
+
results = []
|
| 62 |
+
for sname, setting in SETTINGS[key].items():
|
| 63 |
+
for k in KS[key]:
|
| 64 |
+
preset = dict(ABA, k_max=k, attempts_per_slot=4, repair_retries=1, verify_rounds=2, **setting)
|
| 65 |
+
for qi, q in enumerate(queries):
|
| 66 |
+
a, b = mp.get_context("fork").Pipe()
|
| 67 |
+
p = mp.get_context("fork").Process(target=child, args=(b, q, preset))
|
| 68 |
+
p.start()
|
| 69 |
+
rec = a.recv() if a.poll(900) else {"error": "timeout"}
|
| 70 |
+
p.join(timeout=30)
|
| 71 |
+
rec.update(setting=sname, k=k, query=qi, subset=q["subset"])
|
| 72 |
+
results.append(rec)
|
| 73 |
+
print(json.dumps(rec), flush=True)
|
| 74 |
+
out.write_text(json.dumps(results, indent=1))
|
tools/make_examples.py
ADDED
|
@@ -0,0 +1,121 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Cached demo examples: real runs of the demo pipeline on the paper's figure
|
| 2 |
+
queries (LMMs-Eval-Lite rows), stored with both answers, the imagined demos and
|
| 3 |
+
the reference answer. ``--keep improved`` keeps only runs where SIMIT-ICL fixed
|
| 4 |
+
the base answer.
|
| 5 |
+
|
| 6 |
+
CUDA_VISIBLE_DEVICES=0 python tools/make_examples.py bagel 60 [--keep improved]
|
| 7 |
+
"""
|
| 8 |
+
import argparse
|
| 9 |
+
import json
|
| 10 |
+
import os
|
| 11 |
+
import shutil
|
| 12 |
+
import sys
|
| 13 |
+
import time
|
| 14 |
+
from pathlib import Path
|
| 15 |
+
|
| 16 |
+
os.environ.setdefault("TRITON_CACHE_AUTOTUNING", "1")
|
| 17 |
+
os.environ.setdefault("TRANSFORMERS_DISABLE_DEEPGEMM_LINEAR", "1")
|
| 18 |
+
HERE = Path(__file__).parent.parent
|
| 19 |
+
sys.path.insert(0, str(HERE))
|
| 20 |
+
sys.path.insert(0, str(HERE / "tools"))
|
| 21 |
+
|
| 22 |
+
from datasets import load_dataset # noqa: E402
|
| 23 |
+
from PIL import Image # noqa: E402
|
| 24 |
+
|
| 25 |
+
from demo_models import MODELS, run # noqa: E402
|
| 26 |
+
from general_qa import _list # noqa: E402
|
| 27 |
+
from simit.metrics import get_metric # noqa: E402
|
| 28 |
+
|
| 29 |
+
VQA = "Answer the question using a single word or phrase."
|
| 30 |
+
UNANSWERABLE = "When the provided information is insufficient, respond with 'Unanswerable'."
|
| 31 |
+
# (subset, row, prompt suffix, metric, where it appears in the paper)
|
| 32 |
+
CANDIDATES = [
|
| 33 |
+
("ok_vqa_val2014", 492, f"{UNANSWERABLE}\n{VQA}", "vqa_accuracy", "paper Fig. (qualitative samples)"),
|
| 34 |
+
("infovqa_val", 449, VQA, "anls", "paper Fig. (qualitative samples)"),
|
| 35 |
+
("vizwiz_vqa_val", 345, f"{UNANSWERABLE}\n{VQA}", "vqa_accuracy", "paper teaser"),
|
| 36 |
+
("vizwiz_vqa_val", 47, f"{UNANSWERABLE}\n{VQA}", "vqa_accuracy", "paper appendix"),
|
| 37 |
+
("vizwiz_vqa_val", 87, f"{UNANSWERABLE}\n{VQA}", "vqa_accuracy", "paper appendix"),
|
| 38 |
+
] + [("ok_vqa_val2014", i, VQA, "vqa_accuracy", "paper appendix") for i in (115, 151, 210, 22, 253, 272, 291, 381,
|
| 39 |
+
477, 8)]
|
| 40 |
+
SUBSET_NAME = {"ok_vqa_val2014": "OK-VQA", "infovqa_val": "InfographicVQA", "vizwiz_vqa_val": "VizWiz"}
|
| 41 |
+
|
| 42 |
+
ap = argparse.ArgumentParser()
|
| 43 |
+
ap.add_argument("model")
|
| 44 |
+
ap.add_argument("budget", type=int)
|
| 45 |
+
ap.add_argument("--keep", choices=["all", "improved"], default="all")
|
| 46 |
+
ap.add_argument("--always", action="store_true", help="imagine even when the model is confident (UI checkbox)")
|
| 47 |
+
ap.add_argument("--out", default=str(HERE / "examples"))
|
| 48 |
+
ap.add_argument("--only", nargs="*", default=None, help="subset:row items to run, e.g. ok_vqa_val2014:492")
|
| 49 |
+
ap.add_argument("--image", help="a custom query image (instead of the dataset candidates)")
|
| 50 |
+
ap.add_argument("--question", help="the question for --image")
|
| 51 |
+
ap.add_argument("--reference", nargs="+", help="reference answer(s) for --image")
|
| 52 |
+
ap.add_argument("--metric", default="contains", help="metric for --image (free-form answers: 'contains')")
|
| 53 |
+
ap.add_argument("--id", help="example id (folder name) for --image")
|
| 54 |
+
ap.add_argument("--source", default="custom example", help="source line shown in the UI for --image")
|
| 55 |
+
args = ap.parse_args()
|
| 56 |
+
out = Path(args.out)
|
| 57 |
+
out.mkdir(parents=True, exist_ok=True)
|
| 58 |
+
index_file = out / "index.json"
|
| 59 |
+
index = json.loads(index_file.read_text()) if index_file.exists() else []
|
| 60 |
+
spec = MODELS[args.model]
|
| 61 |
+
spec.load()
|
| 62 |
+
suffix_id = "_always" if args.always else ""
|
| 63 |
+
|
| 64 |
+
|
| 65 |
+
def jobs():
|
| 66 |
+
"""(example id, image, question, references, metric name, source, keep the folder's own files)"""
|
| 67 |
+
if args.image:
|
| 68 |
+
img = Image.open(args.image)
|
| 69 |
+
img.load()
|
| 70 |
+
yield (args.id + suffix_id, img, args.question, args.reference, args.metric, args.source, True)
|
| 71 |
+
return
|
| 72 |
+
for sub, row, suffix, metric_name, where in CANDIDATES:
|
| 73 |
+
if args.only is not None and f"{sub}:{row}" not in args.only:
|
| 74 |
+
continue
|
| 75 |
+
r = load_dataset("lmms-lab/LMMs-Eval-Lite", sub)["lite"][row]
|
| 76 |
+
yield (f"{args.model}_{sub}_{row}{suffix_id}", r["image"].convert("RGB"), f"{r['question']}\n{suffix}",
|
| 77 |
+
[str(a) for a in _list(r.get("answers", r.get("answer")))], metric_name,
|
| 78 |
+
f"{SUBSET_NAME.get(sub, sub)} row {row}, {where}", False)
|
| 79 |
+
|
| 80 |
+
|
| 81 |
+
for ex_id, image, question, refs, metric_name, source, custom in jobs():
|
| 82 |
+
metric = get_metric(metric_name)
|
| 83 |
+
t0 = time.time()
|
| 84 |
+
greedy, demos, final, info = None, [], None, {}
|
| 85 |
+
for ev in run(spec, image, question, args.budget, args.always):
|
| 86 |
+
if ev[0] == "greedy":
|
| 87 |
+
greedy = ev[1]
|
| 88 |
+
elif ev[0] == "demo":
|
| 89 |
+
demos.append(ev[1])
|
| 90 |
+
elif ev[0] == "final":
|
| 91 |
+
final, info = ev[1], ev[2]
|
| 92 |
+
g_ok, s_ok = metric(greedy.answer, refs) >= 0.5, metric(final, refs) >= 0.5
|
| 93 |
+
print(f"{ex_id}: greedy={greedy.answer!r} ({g_ok}) simit={final!r} ({s_ok}) demos={len(demos)} "
|
| 94 |
+
f"{time.time() - t0:.0f}s refs={refs[:3]}", flush=True)
|
| 95 |
+
if args.keep == "improved" and not (s_ok and not g_ok):
|
| 96 |
+
continue
|
| 97 |
+
d = out / ex_id
|
| 98 |
+
if custom and d.exists(): # the user's own files stay; only earlier demo images are replaced
|
| 99 |
+
for f in d.glob("demo_*.png"):
|
| 100 |
+
f.unlink()
|
| 101 |
+
else:
|
| 102 |
+
shutil.rmtree(d, ignore_errors=True)
|
| 103 |
+
d.mkdir()
|
| 104 |
+
if custom and Path(args.image).resolve().parent == d.resolve():
|
| 105 |
+
query_file = Path(args.image).name
|
| 106 |
+
else:
|
| 107 |
+
shown = image.convert("RGB")
|
| 108 |
+
shown.thumbnail((1024, 1024)) # display copy (the run used the original)
|
| 109 |
+
shown.save(d / "query.jpg", quality=90)
|
| 110 |
+
query_file = "query.jpg"
|
| 111 |
+
stored = []
|
| 112 |
+
for i, dm in enumerate(demos):
|
| 113 |
+
dm.image.save(d / f"demo_{i}.png")
|
| 114 |
+
stored.append({"image": f"demo_{i}.png", "question": dm.question, "answer": dm.answer, "skill": dm.skill,
|
| 115 |
+
"verify_score": dm.verify_score, "confidence": dm.confidence})
|
| 116 |
+
entry = {"id": ex_id, "query_file": query_file, "model": args.model, "budget": args.budget,
|
| 117 |
+
"always": args.always, "question": question, "reference": refs[0],
|
| 118 |
+
"greedy": greedy.answer, "p0": greedy.confidence, "simit": final, "greedy_ok": g_ok, "simit_ok": s_ok,
|
| 119 |
+
"demos": stored, "elapsed": info.get("elapsed", time.time() - t0), "source": source}
|
| 120 |
+
index = [e for e in index if e["id"] != ex_id] + [entry]
|
| 121 |
+
index_file.write_text(json.dumps(index, indent=1))
|
tools/results/latency_bagel.json
ADDED
|
@@ -0,0 +1,882 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
[
|
| 2 |
+
{
|
| 3 |
+
"demos": [
|
| 4 |
+
26.002389907836914
|
| 5 |
+
],
|
| 6 |
+
"greedy": 6.846283435821533,
|
| 7 |
+
"total": 26.74322533607483,
|
| 8 |
+
"setting": "think50",
|
| 9 |
+
"k": 1,
|
| 10 |
+
"query": 0,
|
| 11 |
+
"subset": "vizwiz_vqa_val"
|
| 12 |
+
},
|
| 13 |
+
{
|
| 14 |
+
"demos": [
|
| 15 |
+
12.16827392578125
|
| 16 |
+
],
|
| 17 |
+
"greedy": 6.742192268371582,
|
| 18 |
+
"total": 12.760454177856445,
|
| 19 |
+
"setting": "think50",
|
| 20 |
+
"k": 1,
|
| 21 |
+
"query": 1,
|
| 22 |
+
"subset": "vqav2_val"
|
| 23 |
+
},
|
| 24 |
+
{
|
| 25 |
+
"demos": [
|
| 26 |
+
17.263279914855957
|
| 27 |
+
],
|
| 28 |
+
"greedy": 6.47596001625061,
|
| 29 |
+
"total": 17.900477647781372,
|
| 30 |
+
"setting": "think50",
|
| 31 |
+
"k": 1,
|
| 32 |
+
"query": 2,
|
| 33 |
+
"subset": "ok_vqa_val2014"
|
| 34 |
+
},
|
| 35 |
+
{
|
| 36 |
+
"demos": [
|
| 37 |
+
33.245251417160034
|
| 38 |
+
],
|
| 39 |
+
"greedy": 6.677622079849243,
|
| 40 |
+
"total": 34.04009461402893,
|
| 41 |
+
"setting": "think50",
|
| 42 |
+
"k": 1,
|
| 43 |
+
"query": 3,
|
| 44 |
+
"subset": "textvqa_val"
|
| 45 |
+
},
|
| 46 |
+
{
|
| 47 |
+
"demos": [
|
| 48 |
+
21.499918460845947
|
| 49 |
+
],
|
| 50 |
+
"greedy": 6.547531366348267,
|
| 51 |
+
"total": 22.275030374526978,
|
| 52 |
+
"setting": "think50",
|
| 53 |
+
"k": 1,
|
| 54 |
+
"query": 4,
|
| 55 |
+
"subset": "chartqa"
|
| 56 |
+
},
|
| 57 |
+
{
|
| 58 |
+
"demos": [],
|
| 59 |
+
"greedy": 6.649888515472412,
|
| 60 |
+
"total": 57.07210087776184,
|
| 61 |
+
"setting": "think50",
|
| 62 |
+
"k": 1,
|
| 63 |
+
"query": 5,
|
| 64 |
+
"subset": "docvqa_val"
|
| 65 |
+
},
|
| 66 |
+
{
|
| 67 |
+
"demos": [
|
| 68 |
+
14.020853996276855,
|
| 69 |
+
15.747642755508423
|
| 70 |
+
],
|
| 71 |
+
"greedy": 6.6902241706848145,
|
| 72 |
+
"total": 16.58476734161377,
|
| 73 |
+
"setting": "think50",
|
| 74 |
+
"k": 2,
|
| 75 |
+
"query": 0,
|
| 76 |
+
"subset": "vizwiz_vqa_val"
|
| 77 |
+
},
|
| 78 |
+
{
|
| 79 |
+
"demos": [
|
| 80 |
+
14.245272397994995,
|
| 81 |
+
15.97743272781372
|
| 82 |
+
],
|
| 83 |
+
"greedy": 6.408905506134033,
|
| 84 |
+
"total": 16.649274826049805,
|
| 85 |
+
"setting": "think50",
|
| 86 |
+
"k": 2,
|
| 87 |
+
"query": 1,
|
| 88 |
+
"subset": "vqav2_val"
|
| 89 |
+
},
|
| 90 |
+
{
|
| 91 |
+
"demos": [
|
| 92 |
+
14.016824960708618,
|
| 93 |
+
27.909880876541138
|
| 94 |
+
],
|
| 95 |
+
"greedy": 6.539390563964844,
|
| 96 |
+
"total": 28.65375280380249,
|
| 97 |
+
"setting": "think50",
|
| 98 |
+
"k": 2,
|
| 99 |
+
"query": 2,
|
| 100 |
+
"subset": "ok_vqa_val2014"
|
| 101 |
+
},
|
| 102 |
+
{
|
| 103 |
+
"demos": [
|
| 104 |
+
19.595810413360596,
|
| 105 |
+
20.93066096305847
|
| 106 |
+
],
|
| 107 |
+
"greedy": 6.817545413970947,
|
| 108 |
+
"total": 21.97844171524048,
|
| 109 |
+
"setting": "think50",
|
| 110 |
+
"k": 2,
|
| 111 |
+
"query": 3,
|
| 112 |
+
"subset": "textvqa_val"
|
| 113 |
+
},
|
| 114 |
+
{
|
| 115 |
+
"demos": [
|
| 116 |
+
21.255191326141357,
|
| 117 |
+
23.45720887184143
|
| 118 |
+
],
|
| 119 |
+
"greedy": 6.85756778717041,
|
| 120 |
+
"total": 24.484545946121216,
|
| 121 |
+
"setting": "think50",
|
| 122 |
+
"k": 2,
|
| 123 |
+
"query": 4,
|
| 124 |
+
"subset": "chartqa"
|
| 125 |
+
},
|
| 126 |
+
{
|
| 127 |
+
"demos": [
|
| 128 |
+
17.32177734375,
|
| 129 |
+
29.950555562973022
|
| 130 |
+
],
|
| 131 |
+
"greedy": 6.608810186386108,
|
| 132 |
+
"total": 30.91857123374939,
|
| 133 |
+
"setting": "think50",
|
| 134 |
+
"k": 2,
|
| 135 |
+
"query": 5,
|
| 136 |
+
"subset": "docvqa_val"
|
| 137 |
+
},
|
| 138 |
+
{
|
| 139 |
+
"demos": [
|
| 140 |
+
19.331388235092163,
|
| 141 |
+
20.449653387069702,
|
| 142 |
+
20.78369402885437,
|
| 143 |
+
24.906179904937744
|
| 144 |
+
],
|
| 145 |
+
"greedy": 6.746614217758179,
|
| 146 |
+
"total": 25.91768527030945,
|
| 147 |
+
"setting": "think50",
|
| 148 |
+
"k": 4,
|
| 149 |
+
"query": 0,
|
| 150 |
+
"subset": "vizwiz_vqa_val"
|
| 151 |
+
},
|
| 152 |
+
{
|
| 153 |
+
"demos": [
|
| 154 |
+
19.866760969161987,
|
| 155 |
+
21.37235426902771,
|
| 156 |
+
25.51891589164734,
|
| 157 |
+
25.743648767471313
|
| 158 |
+
],
|
| 159 |
+
"greedy": 6.74257230758667,
|
| 160 |
+
"total": 26.58999800682068,
|
| 161 |
+
"setting": "think50",
|
| 162 |
+
"k": 4,
|
| 163 |
+
"query": 1,
|
| 164 |
+
"subset": "vqav2_val"
|
| 165 |
+
},
|
| 166 |
+
{
|
| 167 |
+
"demos": [
|
| 168 |
+
18.303645849227905,
|
| 169 |
+
19.48818588256836,
|
| 170 |
+
19.763233184814453,
|
| 171 |
+
21.081400394439697
|
| 172 |
+
],
|
| 173 |
+
"greedy": 6.863516330718994,
|
| 174 |
+
"total": 21.978867053985596,
|
| 175 |
+
"setting": "think50",
|
| 176 |
+
"k": 4,
|
| 177 |
+
"query": 2,
|
| 178 |
+
"subset": "ok_vqa_val2014"
|
| 179 |
+
},
|
| 180 |
+
{
|
| 181 |
+
"demos": [
|
| 182 |
+
18.599497079849243,
|
| 183 |
+
47.75624179840088,
|
| 184 |
+
48.84687924385071
|
| 185 |
+
],
|
| 186 |
+
"greedy": 6.99056339263916,
|
| 187 |
+
"total": 123.67395544052124,
|
| 188 |
+
"setting": "think50",
|
| 189 |
+
"k": 4,
|
| 190 |
+
"query": 3,
|
| 191 |
+
"subset": "textvqa_val"
|
| 192 |
+
},
|
| 193 |
+
{
|
| 194 |
+
"demos": [
|
| 195 |
+
16.050732851028442,
|
| 196 |
+
17.049232006072998,
|
| 197 |
+
17.45568537712097,
|
| 198 |
+
17.470638751983643
|
| 199 |
+
],
|
| 200 |
+
"greedy": 6.753719091415405,
|
| 201 |
+
"total": 18.682735443115234,
|
| 202 |
+
"setting": "think50",
|
| 203 |
+
"k": 4,
|
| 204 |
+
"query": 4,
|
| 205 |
+
"subset": "chartqa"
|
| 206 |
+
},
|
| 207 |
+
{
|
| 208 |
+
"demos": [
|
| 209 |
+
17.657182216644287,
|
| 210 |
+
17.93081760406494,
|
| 211 |
+
28.86550760269165,
|
| 212 |
+
48.43350315093994
|
| 213 |
+
],
|
| 214 |
+
"greedy": 6.711917877197266,
|
| 215 |
+
"total": 49.75816011428833,
|
| 216 |
+
"setting": "think50",
|
| 217 |
+
"k": 4,
|
| 218 |
+
"query": 5,
|
| 219 |
+
"subset": "docvqa_val"
|
| 220 |
+
},
|
| 221 |
+
{
|
| 222 |
+
"demos": [
|
| 223 |
+
12.481452226638794
|
| 224 |
+
],
|
| 225 |
+
"greedy": 6.7068750858306885,
|
| 226 |
+
"total": 13.248512744903564,
|
| 227 |
+
"setting": "nothink50",
|
| 228 |
+
"k": 1,
|
| 229 |
+
"query": 0,
|
| 230 |
+
"subset": "vizwiz_vqa_val"
|
| 231 |
+
},
|
| 232 |
+
{
|
| 233 |
+
"demos": [
|
| 234 |
+
12.028658628463745
|
| 235 |
+
],
|
| 236 |
+
"greedy": 6.6228907108306885,
|
| 237 |
+
"total": 12.620982646942139,
|
| 238 |
+
"setting": "nothink50",
|
| 239 |
+
"k": 1,
|
| 240 |
+
"query": 1,
|
| 241 |
+
"subset": "vqav2_val"
|
| 242 |
+
},
|
| 243 |
+
{
|
| 244 |
+
"demos": [
|
| 245 |
+
12.994443416595459
|
| 246 |
+
],
|
| 247 |
+
"greedy": 6.687414646148682,
|
| 248 |
+
"total": 13.635976076126099,
|
| 249 |
+
"setting": "nothink50",
|
| 250 |
+
"k": 1,
|
| 251 |
+
"query": 2,
|
| 252 |
+
"subset": "ok_vqa_val2014"
|
| 253 |
+
},
|
| 254 |
+
{
|
| 255 |
+
"demos": [],
|
| 256 |
+
"greedy": 6.697295427322388,
|
| 257 |
+
"total": 31.110349416732788,
|
| 258 |
+
"setting": "nothink50",
|
| 259 |
+
"k": 1,
|
| 260 |
+
"query": 3,
|
| 261 |
+
"subset": "textvqa_val"
|
| 262 |
+
},
|
| 263 |
+
{
|
| 264 |
+
"demos": [
|
| 265 |
+
26.508880138397217
|
| 266 |
+
],
|
| 267 |
+
"greedy": 6.976494312286377,
|
| 268 |
+
"total": 27.29017186164856,
|
| 269 |
+
"setting": "nothink50",
|
| 270 |
+
"k": 1,
|
| 271 |
+
"query": 4,
|
| 272 |
+
"subset": "chartqa"
|
| 273 |
+
},
|
| 274 |
+
{
|
| 275 |
+
"demos": [
|
| 276 |
+
41.90188384056091
|
| 277 |
+
],
|
| 278 |
+
"greedy": 6.816613435745239,
|
| 279 |
+
"total": 42.66916561126709,
|
| 280 |
+
"setting": "nothink50",
|
| 281 |
+
"k": 1,
|
| 282 |
+
"query": 5,
|
| 283 |
+
"subset": "docvqa_val"
|
| 284 |
+
},
|
| 285 |
+
{
|
| 286 |
+
"demos": [
|
| 287 |
+
15.123422622680664,
|
| 288 |
+
15.220020294189453
|
| 289 |
+
],
|
| 290 |
+
"greedy": 7.205089807510376,
|
| 291 |
+
"total": 16.069376230239868,
|
| 292 |
+
"setting": "nothink50",
|
| 293 |
+
"k": 2,
|
| 294 |
+
"query": 0,
|
| 295 |
+
"subset": "vizwiz_vqa_val"
|
| 296 |
+
},
|
| 297 |
+
{
|
| 298 |
+
"demos": [
|
| 299 |
+
16.50620174407959,
|
| 300 |
+
17.572696685791016
|
| 301 |
+
],
|
| 302 |
+
"greedy": 7.146985292434692,
|
| 303 |
+
"total": 18.24684238433838,
|
| 304 |
+
"setting": "nothink50",
|
| 305 |
+
"k": 2,
|
| 306 |
+
"query": 1,
|
| 307 |
+
"subset": "vqav2_val"
|
| 308 |
+
},
|
| 309 |
+
{
|
| 310 |
+
"demos": [
|
| 311 |
+
25.19501304626465,
|
| 312 |
+
32.75030159950256
|
| 313 |
+
],
|
| 314 |
+
"greedy": 7.005849599838257,
|
| 315 |
+
"total": 33.467132806777954,
|
| 316 |
+
"setting": "nothink50",
|
| 317 |
+
"k": 2,
|
| 318 |
+
"query": 2,
|
| 319 |
+
"subset": "ok_vqa_val2014"
|
| 320 |
+
},
|
| 321 |
+
{
|
| 322 |
+
"demos": [
|
| 323 |
+
40.2707622051239,
|
| 324 |
+
67.3787031173706
|
| 325 |
+
],
|
| 326 |
+
"greedy": 7.045415639877319,
|
| 327 |
+
"total": 68.47934651374817,
|
| 328 |
+
"setting": "nothink50",
|
| 329 |
+
"k": 2,
|
| 330 |
+
"query": 3,
|
| 331 |
+
"subset": "textvqa_val"
|
| 332 |
+
},
|
| 333 |
+
{
|
| 334 |
+
"demos": [
|
| 335 |
+
16.73797917366028,
|
| 336 |
+
26.571113348007202
|
| 337 |
+
],
|
| 338 |
+
"greedy": 7.133496999740601,
|
| 339 |
+
"total": 27.498658895492554,
|
| 340 |
+
"setting": "nothink50",
|
| 341 |
+
"k": 2,
|
| 342 |
+
"query": 4,
|
| 343 |
+
"subset": "chartqa"
|
| 344 |
+
},
|
| 345 |
+
{
|
| 346 |
+
"demos": [
|
| 347 |
+
18.925209283828735,
|
| 348 |
+
45.47921824455261
|
| 349 |
+
],
|
| 350 |
+
"greedy": 7.158755540847778,
|
| 351 |
+
"total": 46.42701768875122,
|
| 352 |
+
"setting": "nothink50",
|
| 353 |
+
"k": 2,
|
| 354 |
+
"query": 5,
|
| 355 |
+
"subset": "docvqa_val"
|
| 356 |
+
},
|
| 357 |
+
{
|
| 358 |
+
"demos": [
|
| 359 |
+
18.92365074157715,
|
| 360 |
+
20.19171667098999,
|
| 361 |
+
20.651800394058228,
|
| 362 |
+
24.54624581336975
|
| 363 |
+
],
|
| 364 |
+
"greedy": 6.934945344924927,
|
| 365 |
+
"total": 25.56127405166626,
|
| 366 |
+
"setting": "nothink50",
|
| 367 |
+
"k": 4,
|
| 368 |
+
"query": 0,
|
| 369 |
+
"subset": "vizwiz_vqa_val"
|
| 370 |
+
},
|
| 371 |
+
{
|
| 372 |
+
"demos": [
|
| 373 |
+
20.397387266159058,
|
| 374 |
+
21.484426259994507,
|
| 375 |
+
21.654940843582153,
|
| 376 |
+
39.95197796821594
|
| 377 |
+
],
|
| 378 |
+
"greedy": 6.822445392608643,
|
| 379 |
+
"total": 40.79026508331299,
|
| 380 |
+
"setting": "nothink50",
|
| 381 |
+
"k": 4,
|
| 382 |
+
"query": 1,
|
| 383 |
+
"subset": "vqav2_val"
|
| 384 |
+
},
|
| 385 |
+
{
|
| 386 |
+
"demos": [
|
| 387 |
+
17.846131563186646,
|
| 388 |
+
31.462411880493164,
|
| 389 |
+
31.96941876411438,
|
| 390 |
+
42.511390209198
|
| 391 |
+
],
|
| 392 |
+
"greedy": 7.063258647918701,
|
| 393 |
+
"total": 43.39737248420715,
|
| 394 |
+
"setting": "nothink50",
|
| 395 |
+
"k": 4,
|
| 396 |
+
"query": 2,
|
| 397 |
+
"subset": "ok_vqa_val2014"
|
| 398 |
+
},
|
| 399 |
+
{
|
| 400 |
+
"demos": [
|
| 401 |
+
31.60993504524231,
|
| 402 |
+
32.77745318412781,
|
| 403 |
+
46.97655916213989,
|
| 404 |
+
67.76566171646118
|
| 405 |
+
],
|
| 406 |
+
"greedy": 7.10393762588501,
|
| 407 |
+
"total": 69.44919919967651,
|
| 408 |
+
"setting": "nothink50",
|
| 409 |
+
"k": 4,
|
| 410 |
+
"query": 3,
|
| 411 |
+
"subset": "textvqa_val"
|
| 412 |
+
},
|
| 413 |
+
{
|
| 414 |
+
"demos": [
|
| 415 |
+
13.910687446594238,
|
| 416 |
+
18.546985864639282,
|
| 417 |
+
19.083553314208984,
|
| 418 |
+
19.979618787765503
|
| 419 |
+
],
|
| 420 |
+
"greedy": 6.712128639221191,
|
| 421 |
+
"total": 21.244775533676147,
|
| 422 |
+
"setting": "nothink50",
|
| 423 |
+
"k": 4,
|
| 424 |
+
"query": 4,
|
| 425 |
+
"subset": "chartqa"
|
| 426 |
+
},
|
| 427 |
+
{
|
| 428 |
+
"demos": [
|
| 429 |
+
27.11043953895569,
|
| 430 |
+
36.58712029457092,
|
| 431 |
+
41.17549681663513,
|
| 432 |
+
51.479180574417114
|
| 433 |
+
],
|
| 434 |
+
"greedy": 6.7352728843688965,
|
| 435 |
+
"total": 52.88075041770935,
|
| 436 |
+
"setting": "nothink50",
|
| 437 |
+
"k": 4,
|
| 438 |
+
"query": 5,
|
| 439 |
+
"subset": "docvqa_val"
|
| 440 |
+
},
|
| 441 |
+
{
|
| 442 |
+
"demos": [
|
| 443 |
+
15.192901134490967
|
| 444 |
+
],
|
| 445 |
+
"greedy": 6.91644811630249,
|
| 446 |
+
"total": 15.934749126434326,
|
| 447 |
+
"setting": "nothink25",
|
| 448 |
+
"k": 1,
|
| 449 |
+
"query": 0,
|
| 450 |
+
"subset": "vizwiz_vqa_val"
|
| 451 |
+
},
|
| 452 |
+
{
|
| 453 |
+
"demos": [
|
| 454 |
+
14.546730279922485
|
| 455 |
+
],
|
| 456 |
+
"greedy": 6.616936206817627,
|
| 457 |
+
"total": 15.137541770935059,
|
| 458 |
+
"setting": "nothink25",
|
| 459 |
+
"k": 1,
|
| 460 |
+
"query": 1,
|
| 461 |
+
"subset": "vqav2_val"
|
| 462 |
+
},
|
| 463 |
+
{
|
| 464 |
+
"demos": [
|
| 465 |
+
10.399396657943726
|
| 466 |
+
],
|
| 467 |
+
"greedy": 6.794123411178589,
|
| 468 |
+
"total": 11.040837049484253,
|
| 469 |
+
"setting": "nothink25",
|
| 470 |
+
"k": 1,
|
| 471 |
+
"query": 2,
|
| 472 |
+
"subset": "ok_vqa_val2014"
|
| 473 |
+
},
|
| 474 |
+
{
|
| 475 |
+
"demos": [],
|
| 476 |
+
"greedy": 6.857384443283081,
|
| 477 |
+
"total": 36.08436679840088,
|
| 478 |
+
"setting": "nothink25",
|
| 479 |
+
"k": 1,
|
| 480 |
+
"query": 3,
|
| 481 |
+
"subset": "textvqa_val"
|
| 482 |
+
},
|
| 483 |
+
{
|
| 484 |
+
"demos": [
|
| 485 |
+
17.479740619659424
|
| 486 |
+
],
|
| 487 |
+
"greedy": 6.688612699508667,
|
| 488 |
+
"total": 18.236947536468506,
|
| 489 |
+
"setting": "nothink25",
|
| 490 |
+
"k": 1,
|
| 491 |
+
"query": 4,
|
| 492 |
+
"subset": "chartqa"
|
| 493 |
+
},
|
| 494 |
+
{
|
| 495 |
+
"demos": [
|
| 496 |
+
59.99591255187988
|
| 497 |
+
],
|
| 498 |
+
"greedy": 6.7639524936676025,
|
| 499 |
+
"total": 60.8376727104187,
|
| 500 |
+
"setting": "nothink25",
|
| 501 |
+
"k": 1,
|
| 502 |
+
"query": 5,
|
| 503 |
+
"subset": "docvqa_val"
|
| 504 |
+
},
|
| 505 |
+
{
|
| 506 |
+
"demos": [
|
| 507 |
+
12.354958057403564,
|
| 508 |
+
19.66935706138611
|
| 509 |
+
],
|
| 510 |
+
"greedy": 6.810552358627319,
|
| 511 |
+
"total": 20.500991344451904,
|
| 512 |
+
"setting": "nothink25",
|
| 513 |
+
"k": 2,
|
| 514 |
+
"query": 0,
|
| 515 |
+
"subset": "vizwiz_vqa_val"
|
| 516 |
+
},
|
| 517 |
+
{
|
| 518 |
+
"demos": [
|
| 519 |
+
13.075764179229736,
|
| 520 |
+
13.986951351165771
|
| 521 |
+
],
|
| 522 |
+
"greedy": 6.526533842086792,
|
| 523 |
+
"total": 14.659094333648682,
|
| 524 |
+
"setting": "nothink25",
|
| 525 |
+
"k": 2,
|
| 526 |
+
"query": 1,
|
| 527 |
+
"subset": "vqav2_val"
|
| 528 |
+
},
|
| 529 |
+
{
|
| 530 |
+
"demos": [
|
| 531 |
+
11.25294542312622,
|
| 532 |
+
12.738617658615112
|
| 533 |
+
],
|
| 534 |
+
"greedy": 6.653463363647461,
|
| 535 |
+
"total": 13.45282244682312,
|
| 536 |
+
"setting": "nothink25",
|
| 537 |
+
"k": 2,
|
| 538 |
+
"query": 2,
|
| 539 |
+
"subset": "ok_vqa_val2014"
|
| 540 |
+
},
|
| 541 |
+
{
|
| 542 |
+
"demos": [
|
| 543 |
+
52.51059865951538
|
| 544 |
+
],
|
| 545 |
+
"greedy": 6.916888952255249,
|
| 546 |
+
"total": 69.37151217460632,
|
| 547 |
+
"setting": "nothink25",
|
| 548 |
+
"k": 2,
|
| 549 |
+
"query": 3,
|
| 550 |
+
"subset": "textvqa_val"
|
| 551 |
+
},
|
| 552 |
+
{
|
| 553 |
+
"demos": [
|
| 554 |
+
16.357442378997803,
|
| 555 |
+
22.164559364318848
|
| 556 |
+
],
|
| 557 |
+
"greedy": 6.922367811203003,
|
| 558 |
+
"total": 23.112447500228882,
|
| 559 |
+
"setting": "nothink25",
|
| 560 |
+
"k": 2,
|
| 561 |
+
"query": 4,
|
| 562 |
+
"subset": "chartqa"
|
| 563 |
+
},
|
| 564 |
+
{
|
| 565 |
+
"demos": [
|
| 566 |
+
31.607569217681885,
|
| 567 |
+
64.07539129257202
|
| 568 |
+
],
|
| 569 |
+
"greedy": 7.111017942428589,
|
| 570 |
+
"total": 65.04016399383545,
|
| 571 |
+
"setting": "nothink25",
|
| 572 |
+
"k": 2,
|
| 573 |
+
"query": 5,
|
| 574 |
+
"subset": "docvqa_val"
|
| 575 |
+
},
|
| 576 |
+
{
|
| 577 |
+
"demos": [
|
| 578 |
+
14.144917011260986,
|
| 579 |
+
14.450908422470093,
|
| 580 |
+
14.969972133636475,
|
| 581 |
+
22.031485557556152
|
| 582 |
+
],
|
| 583 |
+
"greedy": 6.845429420471191,
|
| 584 |
+
"total": 23.0630886554718,
|
| 585 |
+
"setting": "nothink25",
|
| 586 |
+
"k": 4,
|
| 587 |
+
"query": 0,
|
| 588 |
+
"subset": "vizwiz_vqa_val"
|
| 589 |
+
},
|
| 590 |
+
{
|
| 591 |
+
"demos": [
|
| 592 |
+
15.669585943222046,
|
| 593 |
+
19.648891925811768,
|
| 594 |
+
20.805134534835815,
|
| 595 |
+
29.888363122940063
|
| 596 |
+
],
|
| 597 |
+
"greedy": 6.622054576873779,
|
| 598 |
+
"total": 30.800080060958862,
|
| 599 |
+
"setting": "nothink25",
|
| 600 |
+
"k": 4,
|
| 601 |
+
"query": 1,
|
| 602 |
+
"subset": "vqav2_val"
|
| 603 |
+
},
|
| 604 |
+
{
|
| 605 |
+
"demos": [
|
| 606 |
+
15.39557409286499,
|
| 607 |
+
16.348963260650635,
|
| 608 |
+
19.663447618484497,
|
| 609 |
+
20.33902072906494
|
| 610 |
+
],
|
| 611 |
+
"greedy": 6.637497663497925,
|
| 612 |
+
"total": 21.22013282775879,
|
| 613 |
+
"setting": "nothink25",
|
| 614 |
+
"k": 4,
|
| 615 |
+
"query": 2,
|
| 616 |
+
"subset": "ok_vqa_val2014"
|
| 617 |
+
},
|
| 618 |
+
{
|
| 619 |
+
"demos": [
|
| 620 |
+
15.362420558929443,
|
| 621 |
+
15.921366691589355,
|
| 622 |
+
24.226126670837402,
|
| 623 |
+
45.1386661529541
|
| 624 |
+
],
|
| 625 |
+
"greedy": 6.875400066375732,
|
| 626 |
+
"total": 46.43324565887451,
|
| 627 |
+
"setting": "nothink25",
|
| 628 |
+
"k": 4,
|
| 629 |
+
"query": 3,
|
| 630 |
+
"subset": "textvqa_val"
|
| 631 |
+
},
|
| 632 |
+
{
|
| 633 |
+
"demos": [
|
| 634 |
+
15.788510084152222,
|
| 635 |
+
16.097867727279663,
|
| 636 |
+
19.899868726730347,
|
| 637 |
+
24.75997519493103
|
| 638 |
+
],
|
| 639 |
+
"greedy": 7.031396865844727,
|
| 640 |
+
"total": 26.027679443359375,
|
| 641 |
+
"setting": "nothink25",
|
| 642 |
+
"k": 4,
|
| 643 |
+
"query": 4,
|
| 644 |
+
"subset": "chartqa"
|
| 645 |
+
},
|
| 646 |
+
{
|
| 647 |
+
"demos": [
|
| 648 |
+
20.376034021377563,
|
| 649 |
+
25.979355096817017,
|
| 650 |
+
39.69830060005188,
|
| 651 |
+
47.3364462852478
|
| 652 |
+
],
|
| 653 |
+
"greedy": 7.084344863891602,
|
| 654 |
+
"total": 48.586371660232544,
|
| 655 |
+
"setting": "nothink25",
|
| 656 |
+
"k": 4,
|
| 657 |
+
"query": 5,
|
| 658 |
+
"subset": "docvqa_val"
|
| 659 |
+
},
|
| 660 |
+
{
|
| 661 |
+
"demos": [
|
| 662 |
+
10.09440541267395
|
| 663 |
+
],
|
| 664 |
+
"greedy": 7.1284401416778564,
|
| 665 |
+
"total": 10.858816146850586,
|
| 666 |
+
"setting": "nocritic25",
|
| 667 |
+
"k": 1,
|
| 668 |
+
"query": 0,
|
| 669 |
+
"subset": "vizwiz_vqa_val"
|
| 670 |
+
},
|
| 671 |
+
{
|
| 672 |
+
"demos": [
|
| 673 |
+
9.839585781097412
|
| 674 |
+
],
|
| 675 |
+
"greedy": 6.931880235671997,
|
| 676 |
+
"total": 10.431512594223022,
|
| 677 |
+
"setting": "nocritic25",
|
| 678 |
+
"k": 1,
|
| 679 |
+
"query": 1,
|
| 680 |
+
"subset": "vqav2_val"
|
| 681 |
+
},
|
| 682 |
+
{
|
| 683 |
+
"demos": [
|
| 684 |
+
9.52512812614441
|
| 685 |
+
],
|
| 686 |
+
"greedy": 6.723513603210449,
|
| 687 |
+
"total": 10.16373586654663,
|
| 688 |
+
"setting": "nocritic25",
|
| 689 |
+
"k": 1,
|
| 690 |
+
"query": 2,
|
| 691 |
+
"subset": "ok_vqa_val2014"
|
| 692 |
+
},
|
| 693 |
+
{
|
| 694 |
+
"demos": [
|
| 695 |
+
10.409209489822388
|
| 696 |
+
],
|
| 697 |
+
"greedy": 6.779156446456909,
|
| 698 |
+
"total": 11.201088905334473,
|
| 699 |
+
"setting": "nocritic25",
|
| 700 |
+
"k": 1,
|
| 701 |
+
"query": 3,
|
| 702 |
+
"subset": "textvqa_val"
|
| 703 |
+
},
|
| 704 |
+
{
|
| 705 |
+
"demos": [
|
| 706 |
+
13.039257526397705
|
| 707 |
+
],
|
| 708 |
+
"greedy": 6.719432830810547,
|
| 709 |
+
"total": 13.820397138595581,
|
| 710 |
+
"setting": "nocritic25",
|
| 711 |
+
"k": 1,
|
| 712 |
+
"query": 4,
|
| 713 |
+
"subset": "chartqa"
|
| 714 |
+
},
|
| 715 |
+
{
|
| 716 |
+
"demos": [
|
| 717 |
+
13.39191722869873
|
| 718 |
+
],
|
| 719 |
+
"greedy": 6.744472503662109,
|
| 720 |
+
"total": 13.756275415420532,
|
| 721 |
+
"setting": "nocritic25",
|
| 722 |
+
"k": 1,
|
| 723 |
+
"query": 5,
|
| 724 |
+
"subset": "docvqa_val"
|
| 725 |
+
},
|
| 726 |
+
{
|
| 727 |
+
"demos": [
|
| 728 |
+
10.131142616271973,
|
| 729 |
+
10.941117525100708
|
| 730 |
+
],
|
| 731 |
+
"greedy": 6.75618314743042,
|
| 732 |
+
"total": 11.777562379837036,
|
| 733 |
+
"setting": "nocritic25",
|
| 734 |
+
"k": 2,
|
| 735 |
+
"query": 0,
|
| 736 |
+
"subset": "vizwiz_vqa_val"
|
| 737 |
+
},
|
| 738 |
+
{
|
| 739 |
+
"demos": [
|
| 740 |
+
9.828729629516602,
|
| 741 |
+
10.91353988647461
|
| 742 |
+
],
|
| 743 |
+
"greedy": 6.636314868927002,
|
| 744 |
+
"total": 11.586650609970093,
|
| 745 |
+
"setting": "nocritic25",
|
| 746 |
+
"k": 2,
|
| 747 |
+
"query": 1,
|
| 748 |
+
"subset": "vqav2_val"
|
| 749 |
+
},
|
| 750 |
+
{
|
| 751 |
+
"demos": [
|
| 752 |
+
10.159326076507568,
|
| 753 |
+
11.352014064788818
|
| 754 |
+
],
|
| 755 |
+
"greedy": 7.0619635581970215,
|
| 756 |
+
"total": 12.062993288040161,
|
| 757 |
+
"setting": "nocritic25",
|
| 758 |
+
"k": 2,
|
| 759 |
+
"query": 2,
|
| 760 |
+
"subset": "ok_vqa_val2014"
|
| 761 |
+
},
|
| 762 |
+
{
|
| 763 |
+
"demos": [
|
| 764 |
+
12.485306739807129,
|
| 765 |
+
37.578511238098145
|
| 766 |
+
],
|
| 767 |
+
"greedy": 6.796743392944336,
|
| 768 |
+
"total": 38.57980442047119,
|
| 769 |
+
"setting": "nocritic25",
|
| 770 |
+
"k": 2,
|
| 771 |
+
"query": 3,
|
| 772 |
+
"subset": "textvqa_val"
|
| 773 |
+
},
|
| 774 |
+
{
|
| 775 |
+
"demos": [
|
| 776 |
+
13.873923778533936,
|
| 777 |
+
15.028538465499878
|
| 778 |
+
],
|
| 779 |
+
"greedy": 6.755588054656982,
|
| 780 |
+
"total": 15.933726072311401,
|
| 781 |
+
"setting": "nocritic25",
|
| 782 |
+
"k": 2,
|
| 783 |
+
"query": 4,
|
| 784 |
+
"subset": "chartqa"
|
| 785 |
+
},
|
| 786 |
+
{
|
| 787 |
+
"demos": [
|
| 788 |
+
13.108296394348145,
|
| 789 |
+
32.831605195999146
|
| 790 |
+
],
|
| 791 |
+
"greedy": 6.732863426208496,
|
| 792 |
+
"total": 33.64893388748169,
|
| 793 |
+
"setting": "nocritic25",
|
| 794 |
+
"k": 2,
|
| 795 |
+
"query": 5,
|
| 796 |
+
"subset": "docvqa_val"
|
| 797 |
+
},
|
| 798 |
+
{
|
| 799 |
+
"demos": [
|
| 800 |
+
10.41283893585205,
|
| 801 |
+
12.226398944854736,
|
| 802 |
+
13.141302824020386,
|
| 803 |
+
13.47109079360962
|
| 804 |
+
],
|
| 805 |
+
"greedy": 6.717079401016235,
|
| 806 |
+
"total": 14.473924160003662,
|
| 807 |
+
"setting": "nocritic25",
|
| 808 |
+
"k": 4,
|
| 809 |
+
"query": 0,
|
| 810 |
+
"subset": "vizwiz_vqa_val"
|
| 811 |
+
},
|
| 812 |
+
{
|
| 813 |
+
"demos": [
|
| 814 |
+
10.277020931243896,
|
| 815 |
+
12.27582597732544,
|
| 816 |
+
13.355355024337769,
|
| 817 |
+
14.00516128540039
|
| 818 |
+
],
|
| 819 |
+
"greedy": 6.814017057418823,
|
| 820 |
+
"total": 14.8407723903656,
|
| 821 |
+
"setting": "nocritic25",
|
| 822 |
+
"k": 4,
|
| 823 |
+
"query": 1,
|
| 824 |
+
"subset": "vqav2_val"
|
| 825 |
+
},
|
| 826 |
+
{
|
| 827 |
+
"demos": [
|
| 828 |
+
11.423258543014526,
|
| 829 |
+
13.428069591522217,
|
| 830 |
+
14.622854709625244,
|
| 831 |
+
16.18061876296997
|
| 832 |
+
],
|
| 833 |
+
"greedy": 6.668004751205444,
|
| 834 |
+
"total": 17.06151270866394,
|
| 835 |
+
"setting": "nocritic25",
|
| 836 |
+
"k": 4,
|
| 837 |
+
"query": 2,
|
| 838 |
+
"subset": "ok_vqa_val2014"
|
| 839 |
+
},
|
| 840 |
+
{
|
| 841 |
+
"demos": [
|
| 842 |
+
10.60582423210144,
|
| 843 |
+
16.0261173248291,
|
| 844 |
+
16.700007915496826,
|
| 845 |
+
33.09624981880188
|
| 846 |
+
],
|
| 847 |
+
"greedy": 6.843786954879761,
|
| 848 |
+
"total": 34.594066858291626,
|
| 849 |
+
"setting": "nocritic25",
|
| 850 |
+
"k": 4,
|
| 851 |
+
"query": 3,
|
| 852 |
+
"subset": "textvqa_val"
|
| 853 |
+
},
|
| 854 |
+
{
|
| 855 |
+
"demos": [
|
| 856 |
+
13.488800764083862,
|
| 857 |
+
17.300123929977417,
|
| 858 |
+
18.207062244415283,
|
| 859 |
+
18.684130907058716
|
| 860 |
+
],
|
| 861 |
+
"greedy": 6.735860824584961,
|
| 862 |
+
"total": 19.977572679519653,
|
| 863 |
+
"setting": "nocritic25",
|
| 864 |
+
"k": 4,
|
| 865 |
+
"query": 4,
|
| 866 |
+
"subset": "chartqa"
|
| 867 |
+
},
|
| 868 |
+
{
|
| 869 |
+
"demos": [
|
| 870 |
+
12.050382614135742,
|
| 871 |
+
14.343275308609009,
|
| 872 |
+
15.249944925308228,
|
| 873 |
+
15.570221662521362
|
| 874 |
+
],
|
| 875 |
+
"greedy": 6.8102805614471436,
|
| 876 |
+
"total": 16.96952247619629,
|
| 877 |
+
"setting": "nocritic25",
|
| 878 |
+
"k": 4,
|
| 879 |
+
"query": 5,
|
| 880 |
+
"subset": "docvqa_val"
|
| 881 |
+
}
|
| 882 |
+
]
|
tools/results/latency_lance.json
ADDED
|
@@ -0,0 +1,446 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
[
|
| 2 |
+
{
|
| 3 |
+
"demos": [
|
| 4 |
+
7.727382183074951
|
| 5 |
+
],
|
| 6 |
+
"greedy": 4.262204647064209,
|
| 7 |
+
"total": 8.35633897781372,
|
| 8 |
+
"setting": "steps30",
|
| 9 |
+
"k": 1,
|
| 10 |
+
"query": 0,
|
| 11 |
+
"subset": "vizwiz_vqa_val"
|
| 12 |
+
},
|
| 13 |
+
{
|
| 14 |
+
"demos": [
|
| 15 |
+
5.997106552124023
|
| 16 |
+
],
|
| 17 |
+
"greedy": 4.037919521331787,
|
| 18 |
+
"total": 6.613722324371338,
|
| 19 |
+
"setting": "steps30",
|
| 20 |
+
"k": 1,
|
| 21 |
+
"query": 1,
|
| 22 |
+
"subset": "vqav2_val"
|
| 23 |
+
},
|
| 24 |
+
{
|
| 25 |
+
"demos": [
|
| 26 |
+
7.009830474853516
|
| 27 |
+
],
|
| 28 |
+
"greedy": 4.070275545120239,
|
| 29 |
+
"total": 7.65753698348999,
|
| 30 |
+
"setting": "steps30",
|
| 31 |
+
"k": 1,
|
| 32 |
+
"query": 2,
|
| 33 |
+
"subset": "ok_vqa_val2014"
|
| 34 |
+
},
|
| 35 |
+
{
|
| 36 |
+
"demos": [
|
| 37 |
+
14.579021215438843
|
| 38 |
+
],
|
| 39 |
+
"greedy": 3.9722044467926025,
|
| 40 |
+
"total": 15.173923015594482,
|
| 41 |
+
"setting": "steps30",
|
| 42 |
+
"k": 1,
|
| 43 |
+
"query": 3,
|
| 44 |
+
"subset": "textvqa_val"
|
| 45 |
+
},
|
| 46 |
+
{
|
| 47 |
+
"demos": [
|
| 48 |
+
7.009682893753052
|
| 49 |
+
],
|
| 50 |
+
"greedy": 4.298367500305176,
|
| 51 |
+
"total": 7.6930906772613525,
|
| 52 |
+
"setting": "steps30",
|
| 53 |
+
"k": 1,
|
| 54 |
+
"query": 4,
|
| 55 |
+
"subset": "chartqa"
|
| 56 |
+
},
|
| 57 |
+
{
|
| 58 |
+
"demos": [
|
| 59 |
+
9.488945960998535
|
| 60 |
+
],
|
| 61 |
+
"greedy": 4.068331241607666,
|
| 62 |
+
"total": 10.129173040390015,
|
| 63 |
+
"setting": "steps30",
|
| 64 |
+
"k": 1,
|
| 65 |
+
"query": 5,
|
| 66 |
+
"subset": "docvqa_val"
|
| 67 |
+
},
|
| 68 |
+
{
|
| 69 |
+
"demos": [
|
| 70 |
+
8.336636781692505,
|
| 71 |
+
10.828430652618408
|
| 72 |
+
],
|
| 73 |
+
"greedy": 4.046021223068237,
|
| 74 |
+
"total": 11.519325733184814,
|
| 75 |
+
"setting": "steps30",
|
| 76 |
+
"k": 2,
|
| 77 |
+
"query": 0,
|
| 78 |
+
"subset": "vizwiz_vqa_val"
|
| 79 |
+
},
|
| 80 |
+
{
|
| 81 |
+
"demos": [
|
| 82 |
+
8.213298082351685,
|
| 83 |
+
10.596574068069458
|
| 84 |
+
],
|
| 85 |
+
"greedy": 4.0463056564331055,
|
| 86 |
+
"total": 11.278587102890015,
|
| 87 |
+
"setting": "steps30",
|
| 88 |
+
"k": 2,
|
| 89 |
+
"query": 1,
|
| 90 |
+
"subset": "vqav2_val"
|
| 91 |
+
},
|
| 92 |
+
{
|
| 93 |
+
"demos": [
|
| 94 |
+
8.57367467880249,
|
| 95 |
+
10.886900663375854
|
| 96 |
+
],
|
| 97 |
+
"greedy": 4.036629676818848,
|
| 98 |
+
"total": 11.565510511398315,
|
| 99 |
+
"setting": "steps30",
|
| 100 |
+
"k": 2,
|
| 101 |
+
"query": 2,
|
| 102 |
+
"subset": "ok_vqa_val2014"
|
| 103 |
+
},
|
| 104 |
+
{
|
| 105 |
+
"demos": [
|
| 106 |
+
11.43087124824524,
|
| 107 |
+
13.566925525665283
|
| 108 |
+
],
|
| 109 |
+
"greedy": 3.9821929931640625,
|
| 110 |
+
"total": 14.221354246139526,
|
| 111 |
+
"setting": "steps30",
|
| 112 |
+
"k": 2,
|
| 113 |
+
"query": 3,
|
| 114 |
+
"subset": "textvqa_val"
|
| 115 |
+
},
|
| 116 |
+
{
|
| 117 |
+
"demos": [
|
| 118 |
+
9.571406841278076,
|
| 119 |
+
10.061784029006958
|
| 120 |
+
],
|
| 121 |
+
"greedy": 4.069300174713135,
|
| 122 |
+
"total": 10.802971124649048,
|
| 123 |
+
"setting": "steps30",
|
| 124 |
+
"k": 2,
|
| 125 |
+
"query": 4,
|
| 126 |
+
"subset": "chartqa"
|
| 127 |
+
},
|
| 128 |
+
{
|
| 129 |
+
"demos": [
|
| 130 |
+
8.667640209197998,
|
| 131 |
+
11.335434436798096
|
| 132 |
+
],
|
| 133 |
+
"greedy": 4.0576910972595215,
|
| 134 |
+
"total": 12.056833982467651,
|
| 135 |
+
"setting": "steps30",
|
| 136 |
+
"k": 2,
|
| 137 |
+
"query": 5,
|
| 138 |
+
"subset": "docvqa_val"
|
| 139 |
+
},
|
| 140 |
+
{
|
| 141 |
+
"demos": [
|
| 142 |
+
9.209861993789673,
|
| 143 |
+
10.122817754745483,
|
| 144 |
+
11.41738224029541,
|
| 145 |
+
11.986809253692627
|
| 146 |
+
],
|
| 147 |
+
"greedy": 4.213792562484741,
|
| 148 |
+
"total": 13.086158275604248,
|
| 149 |
+
"setting": "steps30",
|
| 150 |
+
"k": 4,
|
| 151 |
+
"query": 0,
|
| 152 |
+
"subset": "vizwiz_vqa_val"
|
| 153 |
+
},
|
| 154 |
+
{
|
| 155 |
+
"demos": [
|
| 156 |
+
8.620089054107666,
|
| 157 |
+
11.2899010181427,
|
| 158 |
+
12.844002962112427,
|
| 159 |
+
13.42651104927063
|
| 160 |
+
],
|
| 161 |
+
"greedy": 4.100450038909912,
|
| 162 |
+
"total": 14.261920928955078,
|
| 163 |
+
"setting": "steps30",
|
| 164 |
+
"k": 4,
|
| 165 |
+
"query": 1,
|
| 166 |
+
"subset": "vqav2_val"
|
| 167 |
+
},
|
| 168 |
+
{
|
| 169 |
+
"demos": [
|
| 170 |
+
7.910981893539429,
|
| 171 |
+
10.294513940811157,
|
| 172 |
+
10.294527292251587,
|
| 173 |
+
11.984448909759521
|
| 174 |
+
],
|
| 175 |
+
"greedy": 4.040694236755371,
|
| 176 |
+
"total": 12.851784229278564,
|
| 177 |
+
"setting": "steps30",
|
| 178 |
+
"k": 4,
|
| 179 |
+
"query": 2,
|
| 180 |
+
"subset": "ok_vqa_val2014"
|
| 181 |
+
},
|
| 182 |
+
{
|
| 183 |
+
"demos": [
|
| 184 |
+
8.954644203186035,
|
| 185 |
+
12.185572862625122,
|
| 186 |
+
12.858001470565796,
|
| 187 |
+
15.00017762184143
|
| 188 |
+
],
|
| 189 |
+
"greedy": 3.9677560329437256,
|
| 190 |
+
"total": 15.794921159744263,
|
| 191 |
+
"setting": "steps30",
|
| 192 |
+
"k": 4,
|
| 193 |
+
"query": 3,
|
| 194 |
+
"subset": "textvqa_val"
|
| 195 |
+
},
|
| 196 |
+
{
|
| 197 |
+
"demos": [
|
| 198 |
+
7.161853790283203,
|
| 199 |
+
10.45381784439087,
|
| 200 |
+
11.22620153427124,
|
| 201 |
+
13.835843801498413
|
| 202 |
+
],
|
| 203 |
+
"greedy": 4.065622329711914,
|
| 204 |
+
"total": 14.741939306259155,
|
| 205 |
+
"setting": "steps30",
|
| 206 |
+
"k": 4,
|
| 207 |
+
"query": 4,
|
| 208 |
+
"subset": "chartqa"
|
| 209 |
+
},
|
| 210 |
+
{
|
| 211 |
+
"demos": [
|
| 212 |
+
9.17311716079712,
|
| 213 |
+
9.261759042739868,
|
| 214 |
+
9.878037929534912,
|
| 215 |
+
10.475617170333862
|
| 216 |
+
],
|
| 217 |
+
"greedy": 4.253077268600464,
|
| 218 |
+
"total": 11.615146398544312,
|
| 219 |
+
"setting": "steps30",
|
| 220 |
+
"k": 4,
|
| 221 |
+
"query": 5,
|
| 222 |
+
"subset": "docvqa_val"
|
| 223 |
+
},
|
| 224 |
+
{
|
| 225 |
+
"demos": [
|
| 226 |
+
9.40204405784607
|
| 227 |
+
],
|
| 228 |
+
"greedy": 4.1254284381866455,
|
| 229 |
+
"total": 10.027106285095215,
|
| 230 |
+
"setting": "steps20",
|
| 231 |
+
"k": 1,
|
| 232 |
+
"query": 0,
|
| 233 |
+
"subset": "vizwiz_vqa_val"
|
| 234 |
+
},
|
| 235 |
+
{
|
| 236 |
+
"demos": [
|
| 237 |
+
7.260383367538452
|
| 238 |
+
],
|
| 239 |
+
"greedy": 4.095474481582642,
|
| 240 |
+
"total": 7.883589267730713,
|
| 241 |
+
"setting": "steps20",
|
| 242 |
+
"k": 1,
|
| 243 |
+
"query": 1,
|
| 244 |
+
"subset": "vqav2_val"
|
| 245 |
+
},
|
| 246 |
+
{
|
| 247 |
+
"demos": [
|
| 248 |
+
8.731263637542725
|
| 249 |
+
],
|
| 250 |
+
"greedy": 4.012105703353882,
|
| 251 |
+
"total": 9.378830432891846,
|
| 252 |
+
"setting": "steps20",
|
| 253 |
+
"k": 1,
|
| 254 |
+
"query": 2,
|
| 255 |
+
"subset": "ok_vqa_val2014"
|
| 256 |
+
},
|
| 257 |
+
{
|
| 258 |
+
"demos": [
|
| 259 |
+
11.81164813041687
|
| 260 |
+
],
|
| 261 |
+
"greedy": 4.001217842102051,
|
| 262 |
+
"total": 12.418342351913452,
|
| 263 |
+
"setting": "steps20",
|
| 264 |
+
"k": 1,
|
| 265 |
+
"query": 3,
|
| 266 |
+
"subset": "textvqa_val"
|
| 267 |
+
},
|
| 268 |
+
{
|
| 269 |
+
"demos": [
|
| 270 |
+
6.208858489990234
|
| 271 |
+
],
|
| 272 |
+
"greedy": 4.017943859100342,
|
| 273 |
+
"total": 6.826139211654663,
|
| 274 |
+
"setting": "steps20",
|
| 275 |
+
"k": 1,
|
| 276 |
+
"query": 4,
|
| 277 |
+
"subset": "chartqa"
|
| 278 |
+
},
|
| 279 |
+
{
|
| 280 |
+
"demos": [
|
| 281 |
+
9.363977670669556
|
| 282 |
+
],
|
| 283 |
+
"greedy": 4.0784912109375,
|
| 284 |
+
"total": 10.00316596031189,
|
| 285 |
+
"setting": "steps20",
|
| 286 |
+
"k": 1,
|
| 287 |
+
"query": 5,
|
| 288 |
+
"subset": "docvqa_val"
|
| 289 |
+
},
|
| 290 |
+
{
|
| 291 |
+
"demos": [
|
| 292 |
+
8.895264863967896,
|
| 293 |
+
11.027926921844482
|
| 294 |
+
],
|
| 295 |
+
"greedy": 4.093302011489868,
|
| 296 |
+
"total": 11.739497900009155,
|
| 297 |
+
"setting": "steps20",
|
| 298 |
+
"k": 2,
|
| 299 |
+
"query": 0,
|
| 300 |
+
"subset": "vizwiz_vqa_val"
|
| 301 |
+
},
|
| 302 |
+
{
|
| 303 |
+
"demos": [
|
| 304 |
+
8.46868896484375,
|
| 305 |
+
8.78795862197876
|
| 306 |
+
],
|
| 307 |
+
"greedy": 4.05513858795166,
|
| 308 |
+
"total": 9.470828771591187,
|
| 309 |
+
"setting": "steps20",
|
| 310 |
+
"k": 2,
|
| 311 |
+
"query": 1,
|
| 312 |
+
"subset": "vqav2_val"
|
| 313 |
+
},
|
| 314 |
+
{
|
| 315 |
+
"demos": [
|
| 316 |
+
7.852817535400391,
|
| 317 |
+
8.184361457824707
|
| 318 |
+
],
|
| 319 |
+
"greedy": 4.0655388832092285,
|
| 320 |
+
"total": 8.493193626403809,
|
| 321 |
+
"setting": "steps20",
|
| 322 |
+
"k": 2,
|
| 323 |
+
"query": 2,
|
| 324 |
+
"subset": "ok_vqa_val2014"
|
| 325 |
+
},
|
| 326 |
+
{
|
| 327 |
+
"demos": [
|
| 328 |
+
7.7487473487854,
|
| 329 |
+
9.187946081161499
|
| 330 |
+
],
|
| 331 |
+
"greedy": 4.10131573677063,
|
| 332 |
+
"total": 9.855353355407715,
|
| 333 |
+
"setting": "steps20",
|
| 334 |
+
"k": 2,
|
| 335 |
+
"query": 3,
|
| 336 |
+
"subset": "textvqa_val"
|
| 337 |
+
},
|
| 338 |
+
{
|
| 339 |
+
"demos": [
|
| 340 |
+
8.990880966186523,
|
| 341 |
+
9.044522285461426
|
| 342 |
+
],
|
| 343 |
+
"greedy": 4.088159084320068,
|
| 344 |
+
"total": 9.791136264801025,
|
| 345 |
+
"setting": "steps20",
|
| 346 |
+
"k": 2,
|
| 347 |
+
"query": 4,
|
| 348 |
+
"subset": "chartqa"
|
| 349 |
+
},
|
| 350 |
+
{
|
| 351 |
+
"demos": [
|
| 352 |
+
6.566630125045776,
|
| 353 |
+
8.522355079650879
|
| 354 |
+
],
|
| 355 |
+
"greedy": 4.04767918586731,
|
| 356 |
+
"total": 9.243301391601562,
|
| 357 |
+
"setting": "steps20",
|
| 358 |
+
"k": 2,
|
| 359 |
+
"query": 5,
|
| 360 |
+
"subset": "docvqa_val"
|
| 361 |
+
},
|
| 362 |
+
{
|
| 363 |
+
"demos": [
|
| 364 |
+
8.488969802856445,
|
| 365 |
+
10.305099725723267,
|
| 366 |
+
10.305113554000854,
|
| 367 |
+
10.661272048950195
|
| 368 |
+
],
|
| 369 |
+
"greedy": 3.986391544342041,
|
| 370 |
+
"total": 11.507165431976318,
|
| 371 |
+
"setting": "steps20",
|
| 372 |
+
"k": 4,
|
| 373 |
+
"query": 0,
|
| 374 |
+
"subset": "vizwiz_vqa_val"
|
| 375 |
+
},
|
| 376 |
+
{
|
| 377 |
+
"demos": [
|
| 378 |
+
8.554109811782837,
|
| 379 |
+
10.155081510543823,
|
| 380 |
+
10.1550931930542,
|
| 381 |
+
10.528407335281372
|
| 382 |
+
],
|
| 383 |
+
"greedy": 4.004652976989746,
|
| 384 |
+
"total": 11.368907690048218,
|
| 385 |
+
"setting": "steps20",
|
| 386 |
+
"k": 4,
|
| 387 |
+
"query": 1,
|
| 388 |
+
"subset": "vqav2_val"
|
| 389 |
+
},
|
| 390 |
+
{
|
| 391 |
+
"demos": [
|
| 392 |
+
7.51558256149292,
|
| 393 |
+
8.081176280975342,
|
| 394 |
+
9.272027015686035,
|
| 395 |
+
9.97708535194397
|
| 396 |
+
],
|
| 397 |
+
"greedy": 4.020036935806274,
|
| 398 |
+
"total": 10.843841552734375,
|
| 399 |
+
"setting": "steps20",
|
| 400 |
+
"k": 4,
|
| 401 |
+
"query": 2,
|
| 402 |
+
"subset": "ok_vqa_val2014"
|
| 403 |
+
},
|
| 404 |
+
{
|
| 405 |
+
"demos": [
|
| 406 |
+
7.948291540145874,
|
| 407 |
+
8.477357387542725,
|
| 408 |
+
10.59636402130127,
|
| 409 |
+
12.149571418762207
|
| 410 |
+
],
|
| 411 |
+
"greedy": 3.9414212703704834,
|
| 412 |
+
"total": 12.953968048095703,
|
| 413 |
+
"setting": "steps20",
|
| 414 |
+
"k": 4,
|
| 415 |
+
"query": 3,
|
| 416 |
+
"subset": "textvqa_val"
|
| 417 |
+
},
|
| 418 |
+
{
|
| 419 |
+
"demos": [
|
| 420 |
+
7.972762107849121,
|
| 421 |
+
9.431615591049194,
|
| 422 |
+
9.854741096496582,
|
| 423 |
+
12.329490423202515
|
| 424 |
+
],
|
| 425 |
+
"greedy": 4.051168918609619,
|
| 426 |
+
"total": 13.170568704605103,
|
| 427 |
+
"setting": "steps20",
|
| 428 |
+
"k": 4,
|
| 429 |
+
"query": 4,
|
| 430 |
+
"subset": "chartqa"
|
| 431 |
+
},
|
| 432 |
+
{
|
| 433 |
+
"demos": [
|
| 434 |
+
8.014616250991821,
|
| 435 |
+
9.337750434875488,
|
| 436 |
+
9.904663562774658,
|
| 437 |
+
13.779886245727539
|
| 438 |
+
],
|
| 439 |
+
"greedy": 4.087887763977051,
|
| 440 |
+
"total": 14.651642084121704,
|
| 441 |
+
"setting": "steps20",
|
| 442 |
+
"k": 4,
|
| 443 |
+
"query": 5,
|
| 444 |
+
"subset": "docvqa_val"
|
| 445 |
+
}
|
| 446 |
+
]
|
tools/results/latency_qwen.json
ADDED
|
@@ -0,0 +1,347 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
[
|
| 2 |
+
{
|
| 3 |
+
"demos": [
|
| 4 |
+
30.282771110534668
|
| 5 |
+
],
|
| 6 |
+
"greedy": 12.503609418869019,
|
| 7 |
+
"total": 31.559006452560425,
|
| 8 |
+
"setting": "natural",
|
| 9 |
+
"k": 1,
|
| 10 |
+
"query": 0,
|
| 11 |
+
"subset": "vizwiz_vqa_val"
|
| 12 |
+
},
|
| 13 |
+
{
|
| 14 |
+
"demos": [
|
| 15 |
+
24.56417155265808
|
| 16 |
+
],
|
| 17 |
+
"greedy": 12.057083129882812,
|
| 18 |
+
"total": 25.436967611312866,
|
| 19 |
+
"setting": "natural",
|
| 20 |
+
"k": 1,
|
| 21 |
+
"query": 1,
|
| 22 |
+
"subset": "vqav2_val"
|
| 23 |
+
},
|
| 24 |
+
{
|
| 25 |
+
"demos": [
|
| 26 |
+
27.104141235351562
|
| 27 |
+
],
|
| 28 |
+
"greedy": 11.834922552108765,
|
| 29 |
+
"total": 28.232157707214355,
|
| 30 |
+
"setting": "natural",
|
| 31 |
+
"k": 1,
|
| 32 |
+
"query": 2,
|
| 33 |
+
"subset": "ok_vqa_val2014"
|
| 34 |
+
},
|
| 35 |
+
{
|
| 36 |
+
"demos": [
|
| 37 |
+
30.065296173095703
|
| 38 |
+
],
|
| 39 |
+
"greedy": 11.83328104019165,
|
| 40 |
+
"total": 31.18956470489502,
|
| 41 |
+
"setting": "natural",
|
| 42 |
+
"k": 1,
|
| 43 |
+
"query": 3,
|
| 44 |
+
"subset": "textvqa_val"
|
| 45 |
+
},
|
| 46 |
+
{
|
| 47 |
+
"demos": [
|
| 48 |
+
28.98874258995056
|
| 49 |
+
],
|
| 50 |
+
"greedy": 12.290349006652832,
|
| 51 |
+
"total": 30.4953191280365,
|
| 52 |
+
"setting": "natural",
|
| 53 |
+
"k": 1,
|
| 54 |
+
"query": 4,
|
| 55 |
+
"subset": "chartqa"
|
| 56 |
+
},
|
| 57 |
+
{
|
| 58 |
+
"demos": [
|
| 59 |
+
49.92998290061951,
|
| 60 |
+
49.93000912666321
|
| 61 |
+
],
|
| 62 |
+
"greedy": 12.474754333496094,
|
| 63 |
+
"total": 51.23621082305908,
|
| 64 |
+
"setting": "natural",
|
| 65 |
+
"k": 2,
|
| 66 |
+
"query": 0,
|
| 67 |
+
"subset": "vizwiz_vqa_val"
|
| 68 |
+
},
|
| 69 |
+
{
|
| 70 |
+
"demos": [
|
| 71 |
+
39.91589426994324,
|
| 72 |
+
39.91593027114868
|
| 73 |
+
],
|
| 74 |
+
"greedy": 11.602417945861816,
|
| 75 |
+
"total": 40.8339729309082,
|
| 76 |
+
"setting": "natural",
|
| 77 |
+
"k": 2,
|
| 78 |
+
"query": 1,
|
| 79 |
+
"subset": "vqav2_val"
|
| 80 |
+
},
|
| 81 |
+
{
|
| 82 |
+
"demos": [
|
| 83 |
+
44.16777563095093,
|
| 84 |
+
44.16781187057495
|
| 85 |
+
],
|
| 86 |
+
"greedy": 11.90004849433899,
|
| 87 |
+
"total": 45.30206370353699,
|
| 88 |
+
"setting": "natural",
|
| 89 |
+
"k": 2,
|
| 90 |
+
"query": 2,
|
| 91 |
+
"subset": "ok_vqa_val2014"
|
| 92 |
+
},
|
| 93 |
+
{
|
| 94 |
+
"demos": [
|
| 95 |
+
39.3218777179718,
|
| 96 |
+
39.32190752029419
|
| 97 |
+
],
|
| 98 |
+
"greedy": 12.019229888916016,
|
| 99 |
+
"total": 40.492233753204346,
|
| 100 |
+
"setting": "natural",
|
| 101 |
+
"k": 2,
|
| 102 |
+
"query": 3,
|
| 103 |
+
"subset": "textvqa_val"
|
| 104 |
+
},
|
| 105 |
+
{
|
| 106 |
+
"demos": [
|
| 107 |
+
102.98197150230408,
|
| 108 |
+
102.9820077419281
|
| 109 |
+
],
|
| 110 |
+
"greedy": 11.957572221755981,
|
| 111 |
+
"total": 104.37802076339722,
|
| 112 |
+
"setting": "natural",
|
| 113 |
+
"k": 2,
|
| 114 |
+
"query": 4,
|
| 115 |
+
"subset": "chartqa"
|
| 116 |
+
},
|
| 117 |
+
{
|
| 118 |
+
"demos": [
|
| 119 |
+
34.93216681480408
|
| 120 |
+
],
|
| 121 |
+
"greedy": 12.076660633087158,
|
| 122 |
+
"total": 36.01313900947571,
|
| 123 |
+
"setting": "skills",
|
| 124 |
+
"k": 1,
|
| 125 |
+
"query": 0,
|
| 126 |
+
"subset": "vizwiz_vqa_val"
|
| 127 |
+
},
|
| 128 |
+
{
|
| 129 |
+
"demos": [
|
| 130 |
+
26.665636777877808
|
| 131 |
+
],
|
| 132 |
+
"greedy": 11.77315878868103,
|
| 133 |
+
"total": 27.53944206237793,
|
| 134 |
+
"setting": "skills",
|
| 135 |
+
"k": 1,
|
| 136 |
+
"query": 1,
|
| 137 |
+
"subset": "vqav2_val"
|
| 138 |
+
},
|
| 139 |
+
{
|
| 140 |
+
"demos": [
|
| 141 |
+
30.65152359008789
|
| 142 |
+
],
|
| 143 |
+
"greedy": 11.954692363739014,
|
| 144 |
+
"total": 31.91823172569275,
|
| 145 |
+
"setting": "skills",
|
| 146 |
+
"k": 1,
|
| 147 |
+
"query": 2,
|
| 148 |
+
"subset": "ok_vqa_val2014"
|
| 149 |
+
},
|
| 150 |
+
{
|
| 151 |
+
"demos": [
|
| 152 |
+
32.95020818710327
|
| 153 |
+
],
|
| 154 |
+
"greedy": 11.763028144836426,
|
| 155 |
+
"total": 35.805758476257324,
|
| 156 |
+
"setting": "skills",
|
| 157 |
+
"k": 1,
|
| 158 |
+
"query": 3,
|
| 159 |
+
"subset": "textvqa_val"
|
| 160 |
+
},
|
| 161 |
+
{
|
| 162 |
+
"demos": [
|
| 163 |
+
30.482857704162598
|
| 164 |
+
],
|
| 165 |
+
"greedy": 12.288076162338257,
|
| 166 |
+
"total": 31.9945011138916,
|
| 167 |
+
"setting": "skills",
|
| 168 |
+
"k": 1,
|
| 169 |
+
"query": 4,
|
| 170 |
+
"subset": "chartqa"
|
| 171 |
+
},
|
| 172 |
+
{
|
| 173 |
+
"demos": [
|
| 174 |
+
47.012778997421265,
|
| 175 |
+
47.012813568115234
|
| 176 |
+
],
|
| 177 |
+
"greedy": 11.944422960281372,
|
| 178 |
+
"total": 48.55853629112244,
|
| 179 |
+
"setting": "skills",
|
| 180 |
+
"k": 2,
|
| 181 |
+
"query": 0,
|
| 182 |
+
"subset": "vizwiz_vqa_val"
|
| 183 |
+
},
|
| 184 |
+
{
|
| 185 |
+
"demos": [
|
| 186 |
+
43.62376880645752,
|
| 187 |
+
44.03544592857361
|
| 188 |
+
],
|
| 189 |
+
"greedy": 11.559141635894775,
|
| 190 |
+
"total": 44.94946813583374,
|
| 191 |
+
"setting": "skills",
|
| 192 |
+
"k": 2,
|
| 193 |
+
"query": 1,
|
| 194 |
+
"subset": "vqav2_val"
|
| 195 |
+
},
|
| 196 |
+
{
|
| 197 |
+
"demos": [
|
| 198 |
+
44.78076720237732,
|
| 199 |
+
44.78083920478821
|
| 200 |
+
],
|
| 201 |
+
"greedy": 12.2094886302948,
|
| 202 |
+
"total": 45.610127210617065,
|
| 203 |
+
"setting": "skills",
|
| 204 |
+
"k": 2,
|
| 205 |
+
"query": 2,
|
| 206 |
+
"subset": "ok_vqa_val2014"
|
| 207 |
+
},
|
| 208 |
+
{
|
| 209 |
+
"demos": [
|
| 210 |
+
39.1291127204895,
|
| 211 |
+
39.12914180755615
|
| 212 |
+
],
|
| 213 |
+
"greedy": 11.88507342338562,
|
| 214 |
+
"total": 40.32528471946716,
|
| 215 |
+
"setting": "skills",
|
| 216 |
+
"k": 2,
|
| 217 |
+
"query": 3,
|
| 218 |
+
"subset": "textvqa_val"
|
| 219 |
+
},
|
| 220 |
+
{
|
| 221 |
+
"demos": [
|
| 222 |
+
90.45457100868225,
|
| 223 |
+
90.8869194984436
|
| 224 |
+
],
|
| 225 |
+
"greedy": 12.611650466918945,
|
| 226 |
+
"total": 92.93272161483765,
|
| 227 |
+
"setting": "skills",
|
| 228 |
+
"k": 2,
|
| 229 |
+
"query": 4,
|
| 230 |
+
"subset": "chartqa"
|
| 231 |
+
},
|
| 232 |
+
{
|
| 233 |
+
"demos": [
|
| 234 |
+
75.17404270172119
|
| 235 |
+
],
|
| 236 |
+
"greedy": 12.442910432815552,
|
| 237 |
+
"total": 76.42014288902283,
|
| 238 |
+
"setting": "critic",
|
| 239 |
+
"k": 1,
|
| 240 |
+
"query": 0,
|
| 241 |
+
"subset": "vizwiz_vqa_val"
|
| 242 |
+
},
|
| 243 |
+
{
|
| 244 |
+
"demos": [
|
| 245 |
+
70.33047604560852
|
| 246 |
+
],
|
| 247 |
+
"greedy": 11.583081483840942,
|
| 248 |
+
"total": 71.14328861236572,
|
| 249 |
+
"setting": "critic",
|
| 250 |
+
"k": 1,
|
| 251 |
+
"query": 1,
|
| 252 |
+
"subset": "vqav2_val"
|
| 253 |
+
},
|
| 254 |
+
{
|
| 255 |
+
"demos": [
|
| 256 |
+
71.25398755073547
|
| 257 |
+
],
|
| 258 |
+
"greedy": 12.159323930740356,
|
| 259 |
+
"total": 72.36944890022278,
|
| 260 |
+
"setting": "critic",
|
| 261 |
+
"k": 1,
|
| 262 |
+
"query": 2,
|
| 263 |
+
"subset": "ok_vqa_val2014"
|
| 264 |
+
},
|
| 265 |
+
{
|
| 266 |
+
"demos": [
|
| 267 |
+
349.18033623695374
|
| 268 |
+
],
|
| 269 |
+
"greedy": 11.960462808609009,
|
| 270 |
+
"total": 350.2297682762146,
|
| 271 |
+
"setting": "critic",
|
| 272 |
+
"k": 1,
|
| 273 |
+
"query": 3,
|
| 274 |
+
"subset": "textvqa_val"
|
| 275 |
+
},
|
| 276 |
+
{
|
| 277 |
+
"demos": [
|
| 278 |
+
79.8816294670105
|
| 279 |
+
],
|
| 280 |
+
"greedy": 12.26366114616394,
|
| 281 |
+
"total": 81.34397840499878,
|
| 282 |
+
"setting": "critic",
|
| 283 |
+
"k": 1,
|
| 284 |
+
"query": 4,
|
| 285 |
+
"subset": "chartqa"
|
| 286 |
+
},
|
| 287 |
+
{
|
| 288 |
+
"demos": [
|
| 289 |
+
214.83966445922852,
|
| 290 |
+
246.23991990089417
|
| 291 |
+
],
|
| 292 |
+
"greedy": 11.864954471588135,
|
| 293 |
+
"total": 247.5113034248352,
|
| 294 |
+
"setting": "critic",
|
| 295 |
+
"k": 2,
|
| 296 |
+
"query": 0,
|
| 297 |
+
"subset": "vizwiz_vqa_val"
|
| 298 |
+
},
|
| 299 |
+
{
|
| 300 |
+
"demos": [
|
| 301 |
+
111.94581985473633,
|
| 302 |
+
112.84730768203735
|
| 303 |
+
],
|
| 304 |
+
"greedy": 11.496987581253052,
|
| 305 |
+
"total": 113.7428183555603,
|
| 306 |
+
"setting": "critic",
|
| 307 |
+
"k": 2,
|
| 308 |
+
"query": 1,
|
| 309 |
+
"subset": "vqav2_val"
|
| 310 |
+
},
|
| 311 |
+
{
|
| 312 |
+
"demos": [
|
| 313 |
+
106.2651150226593,
|
| 314 |
+
107.3609848022461
|
| 315 |
+
],
|
| 316 |
+
"greedy": 11.693379402160645,
|
| 317 |
+
"total": 108.45507740974426,
|
| 318 |
+
"setting": "critic",
|
| 319 |
+
"k": 2,
|
| 320 |
+
"query": 2,
|
| 321 |
+
"subset": "ok_vqa_val2014"
|
| 322 |
+
},
|
| 323 |
+
{
|
| 324 |
+
"demos": [
|
| 325 |
+
97.19141793251038,
|
| 326 |
+
402.90441393852234
|
| 327 |
+
],
|
| 328 |
+
"greedy": 11.953308343887329,
|
| 329 |
+
"total": 404.0378088951111,
|
| 330 |
+
"setting": "critic",
|
| 331 |
+
"k": 2,
|
| 332 |
+
"query": 3,
|
| 333 |
+
"subset": "textvqa_val"
|
| 334 |
+
},
|
| 335 |
+
{
|
| 336 |
+
"demos": [
|
| 337 |
+
125.35352444648743,
|
| 338 |
+
164.42774605751038
|
| 339 |
+
],
|
| 340 |
+
"greedy": 11.947067499160767,
|
| 341 |
+
"total": 166.28691792488098,
|
| 342 |
+
"setting": "critic",
|
| 343 |
+
"k": 2,
|
| 344 |
+
"query": 4,
|
| 345 |
+
"subset": "chartqa"
|
| 346 |
+
}
|
| 347 |
+
]
|
tools/results/presets_bagel.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|