monurcan commited on
Commit
a64255e
·
verified ·
1 Parent(s): 908f08c

SIMIT demo: BAGEL, Lance, Qwen3.8-27B-FP8 + FLUX.2-klein on ZeroGPU

Browse files
This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. .gitattributes +21 -0
  2. README.md +70 -6
  3. app.py +297 -0
  4. assets/architect.webp +0 -0
  5. assets/artist.webp +0 -0
  6. assets/critic.webp +0 -0
  7. assets/improved.webp +0 -0
  8. assets/initial.webp +0 -0
  9. assets/router.webp +0 -0
  10. assets/solver.webp +0 -0
  11. assets/style.css +32 -0
  12. assets/synthesizer.webp +0 -0
  13. demo_models.py +251 -0
  14. examples/bagel_amazon/demo_0.png +3 -0
  15. examples/bagel_amazon/demo_1.png +3 -0
  16. examples/bagel_amazon/query.png +3 -0
  17. examples/bagel_infovqa_val_449/demo_0.png +0 -0
  18. examples/bagel_infovqa_val_449/query.jpg +3 -0
  19. examples/bagel_ok_vqa_val2014_492_always/demo_0.png +3 -0
  20. examples/bagel_ok_vqa_val2014_492_always/demo_1.png +3 -0
  21. examples/bagel_ok_vqa_val2014_492_always/query.jpg +0 -0
  22. examples/bagel_vizwiz_vqa_val_345_always/demo_0.png +3 -0
  23. examples/bagel_vizwiz_vqa_val_345_always/demo_1.png +3 -0
  24. examples/bagel_vizwiz_vqa_val_345_always/query.jpg +3 -0
  25. examples/bagel_vizwiz_vqa_val_47_always/demo_0.png +3 -0
  26. examples/bagel_vizwiz_vqa_val_47_always/demo_1.png +3 -0
  27. examples/bagel_vizwiz_vqa_val_47_always/query.jpg +3 -0
  28. examples/bagel_vizwiz_vqa_val_87_always/demo_0.png +3 -0
  29. examples/bagel_vizwiz_vqa_val_87_always/demo_1.png +3 -0
  30. examples/bagel_vizwiz_vqa_val_87_always/query.jpg +3 -0
  31. examples/index.json +274 -0
  32. examples/lance_ok_vqa_val2014_253/demo_0.png +3 -0
  33. examples/lance_ok_vqa_val2014_253/query.jpg +3 -0
  34. examples/lance_ok_vqa_val2014_8_always/demo_0.png +3 -0
  35. examples/lance_ok_vqa_val2014_8_always/demo_1.png +3 -0
  36. examples/lance_ok_vqa_val2014_8_always/demo_2.png +3 -0
  37. examples/lance_ok_vqa_val2014_8_always/demo_3.png +3 -0
  38. examples/lance_ok_vqa_val2014_8_always/query.jpg +0 -0
  39. packages.txt +18 -0
  40. presets.json +233 -0
  41. requirements.txt +4 -0
  42. tools/build_presets.py +91 -0
  43. tools/fork_test.py +55 -0
  44. tools/general_qa.py +70 -0
  45. tools/latency.py +74 -0
  46. tools/make_examples.py +121 -0
  47. tools/results/latency_bagel.json +882 -0
  48. tools/results/latency_lance.json +446 -0
  49. tools/results/latency_qwen.json +347 -0
  50. tools/results/presets_bagel.json +0 -0
.gitattributes CHANGED
@@ -33,3 +33,24 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ examples/bagel_amazon/demo_0.png filter=lfs diff=lfs merge=lfs -text
37
+ examples/bagel_amazon/demo_1.png filter=lfs diff=lfs merge=lfs -text
38
+ examples/bagel_amazon/query.png filter=lfs diff=lfs merge=lfs -text
39
+ examples/bagel_infovqa_val_449/query.jpg filter=lfs diff=lfs merge=lfs -text
40
+ examples/bagel_ok_vqa_val2014_492_always/demo_0.png filter=lfs diff=lfs merge=lfs -text
41
+ examples/bagel_ok_vqa_val2014_492_always/demo_1.png filter=lfs diff=lfs merge=lfs -text
42
+ examples/bagel_vizwiz_vqa_val_345_always/demo_0.png filter=lfs diff=lfs merge=lfs -text
43
+ examples/bagel_vizwiz_vqa_val_345_always/demo_1.png filter=lfs diff=lfs merge=lfs -text
44
+ examples/bagel_vizwiz_vqa_val_345_always/query.jpg filter=lfs diff=lfs merge=lfs -text
45
+ examples/bagel_vizwiz_vqa_val_47_always/demo_0.png filter=lfs diff=lfs merge=lfs -text
46
+ examples/bagel_vizwiz_vqa_val_47_always/demo_1.png filter=lfs diff=lfs merge=lfs -text
47
+ examples/bagel_vizwiz_vqa_val_47_always/query.jpg filter=lfs diff=lfs merge=lfs -text
48
+ examples/bagel_vizwiz_vqa_val_87_always/demo_0.png filter=lfs diff=lfs merge=lfs -text
49
+ examples/bagel_vizwiz_vqa_val_87_always/demo_1.png filter=lfs diff=lfs merge=lfs -text
50
+ examples/bagel_vizwiz_vqa_val_87_always/query.jpg filter=lfs diff=lfs merge=lfs -text
51
+ examples/lance_ok_vqa_val2014_253/demo_0.png filter=lfs diff=lfs merge=lfs -text
52
+ examples/lance_ok_vqa_val2014_253/query.jpg filter=lfs diff=lfs merge=lfs -text
53
+ examples/lance_ok_vqa_val2014_8_always/demo_0.png filter=lfs diff=lfs merge=lfs -text
54
+ examples/lance_ok_vqa_val2014_8_always/demo_1.png filter=lfs diff=lfs merge=lfs -text
55
+ examples/lance_ok_vqa_val2014_8_always/demo_2.png filter=lfs diff=lfs merge=lfs -text
56
+ examples/lance_ok_vqa_val2014_8_always/demo_3.png filter=lfs diff=lfs merge=lfs -text
README.md CHANGED
@@ -1,13 +1,77 @@
1
  ---
2
- title: Simit
3
- emoji: 💻
4
- colorFrom: pink
5
- colorTo: green
6
  sdk: gradio
7
  sdk_version: 6.29.1
8
- python_version: '3.12'
9
  app_file: app.py
10
  pinned: false
 
 
 
 
 
 
 
 
11
  ---
12
 
13
- Check out the configuration reference at https://huggingface.co/docs/hub/spaces-config-reference
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
  ---
2
+ title: SIMIT - models that imagine their own practice examples
3
+ emoji: 🥯
4
+ colorFrom: yellow
5
+ colorTo: red
6
  sdk: gradio
7
  sdk_version: 6.29.1
8
+ python_version: "3.12"
9
  app_file: app.py
10
  pinned: false
11
+ license: apache-2.0
12
+ startup_duration_timeout: 1h
13
+ short_description: A VLM imagines practice examples for your question
14
+ models:
15
+ - ByteDance-Seed/BAGEL-7B-MoT
16
+ - bytedance-research/Lance
17
+ - Qwen/Qwen3.8-27B-FP8
18
+ - black-forest-labs/FLUX.2-klein-4B
19
  ---
20
 
21
+ # SIMIT demo
22
+
23
+ Upload an image, ask a question, and compare the model's plain answer with its
24
+ SIMIT-ICL answer: before answering again, the model imagines a few practice
25
+ examples (question, answer decided first, an image made to fit), checks them,
26
+ and keeps them in context.
27
+
28
+ Built on the [`simit`](https://github.com/monurcan/simit) library. See `tools/` for how the per-budget
29
+ hyperparameters were chosen and how the cached examples were produced.
30
+
31
+ ## Hardware notes (ZeroGPU)
32
+
33
+ * Each request runs in a fresh `@spaces.GPU` process and uses the visitor's own
34
+ ZeroGPU quota for the selected time budget (20-180 s).
35
+ * BAGEL-7B-MoT and Lance run on the default 48 GB size; Qwen3.8-27B-FP8 +
36
+ FLUX.2-klein needs the 96 GB size (2x quota).
37
+ * All weights stay on CPU in the main process (~91 GB of RAM, ~92 GB of disk for
38
+ the model files). If the Space's disk is too small, attach persistent storage
39
+ and set `HF_HOME=/data/.huggingface`, or limit the models with
40
+ `SIMIT_DEMO_MODELS=bagel,lance`.
41
+ * The first request after start-up compiles and autotunes GPU kernels (Qwen:
42
+ ~25 s); later requests reuse the on-disk cache.
43
+
44
+ ## How the time budgets map to settings
45
+
46
+ Each budget (20 / 40 / 60 / 90 / 180 s of GPU time per request) uses its own settings per model
47
+ (`presets.json`, built by `tools/build_presets.py`):
48
+
49
+ 1. **Accuracy per K_max** (`tools/tune_presets.py`): one synthesis cache with 4 candidates per query on a
50
+ mixed 64-question general-QA validation set: 8 each from VizWiz, VQAv2, OK-VQA, TextVQA, ChartQA,
51
+ DocVQA, MMBench and AI2D (LMMs-Eval-Lite rows 300-307), each scored with its own metric. Then 800
52
+ Optuna trials over the ABA/DF thresholds and K_max ∈ {1,2,3,4}. Per K_max, the pick is regularized:
53
+ within one validation query of the best score, the widest difficulty-filter band.
54
+ 2. **Latency** (`tools/latency.py`): per-request timelines on the demo's own code path in fresh forked
55
+ processes (as on ZeroGPU, weight transfer included). Measured on an H100 and scaled by spaces'
56
+ own factor for the ZeroGPU GPU sizes (1.5× on `large`, 1× on `xlarge`).
57
+ 3. **Choice**: the tuned (best-quality) speed setting whenever some K fits the budget for ≥80% of
58
+ requests; among those, the K within one validation query of the best score with the widest band.
59
+ More budget also buys more attempts and verification rounds. If no K beats zero-shot on
60
+ validation, SIMIT keeps the base answer by default; "Imagine even when confident" still runs it.
61
+
62
+ Validation scores (64 general questions):
63
+
64
+ | model | zero-shot | K=1 | K=2 | K=3 | K=4 | used by default |
65
+ |---|---|---|---|---|---|---|
66
+ | BAGEL-7B-MoT | 0.703 | 0.708 | **0.750** | 0.760 | 0.755 | K=2 from 60 s (20-40 s: zero-shot) |
67
+ | Lance | 0.432 | 0.442 | 0.442 | 0.458 | **0.468** | K=1 at 20 s, K=4 from 40 s |
68
+ | Qwen3.8-27B-FP8 + FLUX | **0.854** | 0.839 | 0.823 | 0.839 | 0.844 | zero-shot (no gain found) |
69
+
70
+ Lance's answers are sampled (greedy decoding degenerates on this checkpoint), so its outputs vary
71
+ between runs and SIMIT can also hurt a correct answer. Qwen3.8-27B is strong enough that imagined
72
+ examples did not improve it on these questions.
73
+
74
+ The cached examples are real runs of this app (BAGEL-7B-MoT and Lance) on queries from the paper's
75
+ figures, plus one product-page screenshot (`examples/bagel_amazon`). Several of them used "Imagine even when confident" (shown in the example), because the base
76
+ model was confidently wrong (e.g. "Unanswerable"). Qwen3.8-27B already answered those queries
77
+ correctly, so it has no cached example.
app.py ADDED
@@ -0,0 +1,297 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """SIMIT demo: a vision-language model imagines its own practice examples for
2
+ your question, then answers again with them in context.
3
+
4
+ Runs on Hugging Face ZeroGPU (each request uses the visitor's own GPU quota)
5
+ and on any machine with a CUDA GPU (``python app.py``).
6
+ """
7
+
8
+ import os
9
+
10
+ # ZeroGPU starts a fresh process per request: keep Triton's autotuning results on disk
11
+ # (otherwise every request re-tunes the FP8 / linear-attention kernels, ~20 s).
12
+ os.environ.setdefault("TRITON_CACHE_AUTOTUNING", "1")
13
+ os.environ.setdefault("TRANSFORMERS_DISABLE_DEEPGEMM_LINEAR", "1")
14
+
15
+ import spaces # before anything CUDA-related: ZeroGPU patches torch # noqa: E402 # isort: skip
16
+
17
+ import base64 # noqa: E402
18
+ import html # noqa: E402
19
+ import json # noqa: E402
20
+ from pathlib import Path # noqa: E402
21
+
22
+ import gradio as gr # noqa: E402
23
+ from PIL import Image # noqa: E402
24
+
25
+ from demo_models import BUDGETS, ENABLED, MODELS, code_snippet, run # noqa: E402
26
+
27
+ HERE = Path(__file__).parent
28
+ ASSETS = HERE / "assets"
29
+ EXAMPLES_DIR = HERE / "examples"
30
+
31
+
32
+ def _data_uri(path: Path) -> str:
33
+ return "data:image/webp;base64," + base64.b64encode(path.read_bytes()).decode()
34
+
35
+
36
+ MASCOT = {p.stem: _data_uri(p) for p in ASSETS.glob("*.webp")}
37
+ STAGES = { # stage -> (mascot, message)
38
+ "idle": ("initial", "Ask me anything about an image"),
39
+ "loading": ("initial", "Loading the model weights (first request only)"),
40
+ "warmup": ("initial", "Waking up the model on the GPU"),
41
+ "solving": ("initial", "Solving it on its own first"),
42
+ "synthesizing": ("synthesizer", "Writing practice questions it already knows the answers to"),
43
+ "routing": ("router", "Choosing how to draw each practice image"),
44
+ "drawing": ("artist", "Painting a practice image"),
45
+ "coding": ("architect", "Writing code for a chart, diagram or document"),
46
+ "checking": ("critic", "Checking that each picture matches its question"),
47
+ "improving": ("improved", "Answering again, with its own practice examples in context"),
48
+ }
49
+ LABELS = {MODELS[k].label: k for k in ENABLED}
50
+ BUDGET_LABELS = [f"{b} s" for b in BUDGETS]
51
+ GPU_OVERHEAD = 20 # seconds requested on top of the budget (worker start-up, final answer); weight transfer counts inside it
52
+
53
+
54
+ def _seconds(label: str) -> int:
55
+ return int(str(label).split()[0])
56
+
57
+
58
+ # --------------------------------------------------------------------- HTML
59
+ def status_html(stage: str, elapsed: float = 0.0, limit: float = 0.0, n_demos: int = 0, note: str = "") -> str:
60
+ mascot, message = STAGES.get(stage, STAGES["solving"])
61
+ dots = "" if stage == "idle" else '<span class="simit-dots"></span>'
62
+ bar = ""
63
+ if limit:
64
+ pct = max(2.0, min(100.0, 100.0 * elapsed / limit))
65
+ bar = (f'<div class="simit-bar"><div style="width:{pct:.0f}%"></div></div>'
66
+ f'<div class="simit-sub">{elapsed:.0f} s of {limit:.0f} s'
67
+ + (f' · {n_demos} practice example{"s" if n_demos != 1 else ""} ready' if n_demos else "")
68
+ + "</div>")
69
+ return (f'<div class="simit-status"><img class="simit-mascot" src="{MASCOT[mascot]}" alt="">'
70
+ f'<div><div class="simit-msg">{html.escape(message)}{dots}</div>'
71
+ f'{bar}{f"<div class=simit-sub>{html.escape(note)}</div>" if note else ""}</div></div>')
72
+
73
+
74
+ def done_html(note: str) -> str:
75
+ return (f'<div class="simit-status done"><img class="simit-mascot still" src="{MASCOT["improved"]}" alt="">'
76
+ f'<div><div class="simit-msg">Done</div><div class="simit-sub">{html.escape(note)}</div></div></div>')
77
+
78
+
79
+ def error_html(note: str) -> str:
80
+ return (f'<div class="simit-status error"><img class="simit-mascot still" src="{MASCOT["critic"]}" alt="">'
81
+ f'<div><div class="simit-msg">Something went wrong</div><div class="simit-sub">{html.escape(note)}'
82
+ '</div></div></div>')
83
+
84
+
85
+ def answer_card(kind: str, answer: str = "", sub: str = "", mark: str = "") -> str:
86
+ title = "Base model (greedy)" if kind == "base" else "SIMIT-ICL (self-improved)"
87
+ body = html.escape(answer) if answer else '<span class="simit-wait">…</span>'
88
+ size = " long" if len(answer) > 60 else ""
89
+ badge = {"right": '<span class="simit-ok">✓</span>', "wrong": '<span class="simit-no">✗</span>'}.get(mark, "")
90
+ return (f'<div class="simit-card {kind}"><div class="simit-card-title">{title}{badge}</div>'
91
+ f'<div class="simit-answer{size}">{body}</div><div class="simit-card-sub">{html.escape(sub)}</div></div>')
92
+
93
+
94
+ def demo_caption(d) -> str:
95
+ bits = [f"Q: {d.question}", f"A: {d.answer}"]
96
+ meta = [d.skill] if d.skill else []
97
+ if d.verify_score is not None:
98
+ meta.append(f"critic {d.verify_score}/100")
99
+ if d.confidence is not None:
100
+ meta.append(f"confidence {d.confidence:.2f}")
101
+ return "\n".join(bits) + (f"\n({', '.join(meta)})" if meta else "")
102
+
103
+
104
+ # ------------------------------------------------------------- GPU requests
105
+ def _duration(image, question, key, budget_label, always, *args, **kwargs):
106
+ return _seconds(budget_label) + GPU_OVERHEAD
107
+
108
+
109
+ def _session(image, question, key, budget_label, always):
110
+ """Runs inside the GPU call: turns pipeline events into UI updates."""
111
+ spec, budget = MODELS[key], _seconds(budget_label)
112
+ try:
113
+ yield from _events(spec, image, question, budget, always)
114
+ except Exception as e: # show it in the page instead of a bare error toast
115
+ print(f"[demo] request failed: {type(e).__name__}: {e}", flush=True)
116
+ yield (error_html(f"{type(e).__name__}: {str(e)[:300]}"), answer_card("base"), answer_card("simit"), [], "")
117
+
118
+
119
+ def _events(spec, image, question, budget, always):
120
+ greedy, k_budget, demos, gallery = None, 0, [], []
121
+ base, improved = answer_card("base"), answer_card("simit")
122
+ status = status_html("warmup")
123
+ for event in run(spec, image, question, budget, always):
124
+ kind = event[0]
125
+ if kind == "stage":
126
+ _, stage, info = event
127
+ status = status_html(stage, info.get("elapsed", 0), info.get("limit", 0), len(demos))
128
+ elif kind == "greedy":
129
+ _, zs, k_budget = event
130
+ greedy = zs
131
+ sub = f"confidence p0 = {zs.confidence:.2f}" if zs.confidence is not None else ""
132
+ base = answer_card("base", zs.answer, sub)
133
+ elif kind == "demo":
134
+ demos.append(event[1])
135
+ gallery = [(d.image, demo_caption(d)) for d in demos]
136
+ else: # final
137
+ _, answer, info = event
138
+ reason = info.get("skipped")
139
+ if reason == "confident":
140
+ sub = "The model was already confident, so SIMIT kept its answer (adaptive budget: 0 examples)."
141
+ elif reason == "policy":
142
+ sub = ("On general questions, our tuning found no gain from imagination for this model at this "
143
+ "budget, so SIMIT keeps the base answer. Tick 'Imagine even when confident' to try it anyway.")
144
+ elif reason == "no_time":
145
+ sub = "Not enough time budget left to imagine examples; kept the base answer."
146
+ elif reason == "none_passed":
147
+ sub = "No imagined example passed the checks in time; kept the base answer."
148
+ else:
149
+ changed = "changed" if answer.strip() != greedy.answer.strip() else "kept"
150
+ sub = f"{changed} the answer after {info['demos']} imagined example{'s' if info['demos'] > 1 else ''}"
151
+ improved = answer_card("simit", answer, sub)
152
+ status = done_html(f"{info['elapsed']:.0f} s on the GPU")
153
+ yield status, base, improved, gallery, _details(greedy, k_budget, demos, spec, budget)
154
+
155
+
156
+ @spaces.GPU(duration=_duration, size="large")
157
+ def gpu_large(image, question, key, budget_label, always):
158
+ yield from _session(image, question, key, budget_label, always)
159
+
160
+
161
+ @spaces.GPU(duration=_duration, size="xlarge")
162
+ def gpu_xlarge(image, question, key, budget_label, always):
163
+ yield from _session(image, question, key, budget_label, always)
164
+
165
+
166
+ def _details(greedy, k_budget, demos, spec, budget) -> str:
167
+ if greedy is None:
168
+ return ""
169
+ p0 = f"{greedy.confidence:.2f}" if greedy.confidence is not None else "n/a"
170
+ return (f"**{spec.label}**, {budget} s budget. Zero-shot confidence p0 = {p0}, so the adaptive budget asked "
171
+ f"for **{k_budget}** example{'s' if k_budget != 1 else ''}; **{len(demos)}** passed the checks. "
172
+ "Each example is a question the model wrote, whose answer it decided first, with an image made "
173
+ "to fit that answer.")
174
+
175
+
176
+ def submit(image, question, model_label, budget_label, always):
177
+ if image is None:
178
+ raise gr.Error("Please upload an image.")
179
+ if not question or not question.strip():
180
+ raise gr.Error("Please type a question about the image.")
181
+ key = LABELS[model_label]
182
+ spec = MODELS[key]
183
+ image = image.convert("RGB")
184
+ if spec._weights is None: # main process, CPU: no GPU time is spent here
185
+ yield (status_html("loading"), answer_card("base"), answer_card("simit"), [], "")
186
+ spec.load()
187
+ fn = gpu_xlarge if spec.gpu_size == "xlarge" else gpu_large
188
+ yield from fn(image, question.strip(), key, budget_label, always)
189
+
190
+
191
+ # ------------------------------------------------------------------ cached
192
+ def _load_examples():
193
+ index = EXAMPLES_DIR / "index.json"
194
+ return json.loads(index.read_text()) if index.exists() else []
195
+
196
+
197
+ EXAMPLES = [e for e in _load_examples() if e["model"] in ENABLED]
198
+
199
+
200
+ def show_cached(image, question, model_label, budget_label, always=False):
201
+ """A cached example: the stored outputs of a real run, shown without using the GPU."""
202
+ entry = next((e for e in EXAMPLES if e["question"] == question and MODELS[e["model"]].label == model_label
203
+ and f"{e['budget']} s" == budget_label and bool(e.get("always")) == bool(always)), None)
204
+ if entry is None:
205
+ return status_html("idle", note="Press Submit to run this example."), answer_card("base"), \
206
+ answer_card("simit"), [], ""
207
+ d = EXAMPLES_DIR / entry["id"]
208
+ mark = lambda ok: "right" if ok else "wrong" # noqa: E731
209
+ gt = f"reference answer: {entry['reference']}"
210
+ base = answer_card("base", entry["greedy"], f"confidence p0 = {entry['p0']:.2f} · {gt}", mark(entry["greedy_ok"]))
211
+ n = len(entry["demos"])
212
+ improved = answer_card("simit", entry["simit"], f"after {n} imagined example{'s' if n != 1 else ''} · {gt}",
213
+ mark(entry["simit_ok"]))
214
+ gallery = [(str(d / x["image"]), "\n".join([f"Q: {x['question']}", f"A: {x['answer']}",
215
+ f"({x['skill']}" + (f", critic {x['verify_score']}/100" if
216
+ x.get("verify_score") is not None else "") + ")"]))
217
+ for x in entry["demos"]]
218
+ note = (f"Cached result from a real run ({entry['elapsed']:.0f} s on an H100, {entry['budget']} s budget"
219
+ + (", imagining even when confident" if entry.get("always") else "") + f"). Source: {entry['source']}.")
220
+ details = (f"**{MODELS[entry['model']].label}**, {entry['budget']} s budget. Zero-shot confidence "
221
+ f"p0 = {entry['p0']:.2f}; {n} imagined example{'s' if n != 1 else ''} passed the checks.")
222
+ return done_html(note), base, improved, gallery, details
223
+
224
+
225
+ # ---------------------------------------------------------------------- UI
226
+ CSS = (ASSETS / "style.css").read_text()
227
+ INTRO = """
228
+ <div class="simit-hero">
229
+ <h1>SIMIT: models that imagine their own practice examples</h1>
230
+ <p>Ask any question about an image. The model first answers on its own. Then, without any labels, it writes
231
+ similar practice questions whose answers it decides first, makes an image for each one, keeps only those it
232
+ can verify, and answers your question again with them in context.</p>
233
+ </div>
234
+ """
235
+
236
+ with gr.Blocks(title="SIMIT demo") as demo:
237
+ gr.HTML(INTRO)
238
+ with gr.Row(equal_height=False):
239
+ with gr.Column(scale=5):
240
+ image = gr.Image(type="pil", label="Image", height=360)
241
+ question = gr.Textbox(label="Question", placeholder="e.g. In what country would you find this hat?",
242
+ lines=2)
243
+ model = gr.Dropdown(list(LABELS), value=next(iter(LABELS)), label="Model")
244
+ budget = gr.Radio(BUDGET_LABELS, value="60 s", label="Max GPU time per request",
245
+ info="More time lets the model imagine and check more practice examples.")
246
+ with gr.Accordion("Advanced", open=False):
247
+ always = gr.Checkbox(False, label="Imagine even when the model is already confident",
248
+ info="By default SIMIT skips imagination for confident answers.")
249
+ run_btn = gr.Button("Submit", variant="primary")
250
+ with gr.Column(scale=6):
251
+ status = gr.HTML(status_html("idle", note="Upload an image and type a question, or pick an example below."))
252
+ with gr.Row():
253
+ base_out = gr.HTML(answer_card("base"), min_width=260)
254
+ simit_out = gr.HTML(answer_card("simit"), min_width=260)
255
+ with gr.Accordion("See the imagined practice examples", open=False):
256
+ details = gr.Markdown()
257
+ gallery = gr.Gallery(columns=4, height=300, object_fit="contain", show_label=False)
258
+ with gr.Accordion("Use SIMIT in your own code", open=False):
259
+ code = gr.Code(code_snippet(MODELS[next(iter(LABELS.values()))], 60, ""), language="python",
260
+ interactive=False)
261
+
262
+ outputs = [status, base_out, simit_out, gallery, details]
263
+ # On ZeroGPU each request gets its own GPU process; run locally, requests share this process and GPU.
264
+ from spaces.config import Config as _SpacesConfig
265
+ run_btn.click(submit, [image, question, model, budget, always], outputs,
266
+ concurrency_limit=4 if _SpacesConfig.zero_gpu else 1)
267
+
268
+ def _code(model_label, budget_label, q, force):
269
+ return code_snippet(MODELS[LABELS[model_label]], _seconds(budget_label), q or "", force)
270
+ for comp in (model, budget, question, always):
271
+ comp.change(_code, [model, budget, question, always], code, queue=False)
272
+
273
+ if EXAMPLES:
274
+ gr.Markdown("### Examples\nCached results from real runs of this demo (click one; press Submit to "
275
+ "run it again live).")
276
+ examples = gr.Gallery(
277
+ [(str(EXAMPLES_DIR / e["id"] / e["query_file"]),
278
+ f"{e['question'].splitlines()[0]} ({MODELS[e['model']].label})") for e in EXAMPLES],
279
+ columns=4, height=300 * ((len(EXAMPLES) + 3) // 4) + 20, object_fit="cover", allow_preview=False,
280
+ show_label=False)
281
+
282
+ def pick(evt: gr.SelectData):
283
+ e = EXAMPLES[evt.index]
284
+ inputs = [Image.open(EXAMPLES_DIR / e["id"] / e["query_file"]), e["question"],
285
+ MODELS[e["model"]].label, f"{e['budget']} s", bool(e.get("always"))]
286
+ return (*inputs, *show_cached(*inputs))
287
+ examples.select(pick, None, [image, question, model, budget, always] + outputs)
288
+ gr.Markdown("Models: " + ", ".join(f"[{MODELS[k].label}](https://huggingface.co/{MODELS[k].repo})"
289
+ for k in ENABLED)
290
+ + ". Each request uses your own ZeroGPU quota for the selected time budget.")
291
+
292
+ if os.environ.get("SIMIT_DEMO_PRELOAD", "1") == "1":
293
+ for key in ENABLED: # load every model on CPU before serving (main process, no CUDA)
294
+ MODELS[key].load()
295
+
296
+ if __name__ == "__main__":
297
+ demo.queue(max_size=32).launch(css=CSS, theme=gr.themes.Soft(primary_hue="orange", secondary_hue="amber"))
assets/architect.webp ADDED
assets/artist.webp ADDED
assets/critic.webp ADDED
assets/improved.webp ADDED
assets/initial.webp ADDED
assets/router.webp ADDED
assets/solver.webp ADDED
assets/style.css ADDED
@@ -0,0 +1,32 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ .simit-hero h1 { font-size: 1.7rem; margin: 0.2rem 0 0.4rem; }
2
+ .simit-hero p { margin: 0 0 0.6rem; opacity: 0.85; max-width: 60rem; }
3
+
4
+ .simit-status { display: flex; align-items: center; gap: 1rem; min-height: 132px; padding: 0.6rem 0.8rem;
5
+ border-radius: 14px; background: var(--block-background-fill); border: 1px solid var(--border-color-primary); }
6
+ .simit-mascot { height: 112px; width: auto; animation: simit-bob 1.6s ease-in-out infinite; transform-origin: 50% 90%; }
7
+ .simit-mascot.still { animation: simit-pop 0.5s ease-out 1; }
8
+ .simit-msg { font-size: 1.05rem; font-weight: 600; }
9
+ .simit-sub { font-size: 0.85rem; opacity: 0.75; margin-top: 0.3rem; }
10
+ .simit-dots::after { content: ""; animation: simit-dots 1.4s steps(4, end) infinite; }
11
+ .simit-bar { height: 8px; width: 16rem; max-width: 100%; border-radius: 4px; margin-top: 0.5rem;
12
+ background: var(--border-color-primary); overflow: hidden; }
13
+ .simit-bar > div { height: 100%; border-radius: 4px; transition: width 0.4s linear;
14
+ background: linear-gradient(90deg, #f2a65a, #e07a3f); }
15
+
16
+ .simit-card { border-radius: 14px; padding: 0.8rem 1rem; min-height: 120px;
17
+ border: 1px solid var(--border-color-primary); background: var(--block-background-fill); }
18
+ .simit-card.base { border-top: 5px solid #c9c4bd; }
19
+ .simit-card.simit { border-top: 5px solid #e07a3f; }
20
+ .simit-card-title { font-size: 0.85rem; text-transform: uppercase; letter-spacing: 0.04em; opacity: 0.7;
21
+ display: flex; justify-content: space-between; }
22
+ .simit-answer { font-size: 1.35rem; font-weight: 650; margin: 0.4rem 0; white-space: pre-wrap; }
23
+ .simit-card-sub { font-size: 0.8rem; opacity: 0.7; }
24
+ .simit-wait { opacity: 0.4; }
25
+ .simit-ok { color: #2e9b5a; font-size: 1.1rem; }
26
+ .simit-no { color: #c94a3c; font-size: 1.1rem; }
27
+
28
+ @keyframes simit-bob { 0%, 100% { transform: translateY(0) rotate(-2deg); } 50% { transform: translateY(-8px) rotate(2deg); } }
29
+ @keyframes simit-pop { 0% { transform: scale(0.8); } 70% { transform: scale(1.06); } 100% { transform: scale(1); } }
30
+ @keyframes simit-dots { 0% { content: ""; } 25% { content: "."; } 50% { content: ".."; } 75% { content: "..."; } }
31
+ @media (prefers-reduced-motion: reduce) { .simit-mascot, .simit-dots::after { animation: none; } }
32
+ .simit-answer.long { font-size: 1.0rem; font-weight: 550; line-height: 1.4; }
assets/synthesizer.webp ADDED
demo_models.py ADDED
@@ -0,0 +1,251 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Models of the demo and how one request runs on ZeroGPU.
2
+
3
+ ZeroGPU forks the Space process for every ``@spaces.GPU`` call and kills the
4
+ fork afterwards. So the weights are loaded once, on CPU, in the main process
5
+ (no CUDA there), and each call moves only the chosen model to the GPU and
6
+ builds a fresh SIMIT engine around it.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import contextlib
12
+ import json
13
+ import os
14
+ import queue
15
+ import threading
16
+ import time
17
+ from dataclasses import dataclass, field
18
+ from pathlib import Path
19
+ from typing import Optional
20
+
21
+ import torch
22
+
23
+ from simit import SIMIT, SIMITConfig
24
+
25
+ HERE = Path(__file__).parent
26
+ PRESETS = json.loads((HERE / "presets.json").read_text())
27
+ BUDGETS = [20, 40, 60, 90, 180] # seconds of GPU time per request
28
+ ANSWER_TOKENS = 128
29
+ ANSWER_RESERVE = 5.0 # seconds kept for the final SIMIT-ICL answer
30
+
31
+
32
+ @dataclass
33
+ class ModelSpec:
34
+ key: str
35
+ label: str
36
+ repo: str
37
+ kind: str # "bagel" | "lance" | "hf"
38
+ gpu_size: str # ZeroGPU size: "large" (48 GB) or "xlarge" (96 GB)
39
+ image_generator: Optional[str] = None
40
+ blurb: str = ""
41
+ _weights: object = None
42
+ _lock: threading.Lock = field(default_factory=threading.Lock)
43
+
44
+ # ----------------------------------------------------- main process (CPU)
45
+ def load(self):
46
+ """Load the weights on CPU (main process; nothing touches CUDA)."""
47
+ with self._lock:
48
+ if self._weights is None:
49
+ t = time.time()
50
+ self._weights = _LOADERS[self.kind](self)
51
+ print(f"[demo] loaded {self.label} on CPU in {time.time() - t:.0f}s", flush=True)
52
+ return self._weights
53
+
54
+ # ------------------------------------------------- inside @spaces.GPU call
55
+ def build(self, preset: dict) -> SIMIT:
56
+ """Move the weights to the GPU and wrap them in a SIMIT engine."""
57
+ w = self.load()
58
+ # Without ZeroGPU every request runs in this process: keep one model on the GPU at a time.
59
+ # (On ZeroGPU each request is a fresh fork in which nothing is on the GPU yet.)
60
+ if _ON_GPU[0] not in (None, self.key):
61
+ MODELS[_ON_GPU[0]]._to("cpu")
62
+ torch.cuda.empty_cache()
63
+ _ON_GPU[0] = self.key
64
+ cfg = SIMITConfig(**{k: v for k, v in preset.items() if k in SIMITConfig.__dataclass_fields__})
65
+ if self.kind == "bagel":
66
+ from simit.backends.bagel import BagelBackend
67
+ backend = BagelBackend(w, device="cuda", image_steps=preset.get("image_steps"), use_cuda_graphs=False)
68
+ return SIMIT(backend, config=cfg, model_id=self.repo)
69
+ if self.kind == "lance":
70
+ from simit.backends.lance import LanceBackend
71
+ backend = LanceBackend(w, device="cuda", image_steps=preset.get("image_steps"), use_cuda_graphs=False)
72
+ return SIMIT(backend, config=cfg, model_id=self.repo)
73
+ from simit.backends.hf import HFBackend
74
+ from simit.backends.imagegen import DiffusersImageGenerator
75
+ model, processor, pipe = w
76
+ model.to("cuda")
77
+ pipe.to("cuda")
78
+ backend = HFBackend(model, processor=processor, max_batch_size=8, max_batch_tokens=32768)
79
+ return SIMIT(backend, image_generator=DiffusersImageGenerator(pipe), config=cfg, model_id=self.repo)
80
+
81
+ def _to(self, device: str):
82
+ if self.kind in ("bagel", "lance"):
83
+ self._weights.to(device)
84
+ else:
85
+ model, _, pipe = self._weights
86
+ model.to(device)
87
+ pipe.to(device)
88
+
89
+ def preset(self, budget: int) -> dict:
90
+ return dict(PRESETS[self.key][str(budget)])
91
+
92
+
93
+ _ON_GPU = [None] # key of the model whose weights are on the GPU in this process
94
+
95
+
96
+ @contextlib.contextmanager
97
+ def _fp8_without_gpu():
98
+ """transformers dequantizes FP8 checkpoints to bf16 when it sees no GPU at load
99
+ time. FP8 weights only need a GPU to *run*, so let the load keep them in FP8
100
+ on CPU; they are moved to the GPU inside each request."""
101
+ saved = torch.cuda.is_available, torch.cuda.get_device_capability
102
+ torch.cuda.is_available = lambda: True
103
+ torch.cuda.get_device_capability = lambda *a, **k: (9, 0)
104
+ try:
105
+ yield
106
+ finally:
107
+ torch.cuda.is_available, torch.cuda.get_device_capability = saved
108
+
109
+
110
+ def _load_bagel(spec):
111
+ from simit.backends.bagel import BagelModel
112
+ return BagelModel.load(spec.repo, device="cpu")
113
+
114
+
115
+ def _load_lance(spec):
116
+ from simit.backends.lance import LanceModel
117
+ return LanceModel.load(spec.repo, device="cpu")
118
+
119
+
120
+ def _load_hf(spec):
121
+ from diffusers import DiffusionPipeline
122
+ from transformers import AutoModelForMultimodalLM, AutoProcessor
123
+
124
+ from simit.backends.hf import _dense_fp8_fix
125
+ os.environ.setdefault("TRANSFORMERS_DISABLE_DEEPGEMM_LINEAR", "1") # Triton FP8 kernels: no nvcc JIT at runtime
126
+ processor = AutoProcessor.from_pretrained(spec.repo)
127
+ kw = dict(device_map="cpu", dtype=torch.bfloat16)
128
+ _dense_fp8_fix(spec.repo, kw) # transformers would skip FP8 conversion of gate_proj for this checkpoint
129
+ with _fp8_without_gpu():
130
+ model = AutoModelForMultimodalLM.from_pretrained(spec.repo, **kw)
131
+ model.eval()
132
+ pipe = DiffusionPipeline.from_pretrained(spec.image_generator, dtype=torch.bfloat16)
133
+ return model, processor, pipe
134
+
135
+
136
+ _LOADERS = {"bagel": _load_bagel, "lance": _load_lance, "hf": _load_hf}
137
+
138
+ MODELS = {m.key: m for m in [
139
+ ModelSpec("bagel", "BAGEL-7B-MoT", "ByteDance-Seed/BAGEL-7B-MoT", "bagel", "large",
140
+ blurb="Unified model: draws its own photos and writes code for charts and diagrams."),
141
+ ModelSpec("lance", "Lance", "bytedance-research/Lance", "lance", "large",
142
+ blurb="Small unified model: fast native image generation."),
143
+ ModelSpec("qwen", "Qwen3.8-27B (FP8) + FLUX.2-klein", "Qwen/Qwen3.8-27B-FP8", "hf", "xlarge",
144
+ image_generator="black-forest-labs/FLUX.2-klein-4B",
145
+ blurb="Standard VLM: FLUX.2-klein draws the photos; it writes code for charts and diagrams."),
146
+ ]}
147
+ ENABLED = [k for k in os.environ.get("SIMIT_DEMO_MODELS", "bagel,lance,qwen").split(",") if k in MODELS]
148
+
149
+
150
+ # ---------------------------------------------------------------- a request
151
+ # Pipeline call tags (``on_progress``) -> what the loading screen shows.
152
+ STAGE_OF_TAG = {"synthesis": "synthesizing", "synthesis_fresh": "synthesizing", "revise": "synthesizing",
153
+ "route": "routing", "skill_spec": "coding", "render": "coding", "image": "drawing",
154
+ "critic": "checking", "critic_score": "checking"}
155
+
156
+
157
+ def run(spec: ModelSpec, image, question: str, budget: int, always: bool):
158
+ """Generator of progress events for one request (runs inside the GPU call):
159
+ ("stage", name, info) / ("greedy", ZeroShot, k) / ("demo", Demo) / ("final", answer, info)."""
160
+ t_start = time.time()
161
+ preset = spec.preset(budget)
162
+ yield ("stage", "warmup", {"elapsed": 0.0})
163
+ sim = spec.build(preset)
164
+ try:
165
+ yield ("stage", "solving", {"elapsed": time.time() - t_start})
166
+ zs = sim.zero_shot(image, question, max_new_tokens=ANSWER_TOKENS) # the greedy answer + p0 (ABA input)
167
+ k_budget = sim.config.budget(zs.confidence)
168
+ yield ("greedy", zs, k_budget)
169
+
170
+ # what is left of the budget, keeping room for the final answer (measured: ~1-1.5 s after the last demo)
171
+ time_left = budget - (time.time() - t_start) - ANSWER_RESERVE
172
+ if (k_budget == 0 and not always) or time_left < 3:
173
+ if time_left < 3:
174
+ reason = "no_time"
175
+ else:
176
+ reason = "confident" if preset.get("imagine_by_default", True) else "policy"
177
+ yield ("final", zs.answer, {"demos": 0, "skipped": reason, "elapsed": time.time() - t_start})
178
+ return
179
+
180
+ events: "queue.Queue" = queue.Queue()
181
+ current = ["synthesis"]
182
+ result = {}
183
+
184
+ def work():
185
+ try:
186
+ result["det"] = sim.imagine(image, question, k=preset["k_max"] if always else None,
187
+ return_details=True, max_new_tokens=ANSWER_TOKENS, time_limit=time_left,
188
+ on_demo=events.put, on_progress=lambda tag: current.__setitem__(0, tag))
189
+ except Exception as e: # surfaced below
190
+ result["error"] = e
191
+ th = threading.Thread(target=work, daemon=True)
192
+ th.start()
193
+ while th.is_alive() or not events.empty():
194
+ try:
195
+ yield ("demo", events.get(timeout=0.4))
196
+ continue
197
+ except queue.Empty:
198
+ pass
199
+ stage = STAGE_OF_TAG.get(current[0].split(":")[0], "synthesizing")
200
+ yield ("stage", stage, {"elapsed": time.time() - t_start, "limit": budget})
201
+ if "error" in result:
202
+ raise result["error"]
203
+ det = result["det"]
204
+ if not det.demos:
205
+ yield ("final", zs.answer, {"demos": 0, "skipped": "none_passed", "elapsed": time.time() - t_start,
206
+ "stats": det.stats})
207
+ return
208
+ yield ("stage", "improving", {"elapsed": time.time() - t_start})
209
+ answer = sim.answer(image, question, det.demos, max_new_tokens=ANSWER_TOKENS)
210
+ yield ("final", answer, {"demos": len(det.demos), "elapsed": time.time() - t_start, "stats": det.stats})
211
+ finally:
212
+ sim.close()
213
+
214
+
215
+ def code_snippet(spec: ModelSpec, budget: int, question: str, always: bool = False) -> str:
216
+ """Copy-paste code reproducing this request with the library."""
217
+ p = spec.preset(budget)
218
+ force = always or not p.get("imagine_by_default", True)
219
+ keys = ["k_max", "epsilon", "A0", "A1", "B", "t_low", "t_high", "verify", "verify_think", "use_skills",
220
+ "attempts_per_slot", "verify_rounds"]
221
+ args = [f"{k}={round(p[k], 3) if isinstance(p[k], float) else p[k]!r}" for k in keys if k in p]
222
+ lines, line = [], ""
223
+ for a in args: # wrap the config arguments, a few per line
224
+ if line and len(line) + len(a) > 62:
225
+ lines.append(line.rstrip())
226
+ line = ""
227
+ line += a + ", "
228
+ lines.append(line.rstrip(", "))
229
+ cfg = "\n " + "\n ".join(lines) + ",\n "
230
+ gen = f',\n image_generator="{spec.image_generator}",' if spec.image_generator else ","
231
+ q = question.strip().replace('"""', "'''") or "What is shown in this image?"
232
+ steps = f"\n image_steps={p['image_steps']}," if p.get("image_steps") else ""
233
+ return f'''# pip install simit
234
+ from simit import SIMIT, SIMITConfig
235
+
236
+ model = SIMIT.from_pretrained(
237
+ "{spec.repo}"{gen}
238
+ config=SIMITConfig({cfg}),{steps}
239
+ )
240
+
241
+ image = "my_image.jpg"
242
+ question = """{q}"""
243
+
244
+ greedy = model.greedy(image, question) # base model, zero-shot
245
+ demos = model.imagine(image, question{f", k={p['k_max']}" if force else ""}, time_limit={max(5, budget - 15)}) # imagined examples
246
+ improved = model.answer(image, question, demos) # SIMIT-ICL
247
+ print(greedy, "->", improved)
248
+
249
+ from simit import save_demos
250
+ save_demos(demos, "imagined/") # PNGs + jsonl, e.g. to fine-tune on
251
+ '''
examples/bagel_amazon/demo_0.png ADDED

Git LFS Details

  • SHA256: 802a7419911b15d848b369605e1dc5b4c90216038f7aaf73a550965289212b44
  • Pointer size: 131 Bytes
  • Size of remote file: 243 kB
examples/bagel_amazon/demo_1.png ADDED

Git LFS Details

  • SHA256: ba64ba64781dd4553d10fb1d92e24ac89c6e7e91a09b74093d7b0eb162257073
  • Pointer size: 131 Bytes
  • Size of remote file: 260 kB
examples/bagel_amazon/query.png ADDED

Git LFS Details

  • SHA256: d3db1a26a5450a23396235a38c9d79b1a91c15848ca4ad4e7e60594b28656944
  • Pointer size: 132 Bytes
  • Size of remote file: 1.3 MB
examples/bagel_infovqa_val_449/demo_0.png ADDED
examples/bagel_infovqa_val_449/query.jpg ADDED

Git LFS Details

  • SHA256: 1a95bab7d83c7801f2a57a69f6b6988e4d07f0cae90f4c5d8cd77f87b6753f67
  • Pointer size: 131 Bytes
  • Size of remote file: 186 kB
examples/bagel_ok_vqa_val2014_492_always/demo_0.png ADDED

Git LFS Details

  • SHA256: aa7b8fbec13af083e294c23f84d9d56a3d6771f94646b8fbf16085757ed1c6de
  • Pointer size: 131 Bytes
  • Size of remote file: 121 kB
examples/bagel_ok_vqa_val2014_492_always/demo_1.png ADDED

Git LFS Details

  • SHA256: 3649882f909434f1178e296f63b10427b0a67dbe345fd1357f2ad993e5b8b59a
  • Pointer size: 131 Bytes
  • Size of remote file: 185 kB
examples/bagel_ok_vqa_val2014_492_always/query.jpg ADDED
examples/bagel_vizwiz_vqa_val_345_always/demo_0.png ADDED

Git LFS Details

  • SHA256: 7f852ee1af0bb71310fbd7691abab9638ba7eec68b754cca12beda296e46c546
  • Pointer size: 131 Bytes
  • Size of remote file: 170 kB
examples/bagel_vizwiz_vqa_val_345_always/demo_1.png ADDED

Git LFS Details

  • SHA256: 84c5ca4269046ead10ee015c9ecac651702e8ea2f0c2001579758d6ea72e743c
  • Pointer size: 131 Bytes
  • Size of remote file: 250 kB
examples/bagel_vizwiz_vqa_val_345_always/query.jpg ADDED

Git LFS Details

  • SHA256: 99044d35092967a8b9570c16b5ed4f964d7d4365f362fc1d7de996ba2af33631
  • Pointer size: 131 Bytes
  • Size of remote file: 146 kB
examples/bagel_vizwiz_vqa_val_47_always/demo_0.png ADDED

Git LFS Details

  • SHA256: 04a1048fab70d940c497e2c0fc33c6b46c92b96ae626f7e29c6229f5e385fea1
  • Pointer size: 131 Bytes
  • Size of remote file: 116 kB
examples/bagel_vizwiz_vqa_val_47_always/demo_1.png ADDED

Git LFS Details

  • SHA256: d3692679b2e62167cf37f69ca5a8d6ca50b7a2277443414791a8e35147080b1d
  • Pointer size: 131 Bytes
  • Size of remote file: 164 kB
examples/bagel_vizwiz_vqa_val_47_always/query.jpg ADDED

Git LFS Details

  • SHA256: 3435095742bc8c189a762e45ab17a1558385c9dc53bb0101d4479bab914fa45b
  • Pointer size: 131 Bytes
  • Size of remote file: 129 kB
examples/bagel_vizwiz_vqa_val_87_always/demo_0.png ADDED

Git LFS Details

  • SHA256: 7b587387d0b4a923927ae2f970137b984b1bc33459820bedbe1af4fbd37088f6
  • Pointer size: 131 Bytes
  • Size of remote file: 179 kB
examples/bagel_vizwiz_vqa_val_87_always/demo_1.png ADDED

Git LFS Details

  • SHA256: 72346bf6474dca47365121f6f6367f159531e610547a772e0906bc59657f8c8b
  • Pointer size: 131 Bytes
  • Size of remote file: 191 kB
examples/bagel_vizwiz_vqa_val_87_always/query.jpg ADDED

Git LFS Details

  • SHA256: e1985db521432264f72435cfd14fb58104b156896492985a3b388e58fb088b2f
  • Pointer size: 131 Bytes
  • Size of remote file: 147 kB
examples/index.json ADDED
@@ -0,0 +1,274 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ [
2
+ {
3
+ "id": "bagel_ok_vqa_val2014_492_always",
4
+ "query_file": "query.jpg",
5
+ "model": "bagel",
6
+ "budget": 60,
7
+ "always": true,
8
+ "question": "In what country would you find this hat?\nWhen the provided information is insufficient, respond with 'Unanswerable'.\nAnswer the question using a single word or phrase.",
9
+ "reference": "vietnam",
10
+ "greedy": "Unanswerable",
11
+ "p0": 0.8883122759482616,
12
+ "simit": "Vietnam",
13
+ "greedy_ok": false,
14
+ "simit_ok": true,
15
+ "demos": [
16
+ {
17
+ "image": "demo_0.png",
18
+ "question": "How many wheels does the bicycle have?",
19
+ "answer": "Two",
20
+ "skill": "natural",
21
+ "verify_score": 100,
22
+ "confidence": null
23
+ },
24
+ {
25
+ "image": "demo_1.png",
26
+ "question": "What color is the hat the person is wearing?",
27
+ "answer": "Brown",
28
+ "skill": "natural",
29
+ "verify_score": 70,
30
+ "confidence": null
31
+ }
32
+ ],
33
+ "elapsed": 16.20292091369629,
34
+ "source": "OK-VQA row 492, paper Fig. (qualitative samples)"
35
+ },
36
+ {
37
+ "id": "bagel_infovqa_val_449",
38
+ "query_file": "query.jpg",
39
+ "model": "bagel",
40
+ "budget": 60,
41
+ "always": false,
42
+ "question": "How many points are listed under when to wash your hands?\nAnswer the question using a single word or phrase.",
43
+ "reference": "6",
44
+ "greedy": "7",
45
+ "p0": 0.6461464658834921,
46
+ "simit": "6",
47
+ "greedy_ok": false,
48
+ "simit_ok": true,
49
+ "demos": [
50
+ {
51
+ "image": "demo_0.png",
52
+ "question": "How many steps are there in the handwashing process?",
53
+ "answer": "11",
54
+ "skill": "figure",
55
+ "verify_score": 70,
56
+ "confidence": 0.38085715656328817
57
+ }
58
+ ],
59
+ "elapsed": 55.53464651107788,
60
+ "source": "InfographicVQA row 449, paper Fig. (qualitative samples)"
61
+ },
62
+ {
63
+ "id": "bagel_amazon",
64
+ "query_file": "query.png",
65
+ "model": "bagel",
66
+ "budget": 60,
67
+ "always": false,
68
+ "question": "Does this contain any trans fat?",
69
+ "reference": "No",
70
+ "greedy": "Yes, the bagel contains trans fat. According to the nutrition facts label on the packaging, the bagel has 0.5 grams of trans fat per serving.",
71
+ "p0": 0.7576221457584613,
72
+ "simit": "No",
73
+ "greedy_ok": false,
74
+ "simit_ok": true,
75
+ "demos": [
76
+ {
77
+ "image": "demo_0.png",
78
+ "question": "How many calories are in one serving of the Turkish Sesame Bagel?",
79
+ "answer": "290",
80
+ "skill": "natural",
81
+ "verify_score": 75,
82
+ "confidence": 0.3624771327791413
83
+ },
84
+ {
85
+ "image": "demo_1.png",
86
+ "question": "How many calories are listed for each serving of Turkish Sessame Bagel?",
87
+ "answer": "290",
88
+ "skill": "natural",
89
+ "verify_score": 100,
90
+ "confidence": 0.147100984163098
91
+ }
92
+ ],
93
+ "elapsed": 27.87025022506714,
94
+ "source": "Amazon product page (screenshot)"
95
+ },
96
+ {
97
+ "id": "bagel_vizwiz_vqa_val_345_always",
98
+ "query_file": "query.jpg",
99
+ "model": "bagel",
100
+ "budget": 60,
101
+ "always": true,
102
+ "question": "What color is this cat?\nWhen the provided information is insufficient, respond with 'Unanswerable'.\nAnswer the question using a single word or phrase.",
103
+ "reference": "brown",
104
+ "greedy": "Unanswerable",
105
+ "p0": 0.6977114977650781,
106
+ "simit": "Brown",
107
+ "greedy_ok": false,
108
+ "simit_ok": true,
109
+ "demos": [
110
+ {
111
+ "image": "demo_0.png",
112
+ "question": "How many cats are in the image?",
113
+ "answer": "One",
114
+ "skill": "natural",
115
+ "verify_score": 70,
116
+ "confidence": null
117
+ },
118
+ {
119
+ "image": "demo_1.png",
120
+ "question": "What type of furniture is visible in the background?",
121
+ "answer": "A chair",
122
+ "skill": "natural",
123
+ "verify_score": 70,
124
+ "confidence": null
125
+ }
126
+ ],
127
+ "elapsed": 10.399970293045044,
128
+ "source": "VizWiz row 345, paper teaser"
129
+ },
130
+ {
131
+ "id": "bagel_vizwiz_vqa_val_47_always",
132
+ "query_file": "query.jpg",
133
+ "model": "bagel",
134
+ "budget": 60,
135
+ "always": true,
136
+ "question": "What is this item?\nWhen the provided information is insufficient, respond with 'Unanswerable'.\nAnswer the question using a single word or phrase.",
137
+ "reference": "angel",
138
+ "greedy": "Unanswerable.",
139
+ "p0": 0.8626286828795229,
140
+ "simit": "Angel",
141
+ "greedy_ok": false,
142
+ "simit_ok": true,
143
+ "demos": [
144
+ {
145
+ "image": "demo_0.png",
146
+ "question": "How many cats are depicted in the image?",
147
+ "answer": "One",
148
+ "skill": "natural",
149
+ "verify_score": 80,
150
+ "confidence": null
151
+ },
152
+ {
153
+ "image": "demo_1.png",
154
+ "question": "How many cats are featured in the image?",
155
+ "answer": "One",
156
+ "skill": "natural",
157
+ "verify_score": 70,
158
+ "confidence": null
159
+ }
160
+ ],
161
+ "elapsed": 15.055564641952515,
162
+ "source": "VizWiz row 47, paper appendix"
163
+ },
164
+ {
165
+ "id": "bagel_vizwiz_vqa_val_87_always",
166
+ "query_file": "query.jpg",
167
+ "model": "bagel",
168
+ "budget": 60,
169
+ "always": true,
170
+ "question": "What is this a picture of?\nWhen the provided information is insufficient, respond with 'Unanswerable'.\nAnswer the question using a single word or phrase.",
171
+ "reference": "dog in lace dress",
172
+ "greedy": "Unanswerable.",
173
+ "p0": 0.8959448319074929,
174
+ "simit": "Dog",
175
+ "greedy_ok": false,
176
+ "simit_ok": true,
177
+ "demos": [
178
+ {
179
+ "image": "demo_0.png",
180
+ "question": "How many legs does the dog have?",
181
+ "answer": "Four",
182
+ "skill": "natural",
183
+ "verify_score": 100,
184
+ "confidence": null
185
+ },
186
+ {
187
+ "image": "demo_1.png",
188
+ "question": "What color is the dog's belly?",
189
+ "answer": "Pink",
190
+ "skill": "natural",
191
+ "verify_score": 100,
192
+ "confidence": null
193
+ }
194
+ ],
195
+ "elapsed": 9.160282850265503,
196
+ "source": "VizWiz row 87, paper appendix"
197
+ },
198
+ {
199
+ "id": "lance_ok_vqa_val2014_253",
200
+ "query_file": "query.jpg",
201
+ "model": "lance",
202
+ "budget": 60,
203
+ "always": false,
204
+ "question": "What time period is the action pictured here based off of?\nAnswer the question using a single word or phrase.",
205
+ "reference": "medieval",
206
+ "greedy": "Polo",
207
+ "p0": 0.6023501950034147,
208
+ "simit": "Medieval",
209
+ "greedy_ok": false,
210
+ "simit_ok": true,
211
+ "demos": [
212
+ {
213
+ "image": "demo_0.png",
214
+ "question": "What color does the umbrella look like?",
215
+ "answer": "Red",
216
+ "skill": "natural",
217
+ "verify_score": null,
218
+ "confidence": 0.36976529385784346
219
+ }
220
+ ],
221
+ "elapsed": 16.72637391090393,
222
+ "source": "OK-VQA row 253, paper appendix"
223
+ },
224
+ {
225
+ "id": "lance_ok_vqa_val2014_8_always",
226
+ "query_file": "query.jpg",
227
+ "model": "lance",
228
+ "budget": 60,
229
+ "always": true,
230
+ "question": "What kind of house is next to the stop sign?\nAnswer the question using a single word or phrase.",
231
+ "reference": "apartment",
232
+ "greedy": "Single story",
233
+ "p0": 0.5895590491401231,
234
+ "simit": "Apartment",
235
+ "greedy_ok": false,
236
+ "simit_ok": true,
237
+ "demos": [
238
+ {
239
+ "image": "demo_0.png",
240
+ "question": "What are the colors of the umbrellas in the painting?",
241
+ "answer": "No",
242
+ "skill": "natural",
243
+ "verify_score": null,
244
+ "confidence": null
245
+ },
246
+ {
247
+ "image": "demo_1.png",
248
+ "question": "What color does the door have?",
249
+ "answer": "Brown",
250
+ "skill": "natural",
251
+ "verify_score": null,
252
+ "confidence": null
253
+ },
254
+ {
255
+ "image": "demo_2.png",
256
+ "question": "What color is the umbrella?",
257
+ "answer": "No",
258
+ "skill": "natural",
259
+ "verify_score": null,
260
+ "confidence": null
261
+ },
262
+ {
263
+ "image": "demo_3.png",
264
+ "question": "What is the color of the stop sign?",
265
+ "answer": "Red",
266
+ "skill": "natural",
267
+ "verify_score": null,
268
+ "confidence": null
269
+ }
270
+ ],
271
+ "elapsed": 7.836957216262817,
272
+ "source": "OK-VQA row 8, paper appendix"
273
+ }
274
+ ]
examples/lance_ok_vqa_val2014_253/demo_0.png ADDED

Git LFS Details

  • SHA256: 6704f5a226fe74337310684ae4abd162acf5549cb9fbecbcf9c9ea12e4d33a6d
  • Pointer size: 131 Bytes
  • Size of remote file: 197 kB
examples/lance_ok_vqa_val2014_253/query.jpg ADDED

Git LFS Details

  • SHA256: 5b851b94ac959ab4c6462c27af0b756a18712c82ecd9759b5d20e040b2dbe80c
  • Pointer size: 131 Bytes
  • Size of remote file: 117 kB
examples/lance_ok_vqa_val2014_8_always/demo_0.png ADDED

Git LFS Details

  • SHA256: 322852a957c987b70f879e2bb2b9683318699576ef0cf0a9fd7403324d06f070
  • Pointer size: 131 Bytes
  • Size of remote file: 237 kB
examples/lance_ok_vqa_val2014_8_always/demo_1.png ADDED

Git LFS Details

  • SHA256: ca59133eb7a74dda75dd9b0244ada8a969497eba7ad5efd999c657bddaea4790
  • Pointer size: 131 Bytes
  • Size of remote file: 209 kB
examples/lance_ok_vqa_val2014_8_always/demo_2.png ADDED

Git LFS Details

  • SHA256: 42b888d177f87e5b48c8d4703353126ad0969af5cf12f9d8f85b7ba28ee2f50c
  • Pointer size: 131 Bytes
  • Size of remote file: 212 kB
examples/lance_ok_vqa_val2014_8_always/demo_3.png ADDED

Git LFS Details

  • SHA256: cd6ff8322a257d9adaf3ae63a345e1470b4a747e44a34ab259c3385cf333eeba
  • Pointer size: 131 Bytes
  • Size of remote file: 202 kB
examples/lance_ok_vqa_val2014_8_always/query.jpg ADDED
packages.txt ADDED
@@ -0,0 +1,18 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ libnss3
2
+ libnspr4
3
+ libatk1.0-0
4
+ libatk-bridge2.0-0
5
+ libcups2
6
+ libdrm2
7
+ libxkbcommon0
8
+ libxcomposite1
9
+ libxdamage1
10
+ libxfixes3
11
+ libxrandr2
12
+ libgbm1
13
+ libasound2
14
+ libpango-1.0-0
15
+ libcairo2
16
+ libxrender1
17
+ libxext6
18
+ fonts-dejavu-core
presets.json ADDED
@@ -0,0 +1,233 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "_note": "built by tools/build_presets.py from tune_presets.py + latency.py runs",
3
+ "bagel": {
4
+ "20": {
5
+ "epsilon": 1.0,
6
+ "A0": 0.5631057319035666,
7
+ "A1": 0.6871778307441961,
8
+ "B": 0.3818684686169197,
9
+ "t_low": 0.01568469261109281,
10
+ "t_high": 0.9941888593223634,
11
+ "k_max": 1,
12
+ "attempts_per_slot": 2,
13
+ "repair_retries": 0,
14
+ "verify_rounds": 1,
15
+ "imagine_by_default": false,
16
+ "verify": false,
17
+ "image_steps": 25
18
+ },
19
+ "40": {
20
+ "epsilon": 1.0,
21
+ "A0": 0.5631057319035666,
22
+ "A1": 0.6871778307441961,
23
+ "B": 0.3818684686169197,
24
+ "t_low": 0.01568469261109281,
25
+ "t_high": 0.9941888593223634,
26
+ "k_max": 1,
27
+ "attempts_per_slot": 2,
28
+ "repair_retries": 1,
29
+ "verify_rounds": 1,
30
+ "imagine_by_default": false,
31
+ "verify": true,
32
+ "verify_think": true,
33
+ "image_steps": 50
34
+ },
35
+ "60": {
36
+ "epsilon": 0.13150801401314569,
37
+ "A0": 0.5697034469957714,
38
+ "A1": 0.6928285171478893,
39
+ "B": 0.4731007761788684,
40
+ "t_low": 0.012450289537920928,
41
+ "t_high": 0.702696037286609,
42
+ "k_max": 2,
43
+ "attempts_per_slot": 3,
44
+ "repair_retries": 1,
45
+ "verify_rounds": 2,
46
+ "imagine_by_default": true,
47
+ "verify": true,
48
+ "verify_think": true,
49
+ "image_steps": 50
50
+ },
51
+ "90": {
52
+ "epsilon": 0.13150801401314569,
53
+ "A0": 0.5697034469957714,
54
+ "A1": 0.6928285171478893,
55
+ "B": 0.4731007761788684,
56
+ "t_low": 0.012450289537920928,
57
+ "t_high": 0.702696037286609,
58
+ "k_max": 2,
59
+ "attempts_per_slot": 4,
60
+ "repair_retries": 2,
61
+ "verify_rounds": 3,
62
+ "imagine_by_default": true,
63
+ "verify": true,
64
+ "verify_think": true,
65
+ "image_steps": 50
66
+ },
67
+ "180": {
68
+ "epsilon": 0.13150801401314569,
69
+ "A0": 0.5697034469957714,
70
+ "A1": 0.6928285171478893,
71
+ "B": 0.4731007761788684,
72
+ "t_low": 0.012450289537920928,
73
+ "t_high": 0.702696037286609,
74
+ "k_max": 2,
75
+ "attempts_per_slot": 4,
76
+ "repair_retries": 2,
77
+ "verify_rounds": 5,
78
+ "imagine_by_default": true,
79
+ "verify": true,
80
+ "verify_think": true,
81
+ "image_steps": 50
82
+ }
83
+ },
84
+ "lance": {
85
+ "20": {
86
+ "epsilon": 0.10540046109011536,
87
+ "A0": 0.5284418351778891,
88
+ "A1": 0.7185259059153423,
89
+ "B": 0.20052864010313778,
90
+ "t_low": 0.24157862499461874,
91
+ "t_high": 0.8741991398657891,
92
+ "k_max": 1,
93
+ "attempts_per_slot": 2,
94
+ "repair_retries": 0,
95
+ "verify_rounds": 1,
96
+ "imagine_by_default": true,
97
+ "image_steps": 30
98
+ },
99
+ "40": {
100
+ "epsilon": 0.11989932190804062,
101
+ "A0": 0.35382385806255806,
102
+ "A1": 0.4638199725338076,
103
+ "B": 0.2584592356693373,
104
+ "t_low": 0.25642426481228886,
105
+ "t_high": 0.9450464322610527,
106
+ "k_max": 4,
107
+ "attempts_per_slot": 2,
108
+ "repair_retries": 1,
109
+ "verify_rounds": 1,
110
+ "imagine_by_default": true,
111
+ "image_steps": 30
112
+ },
113
+ "60": {
114
+ "epsilon": 0.11989932190804062,
115
+ "A0": 0.35382385806255806,
116
+ "A1": 0.4638199725338076,
117
+ "B": 0.2584592356693373,
118
+ "t_low": 0.25642426481228886,
119
+ "t_high": 0.9450464322610527,
120
+ "k_max": 4,
121
+ "attempts_per_slot": 3,
122
+ "repair_retries": 1,
123
+ "verify_rounds": 2,
124
+ "imagine_by_default": true,
125
+ "image_steps": 30
126
+ },
127
+ "90": {
128
+ "epsilon": 0.11989932190804062,
129
+ "A0": 0.35382385806255806,
130
+ "A1": 0.4638199725338076,
131
+ "B": 0.2584592356693373,
132
+ "t_low": 0.25642426481228886,
133
+ "t_high": 0.9450464322610527,
134
+ "k_max": 4,
135
+ "attempts_per_slot": 4,
136
+ "repair_retries": 2,
137
+ "verify_rounds": 3,
138
+ "imagine_by_default": true,
139
+ "image_steps": 30
140
+ },
141
+ "180": {
142
+ "epsilon": 0.11989932190804062,
143
+ "A0": 0.35382385806255806,
144
+ "A1": 0.4638199725338076,
145
+ "B": 0.2584592356693373,
146
+ "t_low": 0.25642426481228886,
147
+ "t_high": 0.9450464322610527,
148
+ "k_max": 4,
149
+ "attempts_per_slot": 4,
150
+ "repair_retries": 2,
151
+ "verify_rounds": 5,
152
+ "imagine_by_default": true,
153
+ "image_steps": 30
154
+ }
155
+ },
156
+ "qwen": {
157
+ "20": {
158
+ "epsilon": 1.0,
159
+ "A0": 0.29123083240427616,
160
+ "A1": 0.892922351072982,
161
+ "B": 0.051022405374014035,
162
+ "t_low": 0.10443837804741735,
163
+ "t_high": 0.24890099152089373,
164
+ "k_max": 1,
165
+ "attempts_per_slot": 2,
166
+ "repair_retries": 0,
167
+ "verify_rounds": 1,
168
+ "imagine_by_default": false,
169
+ "verify": false,
170
+ "use_skills": false
171
+ },
172
+ "40": {
173
+ "epsilon": 1.0,
174
+ "A0": 0.29123083240427616,
175
+ "A1": 0.892922351072982,
176
+ "B": 0.051022405374014035,
177
+ "t_low": 0.10443837804741735,
178
+ "t_high": 0.24890099152089373,
179
+ "k_max": 1,
180
+ "attempts_per_slot": 2,
181
+ "repair_retries": 1,
182
+ "verify_rounds": 1,
183
+ "imagine_by_default": false,
184
+ "verify": false,
185
+ "use_skills": false
186
+ },
187
+ "60": {
188
+ "epsilon": 1.0,
189
+ "A0": 0.4895372103191979,
190
+ "A1": 0.6789573167296538,
191
+ "B": 0.39026458814322773,
192
+ "t_low": 0.05913721293446661,
193
+ "t_high": 0.6612150885625033,
194
+ "k_max": 2,
195
+ "attempts_per_slot": 3,
196
+ "repair_retries": 1,
197
+ "verify_rounds": 2,
198
+ "imagine_by_default": false,
199
+ "verify": false,
200
+ "use_skills": true
201
+ },
202
+ "90": {
203
+ "epsilon": 1.0,
204
+ "A0": 0.4895372103191979,
205
+ "A1": 0.6789573167296538,
206
+ "B": 0.39026458814322773,
207
+ "t_low": 0.05913721293446661,
208
+ "t_high": 0.6612150885625033,
209
+ "k_max": 2,
210
+ "attempts_per_slot": 4,
211
+ "repair_retries": 2,
212
+ "verify_rounds": 3,
213
+ "imagine_by_default": false,
214
+ "verify": false,
215
+ "use_skills": true
216
+ },
217
+ "180": {
218
+ "epsilon": 1.0,
219
+ "A0": 0.29123083240427616,
220
+ "A1": 0.892922351072982,
221
+ "B": 0.051022405374014035,
222
+ "t_low": 0.10443837804741735,
223
+ "t_high": 0.24890099152089373,
224
+ "k_max": 1,
225
+ "attempts_per_slot": 4,
226
+ "repair_retries": 2,
227
+ "verify_rounds": 5,
228
+ "imagine_by_default": false,
229
+ "verify": true,
230
+ "use_skills": true
231
+ }
232
+ }
233
+ }
requirements.txt ADDED
@@ -0,0 +1,4 @@
 
 
 
 
 
1
+ # gradio, spaces and huggingface_hub are preinstalled and managed by the Spaces platform: don't list them.
2
+ simit>=0.1.1
3
+ # simit needs torch >= 2.12 (torch.nn.attention.varlen); 2.13.0 is on ZeroGPU's supported list
4
+ torch==2.13.0
tools/build_presets.py ADDED
@@ -0,0 +1,91 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Combine accuracy (tune_presets.py) and latency (latency.py) into presets.json:
2
+ for every model and time budget, the largest K_max whose demos arrive in time
3
+ for most requests, at the best-quality speed setting that achieves it, with the
4
+ ABA/DF thresholds tuned for that K_max.
5
+
6
+ python tools/build_presets.py RUN_DIR
7
+ """
8
+ import json
9
+ import sys
10
+ from pathlib import Path
11
+
12
+ HERE = Path(__file__).parent.parent
13
+ sys.path.insert(0, str(HERE))
14
+ from demo_models import ANSWER_RESERVE, BUDGETS, MODELS # noqa: E402
15
+
16
+ run = Path(sys.argv[1])
17
+ # ZeroGPU vs. our H100 (spaces' own duration_factor: 1.5 on the half GPU, none on the full one)
18
+ SLOWDOWN = {"large": 1.5, "xlarge": 1.0}
19
+ QUALITY = { # speed settings, best quality first (as timed by latency.py)
20
+ "bagel": ["think50", "nothink50", "nothink25", "nocritic25"],
21
+ "lance": ["steps30", "steps20"],
22
+ "qwen": ["critic", "skills", "natural"],
23
+ }
24
+ SETTING = {
25
+ "think50": dict(verify=True, verify_think=True, image_steps=50),
26
+ "nothink50": dict(verify=True, verify_think=False, image_steps=50),
27
+ "nothink25": dict(verify=True, verify_think=False, image_steps=25),
28
+ "nocritic25": dict(verify=False, image_steps=25),
29
+ "steps30": dict(image_steps=30), "steps20": dict(image_steps=20),
30
+ "critic": dict(verify=True, use_skills=True), "skills": dict(verify=False, use_skills=True),
31
+ "natural": dict(verify=False, use_skills=False),
32
+ }
33
+ SHARE = 0.8 # fraction of requests that must get all K demos in time
34
+
35
+ out_file = HERE / "presets.json"
36
+ presets = json.loads(out_file.read_text()) if out_file.exists() else {} # models without data keep theirs
37
+ presets["_note"] = "built by tools/build_presets.py from tune_presets.py + latency.py runs"
38
+ report = []
39
+ for key in MODELS:
40
+ tune_file, lat_file = run / f"presets_{key}.json", run / f"latency_{key}.json"
41
+ if not (tune_file.exists() and lat_file.exists()):
42
+ print(f"skip {key}: missing {tune_file.name if not tune_file.exists() else lat_file.name}")
43
+ continue
44
+ tuning = json.loads(tune_file.read_text())
45
+ tuned, zero_shot = tuning["robust_per_k_max"], tuning["zero_shot"]
46
+ timelines = json.loads(lat_file.read_text())
47
+ f = SLOWDOWN[MODELS[key].gpu_size]
48
+ scores = {int(k): v["score"] for k, v in tuned.items()}
49
+ presets[key] = {}
50
+ n_val = tuning["n_val"]
51
+ band = {int(k): v["params"]["t_high"] - v["params"]["t_low"] for k, v in tuned.items()}
52
+
53
+ def fits(setting, k, deadline):
54
+ """Most requests have k demos before the deadline. Pooled over every run that asked for at least
55
+ k demos (when its k-th demo arrived): 6 requests per cell alone are too noisy."""
56
+ runs = [r for r in timelines if r["setting"] == setting and r["k"] >= k]
57
+ if not runs:
58
+ return False
59
+ return sum(len(r.get("demos", [])) >= k and r["demos"][k - 1] <= deadline for r in runs) >= SHARE * len(runs)
60
+
61
+ for budget in BUDGETS:
62
+ deadline = (budget - ANSWER_RESERVE) / f # the demo's imagination deadline, in H100 seconds
63
+ choice = None
64
+ # The tuned accuracies hold for the setting the tuning cache was made with (the first, best-quality
65
+ # one), so use it whenever any K fits; faster settings only when it cannot fit at all.
66
+ for setting in QUALITY[key]:
67
+ ks = [k for k in sorted(scores) if fits(setting, k, deadline)]
68
+ if not ks:
69
+ continue
70
+ top = max(scores[k] for k in ks)
71
+ near = [k for k in ks if scores[k] >= top - 1.0 / n_val] # within one validation query
72
+ k = min(near, key=lambda k: (-band[k], k)) # widest band, then fewest demos
73
+ choice = (k, setting)
74
+ break
75
+ if choice is None: # nothing fits: the fastest setting, one demo; the deadline decides
76
+ choice = (1, QUALITY[key][-1])
77
+ k, setting = choice
78
+ # more time: more attempts and checking rounds (the paper's 4 / 2 / 5 at the largest budget)
79
+ effort = {20: (2, 0, 1), 40: (2, 1, 1), 60: (3, 1, 2), 90: (4, 2, 3), 180: (4, 2, 5)}[budget]
80
+ params = {p: v for p, v in tuned[str(k)]["params"].items() if p != "k_max"}
81
+ helps = scores[k] > zero_shot + 0.5 / n_val
82
+ if not helps: # imagination does not beat zero-shot here: ABA never imagines (the UI's checkbox still can)
83
+ params["epsilon"] = 1.0
84
+ presets[key][str(budget)] = dict(params, k_max=k, attempts_per_slot=effort[0], repair_retries=effort[1],
85
+ verify_rounds=effort[2], imagine_by_default=helps, **SETTING[setting])
86
+ report.append((key, budget, k, setting, round(scores[k], 3), round(zero_shot, 3), round(band[k], 2),
87
+ "" if helps else " -> no gain: zero-shot by default"))
88
+
89
+ out_file.write_text(json.dumps(presets, indent=2))
90
+ for row in report:
91
+ print("%-6s %4ss K_max=%d %-11s val %.3f (zero-shot %.3f) DF band %.2f%s" % row)
tools/fork_test.py ADDED
@@ -0,0 +1,55 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Simulate ZeroGPU locally: load weights on CPU in the parent (CUDA must stay
2
+ uninitialized there), then run whole requests in forked children, which is
3
+ what ``@spaces.GPU`` does on ZeroGPU.
4
+
5
+ CUDA_VISIBLE_DEVICES=0 python tools/fork_test.py bagel [n_requests]
6
+ """
7
+ import multiprocessing as mp
8
+ import sys
9
+ import time
10
+ from pathlib import Path
11
+
12
+ sys.path.insert(0, str(Path(__file__).parent.parent))
13
+ sys.path.insert(0, str(Path(__file__).parent))
14
+
15
+ # ZeroGPU's own main-process emulation (fake CUDA queries, CUDA init forbidden); children undo it,
16
+ # as spaces' worker does before attaching the GPU.
17
+ from spaces.zero.torch import patching # noqa: E402
18
+ import torch # noqa: E402
19
+
20
+ patching.patch()
21
+ from demo_models import MODELS, run # noqa: E402
22
+ from general_qa import general_qa # noqa: E402
23
+
24
+ key = sys.argv[1]
25
+ n = int(sys.argv[2]) if len(sys.argv) > 2 else 2
26
+ spec = MODELS[key]
27
+ t = time.time()
28
+ spec.load()
29
+ print(f"parent: loaded on CPU in {time.time() - t:.0f}s; CUDA initialized in parent: {torch.cuda.is_initialized()}",
30
+ flush=True)
31
+ assert not torch.cuda.is_initialized(), "the main process must not initialize CUDA (ZeroGPU forks it)"
32
+ queries = general_qa(per_subset=1, offset=420)[:n]
33
+
34
+
35
+ def child(i, q):
36
+ patching.unpatch()
37
+ t0 = time.time()
38
+ for ev in run(spec, q["image"], q["question"], 60, False):
39
+ if ev[0] == "greedy":
40
+ print(f" [{i}] greedy={ev[1].answer!r} p0={ev[1].confidence:.2f} k={ev[2]} at {time.time() - t0:.1f}s",
41
+ flush=True)
42
+ elif ev[0] == "demo":
43
+ print(f" [{i}] demo [{ev[1].skill}] {ev[1].question!r} -> {ev[1].answer!r} at {time.time() - t0:.1f}s",
44
+ flush=True)
45
+ elif ev[0] == "final":
46
+ print(f" [{i}] final={ev[1]!r} {ev[2]} refs={q['answers'][:3]} total {time.time() - t0:.1f}s", flush=True)
47
+
48
+
49
+ ctx = mp.get_context("fork")
50
+ for i, q in enumerate(queries):
51
+ p = ctx.Process(target=child, args=(i, q))
52
+ t0 = time.time()
53
+ p.start()
54
+ p.join()
55
+ print(f"request {i} ({q['subset']}): exit={p.exitcode} in {time.time() - t0:.1f}s", flush=True)
tools/general_qa.py ADDED
@@ -0,0 +1,70 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """A mixed "general visual QA" set from LMMs-Eval-Lite, close to what demo
2
+ visitors ask: everyday photos (VizWiz, VQAv2), knowledge (OK-VQA), reading text
3
+ (TextVQA), charts and documents (ChartQA, DocVQA) and multiple-choice reasoning
4
+ (MMBench, AI2D). Each example carries its benchmark's own metric.
5
+ """
6
+
7
+ import ast
8
+
9
+ from datasets import load_dataset
10
+
11
+ VQA = "Answer the question using a single word or phrase."
12
+
13
+
14
+ def _list(v):
15
+ if isinstance(v, list):
16
+ return v
17
+ try:
18
+ v = ast.literal_eval(v)
19
+ except Exception:
20
+ pass
21
+ return v if isinstance(v, list) else [v]
22
+
23
+
24
+ def _options(row, keys):
25
+ opts = [(k, str(row[k])) for k in keys if str(row.get(k, "nan")) not in ("nan", "None", "")]
26
+ return "\n".join(f"{k}. {v}" for k, v in opts)
27
+
28
+
29
+ def _mmbench(r):
30
+ hint = str(r.get("hint", "nan"))
31
+ q = (hint + "\n" if hint not in ("nan", "None", "") else "") + r["question"]
32
+ return (q + "\n" + _options(r, "ABCD") + "\nAnswer with the option's letter from the given choices directly.",
33
+ [str(r["answer"])], "multiple_choice")
34
+
35
+
36
+ def _ai2d(r):
37
+ opts = _list(r["options"])
38
+ q = r["question"] + "\n" + "\n".join(f"{chr(65 + i)}. {o}" for i, o in enumerate(opts))
39
+ return q + "\nAnswer with the option's letter from the given choices directly.", [chr(65 + int(r["answer"]))], \
40
+ "multiple_choice"
41
+
42
+
43
+ SUBSETS = {
44
+ "vizwiz_vqa_val": lambda r: (f"{r['question']}\nWhen the provided information is insufficient, respond with "
45
+ f"'Unanswerable'.\n{VQA}", [str(a) for a in _list(r["answers"])], "vqa_accuracy"),
46
+ "vqav2_val": lambda r: (f"{r['question']}\n{VQA}",
47
+ [a["answer"] if isinstance(a, dict) else str(a) for a in _list(r["answers"])],
48
+ "vqa_accuracy"),
49
+ "ok_vqa_val2014": lambda r: (f"{r['question']}\n{VQA}", [str(a) for a in _list(r["answers"])], "vqa_accuracy"),
50
+ "textvqa_val": lambda r: (f"{r['question']}\n{VQA}", [str(a) for a in _list(r["answers"])], "vqa_accuracy"),
51
+ "chartqa": lambda r: (f"{r['question']}\nAnswer the question with a single word.",
52
+ [str(a) for a in _list(r["answer"])], "relaxed_accuracy"),
53
+ "docvqa_val": lambda r: (f"{r['question']}\n{VQA}", [str(a) for a in _list(r["answers"])], "anls"),
54
+ "mmbench_en_dev": _mmbench,
55
+ "ai2d": _ai2d,
56
+ }
57
+
58
+
59
+ def general_qa(per_subset: int = 8, offset: int = 300) -> list[dict]:
60
+ """``per_subset`` examples from each subset, starting at row ``offset``
61
+ (rows below 300 were used for the package benchmarks)."""
62
+ out = []
63
+ for sub, fn in SUBSETS.items():
64
+ ds = load_dataset("lmms-lab/LMMs-Eval-Lite", sub)["lite"]
65
+ for i in range(offset, offset + per_subset):
66
+ r = ds[i]
67
+ question, answers, metric = fn(r)
68
+ out.append({"image": r["image"].convert("RGB"), "question": question, "answers": answers,
69
+ "metric": metric, "subset": sub})
70
+ return out
tools/latency.py ADDED
@@ -0,0 +1,74 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Per-request timelines on the demo's own code path, as ZeroGPU runs it (a fresh
2
+ forked process per request, weights moved to the GPU each time): when the greedy
3
+ answer and each imagined demo arrive, for forced K and a given speed setting.
4
+
5
+ CUDA_VISIBLE_DEVICES=1 python tools/latency.py lance OUT.json
6
+ """
7
+ import json
8
+ import multiprocessing as mp
9
+ import os
10
+ import sys
11
+ import time
12
+ from pathlib import Path
13
+
14
+ os.environ.setdefault("TRITON_CACHE_AUTOTUNING", "1")
15
+ os.environ.setdefault("TRANSFORMERS_DISABLE_DEEPGEMM_LINEAR", "1")
16
+ sys.path.insert(0, str(Path(__file__).parent.parent))
17
+ sys.path.insert(0, str(Path(__file__).parent))
18
+ from spaces.zero.torch import patching # noqa: E402
19
+ import torch # noqa: E402
20
+
21
+ patching.patch()
22
+ from demo_models import MODELS, run # noqa: E402
23
+ from general_qa import general_qa # noqa: E402
24
+
25
+ ABA = dict(epsilon=0.06, A0=0.5, A1=0.85, B=0.4, t_low=0.0, t_high=1.0)
26
+ SETTINGS = { # speed settings to time (accuracy comes from tune_presets.py)
27
+ "bagel": {"think50": dict(verify=True, verify_think=True, image_steps=50),
28
+ "nothink50": dict(verify=True, verify_think=False, image_steps=50),
29
+ "nothink25": dict(verify=True, verify_think=False, image_steps=25),
30
+ "nocritic25": dict(verify=False, image_steps=25)},
31
+ "lance": {"steps30": dict(image_steps=30), "steps20": dict(image_steps=20)},
32
+ "qwen": {"natural": dict(verify=False, use_skills=False), "skills": dict(verify=False, use_skills=True),
33
+ "critic": dict(verify=True, use_skills=True)},
34
+ }
35
+ KS = {"bagel": [1, 2, 4], "lance": [1, 2, 4], "qwen": [1, 2]}
36
+
37
+ key, out = sys.argv[1], Path(sys.argv[2])
38
+ n_q = int(sys.argv[3]) if len(sys.argv) > 3 else 6
39
+ spec = MODELS[key]
40
+ spec.load()
41
+ assert not torch.cuda.is_initialized()
42
+ queries = general_qa(per_subset=1, offset=440)[:n_q]
43
+
44
+
45
+ def child(conn, q, preset):
46
+ patching.unpatch()
47
+ t0 = time.time()
48
+ rec = {"demos": []}
49
+ spec.preset = lambda budget: dict(preset) # this process only
50
+ for ev in run(spec, q["image"], q["question"], 1000, True):
51
+ t = time.time() - t0
52
+ if ev[0] == "greedy":
53
+ rec["greedy"] = t
54
+ elif ev[0] == "demo":
55
+ rec["demos"].append(t)
56
+ elif ev[0] == "final":
57
+ rec["total"] = t
58
+ conn.send(rec)
59
+
60
+
61
+ results = []
62
+ for sname, setting in SETTINGS[key].items():
63
+ for k in KS[key]:
64
+ preset = dict(ABA, k_max=k, attempts_per_slot=4, repair_retries=1, verify_rounds=2, **setting)
65
+ for qi, q in enumerate(queries):
66
+ a, b = mp.get_context("fork").Pipe()
67
+ p = mp.get_context("fork").Process(target=child, args=(b, q, preset))
68
+ p.start()
69
+ rec = a.recv() if a.poll(900) else {"error": "timeout"}
70
+ p.join(timeout=30)
71
+ rec.update(setting=sname, k=k, query=qi, subset=q["subset"])
72
+ results.append(rec)
73
+ print(json.dumps(rec), flush=True)
74
+ out.write_text(json.dumps(results, indent=1))
tools/make_examples.py ADDED
@@ -0,0 +1,121 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Cached demo examples: real runs of the demo pipeline on the paper's figure
2
+ queries (LMMs-Eval-Lite rows), stored with both answers, the imagined demos and
3
+ the reference answer. ``--keep improved`` keeps only runs where SIMIT-ICL fixed
4
+ the base answer.
5
+
6
+ CUDA_VISIBLE_DEVICES=0 python tools/make_examples.py bagel 60 [--keep improved]
7
+ """
8
+ import argparse
9
+ import json
10
+ import os
11
+ import shutil
12
+ import sys
13
+ import time
14
+ from pathlib import Path
15
+
16
+ os.environ.setdefault("TRITON_CACHE_AUTOTUNING", "1")
17
+ os.environ.setdefault("TRANSFORMERS_DISABLE_DEEPGEMM_LINEAR", "1")
18
+ HERE = Path(__file__).parent.parent
19
+ sys.path.insert(0, str(HERE))
20
+ sys.path.insert(0, str(HERE / "tools"))
21
+
22
+ from datasets import load_dataset # noqa: E402
23
+ from PIL import Image # noqa: E402
24
+
25
+ from demo_models import MODELS, run # noqa: E402
26
+ from general_qa import _list # noqa: E402
27
+ from simit.metrics import get_metric # noqa: E402
28
+
29
+ VQA = "Answer the question using a single word or phrase."
30
+ UNANSWERABLE = "When the provided information is insufficient, respond with 'Unanswerable'."
31
+ # (subset, row, prompt suffix, metric, where it appears in the paper)
32
+ CANDIDATES = [
33
+ ("ok_vqa_val2014", 492, f"{UNANSWERABLE}\n{VQA}", "vqa_accuracy", "paper Fig. (qualitative samples)"),
34
+ ("infovqa_val", 449, VQA, "anls", "paper Fig. (qualitative samples)"),
35
+ ("vizwiz_vqa_val", 345, f"{UNANSWERABLE}\n{VQA}", "vqa_accuracy", "paper teaser"),
36
+ ("vizwiz_vqa_val", 47, f"{UNANSWERABLE}\n{VQA}", "vqa_accuracy", "paper appendix"),
37
+ ("vizwiz_vqa_val", 87, f"{UNANSWERABLE}\n{VQA}", "vqa_accuracy", "paper appendix"),
38
+ ] + [("ok_vqa_val2014", i, VQA, "vqa_accuracy", "paper appendix") for i in (115, 151, 210, 22, 253, 272, 291, 381,
39
+ 477, 8)]
40
+ SUBSET_NAME = {"ok_vqa_val2014": "OK-VQA", "infovqa_val": "InfographicVQA", "vizwiz_vqa_val": "VizWiz"}
41
+
42
+ ap = argparse.ArgumentParser()
43
+ ap.add_argument("model")
44
+ ap.add_argument("budget", type=int)
45
+ ap.add_argument("--keep", choices=["all", "improved"], default="all")
46
+ ap.add_argument("--always", action="store_true", help="imagine even when the model is confident (UI checkbox)")
47
+ ap.add_argument("--out", default=str(HERE / "examples"))
48
+ ap.add_argument("--only", nargs="*", default=None, help="subset:row items to run, e.g. ok_vqa_val2014:492")
49
+ ap.add_argument("--image", help="a custom query image (instead of the dataset candidates)")
50
+ ap.add_argument("--question", help="the question for --image")
51
+ ap.add_argument("--reference", nargs="+", help="reference answer(s) for --image")
52
+ ap.add_argument("--metric", default="contains", help="metric for --image (free-form answers: 'contains')")
53
+ ap.add_argument("--id", help="example id (folder name) for --image")
54
+ ap.add_argument("--source", default="custom example", help="source line shown in the UI for --image")
55
+ args = ap.parse_args()
56
+ out = Path(args.out)
57
+ out.mkdir(parents=True, exist_ok=True)
58
+ index_file = out / "index.json"
59
+ index = json.loads(index_file.read_text()) if index_file.exists() else []
60
+ spec = MODELS[args.model]
61
+ spec.load()
62
+ suffix_id = "_always" if args.always else ""
63
+
64
+
65
+ def jobs():
66
+ """(example id, image, question, references, metric name, source, keep the folder's own files)"""
67
+ if args.image:
68
+ img = Image.open(args.image)
69
+ img.load()
70
+ yield (args.id + suffix_id, img, args.question, args.reference, args.metric, args.source, True)
71
+ return
72
+ for sub, row, suffix, metric_name, where in CANDIDATES:
73
+ if args.only is not None and f"{sub}:{row}" not in args.only:
74
+ continue
75
+ r = load_dataset("lmms-lab/LMMs-Eval-Lite", sub)["lite"][row]
76
+ yield (f"{args.model}_{sub}_{row}{suffix_id}", r["image"].convert("RGB"), f"{r['question']}\n{suffix}",
77
+ [str(a) for a in _list(r.get("answers", r.get("answer")))], metric_name,
78
+ f"{SUBSET_NAME.get(sub, sub)} row {row}, {where}", False)
79
+
80
+
81
+ for ex_id, image, question, refs, metric_name, source, custom in jobs():
82
+ metric = get_metric(metric_name)
83
+ t0 = time.time()
84
+ greedy, demos, final, info = None, [], None, {}
85
+ for ev in run(spec, image, question, args.budget, args.always):
86
+ if ev[0] == "greedy":
87
+ greedy = ev[1]
88
+ elif ev[0] == "demo":
89
+ demos.append(ev[1])
90
+ elif ev[0] == "final":
91
+ final, info = ev[1], ev[2]
92
+ g_ok, s_ok = metric(greedy.answer, refs) >= 0.5, metric(final, refs) >= 0.5
93
+ print(f"{ex_id}: greedy={greedy.answer!r} ({g_ok}) simit={final!r} ({s_ok}) demos={len(demos)} "
94
+ f"{time.time() - t0:.0f}s refs={refs[:3]}", flush=True)
95
+ if args.keep == "improved" and not (s_ok and not g_ok):
96
+ continue
97
+ d = out / ex_id
98
+ if custom and d.exists(): # the user's own files stay; only earlier demo images are replaced
99
+ for f in d.glob("demo_*.png"):
100
+ f.unlink()
101
+ else:
102
+ shutil.rmtree(d, ignore_errors=True)
103
+ d.mkdir()
104
+ if custom and Path(args.image).resolve().parent == d.resolve():
105
+ query_file = Path(args.image).name
106
+ else:
107
+ shown = image.convert("RGB")
108
+ shown.thumbnail((1024, 1024)) # display copy (the run used the original)
109
+ shown.save(d / "query.jpg", quality=90)
110
+ query_file = "query.jpg"
111
+ stored = []
112
+ for i, dm in enumerate(demos):
113
+ dm.image.save(d / f"demo_{i}.png")
114
+ stored.append({"image": f"demo_{i}.png", "question": dm.question, "answer": dm.answer, "skill": dm.skill,
115
+ "verify_score": dm.verify_score, "confidence": dm.confidence})
116
+ entry = {"id": ex_id, "query_file": query_file, "model": args.model, "budget": args.budget,
117
+ "always": args.always, "question": question, "reference": refs[0],
118
+ "greedy": greedy.answer, "p0": greedy.confidence, "simit": final, "greedy_ok": g_ok, "simit_ok": s_ok,
119
+ "demos": stored, "elapsed": info.get("elapsed", time.time() - t0), "source": source}
120
+ index = [e for e in index if e["id"] != ex_id] + [entry]
121
+ index_file.write_text(json.dumps(index, indent=1))
tools/results/latency_bagel.json ADDED
@@ -0,0 +1,882 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ [
2
+ {
3
+ "demos": [
4
+ 26.002389907836914
5
+ ],
6
+ "greedy": 6.846283435821533,
7
+ "total": 26.74322533607483,
8
+ "setting": "think50",
9
+ "k": 1,
10
+ "query": 0,
11
+ "subset": "vizwiz_vqa_val"
12
+ },
13
+ {
14
+ "demos": [
15
+ 12.16827392578125
16
+ ],
17
+ "greedy": 6.742192268371582,
18
+ "total": 12.760454177856445,
19
+ "setting": "think50",
20
+ "k": 1,
21
+ "query": 1,
22
+ "subset": "vqav2_val"
23
+ },
24
+ {
25
+ "demos": [
26
+ 17.263279914855957
27
+ ],
28
+ "greedy": 6.47596001625061,
29
+ "total": 17.900477647781372,
30
+ "setting": "think50",
31
+ "k": 1,
32
+ "query": 2,
33
+ "subset": "ok_vqa_val2014"
34
+ },
35
+ {
36
+ "demos": [
37
+ 33.245251417160034
38
+ ],
39
+ "greedy": 6.677622079849243,
40
+ "total": 34.04009461402893,
41
+ "setting": "think50",
42
+ "k": 1,
43
+ "query": 3,
44
+ "subset": "textvqa_val"
45
+ },
46
+ {
47
+ "demos": [
48
+ 21.499918460845947
49
+ ],
50
+ "greedy": 6.547531366348267,
51
+ "total": 22.275030374526978,
52
+ "setting": "think50",
53
+ "k": 1,
54
+ "query": 4,
55
+ "subset": "chartqa"
56
+ },
57
+ {
58
+ "demos": [],
59
+ "greedy": 6.649888515472412,
60
+ "total": 57.07210087776184,
61
+ "setting": "think50",
62
+ "k": 1,
63
+ "query": 5,
64
+ "subset": "docvqa_val"
65
+ },
66
+ {
67
+ "demos": [
68
+ 14.020853996276855,
69
+ 15.747642755508423
70
+ ],
71
+ "greedy": 6.6902241706848145,
72
+ "total": 16.58476734161377,
73
+ "setting": "think50",
74
+ "k": 2,
75
+ "query": 0,
76
+ "subset": "vizwiz_vqa_val"
77
+ },
78
+ {
79
+ "demos": [
80
+ 14.245272397994995,
81
+ 15.97743272781372
82
+ ],
83
+ "greedy": 6.408905506134033,
84
+ "total": 16.649274826049805,
85
+ "setting": "think50",
86
+ "k": 2,
87
+ "query": 1,
88
+ "subset": "vqav2_val"
89
+ },
90
+ {
91
+ "demos": [
92
+ 14.016824960708618,
93
+ 27.909880876541138
94
+ ],
95
+ "greedy": 6.539390563964844,
96
+ "total": 28.65375280380249,
97
+ "setting": "think50",
98
+ "k": 2,
99
+ "query": 2,
100
+ "subset": "ok_vqa_val2014"
101
+ },
102
+ {
103
+ "demos": [
104
+ 19.595810413360596,
105
+ 20.93066096305847
106
+ ],
107
+ "greedy": 6.817545413970947,
108
+ "total": 21.97844171524048,
109
+ "setting": "think50",
110
+ "k": 2,
111
+ "query": 3,
112
+ "subset": "textvqa_val"
113
+ },
114
+ {
115
+ "demos": [
116
+ 21.255191326141357,
117
+ 23.45720887184143
118
+ ],
119
+ "greedy": 6.85756778717041,
120
+ "total": 24.484545946121216,
121
+ "setting": "think50",
122
+ "k": 2,
123
+ "query": 4,
124
+ "subset": "chartqa"
125
+ },
126
+ {
127
+ "demos": [
128
+ 17.32177734375,
129
+ 29.950555562973022
130
+ ],
131
+ "greedy": 6.608810186386108,
132
+ "total": 30.91857123374939,
133
+ "setting": "think50",
134
+ "k": 2,
135
+ "query": 5,
136
+ "subset": "docvqa_val"
137
+ },
138
+ {
139
+ "demos": [
140
+ 19.331388235092163,
141
+ 20.449653387069702,
142
+ 20.78369402885437,
143
+ 24.906179904937744
144
+ ],
145
+ "greedy": 6.746614217758179,
146
+ "total": 25.91768527030945,
147
+ "setting": "think50",
148
+ "k": 4,
149
+ "query": 0,
150
+ "subset": "vizwiz_vqa_val"
151
+ },
152
+ {
153
+ "demos": [
154
+ 19.866760969161987,
155
+ 21.37235426902771,
156
+ 25.51891589164734,
157
+ 25.743648767471313
158
+ ],
159
+ "greedy": 6.74257230758667,
160
+ "total": 26.58999800682068,
161
+ "setting": "think50",
162
+ "k": 4,
163
+ "query": 1,
164
+ "subset": "vqav2_val"
165
+ },
166
+ {
167
+ "demos": [
168
+ 18.303645849227905,
169
+ 19.48818588256836,
170
+ 19.763233184814453,
171
+ 21.081400394439697
172
+ ],
173
+ "greedy": 6.863516330718994,
174
+ "total": 21.978867053985596,
175
+ "setting": "think50",
176
+ "k": 4,
177
+ "query": 2,
178
+ "subset": "ok_vqa_val2014"
179
+ },
180
+ {
181
+ "demos": [
182
+ 18.599497079849243,
183
+ 47.75624179840088,
184
+ 48.84687924385071
185
+ ],
186
+ "greedy": 6.99056339263916,
187
+ "total": 123.67395544052124,
188
+ "setting": "think50",
189
+ "k": 4,
190
+ "query": 3,
191
+ "subset": "textvqa_val"
192
+ },
193
+ {
194
+ "demos": [
195
+ 16.050732851028442,
196
+ 17.049232006072998,
197
+ 17.45568537712097,
198
+ 17.470638751983643
199
+ ],
200
+ "greedy": 6.753719091415405,
201
+ "total": 18.682735443115234,
202
+ "setting": "think50",
203
+ "k": 4,
204
+ "query": 4,
205
+ "subset": "chartqa"
206
+ },
207
+ {
208
+ "demos": [
209
+ 17.657182216644287,
210
+ 17.93081760406494,
211
+ 28.86550760269165,
212
+ 48.43350315093994
213
+ ],
214
+ "greedy": 6.711917877197266,
215
+ "total": 49.75816011428833,
216
+ "setting": "think50",
217
+ "k": 4,
218
+ "query": 5,
219
+ "subset": "docvqa_val"
220
+ },
221
+ {
222
+ "demos": [
223
+ 12.481452226638794
224
+ ],
225
+ "greedy": 6.7068750858306885,
226
+ "total": 13.248512744903564,
227
+ "setting": "nothink50",
228
+ "k": 1,
229
+ "query": 0,
230
+ "subset": "vizwiz_vqa_val"
231
+ },
232
+ {
233
+ "demos": [
234
+ 12.028658628463745
235
+ ],
236
+ "greedy": 6.6228907108306885,
237
+ "total": 12.620982646942139,
238
+ "setting": "nothink50",
239
+ "k": 1,
240
+ "query": 1,
241
+ "subset": "vqav2_val"
242
+ },
243
+ {
244
+ "demos": [
245
+ 12.994443416595459
246
+ ],
247
+ "greedy": 6.687414646148682,
248
+ "total": 13.635976076126099,
249
+ "setting": "nothink50",
250
+ "k": 1,
251
+ "query": 2,
252
+ "subset": "ok_vqa_val2014"
253
+ },
254
+ {
255
+ "demos": [],
256
+ "greedy": 6.697295427322388,
257
+ "total": 31.110349416732788,
258
+ "setting": "nothink50",
259
+ "k": 1,
260
+ "query": 3,
261
+ "subset": "textvqa_val"
262
+ },
263
+ {
264
+ "demos": [
265
+ 26.508880138397217
266
+ ],
267
+ "greedy": 6.976494312286377,
268
+ "total": 27.29017186164856,
269
+ "setting": "nothink50",
270
+ "k": 1,
271
+ "query": 4,
272
+ "subset": "chartqa"
273
+ },
274
+ {
275
+ "demos": [
276
+ 41.90188384056091
277
+ ],
278
+ "greedy": 6.816613435745239,
279
+ "total": 42.66916561126709,
280
+ "setting": "nothink50",
281
+ "k": 1,
282
+ "query": 5,
283
+ "subset": "docvqa_val"
284
+ },
285
+ {
286
+ "demos": [
287
+ 15.123422622680664,
288
+ 15.220020294189453
289
+ ],
290
+ "greedy": 7.205089807510376,
291
+ "total": 16.069376230239868,
292
+ "setting": "nothink50",
293
+ "k": 2,
294
+ "query": 0,
295
+ "subset": "vizwiz_vqa_val"
296
+ },
297
+ {
298
+ "demos": [
299
+ 16.50620174407959,
300
+ 17.572696685791016
301
+ ],
302
+ "greedy": 7.146985292434692,
303
+ "total": 18.24684238433838,
304
+ "setting": "nothink50",
305
+ "k": 2,
306
+ "query": 1,
307
+ "subset": "vqav2_val"
308
+ },
309
+ {
310
+ "demos": [
311
+ 25.19501304626465,
312
+ 32.75030159950256
313
+ ],
314
+ "greedy": 7.005849599838257,
315
+ "total": 33.467132806777954,
316
+ "setting": "nothink50",
317
+ "k": 2,
318
+ "query": 2,
319
+ "subset": "ok_vqa_val2014"
320
+ },
321
+ {
322
+ "demos": [
323
+ 40.2707622051239,
324
+ 67.3787031173706
325
+ ],
326
+ "greedy": 7.045415639877319,
327
+ "total": 68.47934651374817,
328
+ "setting": "nothink50",
329
+ "k": 2,
330
+ "query": 3,
331
+ "subset": "textvqa_val"
332
+ },
333
+ {
334
+ "demos": [
335
+ 16.73797917366028,
336
+ 26.571113348007202
337
+ ],
338
+ "greedy": 7.133496999740601,
339
+ "total": 27.498658895492554,
340
+ "setting": "nothink50",
341
+ "k": 2,
342
+ "query": 4,
343
+ "subset": "chartqa"
344
+ },
345
+ {
346
+ "demos": [
347
+ 18.925209283828735,
348
+ 45.47921824455261
349
+ ],
350
+ "greedy": 7.158755540847778,
351
+ "total": 46.42701768875122,
352
+ "setting": "nothink50",
353
+ "k": 2,
354
+ "query": 5,
355
+ "subset": "docvqa_val"
356
+ },
357
+ {
358
+ "demos": [
359
+ 18.92365074157715,
360
+ 20.19171667098999,
361
+ 20.651800394058228,
362
+ 24.54624581336975
363
+ ],
364
+ "greedy": 6.934945344924927,
365
+ "total": 25.56127405166626,
366
+ "setting": "nothink50",
367
+ "k": 4,
368
+ "query": 0,
369
+ "subset": "vizwiz_vqa_val"
370
+ },
371
+ {
372
+ "demos": [
373
+ 20.397387266159058,
374
+ 21.484426259994507,
375
+ 21.654940843582153,
376
+ 39.95197796821594
377
+ ],
378
+ "greedy": 6.822445392608643,
379
+ "total": 40.79026508331299,
380
+ "setting": "nothink50",
381
+ "k": 4,
382
+ "query": 1,
383
+ "subset": "vqav2_val"
384
+ },
385
+ {
386
+ "demos": [
387
+ 17.846131563186646,
388
+ 31.462411880493164,
389
+ 31.96941876411438,
390
+ 42.511390209198
391
+ ],
392
+ "greedy": 7.063258647918701,
393
+ "total": 43.39737248420715,
394
+ "setting": "nothink50",
395
+ "k": 4,
396
+ "query": 2,
397
+ "subset": "ok_vqa_val2014"
398
+ },
399
+ {
400
+ "demos": [
401
+ 31.60993504524231,
402
+ 32.77745318412781,
403
+ 46.97655916213989,
404
+ 67.76566171646118
405
+ ],
406
+ "greedy": 7.10393762588501,
407
+ "total": 69.44919919967651,
408
+ "setting": "nothink50",
409
+ "k": 4,
410
+ "query": 3,
411
+ "subset": "textvqa_val"
412
+ },
413
+ {
414
+ "demos": [
415
+ 13.910687446594238,
416
+ 18.546985864639282,
417
+ 19.083553314208984,
418
+ 19.979618787765503
419
+ ],
420
+ "greedy": 6.712128639221191,
421
+ "total": 21.244775533676147,
422
+ "setting": "nothink50",
423
+ "k": 4,
424
+ "query": 4,
425
+ "subset": "chartqa"
426
+ },
427
+ {
428
+ "demos": [
429
+ 27.11043953895569,
430
+ 36.58712029457092,
431
+ 41.17549681663513,
432
+ 51.479180574417114
433
+ ],
434
+ "greedy": 6.7352728843688965,
435
+ "total": 52.88075041770935,
436
+ "setting": "nothink50",
437
+ "k": 4,
438
+ "query": 5,
439
+ "subset": "docvqa_val"
440
+ },
441
+ {
442
+ "demos": [
443
+ 15.192901134490967
444
+ ],
445
+ "greedy": 6.91644811630249,
446
+ "total": 15.934749126434326,
447
+ "setting": "nothink25",
448
+ "k": 1,
449
+ "query": 0,
450
+ "subset": "vizwiz_vqa_val"
451
+ },
452
+ {
453
+ "demos": [
454
+ 14.546730279922485
455
+ ],
456
+ "greedy": 6.616936206817627,
457
+ "total": 15.137541770935059,
458
+ "setting": "nothink25",
459
+ "k": 1,
460
+ "query": 1,
461
+ "subset": "vqav2_val"
462
+ },
463
+ {
464
+ "demos": [
465
+ 10.399396657943726
466
+ ],
467
+ "greedy": 6.794123411178589,
468
+ "total": 11.040837049484253,
469
+ "setting": "nothink25",
470
+ "k": 1,
471
+ "query": 2,
472
+ "subset": "ok_vqa_val2014"
473
+ },
474
+ {
475
+ "demos": [],
476
+ "greedy": 6.857384443283081,
477
+ "total": 36.08436679840088,
478
+ "setting": "nothink25",
479
+ "k": 1,
480
+ "query": 3,
481
+ "subset": "textvqa_val"
482
+ },
483
+ {
484
+ "demos": [
485
+ 17.479740619659424
486
+ ],
487
+ "greedy": 6.688612699508667,
488
+ "total": 18.236947536468506,
489
+ "setting": "nothink25",
490
+ "k": 1,
491
+ "query": 4,
492
+ "subset": "chartqa"
493
+ },
494
+ {
495
+ "demos": [
496
+ 59.99591255187988
497
+ ],
498
+ "greedy": 6.7639524936676025,
499
+ "total": 60.8376727104187,
500
+ "setting": "nothink25",
501
+ "k": 1,
502
+ "query": 5,
503
+ "subset": "docvqa_val"
504
+ },
505
+ {
506
+ "demos": [
507
+ 12.354958057403564,
508
+ 19.66935706138611
509
+ ],
510
+ "greedy": 6.810552358627319,
511
+ "total": 20.500991344451904,
512
+ "setting": "nothink25",
513
+ "k": 2,
514
+ "query": 0,
515
+ "subset": "vizwiz_vqa_val"
516
+ },
517
+ {
518
+ "demos": [
519
+ 13.075764179229736,
520
+ 13.986951351165771
521
+ ],
522
+ "greedy": 6.526533842086792,
523
+ "total": 14.659094333648682,
524
+ "setting": "nothink25",
525
+ "k": 2,
526
+ "query": 1,
527
+ "subset": "vqav2_val"
528
+ },
529
+ {
530
+ "demos": [
531
+ 11.25294542312622,
532
+ 12.738617658615112
533
+ ],
534
+ "greedy": 6.653463363647461,
535
+ "total": 13.45282244682312,
536
+ "setting": "nothink25",
537
+ "k": 2,
538
+ "query": 2,
539
+ "subset": "ok_vqa_val2014"
540
+ },
541
+ {
542
+ "demos": [
543
+ 52.51059865951538
544
+ ],
545
+ "greedy": 6.916888952255249,
546
+ "total": 69.37151217460632,
547
+ "setting": "nothink25",
548
+ "k": 2,
549
+ "query": 3,
550
+ "subset": "textvqa_val"
551
+ },
552
+ {
553
+ "demos": [
554
+ 16.357442378997803,
555
+ 22.164559364318848
556
+ ],
557
+ "greedy": 6.922367811203003,
558
+ "total": 23.112447500228882,
559
+ "setting": "nothink25",
560
+ "k": 2,
561
+ "query": 4,
562
+ "subset": "chartqa"
563
+ },
564
+ {
565
+ "demos": [
566
+ 31.607569217681885,
567
+ 64.07539129257202
568
+ ],
569
+ "greedy": 7.111017942428589,
570
+ "total": 65.04016399383545,
571
+ "setting": "nothink25",
572
+ "k": 2,
573
+ "query": 5,
574
+ "subset": "docvqa_val"
575
+ },
576
+ {
577
+ "demos": [
578
+ 14.144917011260986,
579
+ 14.450908422470093,
580
+ 14.969972133636475,
581
+ 22.031485557556152
582
+ ],
583
+ "greedy": 6.845429420471191,
584
+ "total": 23.0630886554718,
585
+ "setting": "nothink25",
586
+ "k": 4,
587
+ "query": 0,
588
+ "subset": "vizwiz_vqa_val"
589
+ },
590
+ {
591
+ "demos": [
592
+ 15.669585943222046,
593
+ 19.648891925811768,
594
+ 20.805134534835815,
595
+ 29.888363122940063
596
+ ],
597
+ "greedy": 6.622054576873779,
598
+ "total": 30.800080060958862,
599
+ "setting": "nothink25",
600
+ "k": 4,
601
+ "query": 1,
602
+ "subset": "vqav2_val"
603
+ },
604
+ {
605
+ "demos": [
606
+ 15.39557409286499,
607
+ 16.348963260650635,
608
+ 19.663447618484497,
609
+ 20.33902072906494
610
+ ],
611
+ "greedy": 6.637497663497925,
612
+ "total": 21.22013282775879,
613
+ "setting": "nothink25",
614
+ "k": 4,
615
+ "query": 2,
616
+ "subset": "ok_vqa_val2014"
617
+ },
618
+ {
619
+ "demos": [
620
+ 15.362420558929443,
621
+ 15.921366691589355,
622
+ 24.226126670837402,
623
+ 45.1386661529541
624
+ ],
625
+ "greedy": 6.875400066375732,
626
+ "total": 46.43324565887451,
627
+ "setting": "nothink25",
628
+ "k": 4,
629
+ "query": 3,
630
+ "subset": "textvqa_val"
631
+ },
632
+ {
633
+ "demos": [
634
+ 15.788510084152222,
635
+ 16.097867727279663,
636
+ 19.899868726730347,
637
+ 24.75997519493103
638
+ ],
639
+ "greedy": 7.031396865844727,
640
+ "total": 26.027679443359375,
641
+ "setting": "nothink25",
642
+ "k": 4,
643
+ "query": 4,
644
+ "subset": "chartqa"
645
+ },
646
+ {
647
+ "demos": [
648
+ 20.376034021377563,
649
+ 25.979355096817017,
650
+ 39.69830060005188,
651
+ 47.3364462852478
652
+ ],
653
+ "greedy": 7.084344863891602,
654
+ "total": 48.586371660232544,
655
+ "setting": "nothink25",
656
+ "k": 4,
657
+ "query": 5,
658
+ "subset": "docvqa_val"
659
+ },
660
+ {
661
+ "demos": [
662
+ 10.09440541267395
663
+ ],
664
+ "greedy": 7.1284401416778564,
665
+ "total": 10.858816146850586,
666
+ "setting": "nocritic25",
667
+ "k": 1,
668
+ "query": 0,
669
+ "subset": "vizwiz_vqa_val"
670
+ },
671
+ {
672
+ "demos": [
673
+ 9.839585781097412
674
+ ],
675
+ "greedy": 6.931880235671997,
676
+ "total": 10.431512594223022,
677
+ "setting": "nocritic25",
678
+ "k": 1,
679
+ "query": 1,
680
+ "subset": "vqav2_val"
681
+ },
682
+ {
683
+ "demos": [
684
+ 9.52512812614441
685
+ ],
686
+ "greedy": 6.723513603210449,
687
+ "total": 10.16373586654663,
688
+ "setting": "nocritic25",
689
+ "k": 1,
690
+ "query": 2,
691
+ "subset": "ok_vqa_val2014"
692
+ },
693
+ {
694
+ "demos": [
695
+ 10.409209489822388
696
+ ],
697
+ "greedy": 6.779156446456909,
698
+ "total": 11.201088905334473,
699
+ "setting": "nocritic25",
700
+ "k": 1,
701
+ "query": 3,
702
+ "subset": "textvqa_val"
703
+ },
704
+ {
705
+ "demos": [
706
+ 13.039257526397705
707
+ ],
708
+ "greedy": 6.719432830810547,
709
+ "total": 13.820397138595581,
710
+ "setting": "nocritic25",
711
+ "k": 1,
712
+ "query": 4,
713
+ "subset": "chartqa"
714
+ },
715
+ {
716
+ "demos": [
717
+ 13.39191722869873
718
+ ],
719
+ "greedy": 6.744472503662109,
720
+ "total": 13.756275415420532,
721
+ "setting": "nocritic25",
722
+ "k": 1,
723
+ "query": 5,
724
+ "subset": "docvqa_val"
725
+ },
726
+ {
727
+ "demos": [
728
+ 10.131142616271973,
729
+ 10.941117525100708
730
+ ],
731
+ "greedy": 6.75618314743042,
732
+ "total": 11.777562379837036,
733
+ "setting": "nocritic25",
734
+ "k": 2,
735
+ "query": 0,
736
+ "subset": "vizwiz_vqa_val"
737
+ },
738
+ {
739
+ "demos": [
740
+ 9.828729629516602,
741
+ 10.91353988647461
742
+ ],
743
+ "greedy": 6.636314868927002,
744
+ "total": 11.586650609970093,
745
+ "setting": "nocritic25",
746
+ "k": 2,
747
+ "query": 1,
748
+ "subset": "vqav2_val"
749
+ },
750
+ {
751
+ "demos": [
752
+ 10.159326076507568,
753
+ 11.352014064788818
754
+ ],
755
+ "greedy": 7.0619635581970215,
756
+ "total": 12.062993288040161,
757
+ "setting": "nocritic25",
758
+ "k": 2,
759
+ "query": 2,
760
+ "subset": "ok_vqa_val2014"
761
+ },
762
+ {
763
+ "demos": [
764
+ 12.485306739807129,
765
+ 37.578511238098145
766
+ ],
767
+ "greedy": 6.796743392944336,
768
+ "total": 38.57980442047119,
769
+ "setting": "nocritic25",
770
+ "k": 2,
771
+ "query": 3,
772
+ "subset": "textvqa_val"
773
+ },
774
+ {
775
+ "demos": [
776
+ 13.873923778533936,
777
+ 15.028538465499878
778
+ ],
779
+ "greedy": 6.755588054656982,
780
+ "total": 15.933726072311401,
781
+ "setting": "nocritic25",
782
+ "k": 2,
783
+ "query": 4,
784
+ "subset": "chartqa"
785
+ },
786
+ {
787
+ "demos": [
788
+ 13.108296394348145,
789
+ 32.831605195999146
790
+ ],
791
+ "greedy": 6.732863426208496,
792
+ "total": 33.64893388748169,
793
+ "setting": "nocritic25",
794
+ "k": 2,
795
+ "query": 5,
796
+ "subset": "docvqa_val"
797
+ },
798
+ {
799
+ "demos": [
800
+ 10.41283893585205,
801
+ 12.226398944854736,
802
+ 13.141302824020386,
803
+ 13.47109079360962
804
+ ],
805
+ "greedy": 6.717079401016235,
806
+ "total": 14.473924160003662,
807
+ "setting": "nocritic25",
808
+ "k": 4,
809
+ "query": 0,
810
+ "subset": "vizwiz_vqa_val"
811
+ },
812
+ {
813
+ "demos": [
814
+ 10.277020931243896,
815
+ 12.27582597732544,
816
+ 13.355355024337769,
817
+ 14.00516128540039
818
+ ],
819
+ "greedy": 6.814017057418823,
820
+ "total": 14.8407723903656,
821
+ "setting": "nocritic25",
822
+ "k": 4,
823
+ "query": 1,
824
+ "subset": "vqav2_val"
825
+ },
826
+ {
827
+ "demos": [
828
+ 11.423258543014526,
829
+ 13.428069591522217,
830
+ 14.622854709625244,
831
+ 16.18061876296997
832
+ ],
833
+ "greedy": 6.668004751205444,
834
+ "total": 17.06151270866394,
835
+ "setting": "nocritic25",
836
+ "k": 4,
837
+ "query": 2,
838
+ "subset": "ok_vqa_val2014"
839
+ },
840
+ {
841
+ "demos": [
842
+ 10.60582423210144,
843
+ 16.0261173248291,
844
+ 16.700007915496826,
845
+ 33.09624981880188
846
+ ],
847
+ "greedy": 6.843786954879761,
848
+ "total": 34.594066858291626,
849
+ "setting": "nocritic25",
850
+ "k": 4,
851
+ "query": 3,
852
+ "subset": "textvqa_val"
853
+ },
854
+ {
855
+ "demos": [
856
+ 13.488800764083862,
857
+ 17.300123929977417,
858
+ 18.207062244415283,
859
+ 18.684130907058716
860
+ ],
861
+ "greedy": 6.735860824584961,
862
+ "total": 19.977572679519653,
863
+ "setting": "nocritic25",
864
+ "k": 4,
865
+ "query": 4,
866
+ "subset": "chartqa"
867
+ },
868
+ {
869
+ "demos": [
870
+ 12.050382614135742,
871
+ 14.343275308609009,
872
+ 15.249944925308228,
873
+ 15.570221662521362
874
+ ],
875
+ "greedy": 6.8102805614471436,
876
+ "total": 16.96952247619629,
877
+ "setting": "nocritic25",
878
+ "k": 4,
879
+ "query": 5,
880
+ "subset": "docvqa_val"
881
+ }
882
+ ]
tools/results/latency_lance.json ADDED
@@ -0,0 +1,446 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ [
2
+ {
3
+ "demos": [
4
+ 7.727382183074951
5
+ ],
6
+ "greedy": 4.262204647064209,
7
+ "total": 8.35633897781372,
8
+ "setting": "steps30",
9
+ "k": 1,
10
+ "query": 0,
11
+ "subset": "vizwiz_vqa_val"
12
+ },
13
+ {
14
+ "demos": [
15
+ 5.997106552124023
16
+ ],
17
+ "greedy": 4.037919521331787,
18
+ "total": 6.613722324371338,
19
+ "setting": "steps30",
20
+ "k": 1,
21
+ "query": 1,
22
+ "subset": "vqav2_val"
23
+ },
24
+ {
25
+ "demos": [
26
+ 7.009830474853516
27
+ ],
28
+ "greedy": 4.070275545120239,
29
+ "total": 7.65753698348999,
30
+ "setting": "steps30",
31
+ "k": 1,
32
+ "query": 2,
33
+ "subset": "ok_vqa_val2014"
34
+ },
35
+ {
36
+ "demos": [
37
+ 14.579021215438843
38
+ ],
39
+ "greedy": 3.9722044467926025,
40
+ "total": 15.173923015594482,
41
+ "setting": "steps30",
42
+ "k": 1,
43
+ "query": 3,
44
+ "subset": "textvqa_val"
45
+ },
46
+ {
47
+ "demos": [
48
+ 7.009682893753052
49
+ ],
50
+ "greedy": 4.298367500305176,
51
+ "total": 7.6930906772613525,
52
+ "setting": "steps30",
53
+ "k": 1,
54
+ "query": 4,
55
+ "subset": "chartqa"
56
+ },
57
+ {
58
+ "demos": [
59
+ 9.488945960998535
60
+ ],
61
+ "greedy": 4.068331241607666,
62
+ "total": 10.129173040390015,
63
+ "setting": "steps30",
64
+ "k": 1,
65
+ "query": 5,
66
+ "subset": "docvqa_val"
67
+ },
68
+ {
69
+ "demos": [
70
+ 8.336636781692505,
71
+ 10.828430652618408
72
+ ],
73
+ "greedy": 4.046021223068237,
74
+ "total": 11.519325733184814,
75
+ "setting": "steps30",
76
+ "k": 2,
77
+ "query": 0,
78
+ "subset": "vizwiz_vqa_val"
79
+ },
80
+ {
81
+ "demos": [
82
+ 8.213298082351685,
83
+ 10.596574068069458
84
+ ],
85
+ "greedy": 4.0463056564331055,
86
+ "total": 11.278587102890015,
87
+ "setting": "steps30",
88
+ "k": 2,
89
+ "query": 1,
90
+ "subset": "vqav2_val"
91
+ },
92
+ {
93
+ "demos": [
94
+ 8.57367467880249,
95
+ 10.886900663375854
96
+ ],
97
+ "greedy": 4.036629676818848,
98
+ "total": 11.565510511398315,
99
+ "setting": "steps30",
100
+ "k": 2,
101
+ "query": 2,
102
+ "subset": "ok_vqa_val2014"
103
+ },
104
+ {
105
+ "demos": [
106
+ 11.43087124824524,
107
+ 13.566925525665283
108
+ ],
109
+ "greedy": 3.9821929931640625,
110
+ "total": 14.221354246139526,
111
+ "setting": "steps30",
112
+ "k": 2,
113
+ "query": 3,
114
+ "subset": "textvqa_val"
115
+ },
116
+ {
117
+ "demos": [
118
+ 9.571406841278076,
119
+ 10.061784029006958
120
+ ],
121
+ "greedy": 4.069300174713135,
122
+ "total": 10.802971124649048,
123
+ "setting": "steps30",
124
+ "k": 2,
125
+ "query": 4,
126
+ "subset": "chartqa"
127
+ },
128
+ {
129
+ "demos": [
130
+ 8.667640209197998,
131
+ 11.335434436798096
132
+ ],
133
+ "greedy": 4.0576910972595215,
134
+ "total": 12.056833982467651,
135
+ "setting": "steps30",
136
+ "k": 2,
137
+ "query": 5,
138
+ "subset": "docvqa_val"
139
+ },
140
+ {
141
+ "demos": [
142
+ 9.209861993789673,
143
+ 10.122817754745483,
144
+ 11.41738224029541,
145
+ 11.986809253692627
146
+ ],
147
+ "greedy": 4.213792562484741,
148
+ "total": 13.086158275604248,
149
+ "setting": "steps30",
150
+ "k": 4,
151
+ "query": 0,
152
+ "subset": "vizwiz_vqa_val"
153
+ },
154
+ {
155
+ "demos": [
156
+ 8.620089054107666,
157
+ 11.2899010181427,
158
+ 12.844002962112427,
159
+ 13.42651104927063
160
+ ],
161
+ "greedy": 4.100450038909912,
162
+ "total": 14.261920928955078,
163
+ "setting": "steps30",
164
+ "k": 4,
165
+ "query": 1,
166
+ "subset": "vqav2_val"
167
+ },
168
+ {
169
+ "demos": [
170
+ 7.910981893539429,
171
+ 10.294513940811157,
172
+ 10.294527292251587,
173
+ 11.984448909759521
174
+ ],
175
+ "greedy": 4.040694236755371,
176
+ "total": 12.851784229278564,
177
+ "setting": "steps30",
178
+ "k": 4,
179
+ "query": 2,
180
+ "subset": "ok_vqa_val2014"
181
+ },
182
+ {
183
+ "demos": [
184
+ 8.954644203186035,
185
+ 12.185572862625122,
186
+ 12.858001470565796,
187
+ 15.00017762184143
188
+ ],
189
+ "greedy": 3.9677560329437256,
190
+ "total": 15.794921159744263,
191
+ "setting": "steps30",
192
+ "k": 4,
193
+ "query": 3,
194
+ "subset": "textvqa_val"
195
+ },
196
+ {
197
+ "demos": [
198
+ 7.161853790283203,
199
+ 10.45381784439087,
200
+ 11.22620153427124,
201
+ 13.835843801498413
202
+ ],
203
+ "greedy": 4.065622329711914,
204
+ "total": 14.741939306259155,
205
+ "setting": "steps30",
206
+ "k": 4,
207
+ "query": 4,
208
+ "subset": "chartqa"
209
+ },
210
+ {
211
+ "demos": [
212
+ 9.17311716079712,
213
+ 9.261759042739868,
214
+ 9.878037929534912,
215
+ 10.475617170333862
216
+ ],
217
+ "greedy": 4.253077268600464,
218
+ "total": 11.615146398544312,
219
+ "setting": "steps30",
220
+ "k": 4,
221
+ "query": 5,
222
+ "subset": "docvqa_val"
223
+ },
224
+ {
225
+ "demos": [
226
+ 9.40204405784607
227
+ ],
228
+ "greedy": 4.1254284381866455,
229
+ "total": 10.027106285095215,
230
+ "setting": "steps20",
231
+ "k": 1,
232
+ "query": 0,
233
+ "subset": "vizwiz_vqa_val"
234
+ },
235
+ {
236
+ "demos": [
237
+ 7.260383367538452
238
+ ],
239
+ "greedy": 4.095474481582642,
240
+ "total": 7.883589267730713,
241
+ "setting": "steps20",
242
+ "k": 1,
243
+ "query": 1,
244
+ "subset": "vqav2_val"
245
+ },
246
+ {
247
+ "demos": [
248
+ 8.731263637542725
249
+ ],
250
+ "greedy": 4.012105703353882,
251
+ "total": 9.378830432891846,
252
+ "setting": "steps20",
253
+ "k": 1,
254
+ "query": 2,
255
+ "subset": "ok_vqa_val2014"
256
+ },
257
+ {
258
+ "demos": [
259
+ 11.81164813041687
260
+ ],
261
+ "greedy": 4.001217842102051,
262
+ "total": 12.418342351913452,
263
+ "setting": "steps20",
264
+ "k": 1,
265
+ "query": 3,
266
+ "subset": "textvqa_val"
267
+ },
268
+ {
269
+ "demos": [
270
+ 6.208858489990234
271
+ ],
272
+ "greedy": 4.017943859100342,
273
+ "total": 6.826139211654663,
274
+ "setting": "steps20",
275
+ "k": 1,
276
+ "query": 4,
277
+ "subset": "chartqa"
278
+ },
279
+ {
280
+ "demos": [
281
+ 9.363977670669556
282
+ ],
283
+ "greedy": 4.0784912109375,
284
+ "total": 10.00316596031189,
285
+ "setting": "steps20",
286
+ "k": 1,
287
+ "query": 5,
288
+ "subset": "docvqa_val"
289
+ },
290
+ {
291
+ "demos": [
292
+ 8.895264863967896,
293
+ 11.027926921844482
294
+ ],
295
+ "greedy": 4.093302011489868,
296
+ "total": 11.739497900009155,
297
+ "setting": "steps20",
298
+ "k": 2,
299
+ "query": 0,
300
+ "subset": "vizwiz_vqa_val"
301
+ },
302
+ {
303
+ "demos": [
304
+ 8.46868896484375,
305
+ 8.78795862197876
306
+ ],
307
+ "greedy": 4.05513858795166,
308
+ "total": 9.470828771591187,
309
+ "setting": "steps20",
310
+ "k": 2,
311
+ "query": 1,
312
+ "subset": "vqav2_val"
313
+ },
314
+ {
315
+ "demos": [
316
+ 7.852817535400391,
317
+ 8.184361457824707
318
+ ],
319
+ "greedy": 4.0655388832092285,
320
+ "total": 8.493193626403809,
321
+ "setting": "steps20",
322
+ "k": 2,
323
+ "query": 2,
324
+ "subset": "ok_vqa_val2014"
325
+ },
326
+ {
327
+ "demos": [
328
+ 7.7487473487854,
329
+ 9.187946081161499
330
+ ],
331
+ "greedy": 4.10131573677063,
332
+ "total": 9.855353355407715,
333
+ "setting": "steps20",
334
+ "k": 2,
335
+ "query": 3,
336
+ "subset": "textvqa_val"
337
+ },
338
+ {
339
+ "demos": [
340
+ 8.990880966186523,
341
+ 9.044522285461426
342
+ ],
343
+ "greedy": 4.088159084320068,
344
+ "total": 9.791136264801025,
345
+ "setting": "steps20",
346
+ "k": 2,
347
+ "query": 4,
348
+ "subset": "chartqa"
349
+ },
350
+ {
351
+ "demos": [
352
+ 6.566630125045776,
353
+ 8.522355079650879
354
+ ],
355
+ "greedy": 4.04767918586731,
356
+ "total": 9.243301391601562,
357
+ "setting": "steps20",
358
+ "k": 2,
359
+ "query": 5,
360
+ "subset": "docvqa_val"
361
+ },
362
+ {
363
+ "demos": [
364
+ 8.488969802856445,
365
+ 10.305099725723267,
366
+ 10.305113554000854,
367
+ 10.661272048950195
368
+ ],
369
+ "greedy": 3.986391544342041,
370
+ "total": 11.507165431976318,
371
+ "setting": "steps20",
372
+ "k": 4,
373
+ "query": 0,
374
+ "subset": "vizwiz_vqa_val"
375
+ },
376
+ {
377
+ "demos": [
378
+ 8.554109811782837,
379
+ 10.155081510543823,
380
+ 10.1550931930542,
381
+ 10.528407335281372
382
+ ],
383
+ "greedy": 4.004652976989746,
384
+ "total": 11.368907690048218,
385
+ "setting": "steps20",
386
+ "k": 4,
387
+ "query": 1,
388
+ "subset": "vqav2_val"
389
+ },
390
+ {
391
+ "demos": [
392
+ 7.51558256149292,
393
+ 8.081176280975342,
394
+ 9.272027015686035,
395
+ 9.97708535194397
396
+ ],
397
+ "greedy": 4.020036935806274,
398
+ "total": 10.843841552734375,
399
+ "setting": "steps20",
400
+ "k": 4,
401
+ "query": 2,
402
+ "subset": "ok_vqa_val2014"
403
+ },
404
+ {
405
+ "demos": [
406
+ 7.948291540145874,
407
+ 8.477357387542725,
408
+ 10.59636402130127,
409
+ 12.149571418762207
410
+ ],
411
+ "greedy": 3.9414212703704834,
412
+ "total": 12.953968048095703,
413
+ "setting": "steps20",
414
+ "k": 4,
415
+ "query": 3,
416
+ "subset": "textvqa_val"
417
+ },
418
+ {
419
+ "demos": [
420
+ 7.972762107849121,
421
+ 9.431615591049194,
422
+ 9.854741096496582,
423
+ 12.329490423202515
424
+ ],
425
+ "greedy": 4.051168918609619,
426
+ "total": 13.170568704605103,
427
+ "setting": "steps20",
428
+ "k": 4,
429
+ "query": 4,
430
+ "subset": "chartqa"
431
+ },
432
+ {
433
+ "demos": [
434
+ 8.014616250991821,
435
+ 9.337750434875488,
436
+ 9.904663562774658,
437
+ 13.779886245727539
438
+ ],
439
+ "greedy": 4.087887763977051,
440
+ "total": 14.651642084121704,
441
+ "setting": "steps20",
442
+ "k": 4,
443
+ "query": 5,
444
+ "subset": "docvqa_val"
445
+ }
446
+ ]
tools/results/latency_qwen.json ADDED
@@ -0,0 +1,347 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ [
2
+ {
3
+ "demos": [
4
+ 30.282771110534668
5
+ ],
6
+ "greedy": 12.503609418869019,
7
+ "total": 31.559006452560425,
8
+ "setting": "natural",
9
+ "k": 1,
10
+ "query": 0,
11
+ "subset": "vizwiz_vqa_val"
12
+ },
13
+ {
14
+ "demos": [
15
+ 24.56417155265808
16
+ ],
17
+ "greedy": 12.057083129882812,
18
+ "total": 25.436967611312866,
19
+ "setting": "natural",
20
+ "k": 1,
21
+ "query": 1,
22
+ "subset": "vqav2_val"
23
+ },
24
+ {
25
+ "demos": [
26
+ 27.104141235351562
27
+ ],
28
+ "greedy": 11.834922552108765,
29
+ "total": 28.232157707214355,
30
+ "setting": "natural",
31
+ "k": 1,
32
+ "query": 2,
33
+ "subset": "ok_vqa_val2014"
34
+ },
35
+ {
36
+ "demos": [
37
+ 30.065296173095703
38
+ ],
39
+ "greedy": 11.83328104019165,
40
+ "total": 31.18956470489502,
41
+ "setting": "natural",
42
+ "k": 1,
43
+ "query": 3,
44
+ "subset": "textvqa_val"
45
+ },
46
+ {
47
+ "demos": [
48
+ 28.98874258995056
49
+ ],
50
+ "greedy": 12.290349006652832,
51
+ "total": 30.4953191280365,
52
+ "setting": "natural",
53
+ "k": 1,
54
+ "query": 4,
55
+ "subset": "chartqa"
56
+ },
57
+ {
58
+ "demos": [
59
+ 49.92998290061951,
60
+ 49.93000912666321
61
+ ],
62
+ "greedy": 12.474754333496094,
63
+ "total": 51.23621082305908,
64
+ "setting": "natural",
65
+ "k": 2,
66
+ "query": 0,
67
+ "subset": "vizwiz_vqa_val"
68
+ },
69
+ {
70
+ "demos": [
71
+ 39.91589426994324,
72
+ 39.91593027114868
73
+ ],
74
+ "greedy": 11.602417945861816,
75
+ "total": 40.8339729309082,
76
+ "setting": "natural",
77
+ "k": 2,
78
+ "query": 1,
79
+ "subset": "vqav2_val"
80
+ },
81
+ {
82
+ "demos": [
83
+ 44.16777563095093,
84
+ 44.16781187057495
85
+ ],
86
+ "greedy": 11.90004849433899,
87
+ "total": 45.30206370353699,
88
+ "setting": "natural",
89
+ "k": 2,
90
+ "query": 2,
91
+ "subset": "ok_vqa_val2014"
92
+ },
93
+ {
94
+ "demos": [
95
+ 39.3218777179718,
96
+ 39.32190752029419
97
+ ],
98
+ "greedy": 12.019229888916016,
99
+ "total": 40.492233753204346,
100
+ "setting": "natural",
101
+ "k": 2,
102
+ "query": 3,
103
+ "subset": "textvqa_val"
104
+ },
105
+ {
106
+ "demos": [
107
+ 102.98197150230408,
108
+ 102.9820077419281
109
+ ],
110
+ "greedy": 11.957572221755981,
111
+ "total": 104.37802076339722,
112
+ "setting": "natural",
113
+ "k": 2,
114
+ "query": 4,
115
+ "subset": "chartqa"
116
+ },
117
+ {
118
+ "demos": [
119
+ 34.93216681480408
120
+ ],
121
+ "greedy": 12.076660633087158,
122
+ "total": 36.01313900947571,
123
+ "setting": "skills",
124
+ "k": 1,
125
+ "query": 0,
126
+ "subset": "vizwiz_vqa_val"
127
+ },
128
+ {
129
+ "demos": [
130
+ 26.665636777877808
131
+ ],
132
+ "greedy": 11.77315878868103,
133
+ "total": 27.53944206237793,
134
+ "setting": "skills",
135
+ "k": 1,
136
+ "query": 1,
137
+ "subset": "vqav2_val"
138
+ },
139
+ {
140
+ "demos": [
141
+ 30.65152359008789
142
+ ],
143
+ "greedy": 11.954692363739014,
144
+ "total": 31.91823172569275,
145
+ "setting": "skills",
146
+ "k": 1,
147
+ "query": 2,
148
+ "subset": "ok_vqa_val2014"
149
+ },
150
+ {
151
+ "demos": [
152
+ 32.95020818710327
153
+ ],
154
+ "greedy": 11.763028144836426,
155
+ "total": 35.805758476257324,
156
+ "setting": "skills",
157
+ "k": 1,
158
+ "query": 3,
159
+ "subset": "textvqa_val"
160
+ },
161
+ {
162
+ "demos": [
163
+ 30.482857704162598
164
+ ],
165
+ "greedy": 12.288076162338257,
166
+ "total": 31.9945011138916,
167
+ "setting": "skills",
168
+ "k": 1,
169
+ "query": 4,
170
+ "subset": "chartqa"
171
+ },
172
+ {
173
+ "demos": [
174
+ 47.012778997421265,
175
+ 47.012813568115234
176
+ ],
177
+ "greedy": 11.944422960281372,
178
+ "total": 48.55853629112244,
179
+ "setting": "skills",
180
+ "k": 2,
181
+ "query": 0,
182
+ "subset": "vizwiz_vqa_val"
183
+ },
184
+ {
185
+ "demos": [
186
+ 43.62376880645752,
187
+ 44.03544592857361
188
+ ],
189
+ "greedy": 11.559141635894775,
190
+ "total": 44.94946813583374,
191
+ "setting": "skills",
192
+ "k": 2,
193
+ "query": 1,
194
+ "subset": "vqav2_val"
195
+ },
196
+ {
197
+ "demos": [
198
+ 44.78076720237732,
199
+ 44.78083920478821
200
+ ],
201
+ "greedy": 12.2094886302948,
202
+ "total": 45.610127210617065,
203
+ "setting": "skills",
204
+ "k": 2,
205
+ "query": 2,
206
+ "subset": "ok_vqa_val2014"
207
+ },
208
+ {
209
+ "demos": [
210
+ 39.1291127204895,
211
+ 39.12914180755615
212
+ ],
213
+ "greedy": 11.88507342338562,
214
+ "total": 40.32528471946716,
215
+ "setting": "skills",
216
+ "k": 2,
217
+ "query": 3,
218
+ "subset": "textvqa_val"
219
+ },
220
+ {
221
+ "demos": [
222
+ 90.45457100868225,
223
+ 90.8869194984436
224
+ ],
225
+ "greedy": 12.611650466918945,
226
+ "total": 92.93272161483765,
227
+ "setting": "skills",
228
+ "k": 2,
229
+ "query": 4,
230
+ "subset": "chartqa"
231
+ },
232
+ {
233
+ "demos": [
234
+ 75.17404270172119
235
+ ],
236
+ "greedy": 12.442910432815552,
237
+ "total": 76.42014288902283,
238
+ "setting": "critic",
239
+ "k": 1,
240
+ "query": 0,
241
+ "subset": "vizwiz_vqa_val"
242
+ },
243
+ {
244
+ "demos": [
245
+ 70.33047604560852
246
+ ],
247
+ "greedy": 11.583081483840942,
248
+ "total": 71.14328861236572,
249
+ "setting": "critic",
250
+ "k": 1,
251
+ "query": 1,
252
+ "subset": "vqav2_val"
253
+ },
254
+ {
255
+ "demos": [
256
+ 71.25398755073547
257
+ ],
258
+ "greedy": 12.159323930740356,
259
+ "total": 72.36944890022278,
260
+ "setting": "critic",
261
+ "k": 1,
262
+ "query": 2,
263
+ "subset": "ok_vqa_val2014"
264
+ },
265
+ {
266
+ "demos": [
267
+ 349.18033623695374
268
+ ],
269
+ "greedy": 11.960462808609009,
270
+ "total": 350.2297682762146,
271
+ "setting": "critic",
272
+ "k": 1,
273
+ "query": 3,
274
+ "subset": "textvqa_val"
275
+ },
276
+ {
277
+ "demos": [
278
+ 79.8816294670105
279
+ ],
280
+ "greedy": 12.26366114616394,
281
+ "total": 81.34397840499878,
282
+ "setting": "critic",
283
+ "k": 1,
284
+ "query": 4,
285
+ "subset": "chartqa"
286
+ },
287
+ {
288
+ "demos": [
289
+ 214.83966445922852,
290
+ 246.23991990089417
291
+ ],
292
+ "greedy": 11.864954471588135,
293
+ "total": 247.5113034248352,
294
+ "setting": "critic",
295
+ "k": 2,
296
+ "query": 0,
297
+ "subset": "vizwiz_vqa_val"
298
+ },
299
+ {
300
+ "demos": [
301
+ 111.94581985473633,
302
+ 112.84730768203735
303
+ ],
304
+ "greedy": 11.496987581253052,
305
+ "total": 113.7428183555603,
306
+ "setting": "critic",
307
+ "k": 2,
308
+ "query": 1,
309
+ "subset": "vqav2_val"
310
+ },
311
+ {
312
+ "demos": [
313
+ 106.2651150226593,
314
+ 107.3609848022461
315
+ ],
316
+ "greedy": 11.693379402160645,
317
+ "total": 108.45507740974426,
318
+ "setting": "critic",
319
+ "k": 2,
320
+ "query": 2,
321
+ "subset": "ok_vqa_val2014"
322
+ },
323
+ {
324
+ "demos": [
325
+ 97.19141793251038,
326
+ 402.90441393852234
327
+ ],
328
+ "greedy": 11.953308343887329,
329
+ "total": 404.0378088951111,
330
+ "setting": "critic",
331
+ "k": 2,
332
+ "query": 3,
333
+ "subset": "textvqa_val"
334
+ },
335
+ {
336
+ "demos": [
337
+ 125.35352444648743,
338
+ 164.42774605751038
339
+ ],
340
+ "greedy": 11.947067499160767,
341
+ "total": 166.28691792488098,
342
+ "setting": "critic",
343
+ "k": 2,
344
+ "query": 4,
345
+ "subset": "chartqa"
346
+ }
347
+ ]
tools/results/presets_bagel.json ADDED
The diff for this file is too large to render. See raw diff