File size: 20,070 Bytes
18104f9 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274 275 276 277 278 279 280 281 282 283 284 285 286 287 288 289 290 291 292 293 294 295 296 297 298 299 300 301 302 303 304 305 306 307 308 309 310 311 312 313 314 315 316 317 318 319 320 321 322 323 324 325 326 327 328 329 330 331 332 333 334 335 336 337 338 339 340 341 342 343 344 345 346 347 348 349 350 351 352 353 354 355 356 357 358 359 360 361 362 363 364 365 366 367 368 369 370 371 372 373 374 375 376 377 378 379 380 381 382 383 384 385 386 387 388 389 390 391 392 393 394 395 396 397 398 399 400 401 402 403 404 405 406 407 408 409 410 411 412 413 414 415 416 417 418 419 420 421 422 423 424 425 426 427 428 429 430 431 432 433 434 435 436 437 438 439 440 441 442 443 444 445 446 447 448 449 450 451 452 453 454 455 456 457 458 459 460 461 462 463 464 | #!/usr/bin/env python3
"""Build dataset_difficulty.ipynb from the cell sources below."""
import nbformat as nbf, os
OUT = ("/HDD2/disheng/Spatial_VLM_Stereo/June/Spatial_VLM_Stereo/"
"01_3_prelimilary_intervention_different_baseline/analysis/"
"dataset_level_analysis/dataset_difficulty.ipynb")
C = []
md = lambda s: C.append(nbf.v4.new_markdown_cell(s.strip("\n")))
co = lambda s: C.append(nbf.v4.new_code_cell(s.strip("\n")))
md(r"""
# How hard is each split, and do the models agree?
Four models answer four splits at six stereo baselines. This notebook asks two
things the raw accuracy table cannot: **how hard is each split once its own
guessing floor is taken out**, and **do the models find the same items hard**.
Raw accuracy is not comparable across splits. VSR is a two-way true/false
question, so guessing already scores 0.500; the three Youtube levels are
four-way, so guessing scores 0.250. Everything below is therefore reported as
$$\text{headroom captured} = \frac{\text{acc} - \text{floor}}{1 - \text{floor}}$$
the share of the distance between guessing and perfect that a model actually
covers. It goes negative when a model does worse than free.
""")
md("## Setup")
co(r"""
import json, pathlib, itertools, warnings
import numpy as np
import pandas as pd
import matplotlib as mpl
import matplotlib.pyplot as plt
warnings.filterwarnings("ignore")
pd.set_option("display.width", 200)
HERE = pathlib.Path.cwd()
ROOT = HERE if (HERE / "result").is_dir() else HERE.parents[1]
RESULT = ROOT / "result"
REPO = ROOT.parent
FIGDIR = ROOT / "analysis" / "dataset_level_analysis" / "figures"
FIGDIR.mkdir(parents=True, exist_ok=True)
BASELINES = ["0.00", "0.05", "0.10", "0.15", "0.20", "0.25"]
TASKS = ["vsr", "youtube_level1", "youtube_level2", "youtube_level3"]
NICE = {"vsr": "VSR", "youtube_level1": "YT L1", "youtube_level2": "YT L2",
"youtube_level3": "YT L3"}
# guessing floor per split -- all four are plain chance among the options
FLOOR = {"vsr": 0.500, "youtube_level1": 0.250,
"youtube_level2": 0.250, "youtube_level3": 0.250}
CHANCE = dict(FLOOR)
MODELS = sorted(p.name for p in RESULT.iterdir()
if p.is_dir() and not p.name.startswith("_"))
SHORT = {m: m.replace("Qwen_Qwen", "").replace("-Instruct", "") for m in MODELS}
COLOR = dict(zip(MODELS, ["#4C72B0", "#DD8452", "#55A868", "#C44E52"]))
mpl.rcParams.update({
"figure.dpi": 120, "savefig.dpi": 150, "font.size": 10,
"axes.spines.top": False, "axes.spines.right": False,
"axes.grid": True, "grid.alpha": .25, "grid.linewidth": .6,
"axes.axisbelow": True, "legend.frameon": False,
})
print(f"root : {ROOT}")
print(f"models : {[SHORT[m] for m in MODELS]}")
""")
md("### Load every run")
co(r"""
rows = []
for m in MODELS:
for t in TASKS:
for b in BASELINES:
d = json.loads((RESULT / m / t / f"baseline_{b}" / "metrics.json").read_text())
rows.append(dict(model=m, short=SHORT[m], task=t, baseline=float(b),
n=d["num_samples"], accuracy=d["accuracy"],
scoring=d["scoring"], unparsed=d["unparsed_rate"]))
runs = pd.DataFrame(rows)
runs["floor"] = runs["task"].map(FLOOR)
runs["headroom"] = (runs["accuracy"] - runs["floor"]) / (1 - runs["floor"])
assert len(runs) == len(MODELS) * len(TASKS) * len(BASELINES), "missing runs"
# An unparsed answer is scored wrong, so a model that narrates instead of naming
# a letter is penalised twice. Report the rate rather than assume it is zero.
bad = runs[runs.unparsed > 0]
if len(bad):
print("runs with unparsable answers (counted as wrong):")
print(bad.groupby(["short", "task"]).unparsed.agg(["max", "mean"]).round(4)
.to_string())
else:
print("every answer parsed in every run")
acc = runs.pivot_table(index="short", columns="task", values="accuracy", aggfunc="mean")[TASKS]
head = runs.pivot_table(index="short", columns="task", values="headroom", aggfunc="mean")[TASKS]
acc.columns = head.columns = [NICE[t] for t in TASKS]
display(acc.round(3).rename_axis("accuracy (mean over baselines)"))
display(head.round(3).rename_axis("headroom captured"))
""")
md(r"""
## 1. Where each model sits relative to its free floor
The bar is the accuracy; the dashed rule is what the split hands over for
nothing. Anything below the rule is a model doing worse than a rule that ignores
the picture.
""")
co(r"""
fig, axes = plt.subplots(1, 4, figsize=(13, 3.4), sharey=True)
for ax, t in zip(axes, TASKS):
sub = runs[runs.task == t].groupby("short", sort=False)
means = [runs[(runs.task == t) & (runs.model == m)].accuracy.mean() for m in MODELS]
errs = [runs[(runs.task == t) & (runs.model == m)].accuracy.std() for m in MODELS]
x = np.arange(len(MODELS))
ax.bar(x, means, yerr=errs, capsize=3, color=[COLOR[m] for m in MODELS],
edgecolor="white", linewidth=.8)
ax.axhline(FLOOR[t], ls="--", lw=1.4, color="#333")
ax.text(len(MODELS) - .45, FLOOR[t] + .012, f"floor {FLOOR[t]:.2f}",
ha="right", fontsize=8.5, color="#333")
if CHANCE[t] != FLOOR[t]:
ax.axhline(CHANCE[t], ls=":", lw=1.1, color="#999")
ax.text(-.45, CHANCE[t] + .012, f"guessing {CHANCE[t]:.2f}",
fontsize=8, color="#888")
ax.set_xticks(x); ax.set_xticklabels([SHORT[m] for m in MODELS], rotation=30, ha="right")
ax.set_title(f"{NICE[t]} (n={runs[runs.task==t].n.iloc[0]})", fontsize=11)
axes[0].set_ylabel("accuracy"); axes[0].set_ylim(0, 1)
fig.suptitle("Accuracy against the free floor, averaged over the six baselines",
y=1.04, fontsize=12)
fig.tight_layout(); fig.savefig(FIGDIR / "01_accuracy_vs_floor.png", bbox_inches="tight")
""")
md(r"""
## 2. Difficulty, on one scale
Headroom captured puts the four splits on a common axis. Red is a model losing
to its own floor.
""")
co(r"""
fig, ax = plt.subplots(figsize=(7.2, 3.2))
data = head.values
vmax = np.abs(data).max()
im = ax.imshow(data, cmap="RdYlGn", vmin=-vmax, vmax=vmax, aspect="auto")
ax.set_xticks(range(len(TASKS))); ax.set_xticklabels(head.columns)
ax.set_yticks(range(len(MODELS))); ax.set_yticklabels(head.index)
for i in range(data.shape[0]):
for j in range(data.shape[1]):
ax.text(j, i, f"{data[i,j]:+.2f}", ha="center", va="center", fontsize=10,
color="black")
ax.grid(False)
ax.set_title("Headroom captured (acc - floor) / (1 - floor)")
fig.colorbar(im, ax=ax, shrink=.85, label="share of headroom")
fig.tight_layout(); fig.savefig(FIGDIR / "02_headroom_heatmap.png", bbox_inches="tight")
""")
md(r"""
## 3. Do the models rank the splits the same way?
One line per model across the four splits. Parallel lines mean the models agree
on which split is hard and differ only in overall strength; crossings mean a
model finds a particular split unusually easy or hard.
""")
co(r"""
order = list(head.mean().sort_values(ascending=False).index)
fig, ax = plt.subplots(figsize=(7.6, 4.0))
for m in MODELS:
y = [head.loc[SHORT[m], c] for c in order]
ax.plot(range(len(order)), y, "-o", color=COLOR[m], lw=2, ms=6, label=SHORT[m])
ax.axhline(0, color="#333", lw=1.2, ls="--")
ax.text(len(order) - 1, .015, "floor", ha="right", fontsize=8.5, color="#333")
ax.set_xticks(range(len(order))); ax.set_xticklabels(order)
ax.set_ylabel("headroom captured"); ax.set_title("Difficulty profile per model")
ax.legend(loc="upper right", ncol=2)
fig.tight_layout(); fig.savefig(FIGDIR / "03_difficulty_profile.png", bbox_inches="tight")
ranks = head.rank(axis=1, ascending=False).astype(int)
print("difficulty rank each model assigns (1 = most headroom captured):")
display(ranks)
print("Spearman between models (only four splits, so 1.00 means identical order):")
display(head.T.corr(method="spearman").round(2))
""")
md(r"""
## 4. Does the stereo baseline move anything?
Each panel is one split; each line is one model across the six baselines. A
visual tower that models the scene properly should be flat here — the question
and the answer never change, only the viewpoint.
""")
co(r"""
fig, axes = plt.subplots(1, 4, figsize=(13.5, 3.2))
for ax, t in zip(axes, TASKS):
for m in MODELS:
s = runs[(runs.task == t) & (runs.model == m)].sort_values("baseline")
ax.plot(s.baseline, s.accuracy, "-o", ms=4, color=COLOR[m], label=SHORT[m])
ax.axhline(FLOOR[t], ls="--", lw=1.1, color="#666")
ax.set_title(NICE[t]); ax.set_xlabel("baseline")
lo = min(runs[runs.task == t].accuracy.min(), FLOOR[t]) - .03
hi = max(runs[runs.task == t].accuracy.max(), FLOOR[t]) + .03
ax.set_ylim(lo, hi)
axes[0].set_ylabel("accuracy")
axes[-1].legend(loc="center left", bbox_to_anchor=(1.02, .5))
fig.suptitle("Accuracy across the intervention, per split", y=1.05, fontsize=12)
fig.tight_layout(); fig.savefig(FIGDIR / "04_baseline_sensitivity.png", bbox_inches="tight")
spread = (runs.groupby(["short", "task"]).accuracy.agg(["min", "max"])
.assign(range=lambda d: d["max"] - d["min"])["range"]
.unstack()[TASKS].rename(columns=NICE))
display(spread.round(3).rename_axis("accuracy range over the six baselines"))
""")
md(r"""
## 5. Item-level difficulty: do the models fail on the same pictures?
For every item, count how many of the four models get it right at baseline 0.00.
If difficulty were a property of the item, the counts would pile up at 0 and 4.
If each model failed idiosyncratically, the counts would follow the independent
(Poisson-binomial) null drawn behind the bars.
""")
co(r"""
def load_pred(m, t, b="0.00"):
path = RESULT / m / t / f"baseline_{b}" / "predictions.jsonl"
return {json.loads(l)["sample_id"]: json.loads(l)
for l in path.read_text().splitlines() if l.strip()}
correct_by_task = {}
for t in TASKS:
per = {m: load_pred(m, t) for m in MODELS}
ids = sorted(set.intersection(*(set(v) for v in per.values())))
correct_by_task[t] = pd.DataFrame(
{SHORT[m]: [per[m][i]["correct"] for i in ids] for m in MODELS}, index=ids)
def poisson_binomial(ps):
dist = np.zeros(len(ps) + 1); dist[0] = 1.0
for p in ps:
dist[1:] = dist[1:] * (1 - p) + dist[:-1] * p
dist[0] *= (1 - p)
return dist
fig, axes = plt.subplots(1, 4, figsize=(13.5, 3.2), sharey=True)
conc = {}
for ax, t in zip(axes, TASKS):
df = correct_by_task[t]
k = df.sum(axis=1).value_counts().reindex(range(5), fill_value=0).sort_index()
obs = k / k.sum()
null = poisson_binomial(df.mean().values)
ax.bar(range(5), null, color="#CCCCCC", width=.82, label="independent null")
ax.bar(range(5), obs, color="#4C72B0", width=.45, label="observed")
ax.set_title(f"{NICE[t]} (n={len(df)})"); ax.set_xlabel("# models correct")
ax.set_xticks(range(5))
# how much mass sits at the two extremes, versus the null
conc[NICE[t]] = dict(observed_0_or_4=obs.iloc[[0, 4]].sum(),
null_0_or_4=null[[0, 4]].sum())
axes[0].set_ylabel("share of items"); axes[0].legend(fontsize=8.5)
fig.suptitle("Agreement on which items are hard", y=1.05, fontsize=12)
fig.tight_layout(); fig.savefig(FIGDIR / "05_item_difficulty.png", bbox_inches="tight")
conc = pd.DataFrame(conc).T
conc["excess"] = conc["observed_0_or_4"] - conc["null_0_or_4"]
display(conc.round(3).rename_axis("mass at 0 or 4 models correct"))
""")
md(r"""
## 6. Failure overlap between models
Jaccard overlap of the failure sets, per split. High values mean the models
stumble on the same items — shared difficulty rather than independent noise.
""")
co(r"""
fig, axes = plt.subplots(1, 4, figsize=(14, 3.4))
for ax, t in zip(axes, TASKS):
df = correct_by_task[t]
names = list(df.columns)
M = np.ones((len(names), len(names)))
for i, a in enumerate(names):
for j, b in enumerate(names):
if i == j:
continue
fa, fb = ~df[a].values, ~df[b].values
M[i, j] = (fa & fb).sum() / max((fa | fb).sum(), 1)
im = ax.imshow(M, cmap="Blues", vmin=0, vmax=1)
ax.set_xticks(range(len(names))); ax.set_xticklabels(names, rotation=40, ha="right", fontsize=8)
ax.set_yticks(range(len(names))); ax.set_yticklabels(names, fontsize=8)
for i in range(len(names)):
for j in range(len(names)):
ax.text(j, i, f"{M[i,j]:.2f}", ha="center", va="center", fontsize=8,
color="white" if M[i, j] > .55 else "black")
ax.grid(False); ax.set_title(NICE[t], fontsize=11)
fig.suptitle("Jaccard overlap of failure sets (baseline 0.00)", y=1.06, fontsize=12)
fig.tight_layout(); fig.savefig(FIGDIR / "06_failure_overlap.png", bbox_inches="tight")
""")
md(r"""
## 7. Is the model answering, or just picking a letter?
The reference labels are near-uniform in every split, so a flat prediction
histogram is what an unbiased model produces. A spike means the model has
collapsed onto one option, and that spike caps how well it can possibly do.
""")
co(r"""
fig, axes = plt.subplots(len(MODELS), 4, figsize=(12, 2.3 * len(MODELS)),
sharey="row")
for r, m in enumerate(MODELS):
for c, t in enumerate(TASKS):
ax = axes[r, c]
pred = load_pred(m, t)
pc = pd.Series([v["prediction_norm"] for v in pred.values()]).value_counts()
rc = pd.Series([v["reference_norm"] for v in pred.values()]).value_counts()
labels = sorted(set(pc.index) | set(rc.index))
x = np.arange(len(labels))
ax.bar(x - .2, [rc.get(l, 0) / len(pred) for l in labels], width=.4,
color="#BBBBBB", label="reference")
ax.bar(x + .2, [pc.get(l, 0) / len(pred) for l in labels], width=.4,
color=COLOR[m], label="prediction")
ax.set_xticks(x); ax.set_xticklabels(labels, fontsize=8)
ax.axhline(1 / len(labels), ls=":", lw=1, color="#666")
if r == 0:
ax.set_title(NICE[t])
if c == 0:
ax.set_ylabel(SHORT[m], fontsize=9)
if r == 0 and c == 0:
ax.legend(fontsize=7.5)
fig.suptitle("Predicted vs reference label distribution (baseline 0.00)",
y=1.01, fontsize=12)
fig.tight_layout(); fig.savefig(FIGDIR / "07_answer_bias.png", bbox_inches="tight")
""")
md(r"""
## 8. Level 2 regression guard: is the option text still uninformative?
Level 2's three distractors are drawn uniformly from the 23 wrong orderings, so
the four options should be statistically exchangeable and no text-only rule
should beat chance. Earlier, structured distractor sets did not have that
property — the worst of them let a majority vote on successive positions narrow
every item to two candidates and score 0.500 blind.
This cell re-runs that vote. `in survivor set` near 1.0 means the vote eliminated
nothing, which is what random distractors should produce. If it ever drops
toward 0.5 while accuracy rises, the construction has reacquired a leak.
""")
co(r"""
L2_TEST = REPO / "02_data" / "Youtube_self_depth_QA" / "level2" / "test.jsonl"
rows = []
if L2_TEST.exists():
opts = {}
for line in L2_TEST.read_text().splitlines():
rec = json.loads(line)
if rec.get("options"):
opts[rec["sample_id"]] = {k: v.split(" > ") for k, v in rec["options"].items()}
def survivors(o):
alive = list(o)
for k in range(4):
if len(alive) <= 1:
break
counts = pd.Series([o[a][k] for a in alive]).value_counts()
if len(counts) > 1 and counts.iloc[0] == counts.iloc[1]:
continue
alive = [a for a in alive if o[a][k] == counts.index[0]]
return alive
surv = {sid: survivors(o) for sid, o in opts.items()}
for m in MODELS:
pred = load_pred(m, "youtube_level2")
ins = [p["prediction_norm"] in surv[s] for s, p in pred.items() if s in surv]
hit = [p["correct"] for s, p in pred.items() if s in surv]
rows.append(dict(model=SHORT[m], accuracy=np.mean(hit),
in_survivor_set=np.mean(ins)))
l2 = pd.DataFrame(rows).set_index("model")
surv = np.mean([len(v) for v in surv.values()])
print(f"mean survivor-set size after the vote: {surv:.2f} of 4 options")
display(l2.round(3))
fig, ax = plt.subplots(figsize=(7, 3.2))
x = np.arange(len(l2))
ax.bar(x - .2, l2.in_survivor_set, width=.4, color="#999",
label="answer survives the text-only vote")
ax.bar(x + .2, l2.accuracy, width=.4, color="#C44E52", label="accuracy")
ax.axhline(.25, ls="--", color="#333", lw=1.3)
ax.text(len(l2) - .4, .265, "chance", ha="right", fontsize=8.5)
ax.set_xticks(x); ax.set_xticklabels(l2.index, rotation=25, ha="right")
ax.set_ylim(0, 1); ax.set_ylabel("share")
ax.set_title("Level 2: random distractors leave the text-only vote powerless")
ax.legend()
fig.tight_layout(); fig.savefig(FIGDIR / "08_level2_guard.png", bbox_inches="tight")
else:
print(f"level-2 split not found at {L2_TEST}; skipping")
""")
md(r"""
## Findings
**1. Every model puts VSR first and YT L1 second; the two families then
disagree.** Qwen2.5-VL finds level 3 hardest, Qwen3.5 finds level 2 hardest.
Spearman is 1.00 within a family and 0.80 across, so the ordering of the two
Youtube reasoning levels is the one place where architecture, not just
strength, changes what looks hard.
**2. Only VSR is comfortably solved** (0.52-0.66 of headroom). It is also the
easiest question type: a binary relation that is often inferable from object
identity alone, without metric depth.
**3. Level 3 is the sharpest discriminator.** Qwen2.5-VL captures 0.03 (3B) and
0.07 (7B) — indistinguishable from guessing — against 0.29 and 0.33 for
Qwen3.5. A 4-10x gap, far wider than anywhere else in the suite. Continuous
trajectory reasoning is what separates the two families.
**4. Level 2 sits at 0.17-0.26 of headroom for everyone**, a narrow band that
does not track model size or family. Its distractors are drawn at random from
the 23 wrong orderings, so nothing in the option text helps (figure 8: the
text-only vote leaves ~2 of 4 options standing and scores 0.252 on the split,
i.e. chance). The scores are what four-way depth ranking actually costs these
models: clearly above guessing, nowhere near solved.
**5. The base model beats the instruct model on every split.** Qwen3.5-4B-Base
scored by likelihood outperforms Qwen3.5-4B scored by generation everywhere,
and 4B-Instruct is the only model producing unparsable answers (up to 0.7% on
L1), which are scored wrong. Part of the instruct model's deficit is format
compliance, not perception.
**6. The stereo intervention barely moves the aggregate.** Accuracy ranges over
the six baselines are 0.012-0.044 with no consistent sign across models or
splits. Whether the same *items* flip is a separate question, answered in
`../intervention_analysis`.
**7. Models agree most on which L1 items are hard, least on L3.** Excess mass at
"0 or 4 models correct" over the independent null: +0.36 (L1), +0.27 (VSR),
+0.23 (L2), +0.14 (L3). On L3 the failures are close to independent noise, which
is what near-chance accuracy looks like — there is little shared signal to agree
about.
### A note on the level-2 numbers
Level 2 was rebuilt three times before this run. Two earlier option
constructions leaked: distractors grown outward from the ground truth made it
the option most similar to the others (0.388 blind), and a prefix ladder let a
majority vote narrow every item to two (0.500 blind). Under the leaking
constructions these same models scored 0.40-0.47 on level 2; with random
distractors they score 0.37-0.45 against a real 0.25 floor. The drop of
0.02-0.10 is the part of the old score that came from the option text rather
than the picture.
""")
nb = nbf.v4.new_notebook(cells=C)
nb.metadata.kernelspec = {"display_name": "Spatial_VLM", "language": "python",
"name": "python3"}
nb.metadata.language_info = {"name": "python"}
os.makedirs(os.path.dirname(OUT), exist_ok=True)
nbf.write(nb, OUT)
print("wrote", OUT, f"({len(C)} cells)")
|