LLDDSS's picture
Upload folder using huggingface_hub
18104f9 verified
Raw History Blame Contribute Delete
20.1 kB
#!/usr/bin/env python3
"""Build dataset_difficulty.ipynb from the cell sources below."""
import nbformat as nbf, os
OUT = ("/HDD2/disheng/Spatial_VLM_Stereo/June/Spatial_VLM_Stereo/"
"01_3_prelimilary_intervention_different_baseline/analysis/"
"dataset_level_analysis/dataset_difficulty.ipynb")
C = []
md = lambda s: C.append(nbf.v4.new_markdown_cell(s.strip("\n")))
co = lambda s: C.append(nbf.v4.new_code_cell(s.strip("\n")))
md(r"""
# How hard is each split, and do the models agree?
Four models answer four splits at six stereo baselines. This notebook asks two
things the raw accuracy table cannot: **how hard is each split once its own
guessing floor is taken out**, and **do the models find the same items hard**.
Raw accuracy is not comparable across splits. VSR is a two-way true/false
question, so guessing already scores 0.500; the three Youtube levels are
four-way, so guessing scores 0.250. Everything below is therefore reported as
$$\text{headroom captured} = \frac{\text{acc} - \text{floor}}{1 - \text{floor}}$$
the share of the distance between guessing and perfect that a model actually
covers. It goes negative when a model does worse than free.
""")
md("## Setup")
co(r"""
import json, pathlib, itertools, warnings
import numpy as np
import pandas as pd
import matplotlib as mpl
import matplotlib.pyplot as plt
warnings.filterwarnings("ignore")
pd.set_option("display.width", 200)
HERE = pathlib.Path.cwd()
ROOT = HERE if (HERE / "result").is_dir() else HERE.parents[1]
RESULT = ROOT / "result"
REPO = ROOT.parent
FIGDIR = ROOT / "analysis" / "dataset_level_analysis" / "figures"
FIGDIR.mkdir(parents=True, exist_ok=True)
BASELINES = ["0.00", "0.05", "0.10", "0.15", "0.20", "0.25"]
TASKS = ["vsr", "youtube_level1", "youtube_level2", "youtube_level3"]
NICE = {"vsr": "VSR", "youtube_level1": "YT L1", "youtube_level2": "YT L2",
"youtube_level3": "YT L3"}
# guessing floor per split -- all four are plain chance among the options
FLOOR = {"vsr": 0.500, "youtube_level1": 0.250,
"youtube_level2": 0.250, "youtube_level3": 0.250}
CHANCE = dict(FLOOR)
MODELS = sorted(p.name for p in RESULT.iterdir()
if p.is_dir() and not p.name.startswith("_"))
SHORT = {m: m.replace("Qwen_Qwen", "").replace("-Instruct", "") for m in MODELS}
COLOR = dict(zip(MODELS, ["#4C72B0", "#DD8452", "#55A868", "#C44E52"]))
mpl.rcParams.update({
"figure.dpi": 120, "savefig.dpi": 150, "font.size": 10,
"axes.spines.top": False, "axes.spines.right": False,
"axes.grid": True, "grid.alpha": .25, "grid.linewidth": .6,
"axes.axisbelow": True, "legend.frameon": False,
})
print(f"root : {ROOT}")
print(f"models : {[SHORT[m] for m in MODELS]}")
""")
md("### Load every run")
co(r"""
rows = []
for m in MODELS:
for t in TASKS:
for b in BASELINES:
d = json.loads((RESULT / m / t / f"baseline_{b}" / "metrics.json").read_text())
rows.append(dict(model=m, short=SHORT[m], task=t, baseline=float(b),
n=d["num_samples"], accuracy=d["accuracy"],
scoring=d["scoring"], unparsed=d["unparsed_rate"]))
runs = pd.DataFrame(rows)
runs["floor"] = runs["task"].map(FLOOR)
runs["headroom"] = (runs["accuracy"] - runs["floor"]) / (1 - runs["floor"])
assert len(runs) == len(MODELS) * len(TASKS) * len(BASELINES), "missing runs"
# An unparsed answer is scored wrong, so a model that narrates instead of naming
# a letter is penalised twice. Report the rate rather than assume it is zero.
bad = runs[runs.unparsed > 0]
if len(bad):
print("runs with unparsable answers (counted as wrong):")
print(bad.groupby(["short", "task"]).unparsed.agg(["max", "mean"]).round(4)
.to_string())
else:
print("every answer parsed in every run")
acc = runs.pivot_table(index="short", columns="task", values="accuracy", aggfunc="mean")[TASKS]
head = runs.pivot_table(index="short", columns="task", values="headroom", aggfunc="mean")[TASKS]
acc.columns = head.columns = [NICE[t] for t in TASKS]
display(acc.round(3).rename_axis("accuracy (mean over baselines)"))
display(head.round(3).rename_axis("headroom captured"))
""")
md(r"""
## 1. Where each model sits relative to its free floor
The bar is the accuracy; the dashed rule is what the split hands over for
nothing. Anything below the rule is a model doing worse than a rule that ignores
the picture.
""")
co(r"""
fig, axes = plt.subplots(1, 4, figsize=(13, 3.4), sharey=True)
for ax, t in zip(axes, TASKS):
sub = runs[runs.task == t].groupby("short", sort=False)
means = [runs[(runs.task == t) & (runs.model == m)].accuracy.mean() for m in MODELS]
errs = [runs[(runs.task == t) & (runs.model == m)].accuracy.std() for m in MODELS]
x = np.arange(len(MODELS))
ax.bar(x, means, yerr=errs, capsize=3, color=[COLOR[m] for m in MODELS],
edgecolor="white", linewidth=.8)
ax.axhline(FLOOR[t], ls="--", lw=1.4, color="#333")
ax.text(len(MODELS) - .45, FLOOR[t] + .012, f"floor {FLOOR[t]:.2f}",
ha="right", fontsize=8.5, color="#333")
if CHANCE[t] != FLOOR[t]:
ax.axhline(CHANCE[t], ls=":", lw=1.1, color="#999")
ax.text(-.45, CHANCE[t] + .012, f"guessing {CHANCE[t]:.2f}",
fontsize=8, color="#888")
ax.set_xticks(x); ax.set_xticklabels([SHORT[m] for m in MODELS], rotation=30, ha="right")
ax.set_title(f"{NICE[t]} (n={runs[runs.task==t].n.iloc[0]})", fontsize=11)
axes[0].set_ylabel("accuracy"); axes[0].set_ylim(0, 1)
fig.suptitle("Accuracy against the free floor, averaged over the six baselines",
y=1.04, fontsize=12)
fig.tight_layout(); fig.savefig(FIGDIR / "01_accuracy_vs_floor.png", bbox_inches="tight")
""")
md(r"""
## 2. Difficulty, on one scale
Headroom captured puts the four splits on a common axis. Red is a model losing
to its own floor.
""")
co(r"""
fig, ax = plt.subplots(figsize=(7.2, 3.2))
data = head.values
vmax = np.abs(data).max()
im = ax.imshow(data, cmap="RdYlGn", vmin=-vmax, vmax=vmax, aspect="auto")
ax.set_xticks(range(len(TASKS))); ax.set_xticklabels(head.columns)
ax.set_yticks(range(len(MODELS))); ax.set_yticklabels(head.index)
for i in range(data.shape[0]):
for j in range(data.shape[1]):
ax.text(j, i, f"{data[i,j]:+.2f}", ha="center", va="center", fontsize=10,
color="black")
ax.grid(False)
ax.set_title("Headroom captured (acc - floor) / (1 - floor)")
fig.colorbar(im, ax=ax, shrink=.85, label="share of headroom")
fig.tight_layout(); fig.savefig(FIGDIR / "02_headroom_heatmap.png", bbox_inches="tight")
""")
md(r"""
## 3. Do the models rank the splits the same way?
One line per model across the four splits. Parallel lines mean the models agree
on which split is hard and differ only in overall strength; crossings mean a
model finds a particular split unusually easy or hard.
""")
co(r"""
order = list(head.mean().sort_values(ascending=False).index)
fig, ax = plt.subplots(figsize=(7.6, 4.0))
for m in MODELS:
y = [head.loc[SHORT[m], c] for c in order]
ax.plot(range(len(order)), y, "-o", color=COLOR[m], lw=2, ms=6, label=SHORT[m])
ax.axhline(0, color="#333", lw=1.2, ls="--")
ax.text(len(order) - 1, .015, "floor", ha="right", fontsize=8.5, color="#333")
ax.set_xticks(range(len(order))); ax.set_xticklabels(order)
ax.set_ylabel("headroom captured"); ax.set_title("Difficulty profile per model")
ax.legend(loc="upper right", ncol=2)
fig.tight_layout(); fig.savefig(FIGDIR / "03_difficulty_profile.png", bbox_inches="tight")
ranks = head.rank(axis=1, ascending=False).astype(int)
print("difficulty rank each model assigns (1 = most headroom captured):")
display(ranks)
print("Spearman between models (only four splits, so 1.00 means identical order):")
display(head.T.corr(method="spearman").round(2))
""")
md(r"""
## 4. Does the stereo baseline move anything?
Each panel is one split; each line is one model across the six baselines. A
visual tower that models the scene properly should be flat here — the question
and the answer never change, only the viewpoint.
""")
co(r"""
fig, axes = plt.subplots(1, 4, figsize=(13.5, 3.2))
for ax, t in zip(axes, TASKS):
for m in MODELS:
s = runs[(runs.task == t) & (runs.model == m)].sort_values("baseline")
ax.plot(s.baseline, s.accuracy, "-o", ms=4, color=COLOR[m], label=SHORT[m])
ax.axhline(FLOOR[t], ls="--", lw=1.1, color="#666")
ax.set_title(NICE[t]); ax.set_xlabel("baseline")
lo = min(runs[runs.task == t].accuracy.min(), FLOOR[t]) - .03
hi = max(runs[runs.task == t].accuracy.max(), FLOOR[t]) + .03
ax.set_ylim(lo, hi)
axes[0].set_ylabel("accuracy")
axes[-1].legend(loc="center left", bbox_to_anchor=(1.02, .5))
fig.suptitle("Accuracy across the intervention, per split", y=1.05, fontsize=12)
fig.tight_layout(); fig.savefig(FIGDIR / "04_baseline_sensitivity.png", bbox_inches="tight")
spread = (runs.groupby(["short", "task"]).accuracy.agg(["min", "max"])
.assign(range=lambda d: d["max"] - d["min"])["range"]
.unstack()[TASKS].rename(columns=NICE))
display(spread.round(3).rename_axis("accuracy range over the six baselines"))
""")
md(r"""
## 5. Item-level difficulty: do the models fail on the same pictures?
For every item, count how many of the four models get it right at baseline 0.00.
If difficulty were a property of the item, the counts would pile up at 0 and 4.
If each model failed idiosyncratically, the counts would follow the independent
(Poisson-binomial) null drawn behind the bars.
""")
co(r"""
def load_pred(m, t, b="0.00"):
path = RESULT / m / t / f"baseline_{b}" / "predictions.jsonl"
return {json.loads(l)["sample_id"]: json.loads(l)
for l in path.read_text().splitlines() if l.strip()}
correct_by_task = {}
for t in TASKS:
per = {m: load_pred(m, t) for m in MODELS}
ids = sorted(set.intersection(*(set(v) for v in per.values())))
correct_by_task[t] = pd.DataFrame(
{SHORT[m]: [per[m][i]["correct"] for i in ids] for m in MODELS}, index=ids)
def poisson_binomial(ps):
dist = np.zeros(len(ps) + 1); dist[0] = 1.0
for p in ps:
dist[1:] = dist[1:] * (1 - p) + dist[:-1] * p
dist[0] *= (1 - p)
return dist
fig, axes = plt.subplots(1, 4, figsize=(13.5, 3.2), sharey=True)
conc = {}
for ax, t in zip(axes, TASKS):
df = correct_by_task[t]
k = df.sum(axis=1).value_counts().reindex(range(5), fill_value=0).sort_index()
obs = k / k.sum()
null = poisson_binomial(df.mean().values)
ax.bar(range(5), null, color="#CCCCCC", width=.82, label="independent null")
ax.bar(range(5), obs, color="#4C72B0", width=.45, label="observed")
ax.set_title(f"{NICE[t]} (n={len(df)})"); ax.set_xlabel("# models correct")
ax.set_xticks(range(5))
# how much mass sits at the two extremes, versus the null
conc[NICE[t]] = dict(observed_0_or_4=obs.iloc[[0, 4]].sum(),
null_0_or_4=null[[0, 4]].sum())
axes[0].set_ylabel("share of items"); axes[0].legend(fontsize=8.5)
fig.suptitle("Agreement on which items are hard", y=1.05, fontsize=12)
fig.tight_layout(); fig.savefig(FIGDIR / "05_item_difficulty.png", bbox_inches="tight")
conc = pd.DataFrame(conc).T
conc["excess"] = conc["observed_0_or_4"] - conc["null_0_or_4"]
display(conc.round(3).rename_axis("mass at 0 or 4 models correct"))
""")
md(r"""
## 6. Failure overlap between models
Jaccard overlap of the failure sets, per split. High values mean the models
stumble on the same items — shared difficulty rather than independent noise.
""")
co(r"""
fig, axes = plt.subplots(1, 4, figsize=(14, 3.4))
for ax, t in zip(axes, TASKS):
df = correct_by_task[t]
names = list(df.columns)
M = np.ones((len(names), len(names)))
for i, a in enumerate(names):
for j, b in enumerate(names):
if i == j:
continue
fa, fb = ~df[a].values, ~df[b].values
M[i, j] = (fa & fb).sum() / max((fa | fb).sum(), 1)
im = ax.imshow(M, cmap="Blues", vmin=0, vmax=1)
ax.set_xticks(range(len(names))); ax.set_xticklabels(names, rotation=40, ha="right", fontsize=8)
ax.set_yticks(range(len(names))); ax.set_yticklabels(names, fontsize=8)
for i in range(len(names)):
for j in range(len(names)):
ax.text(j, i, f"{M[i,j]:.2f}", ha="center", va="center", fontsize=8,
color="white" if M[i, j] > .55 else "black")
ax.grid(False); ax.set_title(NICE[t], fontsize=11)
fig.suptitle("Jaccard overlap of failure sets (baseline 0.00)", y=1.06, fontsize=12)
fig.tight_layout(); fig.savefig(FIGDIR / "06_failure_overlap.png", bbox_inches="tight")
""")
md(r"""
## 7. Is the model answering, or just picking a letter?
The reference labels are near-uniform in every split, so a flat prediction
histogram is what an unbiased model produces. A spike means the model has
collapsed onto one option, and that spike caps how well it can possibly do.
""")
co(r"""
fig, axes = plt.subplots(len(MODELS), 4, figsize=(12, 2.3 * len(MODELS)),
sharey="row")
for r, m in enumerate(MODELS):
for c, t in enumerate(TASKS):
ax = axes[r, c]
pred = load_pred(m, t)
pc = pd.Series([v["prediction_norm"] for v in pred.values()]).value_counts()
rc = pd.Series([v["reference_norm"] for v in pred.values()]).value_counts()
labels = sorted(set(pc.index) | set(rc.index))
x = np.arange(len(labels))
ax.bar(x - .2, [rc.get(l, 0) / len(pred) for l in labels], width=.4,
color="#BBBBBB", label="reference")
ax.bar(x + .2, [pc.get(l, 0) / len(pred) for l in labels], width=.4,
color=COLOR[m], label="prediction")
ax.set_xticks(x); ax.set_xticklabels(labels, fontsize=8)
ax.axhline(1 / len(labels), ls=":", lw=1, color="#666")
if r == 0:
ax.set_title(NICE[t])
if c == 0:
ax.set_ylabel(SHORT[m], fontsize=9)
if r == 0 and c == 0:
ax.legend(fontsize=7.5)
fig.suptitle("Predicted vs reference label distribution (baseline 0.00)",
y=1.01, fontsize=12)
fig.tight_layout(); fig.savefig(FIGDIR / "07_answer_bias.png", bbox_inches="tight")
""")
md(r"""
## 8. Level 2 regression guard: is the option text still uninformative?
Level 2's three distractors are drawn uniformly from the 23 wrong orderings, so
the four options should be statistically exchangeable and no text-only rule
should beat chance. Earlier, structured distractor sets did not have that
property — the worst of them let a majority vote on successive positions narrow
every item to two candidates and score 0.500 blind.
This cell re-runs that vote. `in survivor set` near 1.0 means the vote eliminated
nothing, which is what random distractors should produce. If it ever drops
toward 0.5 while accuracy rises, the construction has reacquired a leak.
""")
co(r"""
L2_TEST = REPO / "02_data" / "Youtube_self_depth_QA" / "level2" / "test.jsonl"
rows = []
if L2_TEST.exists():
opts = {}
for line in L2_TEST.read_text().splitlines():
rec = json.loads(line)
if rec.get("options"):
opts[rec["sample_id"]] = {k: v.split(" > ") for k, v in rec["options"].items()}
def survivors(o):
alive = list(o)
for k in range(4):
if len(alive) <= 1:
break
counts = pd.Series([o[a][k] for a in alive]).value_counts()
if len(counts) > 1 and counts.iloc[0] == counts.iloc[1]:
continue
alive = [a for a in alive if o[a][k] == counts.index[0]]
return alive
surv = {sid: survivors(o) for sid, o in opts.items()}
for m in MODELS:
pred = load_pred(m, "youtube_level2")
ins = [p["prediction_norm"] in surv[s] for s, p in pred.items() if s in surv]
hit = [p["correct"] for s, p in pred.items() if s in surv]
rows.append(dict(model=SHORT[m], accuracy=np.mean(hit),
in_survivor_set=np.mean(ins)))
l2 = pd.DataFrame(rows).set_index("model")
surv = np.mean([len(v) for v in surv.values()])
print(f"mean survivor-set size after the vote: {surv:.2f} of 4 options")
display(l2.round(3))
fig, ax = plt.subplots(figsize=(7, 3.2))
x = np.arange(len(l2))
ax.bar(x - .2, l2.in_survivor_set, width=.4, color="#999",
label="answer survives the text-only vote")
ax.bar(x + .2, l2.accuracy, width=.4, color="#C44E52", label="accuracy")
ax.axhline(.25, ls="--", color="#333", lw=1.3)
ax.text(len(l2) - .4, .265, "chance", ha="right", fontsize=8.5)
ax.set_xticks(x); ax.set_xticklabels(l2.index, rotation=25, ha="right")
ax.set_ylim(0, 1); ax.set_ylabel("share")
ax.set_title("Level 2: random distractors leave the text-only vote powerless")
ax.legend()
fig.tight_layout(); fig.savefig(FIGDIR / "08_level2_guard.png", bbox_inches="tight")
else:
print(f"level-2 split not found at {L2_TEST}; skipping")
""")
md(r"""
## Findings
**1. Every model puts VSR first and YT L1 second; the two families then
disagree.** Qwen2.5-VL finds level 3 hardest, Qwen3.5 finds level 2 hardest.
Spearman is 1.00 within a family and 0.80 across, so the ordering of the two
Youtube reasoning levels is the one place where architecture, not just
strength, changes what looks hard.
**2. Only VSR is comfortably solved** (0.52-0.66 of headroom). It is also the
easiest question type: a binary relation that is often inferable from object
identity alone, without metric depth.
**3. Level 3 is the sharpest discriminator.** Qwen2.5-VL captures 0.03 (3B) and
0.07 (7B) — indistinguishable from guessing — against 0.29 and 0.33 for
Qwen3.5. A 4-10x gap, far wider than anywhere else in the suite. Continuous
trajectory reasoning is what separates the two families.
**4. Level 2 sits at 0.17-0.26 of headroom for everyone**, a narrow band that
does not track model size or family. Its distractors are drawn at random from
the 23 wrong orderings, so nothing in the option text helps (figure 8: the
text-only vote leaves ~2 of 4 options standing and scores 0.252 on the split,
i.e. chance). The scores are what four-way depth ranking actually costs these
models: clearly above guessing, nowhere near solved.
**5. The base model beats the instruct model on every split.** Qwen3.5-4B-Base
scored by likelihood outperforms Qwen3.5-4B scored by generation everywhere,
and 4B-Instruct is the only model producing unparsable answers (up to 0.7% on
L1), which are scored wrong. Part of the instruct model's deficit is format
compliance, not perception.
**6. The stereo intervention barely moves the aggregate.** Accuracy ranges over
the six baselines are 0.012-0.044 with no consistent sign across models or
splits. Whether the same *items* flip is a separate question, answered in
`../intervention_analysis`.
**7. Models agree most on which L1 items are hard, least on L3.** Excess mass at
"0 or 4 models correct" over the independent null: +0.36 (L1), +0.27 (VSR),
+0.23 (L2), +0.14 (L3). On L3 the failures are close to independent noise, which
is what near-chance accuracy looks like — there is little shared signal to agree
about.
### A note on the level-2 numbers
Level 2 was rebuilt three times before this run. Two earlier option
constructions leaked: distractors grown outward from the ground truth made it
the option most similar to the others (0.388 blind), and a prefix ladder let a
majority vote narrow every item to two (0.500 blind). Under the leaking
constructions these same models scored 0.40-0.47 on level 2; with random
distractors they score 0.37-0.45 against a real 0.25 floor. The drop of
0.02-0.10 is the part of the old score that came from the option text rather
than the picture.
""")
nb = nbf.v4.new_notebook(cells=C)
nb.metadata.kernelspec = {"display_name": "Spatial_VLM", "language": "python",
"name": "python3"}
nb.metadata.language_info = {"name": "python"}
os.makedirs(os.path.dirname(OUT), exist_ok=True)
nbf.write(nb, OUT)
print("wrote", OUT, f"({len(C)} cells)")