#!/usr/bin/env python3 """Build dataset_difficulty.ipynb from the cell sources below.""" import nbformat as nbf, os OUT = ("/HDD2/disheng/Spatial_VLM_Stereo/June/Spatial_VLM_Stereo/" "01_3_prelimilary_intervention_different_baseline/analysis/" "dataset_level_analysis/dataset_difficulty.ipynb") C = [] md = lambda s: C.append(nbf.v4.new_markdown_cell(s.strip("\n"))) co = lambda s: C.append(nbf.v4.new_code_cell(s.strip("\n"))) md(r""" # How hard is each split, and do the models agree? Four models answer four splits at six stereo baselines. This notebook asks two things the raw accuracy table cannot: **how hard is each split once its own guessing floor is taken out**, and **do the models find the same items hard**. Raw accuracy is not comparable across splits. VSR is a two-way true/false question, so guessing already scores 0.500; the three Youtube levels are four-way, so guessing scores 0.250. Everything below is therefore reported as $$\text{headroom captured} = \frac{\text{acc} - \text{floor}}{1 - \text{floor}}$$ the share of the distance between guessing and perfect that a model actually covers. It goes negative when a model does worse than free. """) md("## Setup") co(r""" import json, pathlib, itertools, warnings import numpy as np import pandas as pd import matplotlib as mpl import matplotlib.pyplot as plt warnings.filterwarnings("ignore") pd.set_option("display.width", 200) HERE = pathlib.Path.cwd() ROOT = HERE if (HERE / "result").is_dir() else HERE.parents[1] RESULT = ROOT / "result" REPO = ROOT.parent FIGDIR = ROOT / "analysis" / "dataset_level_analysis" / "figures" FIGDIR.mkdir(parents=True, exist_ok=True) BASELINES = ["0.00", "0.05", "0.10", "0.15", "0.20", "0.25"] TASKS = ["vsr", "youtube_level1", "youtube_level2", "youtube_level3"] NICE = {"vsr": "VSR", "youtube_level1": "YT L1", "youtube_level2": "YT L2", "youtube_level3": "YT L3"} # guessing floor per split -- all four are plain chance among the options FLOOR = {"vsr": 0.500, "youtube_level1": 0.250, "youtube_level2": 0.250, "youtube_level3": 0.250} CHANCE = dict(FLOOR) MODELS = sorted(p.name for p in RESULT.iterdir() if p.is_dir() and not p.name.startswith("_")) SHORT = {m: m.replace("Qwen_Qwen", "").replace("-Instruct", "") for m in MODELS} COLOR = dict(zip(MODELS, ["#4C72B0", "#DD8452", "#55A868", "#C44E52"])) mpl.rcParams.update({ "figure.dpi": 120, "savefig.dpi": 150, "font.size": 10, "axes.spines.top": False, "axes.spines.right": False, "axes.grid": True, "grid.alpha": .25, "grid.linewidth": .6, "axes.axisbelow": True, "legend.frameon": False, }) print(f"root : {ROOT}") print(f"models : {[SHORT[m] for m in MODELS]}") """) md("### Load every run") co(r""" rows = [] for m in MODELS: for t in TASKS: for b in BASELINES: d = json.loads((RESULT / m / t / f"baseline_{b}" / "metrics.json").read_text()) rows.append(dict(model=m, short=SHORT[m], task=t, baseline=float(b), n=d["num_samples"], accuracy=d["accuracy"], scoring=d["scoring"], unparsed=d["unparsed_rate"])) runs = pd.DataFrame(rows) runs["floor"] = runs["task"].map(FLOOR) runs["headroom"] = (runs["accuracy"] - runs["floor"]) / (1 - runs["floor"]) assert len(runs) == len(MODELS) * len(TASKS) * len(BASELINES), "missing runs" # An unparsed answer is scored wrong, so a model that narrates instead of naming # a letter is penalised twice. Report the rate rather than assume it is zero. bad = runs[runs.unparsed > 0] if len(bad): print("runs with unparsable answers (counted as wrong):") print(bad.groupby(["short", "task"]).unparsed.agg(["max", "mean"]).round(4) .to_string()) else: print("every answer parsed in every run") acc = runs.pivot_table(index="short", columns="task", values="accuracy", aggfunc="mean")[TASKS] head = runs.pivot_table(index="short", columns="task", values="headroom", aggfunc="mean")[TASKS] acc.columns = head.columns = [NICE[t] for t in TASKS] display(acc.round(3).rename_axis("accuracy (mean over baselines)")) display(head.round(3).rename_axis("headroom captured")) """) md(r""" ## 1. Where each model sits relative to its free floor The bar is the accuracy; the dashed rule is what the split hands over for nothing. Anything below the rule is a model doing worse than a rule that ignores the picture. """) co(r""" fig, axes = plt.subplots(1, 4, figsize=(13, 3.4), sharey=True) for ax, t in zip(axes, TASKS): sub = runs[runs.task == t].groupby("short", sort=False) means = [runs[(runs.task == t) & (runs.model == m)].accuracy.mean() for m in MODELS] errs = [runs[(runs.task == t) & (runs.model == m)].accuracy.std() for m in MODELS] x = np.arange(len(MODELS)) ax.bar(x, means, yerr=errs, capsize=3, color=[COLOR[m] for m in MODELS], edgecolor="white", linewidth=.8) ax.axhline(FLOOR[t], ls="--", lw=1.4, color="#333") ax.text(len(MODELS) - .45, FLOOR[t] + .012, f"floor {FLOOR[t]:.2f}", ha="right", fontsize=8.5, color="#333") if CHANCE[t] != FLOOR[t]: ax.axhline(CHANCE[t], ls=":", lw=1.1, color="#999") ax.text(-.45, CHANCE[t] + .012, f"guessing {CHANCE[t]:.2f}", fontsize=8, color="#888") ax.set_xticks(x); ax.set_xticklabels([SHORT[m] for m in MODELS], rotation=30, ha="right") ax.set_title(f"{NICE[t]} (n={runs[runs.task==t].n.iloc[0]})", fontsize=11) axes[0].set_ylabel("accuracy"); axes[0].set_ylim(0, 1) fig.suptitle("Accuracy against the free floor, averaged over the six baselines", y=1.04, fontsize=12) fig.tight_layout(); fig.savefig(FIGDIR / "01_accuracy_vs_floor.png", bbox_inches="tight") """) md(r""" ## 2. Difficulty, on one scale Headroom captured puts the four splits on a common axis. Red is a model losing to its own floor. """) co(r""" fig, ax = plt.subplots(figsize=(7.2, 3.2)) data = head.values vmax = np.abs(data).max() im = ax.imshow(data, cmap="RdYlGn", vmin=-vmax, vmax=vmax, aspect="auto") ax.set_xticks(range(len(TASKS))); ax.set_xticklabels(head.columns) ax.set_yticks(range(len(MODELS))); ax.set_yticklabels(head.index) for i in range(data.shape[0]): for j in range(data.shape[1]): ax.text(j, i, f"{data[i,j]:+.2f}", ha="center", va="center", fontsize=10, color="black") ax.grid(False) ax.set_title("Headroom captured (acc - floor) / (1 - floor)") fig.colorbar(im, ax=ax, shrink=.85, label="share of headroom") fig.tight_layout(); fig.savefig(FIGDIR / "02_headroom_heatmap.png", bbox_inches="tight") """) md(r""" ## 3. Do the models rank the splits the same way? One line per model across the four splits. Parallel lines mean the models agree on which split is hard and differ only in overall strength; crossings mean a model finds a particular split unusually easy or hard. """) co(r""" order = list(head.mean().sort_values(ascending=False).index) fig, ax = plt.subplots(figsize=(7.6, 4.0)) for m in MODELS: y = [head.loc[SHORT[m], c] for c in order] ax.plot(range(len(order)), y, "-o", color=COLOR[m], lw=2, ms=6, label=SHORT[m]) ax.axhline(0, color="#333", lw=1.2, ls="--") ax.text(len(order) - 1, .015, "floor", ha="right", fontsize=8.5, color="#333") ax.set_xticks(range(len(order))); ax.set_xticklabels(order) ax.set_ylabel("headroom captured"); ax.set_title("Difficulty profile per model") ax.legend(loc="upper right", ncol=2) fig.tight_layout(); fig.savefig(FIGDIR / "03_difficulty_profile.png", bbox_inches="tight") ranks = head.rank(axis=1, ascending=False).astype(int) print("difficulty rank each model assigns (1 = most headroom captured):") display(ranks) print("Spearman between models (only four splits, so 1.00 means identical order):") display(head.T.corr(method="spearman").round(2)) """) md(r""" ## 4. Does the stereo baseline move anything? Each panel is one split; each line is one model across the six baselines. A visual tower that models the scene properly should be flat here — the question and the answer never change, only the viewpoint. """) co(r""" fig, axes = plt.subplots(1, 4, figsize=(13.5, 3.2)) for ax, t in zip(axes, TASKS): for m in MODELS: s = runs[(runs.task == t) & (runs.model == m)].sort_values("baseline") ax.plot(s.baseline, s.accuracy, "-o", ms=4, color=COLOR[m], label=SHORT[m]) ax.axhline(FLOOR[t], ls="--", lw=1.1, color="#666") ax.set_title(NICE[t]); ax.set_xlabel("baseline") lo = min(runs[runs.task == t].accuracy.min(), FLOOR[t]) - .03 hi = max(runs[runs.task == t].accuracy.max(), FLOOR[t]) + .03 ax.set_ylim(lo, hi) axes[0].set_ylabel("accuracy") axes[-1].legend(loc="center left", bbox_to_anchor=(1.02, .5)) fig.suptitle("Accuracy across the intervention, per split", y=1.05, fontsize=12) fig.tight_layout(); fig.savefig(FIGDIR / "04_baseline_sensitivity.png", bbox_inches="tight") spread = (runs.groupby(["short", "task"]).accuracy.agg(["min", "max"]) .assign(range=lambda d: d["max"] - d["min"])["range"] .unstack()[TASKS].rename(columns=NICE)) display(spread.round(3).rename_axis("accuracy range over the six baselines")) """) md(r""" ## 5. Item-level difficulty: do the models fail on the same pictures? For every item, count how many of the four models get it right at baseline 0.00. If difficulty were a property of the item, the counts would pile up at 0 and 4. If each model failed idiosyncratically, the counts would follow the independent (Poisson-binomial) null drawn behind the bars. """) co(r""" def load_pred(m, t, b="0.00"): path = RESULT / m / t / f"baseline_{b}" / "predictions.jsonl" return {json.loads(l)["sample_id"]: json.loads(l) for l in path.read_text().splitlines() if l.strip()} correct_by_task = {} for t in TASKS: per = {m: load_pred(m, t) for m in MODELS} ids = sorted(set.intersection(*(set(v) for v in per.values()))) correct_by_task[t] = pd.DataFrame( {SHORT[m]: [per[m][i]["correct"] for i in ids] for m in MODELS}, index=ids) def poisson_binomial(ps): dist = np.zeros(len(ps) + 1); dist[0] = 1.0 for p in ps: dist[1:] = dist[1:] * (1 - p) + dist[:-1] * p dist[0] *= (1 - p) return dist fig, axes = plt.subplots(1, 4, figsize=(13.5, 3.2), sharey=True) conc = {} for ax, t in zip(axes, TASKS): df = correct_by_task[t] k = df.sum(axis=1).value_counts().reindex(range(5), fill_value=0).sort_index() obs = k / k.sum() null = poisson_binomial(df.mean().values) ax.bar(range(5), null, color="#CCCCCC", width=.82, label="independent null") ax.bar(range(5), obs, color="#4C72B0", width=.45, label="observed") ax.set_title(f"{NICE[t]} (n={len(df)})"); ax.set_xlabel("# models correct") ax.set_xticks(range(5)) # how much mass sits at the two extremes, versus the null conc[NICE[t]] = dict(observed_0_or_4=obs.iloc[[0, 4]].sum(), null_0_or_4=null[[0, 4]].sum()) axes[0].set_ylabel("share of items"); axes[0].legend(fontsize=8.5) fig.suptitle("Agreement on which items are hard", y=1.05, fontsize=12) fig.tight_layout(); fig.savefig(FIGDIR / "05_item_difficulty.png", bbox_inches="tight") conc = pd.DataFrame(conc).T conc["excess"] = conc["observed_0_or_4"] - conc["null_0_or_4"] display(conc.round(3).rename_axis("mass at 0 or 4 models correct")) """) md(r""" ## 6. Failure overlap between models Jaccard overlap of the failure sets, per split. High values mean the models stumble on the same items — shared difficulty rather than independent noise. """) co(r""" fig, axes = plt.subplots(1, 4, figsize=(14, 3.4)) for ax, t in zip(axes, TASKS): df = correct_by_task[t] names = list(df.columns) M = np.ones((len(names), len(names))) for i, a in enumerate(names): for j, b in enumerate(names): if i == j: continue fa, fb = ~df[a].values, ~df[b].values M[i, j] = (fa & fb).sum() / max((fa | fb).sum(), 1) im = ax.imshow(M, cmap="Blues", vmin=0, vmax=1) ax.set_xticks(range(len(names))); ax.set_xticklabels(names, rotation=40, ha="right", fontsize=8) ax.set_yticks(range(len(names))); ax.set_yticklabels(names, fontsize=8) for i in range(len(names)): for j in range(len(names)): ax.text(j, i, f"{M[i,j]:.2f}", ha="center", va="center", fontsize=8, color="white" if M[i, j] > .55 else "black") ax.grid(False); ax.set_title(NICE[t], fontsize=11) fig.suptitle("Jaccard overlap of failure sets (baseline 0.00)", y=1.06, fontsize=12) fig.tight_layout(); fig.savefig(FIGDIR / "06_failure_overlap.png", bbox_inches="tight") """) md(r""" ## 7. Is the model answering, or just picking a letter? The reference labels are near-uniform in every split, so a flat prediction histogram is what an unbiased model produces. A spike means the model has collapsed onto one option, and that spike caps how well it can possibly do. """) co(r""" fig, axes = plt.subplots(len(MODELS), 4, figsize=(12, 2.3 * len(MODELS)), sharey="row") for r, m in enumerate(MODELS): for c, t in enumerate(TASKS): ax = axes[r, c] pred = load_pred(m, t) pc = pd.Series([v["prediction_norm"] for v in pred.values()]).value_counts() rc = pd.Series([v["reference_norm"] for v in pred.values()]).value_counts() labels = sorted(set(pc.index) | set(rc.index)) x = np.arange(len(labels)) ax.bar(x - .2, [rc.get(l, 0) / len(pred) for l in labels], width=.4, color="#BBBBBB", label="reference") ax.bar(x + .2, [pc.get(l, 0) / len(pred) for l in labels], width=.4, color=COLOR[m], label="prediction") ax.set_xticks(x); ax.set_xticklabels(labels, fontsize=8) ax.axhline(1 / len(labels), ls=":", lw=1, color="#666") if r == 0: ax.set_title(NICE[t]) if c == 0: ax.set_ylabel(SHORT[m], fontsize=9) if r == 0 and c == 0: ax.legend(fontsize=7.5) fig.suptitle("Predicted vs reference label distribution (baseline 0.00)", y=1.01, fontsize=12) fig.tight_layout(); fig.savefig(FIGDIR / "07_answer_bias.png", bbox_inches="tight") """) md(r""" ## 8. Level 2 regression guard: is the option text still uninformative? Level 2's three distractors are drawn uniformly from the 23 wrong orderings, so the four options should be statistically exchangeable and no text-only rule should beat chance. Earlier, structured distractor sets did not have that property — the worst of them let a majority vote on successive positions narrow every item to two candidates and score 0.500 blind. This cell re-runs that vote. `in survivor set` near 1.0 means the vote eliminated nothing, which is what random distractors should produce. If it ever drops toward 0.5 while accuracy rises, the construction has reacquired a leak. """) co(r""" L2_TEST = REPO / "02_data" / "Youtube_self_depth_QA" / "level2" / "test.jsonl" rows = [] if L2_TEST.exists(): opts = {} for line in L2_TEST.read_text().splitlines(): rec = json.loads(line) if rec.get("options"): opts[rec["sample_id"]] = {k: v.split(" > ") for k, v in rec["options"].items()} def survivors(o): alive = list(o) for k in range(4): if len(alive) <= 1: break counts = pd.Series([o[a][k] for a in alive]).value_counts() if len(counts) > 1 and counts.iloc[0] == counts.iloc[1]: continue alive = [a for a in alive if o[a][k] == counts.index[0]] return alive surv = {sid: survivors(o) for sid, o in opts.items()} for m in MODELS: pred = load_pred(m, "youtube_level2") ins = [p["prediction_norm"] in surv[s] for s, p in pred.items() if s in surv] hit = [p["correct"] for s, p in pred.items() if s in surv] rows.append(dict(model=SHORT[m], accuracy=np.mean(hit), in_survivor_set=np.mean(ins))) l2 = pd.DataFrame(rows).set_index("model") surv = np.mean([len(v) for v in surv.values()]) print(f"mean survivor-set size after the vote: {surv:.2f} of 4 options") display(l2.round(3)) fig, ax = plt.subplots(figsize=(7, 3.2)) x = np.arange(len(l2)) ax.bar(x - .2, l2.in_survivor_set, width=.4, color="#999", label="answer survives the text-only vote") ax.bar(x + .2, l2.accuracy, width=.4, color="#C44E52", label="accuracy") ax.axhline(.25, ls="--", color="#333", lw=1.3) ax.text(len(l2) - .4, .265, "chance", ha="right", fontsize=8.5) ax.set_xticks(x); ax.set_xticklabels(l2.index, rotation=25, ha="right") ax.set_ylim(0, 1); ax.set_ylabel("share") ax.set_title("Level 2: random distractors leave the text-only vote powerless") ax.legend() fig.tight_layout(); fig.savefig(FIGDIR / "08_level2_guard.png", bbox_inches="tight") else: print(f"level-2 split not found at {L2_TEST}; skipping") """) md(r""" ## Findings **1. Every model puts VSR first and YT L1 second; the two families then disagree.** Qwen2.5-VL finds level 3 hardest, Qwen3.5 finds level 2 hardest. Spearman is 1.00 within a family and 0.80 across, so the ordering of the two Youtube reasoning levels is the one place where architecture, not just strength, changes what looks hard. **2. Only VSR is comfortably solved** (0.52-0.66 of headroom). It is also the easiest question type: a binary relation that is often inferable from object identity alone, without metric depth. **3. Level 3 is the sharpest discriminator.** Qwen2.5-VL captures 0.03 (3B) and 0.07 (7B) — indistinguishable from guessing — against 0.29 and 0.33 for Qwen3.5. A 4-10x gap, far wider than anywhere else in the suite. Continuous trajectory reasoning is what separates the two families. **4. Level 2 sits at 0.17-0.26 of headroom for everyone**, a narrow band that does not track model size or family. Its distractors are drawn at random from the 23 wrong orderings, so nothing in the option text helps (figure 8: the text-only vote leaves ~2 of 4 options standing and scores 0.252 on the split, i.e. chance). The scores are what four-way depth ranking actually costs these models: clearly above guessing, nowhere near solved. **5. The base model beats the instruct model on every split.** Qwen3.5-4B-Base scored by likelihood outperforms Qwen3.5-4B scored by generation everywhere, and 4B-Instruct is the only model producing unparsable answers (up to 0.7% on L1), which are scored wrong. Part of the instruct model's deficit is format compliance, not perception. **6. The stereo intervention barely moves the aggregate.** Accuracy ranges over the six baselines are 0.012-0.044 with no consistent sign across models or splits. Whether the same *items* flip is a separate question, answered in `../intervention_analysis`. **7. Models agree most on which L1 items are hard, least on L3.** Excess mass at "0 or 4 models correct" over the independent null: +0.36 (L1), +0.27 (VSR), +0.23 (L2), +0.14 (L3). On L3 the failures are close to independent noise, which is what near-chance accuracy looks like — there is little shared signal to agree about. ### A note on the level-2 numbers Level 2 was rebuilt three times before this run. Two earlier option constructions leaked: distractors grown outward from the ground truth made it the option most similar to the others (0.388 blind), and a prefix ladder let a majority vote narrow every item to two (0.500 blind). Under the leaking constructions these same models scored 0.40-0.47 on level 2; with random distractors they score 0.37-0.45 against a real 0.25 floor. The drop of 0.02-0.10 is the part of the old score that came from the option text rather than the picture. """) nb = nbf.v4.new_notebook(cells=C) nb.metadata.kernelspec = {"display_name": "Spatial_VLM", "language": "python", "name": "python3"} nb.metadata.language_info = {"name": "python"} os.makedirs(os.path.dirname(OUT), exist_ok=True) nbf.write(nb, OUT) print("wrote", OUT, f"({len(C)} cells)")