File size: 22,338 Bytes
18104f9 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274 275 276 277 278 279 280 281 282 283 284 285 286 287 288 289 290 291 292 293 294 295 296 297 298 299 300 301 302 303 304 305 306 307 308 309 310 311 312 313 314 315 316 317 318 319 320 321 322 323 324 325 326 327 328 329 330 331 332 333 334 335 336 337 338 339 340 341 342 343 344 345 346 347 348 349 350 351 352 353 354 355 356 357 358 359 360 361 362 363 364 365 366 367 368 369 370 371 372 373 374 375 376 377 378 379 380 381 382 383 384 385 386 387 388 389 390 391 392 393 394 395 396 397 398 399 400 401 402 403 404 405 406 407 408 409 410 411 412 413 414 415 416 417 418 419 420 421 422 423 424 425 426 427 428 429 430 431 432 433 434 435 436 437 438 439 440 441 442 443 444 445 446 447 448 449 450 451 452 453 454 455 456 457 458 459 460 461 462 463 464 465 466 467 468 469 470 471 472 473 474 475 476 477 478 479 480 481 482 483 484 485 486 487 488 489 490 491 492 493 494 495 496 497 498 499 500 501 502 503 504 505 506 507 508 509 510 511 512 513 514 515 516 517 518 519 520 521 522 523 524 525 526 527 528 529 | """Author the analysis notebook as a list of cells, then emit .ipynb JSON."""
import json, pathlib
C = []
def md(s): C.append(("markdown", s.strip("\n")))
def code(s): C.append(("code", s.strip("\n")))
md(r"""
# Baseline-intervention analysis
Every model answers the **same question about the same scene** at six stereo
baselines (0.00 = untouched original -> 0.25). Decoding is greedy, so any change
in the answer across baselines is the intervention's effect.
The premise under test: *if the visual tower modelled the world perfectly, a small
intervention should not change the output.*
Run top to bottom. Regenerate the inputs first if any inference has been re-run:
```bash
python analysis/analyze_baseline_overlap.py --result_dir result --dump_per_sample
```
See `readme.md` in this directory for what each measure means and for the caveats
attached to the current result set.
""")
md("## Setup")
code(r"""
import json, pathlib, itertools, math, warnings
import numpy as np
import pandas as pd
import matplotlib as mpl
import matplotlib.pyplot as plt
from matplotlib.colors import LinearSegmentedColormap
warnings.filterwarnings("ignore")
pd.set_option("display.width", 200)
pd.set_option("display.max_columns", 40)
# The experiment root is the directory holding both result/ and dataset/. Finding
# it by walking up means the notebook runs from any cwd -- its own directory, the
# experiment root, or anywhere between -- and survives the analysis outputs being
# moved again.
def find_root(start):
for d in [start, *start.parents]:
if (d / "result").is_dir() and (d / "dataset").is_dir():
return d
raise RuntimeError(f"no experiment root (a dir with result/ and dataset/) above {start}")
ROOT = find_root(pathlib.Path.cwd().resolve())
RESULT_DIR = ROOT / "result"
ANALYSIS_DIR = ROOT / "analysis" / "intervention_analysis"
assert (ANALYSIS_DIR / "analysis_summary.json").is_file(), (
f"analysis_summary.json not found in {ANALYSIS_DIR}. Run first:\n"
" python analysis/analyze_baseline_overlap.py --dump_per_sample"
)
print("root:", ROOT, "\nanalysis dir:", ANALYSIS_DIR)
""")
md(r"""
### Palette
Categorical slots carry **model identity** and are assigned in fixed order — never
cycled, never reassigned when a filter changes the series count. Magnitude
(overlap heatmaps) uses a single-hue blue ramp, light -> dark.
Two categorical slots sit below 3:1 against the surface, so every chart below
ships direct labels or a table view beside it rather than relying on the fill
alone.
""")
code(r"""
SURFACE = "#fcfcfb"
INK = "#0b0b0b"
INK_2 = "#52514e"
GRID = "#e4e3df"
# Categorical slots 1-4, fixed order. Validated adjacent-pair on the light surface.
SLOTS = ["#2a78d6", "#eb6834", "#1baf7a", "#eda100"]
# Sequential blue 100 -> 700, for magnitude (Jaccard heatmaps).
SEQ = ["#cde2fb", "#b7d3f6", "#9ec5f4", "#86b6ef", "#6da7ec", "#5598e7", "#3987e5",
"#2a78d6", "#256abf", "#1c5cab", "#184f95", "#104281", "#0d366b"]
CMAP = LinearSegmentedColormap.from_list("seq_blue", SEQ)
mpl.rcParams.update({
"figure.facecolor": SURFACE, "axes.facecolor": SURFACE, "savefig.facecolor": SURFACE,
"axes.edgecolor": GRID, "axes.labelcolor": INK_2, "axes.titlecolor": INK,
"xtick.color": INK_2, "ytick.color": INK_2, "text.color": INK,
"grid.color": GRID, "grid.linewidth": 0.8,
"axes.spines.top": False, "axes.spines.right": False,
"axes.titlesize": 11, "axes.titleweight": "600", "axes.labelsize": 9,
"xtick.labelsize": 9, "ytick.labelsize": 9, "legend.fontsize": 9,
"legend.frameon": False, "lines.linewidth": 2, "lines.markersize": 5,
"font.size": 10, "figure.dpi": 110,
})
""")
md("## Load")
code(r"""
summary = json.loads((ANALYSIS_DIR / "analysis_summary.json").read_text())
runs = summary["runs"]
SHORT = {
"Qwen/Qwen2.5-VL-3B-Instruct": "Qwen2.5-VL-3B",
"Qwen/Qwen2.5-VL-7B-Instruct": "Qwen2.5-VL-7B",
"Qwen/Qwen3.5-4B": "Qwen3.5-4B",
"Qwen/Qwen3.5-4B-Base": "Qwen3.5-4B-Base",
}
TASK_LABEL = {"vsr": "VSR", "youtube_level1": "L1", "youtube_level2": "L2", "youtube_level3": "L3"}
# Fixed display order: colour follows the entity, so these indices never move.
MODELS = ["Qwen2.5-VL-3B", "Qwen2.5-VL-7B", "Qwen3.5-4B", "Qwen3.5-4B-Base"]
TASKS = ["VSR", "L1", "L2", "L3"]
COLOR = dict(zip(MODELS, SLOTS))
# Chance level per split: VSR is True/False, the Youtube levels are 4-way.
CHANCE = {"VSR": 0.5, "L1": 0.25, "L2": 0.25, "L3": 0.25}
BASELINES = ["0.00", "0.05", "0.10", "0.15", "0.20", "0.25"]
per_baseline = pd.DataFrame([
{"model": SHORT.get(r["model_id"], r["model_id"]), "task": TASK_LABEL.get(r["task"], r["task"]), **row}
for r in runs for row in r["per_baseline"]
])
pairwise = pd.DataFrame([
{"model": SHORT.get(r["model_id"], r["model_id"]), "task": TASK_LABEL.get(r["task"], r["task"]), **row}
for r in runs for row in r["pairwise"]
])
stability = pd.DataFrame([
{"model": SHORT.get(r["model_id"], r["model_id"]), "task": TASK_LABEL.get(r["task"], r["task"]),
"n": r["num_common_samples"], **r["stability"]}
for r in runs
])
for df in (per_baseline, pairwise, stability):
df["model"] = pd.Categorical(df["model"], MODELS, ordered=True)
df["task"] = pd.Categorical(df["task"], TASKS, ordered=True)
print(f"{len(runs)} runs | {per_baseline.task.nunique()} splits x {per_baseline.model.nunique()} models")
per_baseline.pivot_table(index=["task", "model"], columns="baseline", values="accuracy", observed=True).round(3)
""")
md(r"""
## 1. Accuracy across the sweep
Accuracy is the headline but the *least* informative measure here: a model can
hold it flat while answering differently on half the samples, because wins and
losses cancel. It is here to establish the level relative to chance (dashed), so
the flip and overlap numbers that follow can be read in context.
""")
code(r"""
def facet_by_task(value, ylabel, title, ylim=None, chance=False, ax_hook=None):
# One panel per split, one line per model, direct-labelled at the right end.
fig, axes = plt.subplots(1, len(TASKS), figsize=(15, 3.5), sharey=(ylim is not None))
x = np.arange(len(BASELINES))
for ax, task in zip(axes, TASKS):
sub = per_baseline[per_baseline.task == task]
for model in MODELS:
s = sub[sub.model == model].set_index("baseline").reindex(BASELINES)[value]
ax.plot(x, s.values, color=COLOR[model], marker="o", label=model,
markeredgecolor=SURFACE, markeredgewidth=1.2, zorder=3)
if chance:
ax.axhline(CHANCE[task], color=INK_2, lw=1, ls=(0, (4, 3)), zorder=1)
# Left-anchored: the right end is where the series land on L3.
ax.annotate(f"chance {CHANCE[task]:.2f}", (-0.3, CHANCE[task]), xytext=(0, 3),
textcoords="offset points", fontsize=7.5, color=INK_2,
va="bottom", ha="left")
ax.set_title(task); ax.set_xticks(x); ax.set_xticklabels(BASELINES, rotation=45)
ax.set_xlabel("baseline"); ax.grid(axis="y", zorder=0)
ax.set_xlim(-0.35, len(BASELINES) - 0.65)
if ylim: ax.set_ylim(*ylim)
if ax_hook: ax_hook(ax, task)
axes[0].set_ylabel(ylabel)
handles, labels = axes[0].get_legend_handles_labels()
fig.tight_layout()
fig.legend(handles, labels, loc="upper center", bbox_to_anchor=(0.5, 1.045), ncol=4)
fig.suptitle(title, y=1.135, fontsize=12, fontweight="600", x=0.5)
return fig
facet_by_task("accuracy", "accuracy", "Accuracy vs stereo baseline", chance=True)
plt.show()
""")
md(r"""
### Bias-corrected view
All four models carry a heavy option-position bias, in different directions, so
raw accuracy is not comparable across models within a split. Cohen's κ nets out
each model's own answer marginal — a model that always answers "B" scores κ = 0
however often "B" happens to be right.
κ is recomputed here from the raw predictions rather than read from the summary,
because it needs the answer *distribution*, which the summary does not carry.
""")
code(r"""
def load_predictions(model_short, task_short):
inv_m = {v: k for k, v in SHORT.items()}
inv_t = {v: k for k, v in TASK_LABEL.items()}
tag = inv_m[model_short].replace("/", "_")
frames = []
for b in BASELINES:
p = RESULT_DIR / tag / inv_t[task_short] / f"baseline_{b}" / "predictions.jsonl"
if not p.is_file():
continue
df = pd.read_json(p, lines=True)
df["baseline"] = b
frames.append(df)
return pd.concat(frames, ignore_index=True) if frames else None
def cohen_kappa(df):
# (acc - pe) / (1 - pe), with pe from the gold x predicted marginals.
n = len(df)
gold = df.reference_norm.value_counts(normalize=True)
pred = df.prediction_norm.value_counts(normalize=True)
pe = sum(gold[k] * pred.get(k, 0.0) for k in gold.index)
acc = df.correct.mean()
return (acc - pe) / (1 - pe) if pe < 1 else float("nan")
rows = []
for model in MODELS:
for task in TASKS:
df = load_predictions(model, task)
if df is None:
continue
for b, g in df.groupby("baseline"):
rows.append({"model": model, "task": task, "baseline": b,
"accuracy": g.correct.mean(), "kappa": cohen_kappa(g)})
kappa_df = pd.DataFrame(rows)
kappa_df["model"] = pd.Categorical(kappa_df["model"], MODELS, ordered=True)
kappa_df["task"] = pd.Categorical(kappa_df["task"], TASKS, ordered=True)
kappa_mean = kappa_df.pivot_table(index="model", columns="task", values="kappa", observed=True)
display(kappa_mean.round(3))
fig, ax = plt.subplots(figsize=(8, 3.6))
w, x = 0.2, np.arange(len(TASKS))
for i, model in enumerate(MODELS):
v = kappa_mean.loc[model, TASKS].values
bars = ax.bar(x + (i - 1.5) * w, v, w * 0.9, color=COLOR[model], label=model,
edgecolor=SURFACE, linewidth=2, zorder=3)
ax.bar_label(bars, fmt="%.2f", fontsize=7, padding=2, color=INK_2)
ax.set_xticks(x); ax.set_xticklabels(TASKS); ax.set_ylabel("Cohen's kappa")
ax.set_title("Bias-corrected skill by split (0 = no better than the model's own answer prior)")
ax.axhline(0, color=INK_2, lw=1); ax.grid(axis="y", zorder=0)
ax.legend(ncol=4, loc="upper center", bbox_to_anchor=(0.5, -0.12))
plt.show()
""")
md(r"""
## 2. Flip rate — the direct test of the premise
The fraction of samples whose **answer changes** relative to baseline 0.00,
independent of whether it was right. Under the premise, this line should be flat
at zero.
Read the **shape**, not the level:
- **rising with baseline** -> a real intervention effect; bigger interventions move
more samples.
- **high but flat from the first step** -> the model is near chance and the "flips"
are its own indifference. Level 3 does this for every model.
""")
code(r"""
facet_by_task("flip_rate_vs_reference", "flip rate vs b=0.00",
"Answer flips relative to the untouched view", ylim=(0, 0.42))
plt.show()
flip = per_baseline[per_baseline.baseline != "0.00"].pivot_table(
index=["task", "model"], columns="baseline", values="flip_rate_vs_reference", observed=True)
flip["slope (0.25 - 0.05)"] = flip["0.25"] - flip["0.05"]
display(flip.round(3))
print("A positive slope is intervention signal; ~0 with a high level is chance churn.")
""")
md(r"""
## 3. Failure and success overlap
For each pair of baselines, Jaccard over the samples each one gets **wrong**, and
over the samples each one gets **right**.
- **High** -> both baselines miss the same items: failure is a property of the item.
- **Low** -> each baseline breaks a different subset: failure is driven by the
intervention.
Both are reported because they are not redundant — with 4-way answers the success
sets are small and the failure sets large, so the two move independently.
""")
code(r"""
def overlap_matrix(model, task, column):
m = pd.DataFrame(np.eye(len(BASELINES)), index=BASELINES, columns=BASELINES)
sub = pairwise[(pairwise.model == model) & (pairwise.task == task)]
for _, r in sub.iterrows():
m.loc[r.baseline_a, r.baseline_b] = m.loc[r.baseline_b, r.baseline_a] = r[column]
return m
def overlap_grid(column, title, vmin, vmax):
# Model names head the columns and split names label the rows, so no panel
# carries a title that could collide with the tick labels of the panel above.
fig, axes = plt.subplots(len(TASKS), len(MODELS), figsize=(14, 13.6))
for i, task in enumerate(TASKS):
for j, model in enumerate(MODELS):
ax = axes[i, j]
m = overlap_matrix(model, task, column)
ax.imshow(m.values, cmap=CMAP, vmin=vmin, vmax=vmax)
for a in range(len(BASELINES)):
for b in range(len(BASELINES)):
v = m.values[a, b]
# Label every cell: the fill alone is below the contrast floor.
ax.text(b, a, f"{v:.2f}", ha="center", va="center", fontsize=7,
color=SURFACE if v > (vmin + vmax) / 2 else INK)
ax.set_xticks(range(len(BASELINES))); ax.set_yticks(range(len(BASELINES)))
ax.set_xticklabels(BASELINES if i == len(TASKS) - 1 else [], rotation=90, fontsize=7)
ax.set_yticklabels(BASELINES if j == 0 else [], fontsize=7)
ax.tick_params(length=0)
if i == 0:
ax.set_title(model, fontsize=10, pad=10)
if j == 0:
ax.set_ylabel(task, fontsize=11, fontweight="600", labelpad=8)
for sp in ax.spines.values(): sp.set_visible(False)
fig.suptitle(title, y=0.965, fontsize=12, fontweight="600")
fig.colorbar(mpl.cm.ScalarMappable(mpl.colors.Normalize(vmin, vmax), CMAP),
ax=axes, orientation="horizontal", fraction=0.02, pad=0.05, label="Jaccard")
return fig
overlap_grid("failure_jaccard", "Failure-set overlap between baselines (Jaccard)", 0.4, 1.0)
plt.show()
""")
code(r"""
overlap_grid("success_jaccard", "Success-set overlap between baselines (Jaccard)", 0.4, 1.0)
plt.show()
""")
md(r"""
### Does overlap decay with the size of the intervention?
The matrices above collapse to one question: as two baselines move further apart,
do their failure sets separate? A flat line means the intervention is not what
drives the disagreement.
""")
code(r"""
pairwise["gap"] = (pairwise.baseline_b.astype(float) - pairwise.baseline_a.astype(float)).round(2)
fig, axes = plt.subplots(1, 2, figsize=(13, 4), sharey=True)
for ax, col, name in zip(axes, ["failure_jaccard", "success_jaccard"], ["failure", "success"]):
ends = {}
for task in TASKS:
g = pairwise[pairwise.task == task].groupby("gap", observed=True)[col].mean()
# One line per split, averaged over models; splits are the entity here.
ax.plot(g.index, g.values, marker="o", label=task,
color=SLOTS[TASKS.index(task)], markeredgecolor=SURFACE, markeredgewidth=1.2)
ends[task] = (g.index[-1], g.values[-1])
# Direct labels can land on top of each other where two splits converge;
# nudge them apart along y so both stay readable.
span = max(v for _, v in ends.values()) - min(v for _, v in ends.values())
placed = []
for task, (gx, gy) in sorted(ends.items(), key=lambda kv: kv[1][1]):
while any(abs(gy - y) < span * 0.05 for y in placed):
gy += span * 0.05
placed.append(gy)
ax.annotate(task, (gx, gy), xytext=(6, 0), textcoords="offset points",
fontsize=8.5, color=INK_2, va="center")
ax.set_xlabel("baseline separation |b_a - b_b|"); ax.set_title(f"{name} overlap")
ax.grid(axis="y", zorder=0); ax.set_xlim(0.03, 0.285)
axes[0].set_ylabel("mean Jaccard (over models)")
axes[0].legend(ncol=4, loc="lower center", bbox_to_anchor=(0.5, -0.42), fontsize=8.5)
fig.suptitle("Overlap vs intervention size", y=1.02, fontsize=12, fontweight="600")
plt.show()
display(pairwise.pivot_table(index="task", columns="gap", values="failure_jaccard",
observed=True).round(3))
""")
md(r"""
## 4. Stability — which samples the intervention actually reaches
Each sample is classified over the six baselines:
| class | meaning |
|---|---|
| `always_correct` | right at every baseline — the intervention never broke it |
| `always_wrong` | wrong at every baseline — beyond the model, intervention irrelevant |
| `unstable` | right at some, wrong at others — **the population under test** |
`unstable` is the number that matters. Pair it with the flip-curve slope from §2:
a high unstable rate with a flat flip curve is chance-level churn, not an effect.
""")
code(r"""
order = stability.sort_values(["task", "model"], ascending=[True, True]).reset_index(drop=True)
labels = [f"{r.task} {r.model}" for _, r in order.iterrows()]
segs = [("always_correct_rate", SLOTS[2], "always correct"),
("always_wrong_rate", SLOTS[1], "always wrong"),
("unstable_rate", SLOTS[0], "unstable")]
fig, ax = plt.subplots(figsize=(11, 7))
left = np.zeros(len(order))
for col, color, name in segs:
v = order[col].values
# 2px surface gap between adjacent segments so the boundary reads.
ax.barh(labels, v, left=left, color=color, label=name, height=0.72,
edgecolor=SURFACE, linewidth=2, zorder=3)
for i, (l, w) in enumerate(zip(left, v)):
if w > 0.06:
ax.text(l + w / 2, i, f"{w:.2f}", ha="center", va="center",
fontsize=7.5, color=SURFACE if color != SLOTS[3] else INK)
left += v
ax.set_xlim(0, 1); ax.set_xlabel("fraction of samples"); ax.invert_yaxis()
ax.grid(axis="x", zorder=0)
ax.set_title("Sample stability across the six baselines")
ax.legend(ncol=3, loc="upper center", bbox_to_anchor=(0.5, -0.07))
plt.show()
display(order.set_index(["task", "model"])[
["n", "always_correct_rate", "always_wrong_rate", "unstable_rate"]].round(3))
""")
md(r"""
## 5. How far the unstable samples move
`unstable` lumps together a sample that flipped once and one that flipped at
every baseline. The per-sample files carry `num_correct` (0-6), so the shape of
that distribution says whether instability is a thin edge or a broad band. A
chance-level model piles up in the middle; a model with real signal is U-shaped.
""")
code(r"""
def per_sample(model, task):
inv_m = {v: k for k, v in SHORT.items()}
inv_t = {v: k for k, v in TASK_LABEL.items()}
p = ANALYSIS_DIR / f"{inv_m[model].replace('/', '_')}__{inv_t[task]}_per_sample.csv"
return pd.read_csv(p) if p.is_file() else None
fig, axes = plt.subplots(1, len(TASKS), figsize=(15, 3.5), sharey=True)
x = np.arange(7)
for ax, task in zip(axes, TASKS):
for i, model in enumerate(MODELS):
df = per_sample(model, task)
if df is None:
continue
frac = df.num_correct.value_counts(normalize=True).reindex(x, fill_value=0)
ax.plot(x, frac.values, marker="o", color=COLOR[model], label=model,
markeredgecolor=SURFACE, markeredgewidth=1.2, zorder=3)
ax.set_title(task); ax.set_xticks(x); ax.set_xlabel("# baselines correct (of 6)")
ax.grid(axis="y", zorder=0)
axes[0].set_ylabel("fraction of samples")
h, l = axes[0].get_legend_handles_labels()
fig.legend(h, l, loc="upper center", bbox_to_anchor=(0.5, 1.10), ncol=4)
fig.suptitle("U-shaped = decided answers; humped in the middle = chance churn",
y=1.20, fontsize=12, fontweight="600")
fig.tight_layout()
plt.show()
""")
md(r"""
## 6. Summary
One row per run. `kappa` is the bias-corrected skill from §1, `flip@0.25` and
`flip slope` the intervention effect from §2, `unstable` the population it reaches
from §4, and `failure J @0.25` the failure overlap between the extremes.
""")
code(r"""
summary_rows = []
for model in MODELS:
for task in TASKS:
pb = per_baseline[(per_baseline.model == model) & (per_baseline.task == task)]
if pb.empty:
continue
pw = pairwise[(pairwise.model == model) & (pairwise.task == task)]
st = stability[(stability.model == model) & (stability.task == task)].iloc[0]
kp = kappa_df[(kappa_df.model == model) & (kappa_df.task == task)].kappa.mean()
f = pb.set_index("baseline").flip_rate_vs_reference
ext = pw[(pw.baseline_a == "0.00") & (pw.baseline_b == "0.25")]
summary_rows.append({
"model": model, "task": task, "n": int(st.n),
"acc": pb.accuracy.mean(), "chance": CHANCE[task], "kappa": kp,
"acc spread": pb.accuracy.max() - pb.accuracy.min(),
"flip@0.25": f["0.25"], "flip slope": f["0.25"] - f["0.05"],
"unstable": st.unstable_rate,
"failure J @0.25": ext.failure_jaccard.iloc[0] if len(ext) else np.nan,
"success J @0.25": ext.success_jaccard.iloc[0] if len(ext) else np.nan,
})
out = pd.DataFrame(summary_rows).set_index(["task", "model"]).round(3)
out.to_csv(ANALYSIS_DIR / "notebook_summary.csv")
print(f"written: {ANALYSIS_DIR / 'notebook_summary.csv'}")
out
""")
md(r"""
### Reading the summary
- **`flip slope` > 0** with a moderate `flip@0.25` is the signature of a real
intervention effect.
- **`flip@0.25` high with `flip slope` ~ 0** is chance churn — check `kappa` before
reading anything into it.
- **`failure J @0.25` well below 1** means the extremes fail on different samples,
i.e. the intervention moved the failure set rather than just its size.
See `readme.md` for the caveats that constrain what these numbers can support —
in particular the non-uniform evaluation modes, the level-2 v2 data boundary, and
level 3 sitting at chance for both Qwen2.5-VL models.
""")
nb = {
"cells": [
{"cell_type": t, "metadata": {}, "source": s.splitlines(keepends=True),
**({"outputs": [], "execution_count": None} if t == "code" else {})}
for t, s in C
],
"metadata": {
"kernelspec": {"display_name": "Spatial_VLM", "language": "python", "name": "python3"},
"language_info": {"name": "python", "version": "3.10"},
},
"nbformat": 4, "nbformat_minor": 5,
}
out = pathlib.Path(__file__).resolve().parent / "baseline_intervention.ipynb"
out.write_text(json.dumps(nb, indent=1))
print("wrote", out, len(C), "cells")
|