Download analysis/dataset_level_analysis/build_notebook.py from LLDDSS/Stereo_Depth: direct link, hf CLI and curl.
- Browser
- Download file 20.1 kB
-
https://huggingface.co/LLDDSS/Stereo_Depth/resolve/main/analysis/dataset_level_analysis/build_notebook.py
- Command line
-
hf download hf://LLDDSS/Stereo_Depth/analysis/dataset_level_analysis/build_notebook.py
-
curl -L -o build_notebook.py https://huggingface.co/LLDDSS/Stereo_Depth/resolve/main/analysis/dataset_level_analysis/build_notebook.py
20.1 kB
| #!/usr/bin/env python3 | |
| """Build dataset_difficulty.ipynb from the cell sources below.""" | |
| import nbformat as nbf, os | |
| OUT = ("/HDD2/disheng/Spatial_VLM_Stereo/June/Spatial_VLM_Stereo/" | |
| "01_3_prelimilary_intervention_different_baseline/analysis/" | |
| "dataset_level_analysis/dataset_difficulty.ipynb") | |
| C = [] | |
| md = lambda s: C.append(nbf.v4.new_markdown_cell(s.strip("\n"))) | |
| co = lambda s: C.append(nbf.v4.new_code_cell(s.strip("\n"))) | |
| md(r""" | |
| # How hard is each split, and do the models agree? | |
| Four models answer four splits at six stereo baselines. This notebook asks two | |
| things the raw accuracy table cannot: **how hard is each split once its own | |
| guessing floor is taken out**, and **do the models find the same items hard**. | |
| Raw accuracy is not comparable across splits. VSR is a two-way true/false | |
| question, so guessing already scores 0.500; the three Youtube levels are | |
| four-way, so guessing scores 0.250. Everything below is therefore reported as | |
| $$\text{headroom captured} = \frac{\text{acc} - \text{floor}}{1 - \text{floor}}$$ | |
| the share of the distance between guessing and perfect that a model actually | |
| covers. It goes negative when a model does worse than free. | |
| """) | |
| md("## Setup") | |
| co(r""" | |
| import json, pathlib, itertools, warnings | |
| import numpy as np | |
| import pandas as pd | |
| import matplotlib as mpl | |
| import matplotlib.pyplot as plt | |
| warnings.filterwarnings("ignore") | |
| pd.set_option("display.width", 200) | |
| HERE = pathlib.Path.cwd() | |
| ROOT = HERE if (HERE / "result").is_dir() else HERE.parents[1] | |
| RESULT = ROOT / "result" | |
| REPO = ROOT.parent | |
| FIGDIR = ROOT / "analysis" / "dataset_level_analysis" / "figures" | |
| FIGDIR.mkdir(parents=True, exist_ok=True) | |
| BASELINES = ["0.00", "0.05", "0.10", "0.15", "0.20", "0.25"] | |
| TASKS = ["vsr", "youtube_level1", "youtube_level2", "youtube_level3"] | |
| NICE = {"vsr": "VSR", "youtube_level1": "YT L1", "youtube_level2": "YT L2", | |
| "youtube_level3": "YT L3"} | |
| # guessing floor per split -- all four are plain chance among the options | |
| FLOOR = {"vsr": 0.500, "youtube_level1": 0.250, | |
| "youtube_level2": 0.250, "youtube_level3": 0.250} | |
| CHANCE = dict(FLOOR) | |
| MODELS = sorted(p.name for p in RESULT.iterdir() | |
| if p.is_dir() and not p.name.startswith("_")) | |
| SHORT = {m: m.replace("Qwen_Qwen", "").replace("-Instruct", "") for m in MODELS} | |
| COLOR = dict(zip(MODELS, ["#4C72B0", "#DD8452", "#55A868", "#C44E52"])) | |
| mpl.rcParams.update({ | |
| "figure.dpi": 120, "savefig.dpi": 150, "font.size": 10, | |
| "axes.spines.top": False, "axes.spines.right": False, | |
| "axes.grid": True, "grid.alpha": .25, "grid.linewidth": .6, | |
| "axes.axisbelow": True, "legend.frameon": False, | |
| }) | |
| print(f"root : {ROOT}") | |
| print(f"models : {[SHORT[m] for m in MODELS]}") | |
| """) | |
| md("### Load every run") | |
| co(r""" | |
| rows = [] | |
| for m in MODELS: | |
| for t in TASKS: | |
| for b in BASELINES: | |
| d = json.loads((RESULT / m / t / f"baseline_{b}" / "metrics.json").read_text()) | |
| rows.append(dict(model=m, short=SHORT[m], task=t, baseline=float(b), | |
| n=d["num_samples"], accuracy=d["accuracy"], | |
| scoring=d["scoring"], unparsed=d["unparsed_rate"])) | |
| runs = pd.DataFrame(rows) | |
| runs["floor"] = runs["task"].map(FLOOR) | |
| runs["headroom"] = (runs["accuracy"] - runs["floor"]) / (1 - runs["floor"]) | |
| assert len(runs) == len(MODELS) * len(TASKS) * len(BASELINES), "missing runs" | |
| # An unparsed answer is scored wrong, so a model that narrates instead of naming | |
| # a letter is penalised twice. Report the rate rather than assume it is zero. | |
| bad = runs[runs.unparsed > 0] | |
| if len(bad): | |
| print("runs with unparsable answers (counted as wrong):") | |
| print(bad.groupby(["short", "task"]).unparsed.agg(["max", "mean"]).round(4) | |
| .to_string()) | |
| else: | |
| print("every answer parsed in every run") | |
| acc = runs.pivot_table(index="short", columns="task", values="accuracy", aggfunc="mean")[TASKS] | |
| head = runs.pivot_table(index="short", columns="task", values="headroom", aggfunc="mean")[TASKS] | |
| acc.columns = head.columns = [NICE[t] for t in TASKS] | |
| display(acc.round(3).rename_axis("accuracy (mean over baselines)")) | |
| display(head.round(3).rename_axis("headroom captured")) | |
| """) | |
| md(r""" | |
| ## 1. Where each model sits relative to its free floor | |
| The bar is the accuracy; the dashed rule is what the split hands over for | |
| nothing. Anything below the rule is a model doing worse than a rule that ignores | |
| the picture. | |
| """) | |
| co(r""" | |
| fig, axes = plt.subplots(1, 4, figsize=(13, 3.4), sharey=True) | |
| for ax, t in zip(axes, TASKS): | |
| sub = runs[runs.task == t].groupby("short", sort=False) | |
| means = [runs[(runs.task == t) & (runs.model == m)].accuracy.mean() for m in MODELS] | |
| errs = [runs[(runs.task == t) & (runs.model == m)].accuracy.std() for m in MODELS] | |
| x = np.arange(len(MODELS)) | |
| ax.bar(x, means, yerr=errs, capsize=3, color=[COLOR[m] for m in MODELS], | |
| edgecolor="white", linewidth=.8) | |
| ax.axhline(FLOOR[t], ls="--", lw=1.4, color="#333") | |
| ax.text(len(MODELS) - .45, FLOOR[t] + .012, f"floor {FLOOR[t]:.2f}", | |
| ha="right", fontsize=8.5, color="#333") | |
| if CHANCE[t] != FLOOR[t]: | |
| ax.axhline(CHANCE[t], ls=":", lw=1.1, color="#999") | |
| ax.text(-.45, CHANCE[t] + .012, f"guessing {CHANCE[t]:.2f}", | |
| fontsize=8, color="#888") | |
| ax.set_xticks(x); ax.set_xticklabels([SHORT[m] for m in MODELS], rotation=30, ha="right") | |
| ax.set_title(f"{NICE[t]} (n={runs[runs.task==t].n.iloc[0]})", fontsize=11) | |
| axes[0].set_ylabel("accuracy"); axes[0].set_ylim(0, 1) | |
| fig.suptitle("Accuracy against the free floor, averaged over the six baselines", | |
| y=1.04, fontsize=12) | |
| fig.tight_layout(); fig.savefig(FIGDIR / "01_accuracy_vs_floor.png", bbox_inches="tight") | |
| """) | |
| md(r""" | |
| ## 2. Difficulty, on one scale | |
| Headroom captured puts the four splits on a common axis. Red is a model losing | |
| to its own floor. | |
| """) | |
| co(r""" | |
| fig, ax = plt.subplots(figsize=(7.2, 3.2)) | |
| data = head.values | |
| vmax = np.abs(data).max() | |
| im = ax.imshow(data, cmap="RdYlGn", vmin=-vmax, vmax=vmax, aspect="auto") | |
| ax.set_xticks(range(len(TASKS))); ax.set_xticklabels(head.columns) | |
| ax.set_yticks(range(len(MODELS))); ax.set_yticklabels(head.index) | |
| for i in range(data.shape[0]): | |
| for j in range(data.shape[1]): | |
| ax.text(j, i, f"{data[i,j]:+.2f}", ha="center", va="center", fontsize=10, | |
| color="black") | |
| ax.grid(False) | |
| ax.set_title("Headroom captured (acc - floor) / (1 - floor)") | |
| fig.colorbar(im, ax=ax, shrink=.85, label="share of headroom") | |
| fig.tight_layout(); fig.savefig(FIGDIR / "02_headroom_heatmap.png", bbox_inches="tight") | |
| """) | |
| md(r""" | |
| ## 3. Do the models rank the splits the same way? | |
| One line per model across the four splits. Parallel lines mean the models agree | |
| on which split is hard and differ only in overall strength; crossings mean a | |
| model finds a particular split unusually easy or hard. | |
| """) | |
| co(r""" | |
| order = list(head.mean().sort_values(ascending=False).index) | |
| fig, ax = plt.subplots(figsize=(7.6, 4.0)) | |
| for m in MODELS: | |
| y = [head.loc[SHORT[m], c] for c in order] | |
| ax.plot(range(len(order)), y, "-o", color=COLOR[m], lw=2, ms=6, label=SHORT[m]) | |
| ax.axhline(0, color="#333", lw=1.2, ls="--") | |
| ax.text(len(order) - 1, .015, "floor", ha="right", fontsize=8.5, color="#333") | |
| ax.set_xticks(range(len(order))); ax.set_xticklabels(order) | |
| ax.set_ylabel("headroom captured"); ax.set_title("Difficulty profile per model") | |
| ax.legend(loc="upper right", ncol=2) | |
| fig.tight_layout(); fig.savefig(FIGDIR / "03_difficulty_profile.png", bbox_inches="tight") | |
| ranks = head.rank(axis=1, ascending=False).astype(int) | |
| print("difficulty rank each model assigns (1 = most headroom captured):") | |
| display(ranks) | |
| print("Spearman between models (only four splits, so 1.00 means identical order):") | |
| display(head.T.corr(method="spearman").round(2)) | |
| """) | |
| md(r""" | |
| ## 4. Does the stereo baseline move anything? | |
| Each panel is one split; each line is one model across the six baselines. A | |
| visual tower that models the scene properly should be flat here — the question | |
| and the answer never change, only the viewpoint. | |
| """) | |
| co(r""" | |
| fig, axes = plt.subplots(1, 4, figsize=(13.5, 3.2)) | |
| for ax, t in zip(axes, TASKS): | |
| for m in MODELS: | |
| s = runs[(runs.task == t) & (runs.model == m)].sort_values("baseline") | |
| ax.plot(s.baseline, s.accuracy, "-o", ms=4, color=COLOR[m], label=SHORT[m]) | |
| ax.axhline(FLOOR[t], ls="--", lw=1.1, color="#666") | |
| ax.set_title(NICE[t]); ax.set_xlabel("baseline") | |
| lo = min(runs[runs.task == t].accuracy.min(), FLOOR[t]) - .03 | |
| hi = max(runs[runs.task == t].accuracy.max(), FLOOR[t]) + .03 | |
| ax.set_ylim(lo, hi) | |
| axes[0].set_ylabel("accuracy") | |
| axes[-1].legend(loc="center left", bbox_to_anchor=(1.02, .5)) | |
| fig.suptitle("Accuracy across the intervention, per split", y=1.05, fontsize=12) | |
| fig.tight_layout(); fig.savefig(FIGDIR / "04_baseline_sensitivity.png", bbox_inches="tight") | |
| spread = (runs.groupby(["short", "task"]).accuracy.agg(["min", "max"]) | |
| .assign(range=lambda d: d["max"] - d["min"])["range"] | |
| .unstack()[TASKS].rename(columns=NICE)) | |
| display(spread.round(3).rename_axis("accuracy range over the six baselines")) | |
| """) | |
| md(r""" | |
| ## 5. Item-level difficulty: do the models fail on the same pictures? | |
| For every item, count how many of the four models get it right at baseline 0.00. | |
| If difficulty were a property of the item, the counts would pile up at 0 and 4. | |
| If each model failed idiosyncratically, the counts would follow the independent | |
| (Poisson-binomial) null drawn behind the bars. | |
| """) | |
| co(r""" | |
| def load_pred(m, t, b="0.00"): | |
| path = RESULT / m / t / f"baseline_{b}" / "predictions.jsonl" | |
| return {json.loads(l)["sample_id"]: json.loads(l) | |
| for l in path.read_text().splitlines() if l.strip()} | |
| correct_by_task = {} | |
| for t in TASKS: | |
| per = {m: load_pred(m, t) for m in MODELS} | |
| ids = sorted(set.intersection(*(set(v) for v in per.values()))) | |
| correct_by_task[t] = pd.DataFrame( | |
| {SHORT[m]: [per[m][i]["correct"] for i in ids] for m in MODELS}, index=ids) | |
| def poisson_binomial(ps): | |
| dist = np.zeros(len(ps) + 1); dist[0] = 1.0 | |
| for p in ps: | |
| dist[1:] = dist[1:] * (1 - p) + dist[:-1] * p | |
| dist[0] *= (1 - p) | |
| return dist | |
| fig, axes = plt.subplots(1, 4, figsize=(13.5, 3.2), sharey=True) | |
| conc = {} | |
| for ax, t in zip(axes, TASKS): | |
| df = correct_by_task[t] | |
| k = df.sum(axis=1).value_counts().reindex(range(5), fill_value=0).sort_index() | |
| obs = k / k.sum() | |
| null = poisson_binomial(df.mean().values) | |
| ax.bar(range(5), null, color="#CCCCCC", width=.82, label="independent null") | |
| ax.bar(range(5), obs, color="#4C72B0", width=.45, label="observed") | |
| ax.set_title(f"{NICE[t]} (n={len(df)})"); ax.set_xlabel("# models correct") | |
| ax.set_xticks(range(5)) | |
| # how much mass sits at the two extremes, versus the null | |
| conc[NICE[t]] = dict(observed_0_or_4=obs.iloc[[0, 4]].sum(), | |
| null_0_or_4=null[[0, 4]].sum()) | |
| axes[0].set_ylabel("share of items"); axes[0].legend(fontsize=8.5) | |
| fig.suptitle("Agreement on which items are hard", y=1.05, fontsize=12) | |
| fig.tight_layout(); fig.savefig(FIGDIR / "05_item_difficulty.png", bbox_inches="tight") | |
| conc = pd.DataFrame(conc).T | |
| conc["excess"] = conc["observed_0_or_4"] - conc["null_0_or_4"] | |
| display(conc.round(3).rename_axis("mass at 0 or 4 models correct")) | |
| """) | |
| md(r""" | |
| ## 6. Failure overlap between models | |
| Jaccard overlap of the failure sets, per split. High values mean the models | |
| stumble on the same items — shared difficulty rather than independent noise. | |
| """) | |
| co(r""" | |
| fig, axes = plt.subplots(1, 4, figsize=(14, 3.4)) | |
| for ax, t in zip(axes, TASKS): | |
| df = correct_by_task[t] | |
| names = list(df.columns) | |
| M = np.ones((len(names), len(names))) | |
| for i, a in enumerate(names): | |
| for j, b in enumerate(names): | |
| if i == j: | |
| continue | |
| fa, fb = ~df[a].values, ~df[b].values | |
| M[i, j] = (fa & fb).sum() / max((fa | fb).sum(), 1) | |
| im = ax.imshow(M, cmap="Blues", vmin=0, vmax=1) | |
| ax.set_xticks(range(len(names))); ax.set_xticklabels(names, rotation=40, ha="right", fontsize=8) | |
| ax.set_yticks(range(len(names))); ax.set_yticklabels(names, fontsize=8) | |
| for i in range(len(names)): | |
| for j in range(len(names)): | |
| ax.text(j, i, f"{M[i,j]:.2f}", ha="center", va="center", fontsize=8, | |
| color="white" if M[i, j] > .55 else "black") | |
| ax.grid(False); ax.set_title(NICE[t], fontsize=11) | |
| fig.suptitle("Jaccard overlap of failure sets (baseline 0.00)", y=1.06, fontsize=12) | |
| fig.tight_layout(); fig.savefig(FIGDIR / "06_failure_overlap.png", bbox_inches="tight") | |
| """) | |
| md(r""" | |
| ## 7. Is the model answering, or just picking a letter? | |
| The reference labels are near-uniform in every split, so a flat prediction | |
| histogram is what an unbiased model produces. A spike means the model has | |
| collapsed onto one option, and that spike caps how well it can possibly do. | |
| """) | |
| co(r""" | |
| fig, axes = plt.subplots(len(MODELS), 4, figsize=(12, 2.3 * len(MODELS)), | |
| sharey="row") | |
| for r, m in enumerate(MODELS): | |
| for c, t in enumerate(TASKS): | |
| ax = axes[r, c] | |
| pred = load_pred(m, t) | |
| pc = pd.Series([v["prediction_norm"] for v in pred.values()]).value_counts() | |
| rc = pd.Series([v["reference_norm"] for v in pred.values()]).value_counts() | |
| labels = sorted(set(pc.index) | set(rc.index)) | |
| x = np.arange(len(labels)) | |
| ax.bar(x - .2, [rc.get(l, 0) / len(pred) for l in labels], width=.4, | |
| color="#BBBBBB", label="reference") | |
| ax.bar(x + .2, [pc.get(l, 0) / len(pred) for l in labels], width=.4, | |
| color=COLOR[m], label="prediction") | |
| ax.set_xticks(x); ax.set_xticklabels(labels, fontsize=8) | |
| ax.axhline(1 / len(labels), ls=":", lw=1, color="#666") | |
| if r == 0: | |
| ax.set_title(NICE[t]) | |
| if c == 0: | |
| ax.set_ylabel(SHORT[m], fontsize=9) | |
| if r == 0 and c == 0: | |
| ax.legend(fontsize=7.5) | |
| fig.suptitle("Predicted vs reference label distribution (baseline 0.00)", | |
| y=1.01, fontsize=12) | |
| fig.tight_layout(); fig.savefig(FIGDIR / "07_answer_bias.png", bbox_inches="tight") | |
| """) | |
| md(r""" | |
| ## 8. Level 2 regression guard: is the option text still uninformative? | |
| Level 2's three distractors are drawn uniformly from the 23 wrong orderings, so | |
| the four options should be statistically exchangeable and no text-only rule | |
| should beat chance. Earlier, structured distractor sets did not have that | |
| property — the worst of them let a majority vote on successive positions narrow | |
| every item to two candidates and score 0.500 blind. | |
| This cell re-runs that vote. `in survivor set` near 1.0 means the vote eliminated | |
| nothing, which is what random distractors should produce. If it ever drops | |
| toward 0.5 while accuracy rises, the construction has reacquired a leak. | |
| """) | |
| co(r""" | |
| L2_TEST = REPO / "02_data" / "Youtube_self_depth_QA" / "level2" / "test.jsonl" | |
| rows = [] | |
| if L2_TEST.exists(): | |
| opts = {} | |
| for line in L2_TEST.read_text().splitlines(): | |
| rec = json.loads(line) | |
| if rec.get("options"): | |
| opts[rec["sample_id"]] = {k: v.split(" > ") for k, v in rec["options"].items()} | |
| def survivors(o): | |
| alive = list(o) | |
| for k in range(4): | |
| if len(alive) <= 1: | |
| break | |
| counts = pd.Series([o[a][k] for a in alive]).value_counts() | |
| if len(counts) > 1 and counts.iloc[0] == counts.iloc[1]: | |
| continue | |
| alive = [a for a in alive if o[a][k] == counts.index[0]] | |
| return alive | |
| surv = {sid: survivors(o) for sid, o in opts.items()} | |
| for m in MODELS: | |
| pred = load_pred(m, "youtube_level2") | |
| ins = [p["prediction_norm"] in surv[s] for s, p in pred.items() if s in surv] | |
| hit = [p["correct"] for s, p in pred.items() if s in surv] | |
| rows.append(dict(model=SHORT[m], accuracy=np.mean(hit), | |
| in_survivor_set=np.mean(ins))) | |
| l2 = pd.DataFrame(rows).set_index("model") | |
| surv = np.mean([len(v) for v in surv.values()]) | |
| print(f"mean survivor-set size after the vote: {surv:.2f} of 4 options") | |
| display(l2.round(3)) | |
| fig, ax = plt.subplots(figsize=(7, 3.2)) | |
| x = np.arange(len(l2)) | |
| ax.bar(x - .2, l2.in_survivor_set, width=.4, color="#999", | |
| label="answer survives the text-only vote") | |
| ax.bar(x + .2, l2.accuracy, width=.4, color="#C44E52", label="accuracy") | |
| ax.axhline(.25, ls="--", color="#333", lw=1.3) | |
| ax.text(len(l2) - .4, .265, "chance", ha="right", fontsize=8.5) | |
| ax.set_xticks(x); ax.set_xticklabels(l2.index, rotation=25, ha="right") | |
| ax.set_ylim(0, 1); ax.set_ylabel("share") | |
| ax.set_title("Level 2: random distractors leave the text-only vote powerless") | |
| ax.legend() | |
| fig.tight_layout(); fig.savefig(FIGDIR / "08_level2_guard.png", bbox_inches="tight") | |
| else: | |
| print(f"level-2 split not found at {L2_TEST}; skipping") | |
| """) | |
| md(r""" | |
| ## Findings | |
| **1. Every model puts VSR first and YT L1 second; the two families then | |
| disagree.** Qwen2.5-VL finds level 3 hardest, Qwen3.5 finds level 2 hardest. | |
| Spearman is 1.00 within a family and 0.80 across, so the ordering of the two | |
| Youtube reasoning levels is the one place where architecture, not just | |
| strength, changes what looks hard. | |
| **2. Only VSR is comfortably solved** (0.52-0.66 of headroom). It is also the | |
| easiest question type: a binary relation that is often inferable from object | |
| identity alone, without metric depth. | |
| **3. Level 3 is the sharpest discriminator.** Qwen2.5-VL captures 0.03 (3B) and | |
| 0.07 (7B) — indistinguishable from guessing — against 0.29 and 0.33 for | |
| Qwen3.5. A 4-10x gap, far wider than anywhere else in the suite. Continuous | |
| trajectory reasoning is what separates the two families. | |
| **4. Level 2 sits at 0.17-0.26 of headroom for everyone**, a narrow band that | |
| does not track model size or family. Its distractors are drawn at random from | |
| the 23 wrong orderings, so nothing in the option text helps (figure 8: the | |
| text-only vote leaves ~2 of 4 options standing and scores 0.252 on the split, | |
| i.e. chance). The scores are what four-way depth ranking actually costs these | |
| models: clearly above guessing, nowhere near solved. | |
| **5. The base model beats the instruct model on every split.** Qwen3.5-4B-Base | |
| scored by likelihood outperforms Qwen3.5-4B scored by generation everywhere, | |
| and 4B-Instruct is the only model producing unparsable answers (up to 0.7% on | |
| L1), which are scored wrong. Part of the instruct model's deficit is format | |
| compliance, not perception. | |
| **6. The stereo intervention barely moves the aggregate.** Accuracy ranges over | |
| the six baselines are 0.012-0.044 with no consistent sign across models or | |
| splits. Whether the same *items* flip is a separate question, answered in | |
| `../intervention_analysis`. | |
| **7. Models agree most on which L1 items are hard, least on L3.** Excess mass at | |
| "0 or 4 models correct" over the independent null: +0.36 (L1), +0.27 (VSR), | |
| +0.23 (L2), +0.14 (L3). On L3 the failures are close to independent noise, which | |
| is what near-chance accuracy looks like — there is little shared signal to agree | |
| about. | |
| ### A note on the level-2 numbers | |
| Level 2 was rebuilt three times before this run. Two earlier option | |
| constructions leaked: distractors grown outward from the ground truth made it | |
| the option most similar to the others (0.388 blind), and a prefix ladder let a | |
| majority vote narrow every item to two (0.500 blind). Under the leaking | |
| constructions these same models scored 0.40-0.47 on level 2; with random | |
| distractors they score 0.37-0.45 against a real 0.25 floor. The drop of | |
| 0.02-0.10 is the part of the old score that came from the option text rather | |
| than the picture. | |
| """) | |
| nb = nbf.v4.new_notebook(cells=C) | |
| nb.metadata.kernelspec = {"display_name": "Spatial_VLM", "language": "python", | |
| "name": "python3"} | |
| nb.metadata.language_info = {"name": "python"} | |
| os.makedirs(os.path.dirname(OUT), exist_ok=True) | |
| nbf.write(nb, OUT) | |
| print("wrote", OUT, f"({len(C)} cells)") | |