"""Author the analysis notebook as a list of cells, then emit .ipynb JSON.""" import json, pathlib C = [] def md(s): C.append(("markdown", s.strip("\n"))) def code(s): C.append(("code", s.strip("\n"))) md(r""" # Baseline-intervention analysis Every model answers the **same question about the same scene** at six stereo baselines (0.00 = untouched original -> 0.25). Decoding is greedy, so any change in the answer across baselines is the intervention's effect. The premise under test: *if the visual tower modelled the world perfectly, a small intervention should not change the output.* Run top to bottom. Regenerate the inputs first if any inference has been re-run: ```bash python analysis/analyze_baseline_overlap.py --result_dir result --dump_per_sample ``` See `readme.md` in this directory for what each measure means and for the caveats attached to the current result set. """) md("## Setup") code(r""" import json, pathlib, itertools, math, warnings import numpy as np import pandas as pd import matplotlib as mpl import matplotlib.pyplot as plt from matplotlib.colors import LinearSegmentedColormap warnings.filterwarnings("ignore") pd.set_option("display.width", 200) pd.set_option("display.max_columns", 40) # The experiment root is the directory holding both result/ and dataset/. Finding # it by walking up means the notebook runs from any cwd -- its own directory, the # experiment root, or anywhere between -- and survives the analysis outputs being # moved again. def find_root(start): for d in [start, *start.parents]: if (d / "result").is_dir() and (d / "dataset").is_dir(): return d raise RuntimeError(f"no experiment root (a dir with result/ and dataset/) above {start}") ROOT = find_root(pathlib.Path.cwd().resolve()) RESULT_DIR = ROOT / "result" ANALYSIS_DIR = ROOT / "analysis" / "intervention_analysis" assert (ANALYSIS_DIR / "analysis_summary.json").is_file(), ( f"analysis_summary.json not found in {ANALYSIS_DIR}. Run first:\n" " python analysis/analyze_baseline_overlap.py --dump_per_sample" ) print("root:", ROOT, "\nanalysis dir:", ANALYSIS_DIR) """) md(r""" ### Palette Categorical slots carry **model identity** and are assigned in fixed order — never cycled, never reassigned when a filter changes the series count. Magnitude (overlap heatmaps) uses a single-hue blue ramp, light -> dark. Two categorical slots sit below 3:1 against the surface, so every chart below ships direct labels or a table view beside it rather than relying on the fill alone. """) code(r""" SURFACE = "#fcfcfb" INK = "#0b0b0b" INK_2 = "#52514e" GRID = "#e4e3df" # Categorical slots 1-4, fixed order. Validated adjacent-pair on the light surface. SLOTS = ["#2a78d6", "#eb6834", "#1baf7a", "#eda100"] # Sequential blue 100 -> 700, for magnitude (Jaccard heatmaps). SEQ = ["#cde2fb", "#b7d3f6", "#9ec5f4", "#86b6ef", "#6da7ec", "#5598e7", "#3987e5", "#2a78d6", "#256abf", "#1c5cab", "#184f95", "#104281", "#0d366b"] CMAP = LinearSegmentedColormap.from_list("seq_blue", SEQ) mpl.rcParams.update({ "figure.facecolor": SURFACE, "axes.facecolor": SURFACE, "savefig.facecolor": SURFACE, "axes.edgecolor": GRID, "axes.labelcolor": INK_2, "axes.titlecolor": INK, "xtick.color": INK_2, "ytick.color": INK_2, "text.color": INK, "grid.color": GRID, "grid.linewidth": 0.8, "axes.spines.top": False, "axes.spines.right": False, "axes.titlesize": 11, "axes.titleweight": "600", "axes.labelsize": 9, "xtick.labelsize": 9, "ytick.labelsize": 9, "legend.fontsize": 9, "legend.frameon": False, "lines.linewidth": 2, "lines.markersize": 5, "font.size": 10, "figure.dpi": 110, }) """) md("## Load") code(r""" summary = json.loads((ANALYSIS_DIR / "analysis_summary.json").read_text()) runs = summary["runs"] SHORT = { "Qwen/Qwen2.5-VL-3B-Instruct": "Qwen2.5-VL-3B", "Qwen/Qwen2.5-VL-7B-Instruct": "Qwen2.5-VL-7B", "Qwen/Qwen3.5-4B": "Qwen3.5-4B", "Qwen/Qwen3.5-4B-Base": "Qwen3.5-4B-Base", } TASK_LABEL = {"vsr": "VSR", "youtube_level1": "L1", "youtube_level2": "L2", "youtube_level3": "L3"} # Fixed display order: colour follows the entity, so these indices never move. MODELS = ["Qwen2.5-VL-3B", "Qwen2.5-VL-7B", "Qwen3.5-4B", "Qwen3.5-4B-Base"] TASKS = ["VSR", "L1", "L2", "L3"] COLOR = dict(zip(MODELS, SLOTS)) # Chance level per split: VSR is True/False, the Youtube levels are 4-way. CHANCE = {"VSR": 0.5, "L1": 0.25, "L2": 0.25, "L3": 0.25} BASELINES = ["0.00", "0.05", "0.10", "0.15", "0.20", "0.25"] per_baseline = pd.DataFrame([ {"model": SHORT.get(r["model_id"], r["model_id"]), "task": TASK_LABEL.get(r["task"], r["task"]), **row} for r in runs for row in r["per_baseline"] ]) pairwise = pd.DataFrame([ {"model": SHORT.get(r["model_id"], r["model_id"]), "task": TASK_LABEL.get(r["task"], r["task"]), **row} for r in runs for row in r["pairwise"] ]) stability = pd.DataFrame([ {"model": SHORT.get(r["model_id"], r["model_id"]), "task": TASK_LABEL.get(r["task"], r["task"]), "n": r["num_common_samples"], **r["stability"]} for r in runs ]) for df in (per_baseline, pairwise, stability): df["model"] = pd.Categorical(df["model"], MODELS, ordered=True) df["task"] = pd.Categorical(df["task"], TASKS, ordered=True) print(f"{len(runs)} runs | {per_baseline.task.nunique()} splits x {per_baseline.model.nunique()} models") per_baseline.pivot_table(index=["task", "model"], columns="baseline", values="accuracy", observed=True).round(3) """) md(r""" ## 1. Accuracy across the sweep Accuracy is the headline but the *least* informative measure here: a model can hold it flat while answering differently on half the samples, because wins and losses cancel. It is here to establish the level relative to chance (dashed), so the flip and overlap numbers that follow can be read in context. """) code(r""" def facet_by_task(value, ylabel, title, ylim=None, chance=False, ax_hook=None): # One panel per split, one line per model, direct-labelled at the right end. fig, axes = plt.subplots(1, len(TASKS), figsize=(15, 3.5), sharey=(ylim is not None)) x = np.arange(len(BASELINES)) for ax, task in zip(axes, TASKS): sub = per_baseline[per_baseline.task == task] for model in MODELS: s = sub[sub.model == model].set_index("baseline").reindex(BASELINES)[value] ax.plot(x, s.values, color=COLOR[model], marker="o", label=model, markeredgecolor=SURFACE, markeredgewidth=1.2, zorder=3) if chance: ax.axhline(CHANCE[task], color=INK_2, lw=1, ls=(0, (4, 3)), zorder=1) # Left-anchored: the right end is where the series land on L3. ax.annotate(f"chance {CHANCE[task]:.2f}", (-0.3, CHANCE[task]), xytext=(0, 3), textcoords="offset points", fontsize=7.5, color=INK_2, va="bottom", ha="left") ax.set_title(task); ax.set_xticks(x); ax.set_xticklabels(BASELINES, rotation=45) ax.set_xlabel("baseline"); ax.grid(axis="y", zorder=0) ax.set_xlim(-0.35, len(BASELINES) - 0.65) if ylim: ax.set_ylim(*ylim) if ax_hook: ax_hook(ax, task) axes[0].set_ylabel(ylabel) handles, labels = axes[0].get_legend_handles_labels() fig.tight_layout() fig.legend(handles, labels, loc="upper center", bbox_to_anchor=(0.5, 1.045), ncol=4) fig.suptitle(title, y=1.135, fontsize=12, fontweight="600", x=0.5) return fig facet_by_task("accuracy", "accuracy", "Accuracy vs stereo baseline", chance=True) plt.show() """) md(r""" ### Bias-corrected view All four models carry a heavy option-position bias, in different directions, so raw accuracy is not comparable across models within a split. Cohen's κ nets out each model's own answer marginal — a model that always answers "B" scores κ = 0 however often "B" happens to be right. κ is recomputed here from the raw predictions rather than read from the summary, because it needs the answer *distribution*, which the summary does not carry. """) code(r""" def load_predictions(model_short, task_short): inv_m = {v: k for k, v in SHORT.items()} inv_t = {v: k for k, v in TASK_LABEL.items()} tag = inv_m[model_short].replace("/", "_") frames = [] for b in BASELINES: p = RESULT_DIR / tag / inv_t[task_short] / f"baseline_{b}" / "predictions.jsonl" if not p.is_file(): continue df = pd.read_json(p, lines=True) df["baseline"] = b frames.append(df) return pd.concat(frames, ignore_index=True) if frames else None def cohen_kappa(df): # (acc - pe) / (1 - pe), with pe from the gold x predicted marginals. n = len(df) gold = df.reference_norm.value_counts(normalize=True) pred = df.prediction_norm.value_counts(normalize=True) pe = sum(gold[k] * pred.get(k, 0.0) for k in gold.index) acc = df.correct.mean() return (acc - pe) / (1 - pe) if pe < 1 else float("nan") rows = [] for model in MODELS: for task in TASKS: df = load_predictions(model, task) if df is None: continue for b, g in df.groupby("baseline"): rows.append({"model": model, "task": task, "baseline": b, "accuracy": g.correct.mean(), "kappa": cohen_kappa(g)}) kappa_df = pd.DataFrame(rows) kappa_df["model"] = pd.Categorical(kappa_df["model"], MODELS, ordered=True) kappa_df["task"] = pd.Categorical(kappa_df["task"], TASKS, ordered=True) kappa_mean = kappa_df.pivot_table(index="model", columns="task", values="kappa", observed=True) display(kappa_mean.round(3)) fig, ax = plt.subplots(figsize=(8, 3.6)) w, x = 0.2, np.arange(len(TASKS)) for i, model in enumerate(MODELS): v = kappa_mean.loc[model, TASKS].values bars = ax.bar(x + (i - 1.5) * w, v, w * 0.9, color=COLOR[model], label=model, edgecolor=SURFACE, linewidth=2, zorder=3) ax.bar_label(bars, fmt="%.2f", fontsize=7, padding=2, color=INK_2) ax.set_xticks(x); ax.set_xticklabels(TASKS); ax.set_ylabel("Cohen's kappa") ax.set_title("Bias-corrected skill by split (0 = no better than the model's own answer prior)") ax.axhline(0, color=INK_2, lw=1); ax.grid(axis="y", zorder=0) ax.legend(ncol=4, loc="upper center", bbox_to_anchor=(0.5, -0.12)) plt.show() """) md(r""" ## 2. Flip rate — the direct test of the premise The fraction of samples whose **answer changes** relative to baseline 0.00, independent of whether it was right. Under the premise, this line should be flat at zero. Read the **shape**, not the level: - **rising with baseline** -> a real intervention effect; bigger interventions move more samples. - **high but flat from the first step** -> the model is near chance and the "flips" are its own indifference. Level 3 does this for every model. """) code(r""" facet_by_task("flip_rate_vs_reference", "flip rate vs b=0.00", "Answer flips relative to the untouched view", ylim=(0, 0.42)) plt.show() flip = per_baseline[per_baseline.baseline != "0.00"].pivot_table( index=["task", "model"], columns="baseline", values="flip_rate_vs_reference", observed=True) flip["slope (0.25 - 0.05)"] = flip["0.25"] - flip["0.05"] display(flip.round(3)) print("A positive slope is intervention signal; ~0 with a high level is chance churn.") """) md(r""" ## 3. Failure and success overlap For each pair of baselines, Jaccard over the samples each one gets **wrong**, and over the samples each one gets **right**. - **High** -> both baselines miss the same items: failure is a property of the item. - **Low** -> each baseline breaks a different subset: failure is driven by the intervention. Both are reported because they are not redundant — with 4-way answers the success sets are small and the failure sets large, so the two move independently. """) code(r""" def overlap_matrix(model, task, column): m = pd.DataFrame(np.eye(len(BASELINES)), index=BASELINES, columns=BASELINES) sub = pairwise[(pairwise.model == model) & (pairwise.task == task)] for _, r in sub.iterrows(): m.loc[r.baseline_a, r.baseline_b] = m.loc[r.baseline_b, r.baseline_a] = r[column] return m def overlap_grid(column, title, vmin, vmax): # Model names head the columns and split names label the rows, so no panel # carries a title that could collide with the tick labels of the panel above. fig, axes = plt.subplots(len(TASKS), len(MODELS), figsize=(14, 13.6)) for i, task in enumerate(TASKS): for j, model in enumerate(MODELS): ax = axes[i, j] m = overlap_matrix(model, task, column) ax.imshow(m.values, cmap=CMAP, vmin=vmin, vmax=vmax) for a in range(len(BASELINES)): for b in range(len(BASELINES)): v = m.values[a, b] # Label every cell: the fill alone is below the contrast floor. ax.text(b, a, f"{v:.2f}", ha="center", va="center", fontsize=7, color=SURFACE if v > (vmin + vmax) / 2 else INK) ax.set_xticks(range(len(BASELINES))); ax.set_yticks(range(len(BASELINES))) ax.set_xticklabels(BASELINES if i == len(TASKS) - 1 else [], rotation=90, fontsize=7) ax.set_yticklabels(BASELINES if j == 0 else [], fontsize=7) ax.tick_params(length=0) if i == 0: ax.set_title(model, fontsize=10, pad=10) if j == 0: ax.set_ylabel(task, fontsize=11, fontweight="600", labelpad=8) for sp in ax.spines.values(): sp.set_visible(False) fig.suptitle(title, y=0.965, fontsize=12, fontweight="600") fig.colorbar(mpl.cm.ScalarMappable(mpl.colors.Normalize(vmin, vmax), CMAP), ax=axes, orientation="horizontal", fraction=0.02, pad=0.05, label="Jaccard") return fig overlap_grid("failure_jaccard", "Failure-set overlap between baselines (Jaccard)", 0.4, 1.0) plt.show() """) code(r""" overlap_grid("success_jaccard", "Success-set overlap between baselines (Jaccard)", 0.4, 1.0) plt.show() """) md(r""" ### Does overlap decay with the size of the intervention? The matrices above collapse to one question: as two baselines move further apart, do their failure sets separate? A flat line means the intervention is not what drives the disagreement. """) code(r""" pairwise["gap"] = (pairwise.baseline_b.astype(float) - pairwise.baseline_a.astype(float)).round(2) fig, axes = plt.subplots(1, 2, figsize=(13, 4), sharey=True) for ax, col, name in zip(axes, ["failure_jaccard", "success_jaccard"], ["failure", "success"]): ends = {} for task in TASKS: g = pairwise[pairwise.task == task].groupby("gap", observed=True)[col].mean() # One line per split, averaged over models; splits are the entity here. ax.plot(g.index, g.values, marker="o", label=task, color=SLOTS[TASKS.index(task)], markeredgecolor=SURFACE, markeredgewidth=1.2) ends[task] = (g.index[-1], g.values[-1]) # Direct labels can land on top of each other where two splits converge; # nudge them apart along y so both stay readable. span = max(v for _, v in ends.values()) - min(v for _, v in ends.values()) placed = [] for task, (gx, gy) in sorted(ends.items(), key=lambda kv: kv[1][1]): while any(abs(gy - y) < span * 0.05 for y in placed): gy += span * 0.05 placed.append(gy) ax.annotate(task, (gx, gy), xytext=(6, 0), textcoords="offset points", fontsize=8.5, color=INK_2, va="center") ax.set_xlabel("baseline separation |b_a - b_b|"); ax.set_title(f"{name} overlap") ax.grid(axis="y", zorder=0); ax.set_xlim(0.03, 0.285) axes[0].set_ylabel("mean Jaccard (over models)") axes[0].legend(ncol=4, loc="lower center", bbox_to_anchor=(0.5, -0.42), fontsize=8.5) fig.suptitle("Overlap vs intervention size", y=1.02, fontsize=12, fontweight="600") plt.show() display(pairwise.pivot_table(index="task", columns="gap", values="failure_jaccard", observed=True).round(3)) """) md(r""" ## 4. Stability — which samples the intervention actually reaches Each sample is classified over the six baselines: | class | meaning | |---|---| | `always_correct` | right at every baseline — the intervention never broke it | | `always_wrong` | wrong at every baseline — beyond the model, intervention irrelevant | | `unstable` | right at some, wrong at others — **the population under test** | `unstable` is the number that matters. Pair it with the flip-curve slope from §2: a high unstable rate with a flat flip curve is chance-level churn, not an effect. """) code(r""" order = stability.sort_values(["task", "model"], ascending=[True, True]).reset_index(drop=True) labels = [f"{r.task} {r.model}" for _, r in order.iterrows()] segs = [("always_correct_rate", SLOTS[2], "always correct"), ("always_wrong_rate", SLOTS[1], "always wrong"), ("unstable_rate", SLOTS[0], "unstable")] fig, ax = plt.subplots(figsize=(11, 7)) left = np.zeros(len(order)) for col, color, name in segs: v = order[col].values # 2px surface gap between adjacent segments so the boundary reads. ax.barh(labels, v, left=left, color=color, label=name, height=0.72, edgecolor=SURFACE, linewidth=2, zorder=3) for i, (l, w) in enumerate(zip(left, v)): if w > 0.06: ax.text(l + w / 2, i, f"{w:.2f}", ha="center", va="center", fontsize=7.5, color=SURFACE if color != SLOTS[3] else INK) left += v ax.set_xlim(0, 1); ax.set_xlabel("fraction of samples"); ax.invert_yaxis() ax.grid(axis="x", zorder=0) ax.set_title("Sample stability across the six baselines") ax.legend(ncol=3, loc="upper center", bbox_to_anchor=(0.5, -0.07)) plt.show() display(order.set_index(["task", "model"])[ ["n", "always_correct_rate", "always_wrong_rate", "unstable_rate"]].round(3)) """) md(r""" ## 5. How far the unstable samples move `unstable` lumps together a sample that flipped once and one that flipped at every baseline. The per-sample files carry `num_correct` (0-6), so the shape of that distribution says whether instability is a thin edge or a broad band. A chance-level model piles up in the middle; a model with real signal is U-shaped. """) code(r""" def per_sample(model, task): inv_m = {v: k for k, v in SHORT.items()} inv_t = {v: k for k, v in TASK_LABEL.items()} p = ANALYSIS_DIR / f"{inv_m[model].replace('/', '_')}__{inv_t[task]}_per_sample.csv" return pd.read_csv(p) if p.is_file() else None fig, axes = plt.subplots(1, len(TASKS), figsize=(15, 3.5), sharey=True) x = np.arange(7) for ax, task in zip(axes, TASKS): for i, model in enumerate(MODELS): df = per_sample(model, task) if df is None: continue frac = df.num_correct.value_counts(normalize=True).reindex(x, fill_value=0) ax.plot(x, frac.values, marker="o", color=COLOR[model], label=model, markeredgecolor=SURFACE, markeredgewidth=1.2, zorder=3) ax.set_title(task); ax.set_xticks(x); ax.set_xlabel("# baselines correct (of 6)") ax.grid(axis="y", zorder=0) axes[0].set_ylabel("fraction of samples") h, l = axes[0].get_legend_handles_labels() fig.legend(h, l, loc="upper center", bbox_to_anchor=(0.5, 1.10), ncol=4) fig.suptitle("U-shaped = decided answers; humped in the middle = chance churn", y=1.20, fontsize=12, fontweight="600") fig.tight_layout() plt.show() """) md(r""" ## 6. Summary One row per run. `kappa` is the bias-corrected skill from §1, `flip@0.25` and `flip slope` the intervention effect from §2, `unstable` the population it reaches from §4, and `failure J @0.25` the failure overlap between the extremes. """) code(r""" summary_rows = [] for model in MODELS: for task in TASKS: pb = per_baseline[(per_baseline.model == model) & (per_baseline.task == task)] if pb.empty: continue pw = pairwise[(pairwise.model == model) & (pairwise.task == task)] st = stability[(stability.model == model) & (stability.task == task)].iloc[0] kp = kappa_df[(kappa_df.model == model) & (kappa_df.task == task)].kappa.mean() f = pb.set_index("baseline").flip_rate_vs_reference ext = pw[(pw.baseline_a == "0.00") & (pw.baseline_b == "0.25")] summary_rows.append({ "model": model, "task": task, "n": int(st.n), "acc": pb.accuracy.mean(), "chance": CHANCE[task], "kappa": kp, "acc spread": pb.accuracy.max() - pb.accuracy.min(), "flip@0.25": f["0.25"], "flip slope": f["0.25"] - f["0.05"], "unstable": st.unstable_rate, "failure J @0.25": ext.failure_jaccard.iloc[0] if len(ext) else np.nan, "success J @0.25": ext.success_jaccard.iloc[0] if len(ext) else np.nan, }) out = pd.DataFrame(summary_rows).set_index(["task", "model"]).round(3) out.to_csv(ANALYSIS_DIR / "notebook_summary.csv") print(f"written: {ANALYSIS_DIR / 'notebook_summary.csv'}") out """) md(r""" ### Reading the summary - **`flip slope` > 0** with a moderate `flip@0.25` is the signature of a real intervention effect. - **`flip@0.25` high with `flip slope` ~ 0** is chance churn — check `kappa` before reading anything into it. - **`failure J @0.25` well below 1** means the extremes fail on different samples, i.e. the intervention moved the failure set rather than just its size. See `readme.md` for the caveats that constrain what these numbers can support — in particular the non-uniform evaluation modes, the level-2 v2 data boundary, and level 3 sitting at chance for both Qwen2.5-VL models. """) nb = { "cells": [ {"cell_type": t, "metadata": {}, "source": s.splitlines(keepends=True), **({"outputs": [], "execution_count": None} if t == "code" else {})} for t, s in C ], "metadata": { "kernelspec": {"display_name": "Spatial_VLM", "language": "python", "name": "python3"}, "language_info": {"name": "python", "version": "3.10"}, }, "nbformat": 4, "nbformat_minor": 5, } out = pathlib.Path(__file__).resolve().parent / "baseline_intervention.ipynb" out.write_text(json.dumps(nb, indent=1)) print("wrote", out, len(C), "cells")