LLDDSS's picture
Upload folder using huggingface_hub
18104f9 verified
Raw History Blame Contribute Delete
22.3 kB
"""Author the analysis notebook as a list of cells, then emit .ipynb JSON."""
import json, pathlib
C = []
def md(s): C.append(("markdown", s.strip("\n")))
def code(s): C.append(("code", s.strip("\n")))
md(r"""
# Baseline-intervention analysis
Every model answers the **same question about the same scene** at six stereo
baselines (0.00 = untouched original -> 0.25). Decoding is greedy, so any change
in the answer across baselines is the intervention's effect.
The premise under test: *if the visual tower modelled the world perfectly, a small
intervention should not change the output.*
Run top to bottom. Regenerate the inputs first if any inference has been re-run:
```bash
python analysis/analyze_baseline_overlap.py --result_dir result --dump_per_sample
```
See `readme.md` in this directory for what each measure means and for the caveats
attached to the current result set.
""")
md("## Setup")
code(r"""
import json, pathlib, itertools, math, warnings
import numpy as np
import pandas as pd
import matplotlib as mpl
import matplotlib.pyplot as plt
from matplotlib.colors import LinearSegmentedColormap
warnings.filterwarnings("ignore")
pd.set_option("display.width", 200)
pd.set_option("display.max_columns", 40)
# The experiment root is the directory holding both result/ and dataset/. Finding
# it by walking up means the notebook runs from any cwd -- its own directory, the
# experiment root, or anywhere between -- and survives the analysis outputs being
# moved again.
def find_root(start):
for d in [start, *start.parents]:
if (d / "result").is_dir() and (d / "dataset").is_dir():
return d
raise RuntimeError(f"no experiment root (a dir with result/ and dataset/) above {start}")
ROOT = find_root(pathlib.Path.cwd().resolve())
RESULT_DIR = ROOT / "result"
ANALYSIS_DIR = ROOT / "analysis" / "intervention_analysis"
assert (ANALYSIS_DIR / "analysis_summary.json").is_file(), (
f"analysis_summary.json not found in {ANALYSIS_DIR}. Run first:\n"
" python analysis/analyze_baseline_overlap.py --dump_per_sample"
)
print("root:", ROOT, "\nanalysis dir:", ANALYSIS_DIR)
""")
md(r"""
### Palette
Categorical slots carry **model identity** and are assigned in fixed order — never
cycled, never reassigned when a filter changes the series count. Magnitude
(overlap heatmaps) uses a single-hue blue ramp, light -> dark.
Two categorical slots sit below 3:1 against the surface, so every chart below
ships direct labels or a table view beside it rather than relying on the fill
alone.
""")
code(r"""
SURFACE = "#fcfcfb"
INK = "#0b0b0b"
INK_2 = "#52514e"
GRID = "#e4e3df"
# Categorical slots 1-4, fixed order. Validated adjacent-pair on the light surface.
SLOTS = ["#2a78d6", "#eb6834", "#1baf7a", "#eda100"]
# Sequential blue 100 -> 700, for magnitude (Jaccard heatmaps).
SEQ = ["#cde2fb", "#b7d3f6", "#9ec5f4", "#86b6ef", "#6da7ec", "#5598e7", "#3987e5",
"#2a78d6", "#256abf", "#1c5cab", "#184f95", "#104281", "#0d366b"]
CMAP = LinearSegmentedColormap.from_list("seq_blue", SEQ)
mpl.rcParams.update({
"figure.facecolor": SURFACE, "axes.facecolor": SURFACE, "savefig.facecolor": SURFACE,
"axes.edgecolor": GRID, "axes.labelcolor": INK_2, "axes.titlecolor": INK,
"xtick.color": INK_2, "ytick.color": INK_2, "text.color": INK,
"grid.color": GRID, "grid.linewidth": 0.8,
"axes.spines.top": False, "axes.spines.right": False,
"axes.titlesize": 11, "axes.titleweight": "600", "axes.labelsize": 9,
"xtick.labelsize": 9, "ytick.labelsize": 9, "legend.fontsize": 9,
"legend.frameon": False, "lines.linewidth": 2, "lines.markersize": 5,
"font.size": 10, "figure.dpi": 110,
})
""")
md("## Load")
code(r"""
summary = json.loads((ANALYSIS_DIR / "analysis_summary.json").read_text())
runs = summary["runs"]
SHORT = {
"Qwen/Qwen2.5-VL-3B-Instruct": "Qwen2.5-VL-3B",
"Qwen/Qwen2.5-VL-7B-Instruct": "Qwen2.5-VL-7B",
"Qwen/Qwen3.5-4B": "Qwen3.5-4B",
"Qwen/Qwen3.5-4B-Base": "Qwen3.5-4B-Base",
}
TASK_LABEL = {"vsr": "VSR", "youtube_level1": "L1", "youtube_level2": "L2", "youtube_level3": "L3"}
# Fixed display order: colour follows the entity, so these indices never move.
MODELS = ["Qwen2.5-VL-3B", "Qwen2.5-VL-7B", "Qwen3.5-4B", "Qwen3.5-4B-Base"]
TASKS = ["VSR", "L1", "L2", "L3"]
COLOR = dict(zip(MODELS, SLOTS))
# Chance level per split: VSR is True/False, the Youtube levels are 4-way.
CHANCE = {"VSR": 0.5, "L1": 0.25, "L2": 0.25, "L3": 0.25}
BASELINES = ["0.00", "0.05", "0.10", "0.15", "0.20", "0.25"]
per_baseline = pd.DataFrame([
{"model": SHORT.get(r["model_id"], r["model_id"]), "task": TASK_LABEL.get(r["task"], r["task"]), **row}
for r in runs for row in r["per_baseline"]
])
pairwise = pd.DataFrame([
{"model": SHORT.get(r["model_id"], r["model_id"]), "task": TASK_LABEL.get(r["task"], r["task"]), **row}
for r in runs for row in r["pairwise"]
])
stability = pd.DataFrame([
{"model": SHORT.get(r["model_id"], r["model_id"]), "task": TASK_LABEL.get(r["task"], r["task"]),
"n": r["num_common_samples"], **r["stability"]}
for r in runs
])
for df in (per_baseline, pairwise, stability):
df["model"] = pd.Categorical(df["model"], MODELS, ordered=True)
df["task"] = pd.Categorical(df["task"], TASKS, ordered=True)
print(f"{len(runs)} runs | {per_baseline.task.nunique()} splits x {per_baseline.model.nunique()} models")
per_baseline.pivot_table(index=["task", "model"], columns="baseline", values="accuracy", observed=True).round(3)
""")
md(r"""
## 1. Accuracy across the sweep
Accuracy is the headline but the *least* informative measure here: a model can
hold it flat while answering differently on half the samples, because wins and
losses cancel. It is here to establish the level relative to chance (dashed), so
the flip and overlap numbers that follow can be read in context.
""")
code(r"""
def facet_by_task(value, ylabel, title, ylim=None, chance=False, ax_hook=None):
# One panel per split, one line per model, direct-labelled at the right end.
fig, axes = plt.subplots(1, len(TASKS), figsize=(15, 3.5), sharey=(ylim is not None))
x = np.arange(len(BASELINES))
for ax, task in zip(axes, TASKS):
sub = per_baseline[per_baseline.task == task]
for model in MODELS:
s = sub[sub.model == model].set_index("baseline").reindex(BASELINES)[value]
ax.plot(x, s.values, color=COLOR[model], marker="o", label=model,
markeredgecolor=SURFACE, markeredgewidth=1.2, zorder=3)
if chance:
ax.axhline(CHANCE[task], color=INK_2, lw=1, ls=(0, (4, 3)), zorder=1)
# Left-anchored: the right end is where the series land on L3.
ax.annotate(f"chance {CHANCE[task]:.2f}", (-0.3, CHANCE[task]), xytext=(0, 3),
textcoords="offset points", fontsize=7.5, color=INK_2,
va="bottom", ha="left")
ax.set_title(task); ax.set_xticks(x); ax.set_xticklabels(BASELINES, rotation=45)
ax.set_xlabel("baseline"); ax.grid(axis="y", zorder=0)
ax.set_xlim(-0.35, len(BASELINES) - 0.65)
if ylim: ax.set_ylim(*ylim)
if ax_hook: ax_hook(ax, task)
axes[0].set_ylabel(ylabel)
handles, labels = axes[0].get_legend_handles_labels()
fig.tight_layout()
fig.legend(handles, labels, loc="upper center", bbox_to_anchor=(0.5, 1.045), ncol=4)
fig.suptitle(title, y=1.135, fontsize=12, fontweight="600", x=0.5)
return fig
facet_by_task("accuracy", "accuracy", "Accuracy vs stereo baseline", chance=True)
plt.show()
""")
md(r"""
### Bias-corrected view
All four models carry a heavy option-position bias, in different directions, so
raw accuracy is not comparable across models within a split. Cohen's κ nets out
each model's own answer marginal — a model that always answers "B" scores κ = 0
however often "B" happens to be right.
κ is recomputed here from the raw predictions rather than read from the summary,
because it needs the answer *distribution*, which the summary does not carry.
""")
code(r"""
def load_predictions(model_short, task_short):
inv_m = {v: k for k, v in SHORT.items()}
inv_t = {v: k for k, v in TASK_LABEL.items()}
tag = inv_m[model_short].replace("/", "_")
frames = []
for b in BASELINES:
p = RESULT_DIR / tag / inv_t[task_short] / f"baseline_{b}" / "predictions.jsonl"
if not p.is_file():
continue
df = pd.read_json(p, lines=True)
df["baseline"] = b
frames.append(df)
return pd.concat(frames, ignore_index=True) if frames else None
def cohen_kappa(df):
# (acc - pe) / (1 - pe), with pe from the gold x predicted marginals.
n = len(df)
gold = df.reference_norm.value_counts(normalize=True)
pred = df.prediction_norm.value_counts(normalize=True)
pe = sum(gold[k] * pred.get(k, 0.0) for k in gold.index)
acc = df.correct.mean()
return (acc - pe) / (1 - pe) if pe < 1 else float("nan")
rows = []
for model in MODELS:
for task in TASKS:
df = load_predictions(model, task)
if df is None:
continue
for b, g in df.groupby("baseline"):
rows.append({"model": model, "task": task, "baseline": b,
"accuracy": g.correct.mean(), "kappa": cohen_kappa(g)})
kappa_df = pd.DataFrame(rows)
kappa_df["model"] = pd.Categorical(kappa_df["model"], MODELS, ordered=True)
kappa_df["task"] = pd.Categorical(kappa_df["task"], TASKS, ordered=True)
kappa_mean = kappa_df.pivot_table(index="model", columns="task", values="kappa", observed=True)
display(kappa_mean.round(3))
fig, ax = plt.subplots(figsize=(8, 3.6))
w, x = 0.2, np.arange(len(TASKS))
for i, model in enumerate(MODELS):
v = kappa_mean.loc[model, TASKS].values
bars = ax.bar(x + (i - 1.5) * w, v, w * 0.9, color=COLOR[model], label=model,
edgecolor=SURFACE, linewidth=2, zorder=3)
ax.bar_label(bars, fmt="%.2f", fontsize=7, padding=2, color=INK_2)
ax.set_xticks(x); ax.set_xticklabels(TASKS); ax.set_ylabel("Cohen's kappa")
ax.set_title("Bias-corrected skill by split (0 = no better than the model's own answer prior)")
ax.axhline(0, color=INK_2, lw=1); ax.grid(axis="y", zorder=0)
ax.legend(ncol=4, loc="upper center", bbox_to_anchor=(0.5, -0.12))
plt.show()
""")
md(r"""
## 2. Flip rate — the direct test of the premise
The fraction of samples whose **answer changes** relative to baseline 0.00,
independent of whether it was right. Under the premise, this line should be flat
at zero.
Read the **shape**, not the level:
- **rising with baseline** -> a real intervention effect; bigger interventions move
more samples.
- **high but flat from the first step** -> the model is near chance and the "flips"
are its own indifference. Level 3 does this for every model.
""")
code(r"""
facet_by_task("flip_rate_vs_reference", "flip rate vs b=0.00",
"Answer flips relative to the untouched view", ylim=(0, 0.42))
plt.show()
flip = per_baseline[per_baseline.baseline != "0.00"].pivot_table(
index=["task", "model"], columns="baseline", values="flip_rate_vs_reference", observed=True)
flip["slope (0.25 - 0.05)"] = flip["0.25"] - flip["0.05"]
display(flip.round(3))
print("A positive slope is intervention signal; ~0 with a high level is chance churn.")
""")
md(r"""
## 3. Failure and success overlap
For each pair of baselines, Jaccard over the samples each one gets **wrong**, and
over the samples each one gets **right**.
- **High** -> both baselines miss the same items: failure is a property of the item.
- **Low** -> each baseline breaks a different subset: failure is driven by the
intervention.
Both are reported because they are not redundant — with 4-way answers the success
sets are small and the failure sets large, so the two move independently.
""")
code(r"""
def overlap_matrix(model, task, column):
m = pd.DataFrame(np.eye(len(BASELINES)), index=BASELINES, columns=BASELINES)
sub = pairwise[(pairwise.model == model) & (pairwise.task == task)]
for _, r in sub.iterrows():
m.loc[r.baseline_a, r.baseline_b] = m.loc[r.baseline_b, r.baseline_a] = r[column]
return m
def overlap_grid(column, title, vmin, vmax):
# Model names head the columns and split names label the rows, so no panel
# carries a title that could collide with the tick labels of the panel above.
fig, axes = plt.subplots(len(TASKS), len(MODELS), figsize=(14, 13.6))
for i, task in enumerate(TASKS):
for j, model in enumerate(MODELS):
ax = axes[i, j]
m = overlap_matrix(model, task, column)
ax.imshow(m.values, cmap=CMAP, vmin=vmin, vmax=vmax)
for a in range(len(BASELINES)):
for b in range(len(BASELINES)):
v = m.values[a, b]
# Label every cell: the fill alone is below the contrast floor.
ax.text(b, a, f"{v:.2f}", ha="center", va="center", fontsize=7,
color=SURFACE if v > (vmin + vmax) / 2 else INK)
ax.set_xticks(range(len(BASELINES))); ax.set_yticks(range(len(BASELINES)))
ax.set_xticklabels(BASELINES if i == len(TASKS) - 1 else [], rotation=90, fontsize=7)
ax.set_yticklabels(BASELINES if j == 0 else [], fontsize=7)
ax.tick_params(length=0)
if i == 0:
ax.set_title(model, fontsize=10, pad=10)
if j == 0:
ax.set_ylabel(task, fontsize=11, fontweight="600", labelpad=8)
for sp in ax.spines.values(): sp.set_visible(False)
fig.suptitle(title, y=0.965, fontsize=12, fontweight="600")
fig.colorbar(mpl.cm.ScalarMappable(mpl.colors.Normalize(vmin, vmax), CMAP),
ax=axes, orientation="horizontal", fraction=0.02, pad=0.05, label="Jaccard")
return fig
overlap_grid("failure_jaccard", "Failure-set overlap between baselines (Jaccard)", 0.4, 1.0)
plt.show()
""")
code(r"""
overlap_grid("success_jaccard", "Success-set overlap between baselines (Jaccard)", 0.4, 1.0)
plt.show()
""")
md(r"""
### Does overlap decay with the size of the intervention?
The matrices above collapse to one question: as two baselines move further apart,
do their failure sets separate? A flat line means the intervention is not what
drives the disagreement.
""")
code(r"""
pairwise["gap"] = (pairwise.baseline_b.astype(float) - pairwise.baseline_a.astype(float)).round(2)
fig, axes = plt.subplots(1, 2, figsize=(13, 4), sharey=True)
for ax, col, name in zip(axes, ["failure_jaccard", "success_jaccard"], ["failure", "success"]):
ends = {}
for task in TASKS:
g = pairwise[pairwise.task == task].groupby("gap", observed=True)[col].mean()
# One line per split, averaged over models; splits are the entity here.
ax.plot(g.index, g.values, marker="o", label=task,
color=SLOTS[TASKS.index(task)], markeredgecolor=SURFACE, markeredgewidth=1.2)
ends[task] = (g.index[-1], g.values[-1])
# Direct labels can land on top of each other where two splits converge;
# nudge them apart along y so both stay readable.
span = max(v for _, v in ends.values()) - min(v for _, v in ends.values())
placed = []
for task, (gx, gy) in sorted(ends.items(), key=lambda kv: kv[1][1]):
while any(abs(gy - y) < span * 0.05 for y in placed):
gy += span * 0.05
placed.append(gy)
ax.annotate(task, (gx, gy), xytext=(6, 0), textcoords="offset points",
fontsize=8.5, color=INK_2, va="center")
ax.set_xlabel("baseline separation |b_a - b_b|"); ax.set_title(f"{name} overlap")
ax.grid(axis="y", zorder=0); ax.set_xlim(0.03, 0.285)
axes[0].set_ylabel("mean Jaccard (over models)")
axes[0].legend(ncol=4, loc="lower center", bbox_to_anchor=(0.5, -0.42), fontsize=8.5)
fig.suptitle("Overlap vs intervention size", y=1.02, fontsize=12, fontweight="600")
plt.show()
display(pairwise.pivot_table(index="task", columns="gap", values="failure_jaccard",
observed=True).round(3))
""")
md(r"""
## 4. Stability — which samples the intervention actually reaches
Each sample is classified over the six baselines:
| class | meaning |
|---|---|
| `always_correct` | right at every baseline — the intervention never broke it |
| `always_wrong` | wrong at every baseline — beyond the model, intervention irrelevant |
| `unstable` | right at some, wrong at others — **the population under test** |
`unstable` is the number that matters. Pair it with the flip-curve slope from §2:
a high unstable rate with a flat flip curve is chance-level churn, not an effect.
""")
code(r"""
order = stability.sort_values(["task", "model"], ascending=[True, True]).reset_index(drop=True)
labels = [f"{r.task} {r.model}" for _, r in order.iterrows()]
segs = [("always_correct_rate", SLOTS[2], "always correct"),
("always_wrong_rate", SLOTS[1], "always wrong"),
("unstable_rate", SLOTS[0], "unstable")]
fig, ax = plt.subplots(figsize=(11, 7))
left = np.zeros(len(order))
for col, color, name in segs:
v = order[col].values
# 2px surface gap between adjacent segments so the boundary reads.
ax.barh(labels, v, left=left, color=color, label=name, height=0.72,
edgecolor=SURFACE, linewidth=2, zorder=3)
for i, (l, w) in enumerate(zip(left, v)):
if w > 0.06:
ax.text(l + w / 2, i, f"{w:.2f}", ha="center", va="center",
fontsize=7.5, color=SURFACE if color != SLOTS[3] else INK)
left += v
ax.set_xlim(0, 1); ax.set_xlabel("fraction of samples"); ax.invert_yaxis()
ax.grid(axis="x", zorder=0)
ax.set_title("Sample stability across the six baselines")
ax.legend(ncol=3, loc="upper center", bbox_to_anchor=(0.5, -0.07))
plt.show()
display(order.set_index(["task", "model"])[
["n", "always_correct_rate", "always_wrong_rate", "unstable_rate"]].round(3))
""")
md(r"""
## 5. How far the unstable samples move
`unstable` lumps together a sample that flipped once and one that flipped at
every baseline. The per-sample files carry `num_correct` (0-6), so the shape of
that distribution says whether instability is a thin edge or a broad band. A
chance-level model piles up in the middle; a model with real signal is U-shaped.
""")
code(r"""
def per_sample(model, task):
inv_m = {v: k for k, v in SHORT.items()}
inv_t = {v: k for k, v in TASK_LABEL.items()}
p = ANALYSIS_DIR / f"{inv_m[model].replace('/', '_')}__{inv_t[task]}_per_sample.csv"
return pd.read_csv(p) if p.is_file() else None
fig, axes = plt.subplots(1, len(TASKS), figsize=(15, 3.5), sharey=True)
x = np.arange(7)
for ax, task in zip(axes, TASKS):
for i, model in enumerate(MODELS):
df = per_sample(model, task)
if df is None:
continue
frac = df.num_correct.value_counts(normalize=True).reindex(x, fill_value=0)
ax.plot(x, frac.values, marker="o", color=COLOR[model], label=model,
markeredgecolor=SURFACE, markeredgewidth=1.2, zorder=3)
ax.set_title(task); ax.set_xticks(x); ax.set_xlabel("# baselines correct (of 6)")
ax.grid(axis="y", zorder=0)
axes[0].set_ylabel("fraction of samples")
h, l = axes[0].get_legend_handles_labels()
fig.legend(h, l, loc="upper center", bbox_to_anchor=(0.5, 1.10), ncol=4)
fig.suptitle("U-shaped = decided answers; humped in the middle = chance churn",
y=1.20, fontsize=12, fontweight="600")
fig.tight_layout()
plt.show()
""")
md(r"""
## 6. Summary
One row per run. `kappa` is the bias-corrected skill from §1, `flip@0.25` and
`flip slope` the intervention effect from §2, `unstable` the population it reaches
from §4, and `failure J @0.25` the failure overlap between the extremes.
""")
code(r"""
summary_rows = []
for model in MODELS:
for task in TASKS:
pb = per_baseline[(per_baseline.model == model) & (per_baseline.task == task)]
if pb.empty:
continue
pw = pairwise[(pairwise.model == model) & (pairwise.task == task)]
st = stability[(stability.model == model) & (stability.task == task)].iloc[0]
kp = kappa_df[(kappa_df.model == model) & (kappa_df.task == task)].kappa.mean()
f = pb.set_index("baseline").flip_rate_vs_reference
ext = pw[(pw.baseline_a == "0.00") & (pw.baseline_b == "0.25")]
summary_rows.append({
"model": model, "task": task, "n": int(st.n),
"acc": pb.accuracy.mean(), "chance": CHANCE[task], "kappa": kp,
"acc spread": pb.accuracy.max() - pb.accuracy.min(),
"flip@0.25": f["0.25"], "flip slope": f["0.25"] - f["0.05"],
"unstable": st.unstable_rate,
"failure J @0.25": ext.failure_jaccard.iloc[0] if len(ext) else np.nan,
"success J @0.25": ext.success_jaccard.iloc[0] if len(ext) else np.nan,
})
out = pd.DataFrame(summary_rows).set_index(["task", "model"]).round(3)
out.to_csv(ANALYSIS_DIR / "notebook_summary.csv")
print(f"written: {ANALYSIS_DIR / 'notebook_summary.csv'}")
out
""")
md(r"""
### Reading the summary
- **`flip slope` > 0** with a moderate `flip@0.25` is the signature of a real
intervention effect.
- **`flip@0.25` high with `flip slope` ~ 0** is chance churn — check `kappa` before
reading anything into it.
- **`failure J @0.25` well below 1** means the extremes fail on different samples,
i.e. the intervention moved the failure set rather than just its size.
See `readme.md` for the caveats that constrain what these numbers can support —
in particular the non-uniform evaluation modes, the level-2 v2 data boundary, and
level 3 sitting at chance for both Qwen2.5-VL models.
""")
nb = {
"cells": [
{"cell_type": t, "metadata": {}, "source": s.splitlines(keepends=True),
**({"outputs": [], "execution_count": None} if t == "code" else {})}
for t, s in C
],
"metadata": {
"kernelspec": {"display_name": "Spatial_VLM", "language": "python", "name": "python3"},
"language_info": {"name": "python", "version": "3.10"},
},
"nbformat": 4, "nbformat_minor": 5,
}
out = pathlib.Path(__file__).resolve().parent / "baseline_intervention.ipynb"
out.write_text(json.dumps(nb, indent=1))
print("wrote", out, len(C), "cells")