Download analysis/intervention_analysis/build_notebook.py from LLDDSS/Stereo_Depth: direct link, hf CLI and curl.
- Browser
- Download file 22.3 kB
-
https://huggingface.co/LLDDSS/Stereo_Depth/resolve/main/analysis/intervention_analysis/build_notebook.py
- Command line
-
hf download hf://LLDDSS/Stereo_Depth/analysis/intervention_analysis/build_notebook.py
-
curl -L -o build_notebook.py https://huggingface.co/LLDDSS/Stereo_Depth/resolve/main/analysis/intervention_analysis/build_notebook.py
22.3 kB
| """Author the analysis notebook as a list of cells, then emit .ipynb JSON.""" | |
| import json, pathlib | |
| C = [] | |
| def md(s): C.append(("markdown", s.strip("\n"))) | |
| def code(s): C.append(("code", s.strip("\n"))) | |
| md(r""" | |
| # Baseline-intervention analysis | |
| Every model answers the **same question about the same scene** at six stereo | |
| baselines (0.00 = untouched original -> 0.25). Decoding is greedy, so any change | |
| in the answer across baselines is the intervention's effect. | |
| The premise under test: *if the visual tower modelled the world perfectly, a small | |
| intervention should not change the output.* | |
| Run top to bottom. Regenerate the inputs first if any inference has been re-run: | |
| ```bash | |
| python analysis/analyze_baseline_overlap.py --result_dir result --dump_per_sample | |
| ``` | |
| See `readme.md` in this directory for what each measure means and for the caveats | |
| attached to the current result set. | |
| """) | |
| md("## Setup") | |
| code(r""" | |
| import json, pathlib, itertools, math, warnings | |
| import numpy as np | |
| import pandas as pd | |
| import matplotlib as mpl | |
| import matplotlib.pyplot as plt | |
| from matplotlib.colors import LinearSegmentedColormap | |
| warnings.filterwarnings("ignore") | |
| pd.set_option("display.width", 200) | |
| pd.set_option("display.max_columns", 40) | |
| # The experiment root is the directory holding both result/ and dataset/. Finding | |
| # it by walking up means the notebook runs from any cwd -- its own directory, the | |
| # experiment root, or anywhere between -- and survives the analysis outputs being | |
| # moved again. | |
| def find_root(start): | |
| for d in [start, *start.parents]: | |
| if (d / "result").is_dir() and (d / "dataset").is_dir(): | |
| return d | |
| raise RuntimeError(f"no experiment root (a dir with result/ and dataset/) above {start}") | |
| ROOT = find_root(pathlib.Path.cwd().resolve()) | |
| RESULT_DIR = ROOT / "result" | |
| ANALYSIS_DIR = ROOT / "analysis" / "intervention_analysis" | |
| assert (ANALYSIS_DIR / "analysis_summary.json").is_file(), ( | |
| f"analysis_summary.json not found in {ANALYSIS_DIR}. Run first:\n" | |
| " python analysis/analyze_baseline_overlap.py --dump_per_sample" | |
| ) | |
| print("root:", ROOT, "\nanalysis dir:", ANALYSIS_DIR) | |
| """) | |
| md(r""" | |
| ### Palette | |
| Categorical slots carry **model identity** and are assigned in fixed order — never | |
| cycled, never reassigned when a filter changes the series count. Magnitude | |
| (overlap heatmaps) uses a single-hue blue ramp, light -> dark. | |
| Two categorical slots sit below 3:1 against the surface, so every chart below | |
| ships direct labels or a table view beside it rather than relying on the fill | |
| alone. | |
| """) | |
| code(r""" | |
| SURFACE = "#fcfcfb" | |
| INK = "#0b0b0b" | |
| INK_2 = "#52514e" | |
| GRID = "#e4e3df" | |
| # Categorical slots 1-4, fixed order. Validated adjacent-pair on the light surface. | |
| SLOTS = ["#2a78d6", "#eb6834", "#1baf7a", "#eda100"] | |
| # Sequential blue 100 -> 700, for magnitude (Jaccard heatmaps). | |
| SEQ = ["#cde2fb", "#b7d3f6", "#9ec5f4", "#86b6ef", "#6da7ec", "#5598e7", "#3987e5", | |
| "#2a78d6", "#256abf", "#1c5cab", "#184f95", "#104281", "#0d366b"] | |
| CMAP = LinearSegmentedColormap.from_list("seq_blue", SEQ) | |
| mpl.rcParams.update({ | |
| "figure.facecolor": SURFACE, "axes.facecolor": SURFACE, "savefig.facecolor": SURFACE, | |
| "axes.edgecolor": GRID, "axes.labelcolor": INK_2, "axes.titlecolor": INK, | |
| "xtick.color": INK_2, "ytick.color": INK_2, "text.color": INK, | |
| "grid.color": GRID, "grid.linewidth": 0.8, | |
| "axes.spines.top": False, "axes.spines.right": False, | |
| "axes.titlesize": 11, "axes.titleweight": "600", "axes.labelsize": 9, | |
| "xtick.labelsize": 9, "ytick.labelsize": 9, "legend.fontsize": 9, | |
| "legend.frameon": False, "lines.linewidth": 2, "lines.markersize": 5, | |
| "font.size": 10, "figure.dpi": 110, | |
| }) | |
| """) | |
| md("## Load") | |
| code(r""" | |
| summary = json.loads((ANALYSIS_DIR / "analysis_summary.json").read_text()) | |
| runs = summary["runs"] | |
| SHORT = { | |
| "Qwen/Qwen2.5-VL-3B-Instruct": "Qwen2.5-VL-3B", | |
| "Qwen/Qwen2.5-VL-7B-Instruct": "Qwen2.5-VL-7B", | |
| "Qwen/Qwen3.5-4B": "Qwen3.5-4B", | |
| "Qwen/Qwen3.5-4B-Base": "Qwen3.5-4B-Base", | |
| } | |
| TASK_LABEL = {"vsr": "VSR", "youtube_level1": "L1", "youtube_level2": "L2", "youtube_level3": "L3"} | |
| # Fixed display order: colour follows the entity, so these indices never move. | |
| MODELS = ["Qwen2.5-VL-3B", "Qwen2.5-VL-7B", "Qwen3.5-4B", "Qwen3.5-4B-Base"] | |
| TASKS = ["VSR", "L1", "L2", "L3"] | |
| COLOR = dict(zip(MODELS, SLOTS)) | |
| # Chance level per split: VSR is True/False, the Youtube levels are 4-way. | |
| CHANCE = {"VSR": 0.5, "L1": 0.25, "L2": 0.25, "L3": 0.25} | |
| BASELINES = ["0.00", "0.05", "0.10", "0.15", "0.20", "0.25"] | |
| per_baseline = pd.DataFrame([ | |
| {"model": SHORT.get(r["model_id"], r["model_id"]), "task": TASK_LABEL.get(r["task"], r["task"]), **row} | |
| for r in runs for row in r["per_baseline"] | |
| ]) | |
| pairwise = pd.DataFrame([ | |
| {"model": SHORT.get(r["model_id"], r["model_id"]), "task": TASK_LABEL.get(r["task"], r["task"]), **row} | |
| for r in runs for row in r["pairwise"] | |
| ]) | |
| stability = pd.DataFrame([ | |
| {"model": SHORT.get(r["model_id"], r["model_id"]), "task": TASK_LABEL.get(r["task"], r["task"]), | |
| "n": r["num_common_samples"], **r["stability"]} | |
| for r in runs | |
| ]) | |
| for df in (per_baseline, pairwise, stability): | |
| df["model"] = pd.Categorical(df["model"], MODELS, ordered=True) | |
| df["task"] = pd.Categorical(df["task"], TASKS, ordered=True) | |
| print(f"{len(runs)} runs | {per_baseline.task.nunique()} splits x {per_baseline.model.nunique()} models") | |
| per_baseline.pivot_table(index=["task", "model"], columns="baseline", values="accuracy", observed=True).round(3) | |
| """) | |
| md(r""" | |
| ## 1. Accuracy across the sweep | |
| Accuracy is the headline but the *least* informative measure here: a model can | |
| hold it flat while answering differently on half the samples, because wins and | |
| losses cancel. It is here to establish the level relative to chance (dashed), so | |
| the flip and overlap numbers that follow can be read in context. | |
| """) | |
| code(r""" | |
| def facet_by_task(value, ylabel, title, ylim=None, chance=False, ax_hook=None): | |
| # One panel per split, one line per model, direct-labelled at the right end. | |
| fig, axes = plt.subplots(1, len(TASKS), figsize=(15, 3.5), sharey=(ylim is not None)) | |
| x = np.arange(len(BASELINES)) | |
| for ax, task in zip(axes, TASKS): | |
| sub = per_baseline[per_baseline.task == task] | |
| for model in MODELS: | |
| s = sub[sub.model == model].set_index("baseline").reindex(BASELINES)[value] | |
| ax.plot(x, s.values, color=COLOR[model], marker="o", label=model, | |
| markeredgecolor=SURFACE, markeredgewidth=1.2, zorder=3) | |
| if chance: | |
| ax.axhline(CHANCE[task], color=INK_2, lw=1, ls=(0, (4, 3)), zorder=1) | |
| # Left-anchored: the right end is where the series land on L3. | |
| ax.annotate(f"chance {CHANCE[task]:.2f}", (-0.3, CHANCE[task]), xytext=(0, 3), | |
| textcoords="offset points", fontsize=7.5, color=INK_2, | |
| va="bottom", ha="left") | |
| ax.set_title(task); ax.set_xticks(x); ax.set_xticklabels(BASELINES, rotation=45) | |
| ax.set_xlabel("baseline"); ax.grid(axis="y", zorder=0) | |
| ax.set_xlim(-0.35, len(BASELINES) - 0.65) | |
| if ylim: ax.set_ylim(*ylim) | |
| if ax_hook: ax_hook(ax, task) | |
| axes[0].set_ylabel(ylabel) | |
| handles, labels = axes[0].get_legend_handles_labels() | |
| fig.tight_layout() | |
| fig.legend(handles, labels, loc="upper center", bbox_to_anchor=(0.5, 1.045), ncol=4) | |
| fig.suptitle(title, y=1.135, fontsize=12, fontweight="600", x=0.5) | |
| return fig | |
| facet_by_task("accuracy", "accuracy", "Accuracy vs stereo baseline", chance=True) | |
| plt.show() | |
| """) | |
| md(r""" | |
| ### Bias-corrected view | |
| All four models carry a heavy option-position bias, in different directions, so | |
| raw accuracy is not comparable across models within a split. Cohen's κ nets out | |
| each model's own answer marginal — a model that always answers "B" scores κ = 0 | |
| however often "B" happens to be right. | |
| κ is recomputed here from the raw predictions rather than read from the summary, | |
| because it needs the answer *distribution*, which the summary does not carry. | |
| """) | |
| code(r""" | |
| def load_predictions(model_short, task_short): | |
| inv_m = {v: k for k, v in SHORT.items()} | |
| inv_t = {v: k for k, v in TASK_LABEL.items()} | |
| tag = inv_m[model_short].replace("/", "_") | |
| frames = [] | |
| for b in BASELINES: | |
| p = RESULT_DIR / tag / inv_t[task_short] / f"baseline_{b}" / "predictions.jsonl" | |
| if not p.is_file(): | |
| continue | |
| df = pd.read_json(p, lines=True) | |
| df["baseline"] = b | |
| frames.append(df) | |
| return pd.concat(frames, ignore_index=True) if frames else None | |
| def cohen_kappa(df): | |
| # (acc - pe) / (1 - pe), with pe from the gold x predicted marginals. | |
| n = len(df) | |
| gold = df.reference_norm.value_counts(normalize=True) | |
| pred = df.prediction_norm.value_counts(normalize=True) | |
| pe = sum(gold[k] * pred.get(k, 0.0) for k in gold.index) | |
| acc = df.correct.mean() | |
| return (acc - pe) / (1 - pe) if pe < 1 else float("nan") | |
| rows = [] | |
| for model in MODELS: | |
| for task in TASKS: | |
| df = load_predictions(model, task) | |
| if df is None: | |
| continue | |
| for b, g in df.groupby("baseline"): | |
| rows.append({"model": model, "task": task, "baseline": b, | |
| "accuracy": g.correct.mean(), "kappa": cohen_kappa(g)}) | |
| kappa_df = pd.DataFrame(rows) | |
| kappa_df["model"] = pd.Categorical(kappa_df["model"], MODELS, ordered=True) | |
| kappa_df["task"] = pd.Categorical(kappa_df["task"], TASKS, ordered=True) | |
| kappa_mean = kappa_df.pivot_table(index="model", columns="task", values="kappa", observed=True) | |
| display(kappa_mean.round(3)) | |
| fig, ax = plt.subplots(figsize=(8, 3.6)) | |
| w, x = 0.2, np.arange(len(TASKS)) | |
| for i, model in enumerate(MODELS): | |
| v = kappa_mean.loc[model, TASKS].values | |
| bars = ax.bar(x + (i - 1.5) * w, v, w * 0.9, color=COLOR[model], label=model, | |
| edgecolor=SURFACE, linewidth=2, zorder=3) | |
| ax.bar_label(bars, fmt="%.2f", fontsize=7, padding=2, color=INK_2) | |
| ax.set_xticks(x); ax.set_xticklabels(TASKS); ax.set_ylabel("Cohen's kappa") | |
| ax.set_title("Bias-corrected skill by split (0 = no better than the model's own answer prior)") | |
| ax.axhline(0, color=INK_2, lw=1); ax.grid(axis="y", zorder=0) | |
| ax.legend(ncol=4, loc="upper center", bbox_to_anchor=(0.5, -0.12)) | |
| plt.show() | |
| """) | |
| md(r""" | |
| ## 2. Flip rate — the direct test of the premise | |
| The fraction of samples whose **answer changes** relative to baseline 0.00, | |
| independent of whether it was right. Under the premise, this line should be flat | |
| at zero. | |
| Read the **shape**, not the level: | |
| - **rising with baseline** -> a real intervention effect; bigger interventions move | |
| more samples. | |
| - **high but flat from the first step** -> the model is near chance and the "flips" | |
| are its own indifference. Level 3 does this for every model. | |
| """) | |
| code(r""" | |
| facet_by_task("flip_rate_vs_reference", "flip rate vs b=0.00", | |
| "Answer flips relative to the untouched view", ylim=(0, 0.42)) | |
| plt.show() | |
| flip = per_baseline[per_baseline.baseline != "0.00"].pivot_table( | |
| index=["task", "model"], columns="baseline", values="flip_rate_vs_reference", observed=True) | |
| flip["slope (0.25 - 0.05)"] = flip["0.25"] - flip["0.05"] | |
| display(flip.round(3)) | |
| print("A positive slope is intervention signal; ~0 with a high level is chance churn.") | |
| """) | |
| md(r""" | |
| ## 3. Failure and success overlap | |
| For each pair of baselines, Jaccard over the samples each one gets **wrong**, and | |
| over the samples each one gets **right**. | |
| - **High** -> both baselines miss the same items: failure is a property of the item. | |
| - **Low** -> each baseline breaks a different subset: failure is driven by the | |
| intervention. | |
| Both are reported because they are not redundant — with 4-way answers the success | |
| sets are small and the failure sets large, so the two move independently. | |
| """) | |
| code(r""" | |
| def overlap_matrix(model, task, column): | |
| m = pd.DataFrame(np.eye(len(BASELINES)), index=BASELINES, columns=BASELINES) | |
| sub = pairwise[(pairwise.model == model) & (pairwise.task == task)] | |
| for _, r in sub.iterrows(): | |
| m.loc[r.baseline_a, r.baseline_b] = m.loc[r.baseline_b, r.baseline_a] = r[column] | |
| return m | |
| def overlap_grid(column, title, vmin, vmax): | |
| # Model names head the columns and split names label the rows, so no panel | |
| # carries a title that could collide with the tick labels of the panel above. | |
| fig, axes = plt.subplots(len(TASKS), len(MODELS), figsize=(14, 13.6)) | |
| for i, task in enumerate(TASKS): | |
| for j, model in enumerate(MODELS): | |
| ax = axes[i, j] | |
| m = overlap_matrix(model, task, column) | |
| ax.imshow(m.values, cmap=CMAP, vmin=vmin, vmax=vmax) | |
| for a in range(len(BASELINES)): | |
| for b in range(len(BASELINES)): | |
| v = m.values[a, b] | |
| # Label every cell: the fill alone is below the contrast floor. | |
| ax.text(b, a, f"{v:.2f}", ha="center", va="center", fontsize=7, | |
| color=SURFACE if v > (vmin + vmax) / 2 else INK) | |
| ax.set_xticks(range(len(BASELINES))); ax.set_yticks(range(len(BASELINES))) | |
| ax.set_xticklabels(BASELINES if i == len(TASKS) - 1 else [], rotation=90, fontsize=7) | |
| ax.set_yticklabels(BASELINES if j == 0 else [], fontsize=7) | |
| ax.tick_params(length=0) | |
| if i == 0: | |
| ax.set_title(model, fontsize=10, pad=10) | |
| if j == 0: | |
| ax.set_ylabel(task, fontsize=11, fontweight="600", labelpad=8) | |
| for sp in ax.spines.values(): sp.set_visible(False) | |
| fig.suptitle(title, y=0.965, fontsize=12, fontweight="600") | |
| fig.colorbar(mpl.cm.ScalarMappable(mpl.colors.Normalize(vmin, vmax), CMAP), | |
| ax=axes, orientation="horizontal", fraction=0.02, pad=0.05, label="Jaccard") | |
| return fig | |
| overlap_grid("failure_jaccard", "Failure-set overlap between baselines (Jaccard)", 0.4, 1.0) | |
| plt.show() | |
| """) | |
| code(r""" | |
| overlap_grid("success_jaccard", "Success-set overlap between baselines (Jaccard)", 0.4, 1.0) | |
| plt.show() | |
| """) | |
| md(r""" | |
| ### Does overlap decay with the size of the intervention? | |
| The matrices above collapse to one question: as two baselines move further apart, | |
| do their failure sets separate? A flat line means the intervention is not what | |
| drives the disagreement. | |
| """) | |
| code(r""" | |
| pairwise["gap"] = (pairwise.baseline_b.astype(float) - pairwise.baseline_a.astype(float)).round(2) | |
| fig, axes = plt.subplots(1, 2, figsize=(13, 4), sharey=True) | |
| for ax, col, name in zip(axes, ["failure_jaccard", "success_jaccard"], ["failure", "success"]): | |
| ends = {} | |
| for task in TASKS: | |
| g = pairwise[pairwise.task == task].groupby("gap", observed=True)[col].mean() | |
| # One line per split, averaged over models; splits are the entity here. | |
| ax.plot(g.index, g.values, marker="o", label=task, | |
| color=SLOTS[TASKS.index(task)], markeredgecolor=SURFACE, markeredgewidth=1.2) | |
| ends[task] = (g.index[-1], g.values[-1]) | |
| # Direct labels can land on top of each other where two splits converge; | |
| # nudge them apart along y so both stay readable. | |
| span = max(v for _, v in ends.values()) - min(v for _, v in ends.values()) | |
| placed = [] | |
| for task, (gx, gy) in sorted(ends.items(), key=lambda kv: kv[1][1]): | |
| while any(abs(gy - y) < span * 0.05 for y in placed): | |
| gy += span * 0.05 | |
| placed.append(gy) | |
| ax.annotate(task, (gx, gy), xytext=(6, 0), textcoords="offset points", | |
| fontsize=8.5, color=INK_2, va="center") | |
| ax.set_xlabel("baseline separation |b_a - b_b|"); ax.set_title(f"{name} overlap") | |
| ax.grid(axis="y", zorder=0); ax.set_xlim(0.03, 0.285) | |
| axes[0].set_ylabel("mean Jaccard (over models)") | |
| axes[0].legend(ncol=4, loc="lower center", bbox_to_anchor=(0.5, -0.42), fontsize=8.5) | |
| fig.suptitle("Overlap vs intervention size", y=1.02, fontsize=12, fontweight="600") | |
| plt.show() | |
| display(pairwise.pivot_table(index="task", columns="gap", values="failure_jaccard", | |
| observed=True).round(3)) | |
| """) | |
| md(r""" | |
| ## 4. Stability — which samples the intervention actually reaches | |
| Each sample is classified over the six baselines: | |
| | class | meaning | | |
| |---|---| | |
| | `always_correct` | right at every baseline — the intervention never broke it | | |
| | `always_wrong` | wrong at every baseline — beyond the model, intervention irrelevant | | |
| | `unstable` | right at some, wrong at others — **the population under test** | | |
| `unstable` is the number that matters. Pair it with the flip-curve slope from §2: | |
| a high unstable rate with a flat flip curve is chance-level churn, not an effect. | |
| """) | |
| code(r""" | |
| order = stability.sort_values(["task", "model"], ascending=[True, True]).reset_index(drop=True) | |
| labels = [f"{r.task} {r.model}" for _, r in order.iterrows()] | |
| segs = [("always_correct_rate", SLOTS[2], "always correct"), | |
| ("always_wrong_rate", SLOTS[1], "always wrong"), | |
| ("unstable_rate", SLOTS[0], "unstable")] | |
| fig, ax = plt.subplots(figsize=(11, 7)) | |
| left = np.zeros(len(order)) | |
| for col, color, name in segs: | |
| v = order[col].values | |
| # 2px surface gap between adjacent segments so the boundary reads. | |
| ax.barh(labels, v, left=left, color=color, label=name, height=0.72, | |
| edgecolor=SURFACE, linewidth=2, zorder=3) | |
| for i, (l, w) in enumerate(zip(left, v)): | |
| if w > 0.06: | |
| ax.text(l + w / 2, i, f"{w:.2f}", ha="center", va="center", | |
| fontsize=7.5, color=SURFACE if color != SLOTS[3] else INK) | |
| left += v | |
| ax.set_xlim(0, 1); ax.set_xlabel("fraction of samples"); ax.invert_yaxis() | |
| ax.grid(axis="x", zorder=0) | |
| ax.set_title("Sample stability across the six baselines") | |
| ax.legend(ncol=3, loc="upper center", bbox_to_anchor=(0.5, -0.07)) | |
| plt.show() | |
| display(order.set_index(["task", "model"])[ | |
| ["n", "always_correct_rate", "always_wrong_rate", "unstable_rate"]].round(3)) | |
| """) | |
| md(r""" | |
| ## 5. How far the unstable samples move | |
| `unstable` lumps together a sample that flipped once and one that flipped at | |
| every baseline. The per-sample files carry `num_correct` (0-6), so the shape of | |
| that distribution says whether instability is a thin edge or a broad band. A | |
| chance-level model piles up in the middle; a model with real signal is U-shaped. | |
| """) | |
| code(r""" | |
| def per_sample(model, task): | |
| inv_m = {v: k for k, v in SHORT.items()} | |
| inv_t = {v: k for k, v in TASK_LABEL.items()} | |
| p = ANALYSIS_DIR / f"{inv_m[model].replace('/', '_')}__{inv_t[task]}_per_sample.csv" | |
| return pd.read_csv(p) if p.is_file() else None | |
| fig, axes = plt.subplots(1, len(TASKS), figsize=(15, 3.5), sharey=True) | |
| x = np.arange(7) | |
| for ax, task in zip(axes, TASKS): | |
| for i, model in enumerate(MODELS): | |
| df = per_sample(model, task) | |
| if df is None: | |
| continue | |
| frac = df.num_correct.value_counts(normalize=True).reindex(x, fill_value=0) | |
| ax.plot(x, frac.values, marker="o", color=COLOR[model], label=model, | |
| markeredgecolor=SURFACE, markeredgewidth=1.2, zorder=3) | |
| ax.set_title(task); ax.set_xticks(x); ax.set_xlabel("# baselines correct (of 6)") | |
| ax.grid(axis="y", zorder=0) | |
| axes[0].set_ylabel("fraction of samples") | |
| h, l = axes[0].get_legend_handles_labels() | |
| fig.legend(h, l, loc="upper center", bbox_to_anchor=(0.5, 1.10), ncol=4) | |
| fig.suptitle("U-shaped = decided answers; humped in the middle = chance churn", | |
| y=1.20, fontsize=12, fontweight="600") | |
| fig.tight_layout() | |
| plt.show() | |
| """) | |
| md(r""" | |
| ## 6. Summary | |
| One row per run. `kappa` is the bias-corrected skill from §1, `flip@0.25` and | |
| `flip slope` the intervention effect from §2, `unstable` the population it reaches | |
| from §4, and `failure J @0.25` the failure overlap between the extremes. | |
| """) | |
| code(r""" | |
| summary_rows = [] | |
| for model in MODELS: | |
| for task in TASKS: | |
| pb = per_baseline[(per_baseline.model == model) & (per_baseline.task == task)] | |
| if pb.empty: | |
| continue | |
| pw = pairwise[(pairwise.model == model) & (pairwise.task == task)] | |
| st = stability[(stability.model == model) & (stability.task == task)].iloc[0] | |
| kp = kappa_df[(kappa_df.model == model) & (kappa_df.task == task)].kappa.mean() | |
| f = pb.set_index("baseline").flip_rate_vs_reference | |
| ext = pw[(pw.baseline_a == "0.00") & (pw.baseline_b == "0.25")] | |
| summary_rows.append({ | |
| "model": model, "task": task, "n": int(st.n), | |
| "acc": pb.accuracy.mean(), "chance": CHANCE[task], "kappa": kp, | |
| "acc spread": pb.accuracy.max() - pb.accuracy.min(), | |
| "flip@0.25": f["0.25"], "flip slope": f["0.25"] - f["0.05"], | |
| "unstable": st.unstable_rate, | |
| "failure J @0.25": ext.failure_jaccard.iloc[0] if len(ext) else np.nan, | |
| "success J @0.25": ext.success_jaccard.iloc[0] if len(ext) else np.nan, | |
| }) | |
| out = pd.DataFrame(summary_rows).set_index(["task", "model"]).round(3) | |
| out.to_csv(ANALYSIS_DIR / "notebook_summary.csv") | |
| print(f"written: {ANALYSIS_DIR / 'notebook_summary.csv'}") | |
| out | |
| """) | |
| md(r""" | |
| ### Reading the summary | |
| - **`flip slope` > 0** with a moderate `flip@0.25` is the signature of a real | |
| intervention effect. | |
| - **`flip@0.25` high with `flip slope` ~ 0** is chance churn — check `kappa` before | |
| reading anything into it. | |
| - **`failure J @0.25` well below 1** means the extremes fail on different samples, | |
| i.e. the intervention moved the failure set rather than just its size. | |
| See `readme.md` for the caveats that constrain what these numbers can support — | |
| in particular the non-uniform evaluation modes, the level-2 v2 data boundary, and | |
| level 3 sitting at chance for both Qwen2.5-VL models. | |
| """) | |
| nb = { | |
| "cells": [ | |
| {"cell_type": t, "metadata": {}, "source": s.splitlines(keepends=True), | |
| **({"outputs": [], "execution_count": None} if t == "code" else {})} | |
| for t, s in C | |
| ], | |
| "metadata": { | |
| "kernelspec": {"display_name": "Spatial_VLM", "language": "python", "name": "python3"}, | |
| "language_info": {"name": "python", "version": "3.10"}, | |
| }, | |
| "nbformat": 4, "nbformat_minor": 5, | |
| } | |
| out = pathlib.Path(__file__).resolve().parent / "baseline_intervention.ipynb" | |
| out.write_text(json.dumps(nb, indent=1)) | |
| print("wrote", out, len(C), "cells") | |