#!/bin/bash # Compare editing-method effects on the toilet readout, per layer (NO steering). # # For each model variant the toilet attention-probe score sigma is read at every # layer on bathroom-only images; each cell = mean_images( sigma_method(l) - sigma_base(l) ). # Every method reads the SAME fixed context per image (FORCED_TEXT) so only the model # weights differ -> the base row is exactly 0 (reference). Bathroom-only images are # split into NON-hallucinating vs HALLUCINATING by the base model's mention flag; # one figure, two panels, 4 methods on the same fig. # # Tunables (override via env): # DEVICE_ID GPU index (use 2-5 only on this box) (default 4) # DTYPE bfloat16 | float16 | float32 (default bfloat16) # METHODS space-separated subset of: base lora efuf nullu (default all) # FORCED_TEXT constant ASSISTANT answer fed to every model (default # "This image features a bathroom with a"); set FORCED_TEXT="" to # instead use each image's base-model caption. # CATEGORY bathroom_only | bathroom_toilet | toilet_only | others (default bathroom_only) # ('others' = unrelated negatives from neg_cc3m_5k.json, validation split) # NUM_IMAGES cap PER population (0 = all) (default 100) # POPULATION all | both | said | unsaid | halluc | non (default both) # ('all' = one combined set per category, no hallucination split) # HOOK_TYPE pre | mid | post (default post) # TOILET_PROBE base-trained toilet SequenceLayerProbes ckpt # LORA_PATH / EFUF_PATH / NULLU_PATH / NULLU_LOWEST / NULLU_HIGHEST edit ckpts # OUT_TAG optional filename suffix (e.g. OUT_TAG=forced -> *_forced.{png,json}) # # Usage: # bash mechanistic_interp/scripts/compare_baselines.sh # METHODS="base lora" DEVICE_ID=2 bash mechanistic_interp/scripts/compare_baselines.sh # FORCED_TEXT="" bash mechanistic_interp/scripts/compare_baselines.sh # use base captions set -euo pipefail REPO_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)" MODEL_NAME="llava-hf/llava-1.5-7b-hf" DEVICE_ID="${DEVICE_ID:-4}" DTYPE="${DTYPE:-bfloat16}" METHODS="${METHODS:-base lora efuf nullu}" FORCED_TEXT="${FORCED_TEXT-This image features a bathroom with a}" CATEGORY="${CATEGORY:-bathroom_only}" NUM_IMAGES="${NUM_IMAGES:-100}" POPULATION="${POPULATION:-both}" HOOK_TYPE="${HOOK_TYPE:-post}" TOILET_PROBE="${TOILET_PROBE:-/data/caotue/latent_probes/seqprobes_toilet/post/seqprobe.pth}" LORA_PATH="${LORA_PATH:-/data/caotue/multilayer-sae/adv_gen_outputs/run_bathroom_toilet_v2/lora_adapter}" EFUF_PATH="${EFUF_PATH:-/data/caotue/multilayer-sae/EFUF/efuf/checkpoints/llava_vicuna_7b/bathroom_toilet_paper_10ep/epoch_002.pth}" NULLU_PATH="${NULLU_PATH:-/data/caotue/nullu/edited_models/LLaVA-7B-top4-0-32-bathroom_toilet}" NULLU_LOWEST="${NULLU_LOWEST:-8}" NULLU_HIGHEST="${NULLU_HIGHEST:-32}" SAMPLES="${SAMPLES:-${REPO_ROOT}/mechanistic_interp/toilet_bathroom/samples.json}" IMAGE_FOLDER="${IMAGE_FOLDER:-/data/caotue/CC3M-Dataset/cc3m_images}" BASE_PROMPT="${BASE_PROMPT:-Describe this image.}" OUT_TAG="${OUT_TAG:-}" TAG_SUFFIX="${OUT_TAG:+_${OUT_TAG}}" OUT="${REPO_ROOT}/mechanistic_interp/graph/compare_baselines_toilet${TAG_SUFFIX}.png" OUT_JSON="${REPO_ROOT}/mechanistic_interp/graph/compare_baselines_toilet${TAG_SUFFIX}.json" mkdir -p "$(dirname "${OUT}")" echo "========================================================" echo " compare baselines: toilet readout, method - base, per layer" echo "========================================================" echo " Methods : ${METHODS}" echo " Context : ${FORCED_TEXT:-}" echo " Toilet probe : ${TOILET_PROBE}" echo " N per pop : ${NUM_IMAGES} hook=${HOOK_TYPE}" echo " Device / dtype : cuda:${DEVICE_ID} / ${DTYPE}" echo " Out graph/json : ${OUT}" echo "========================================================" cd "${REPO_ROOT}" export PYTHONPATH="$(cd .. && pwd):$(pwd):${PYTHONPATH:-}" export PYTHONUNBUFFERED=1 ARGS=( --model_name "${MODEL_NAME}" --device_id "${DEVICE_ID}" --dtype "${DTYPE}" --methods ${METHODS} --toilet_probe "${TOILET_PROBE}" --lora_path "${LORA_PATH}" --efuf_path "${EFUF_PATH}" --nullu_path "${NULLU_PATH}" --nullu_lowest "${NULLU_LOWEST}" --nullu_highest "${NULLU_HIGHEST}" --samples_json "${SAMPLES}" --image_folder "${IMAGE_FOLDER}" --base_prompt "${BASE_PROMPT}" --category "${CATEGORY}" --population "${POPULATION}" --num_images "${NUM_IMAGES}" --hook_type "${HOOK_TYPE}" --out "${OUT}" --out_json "${OUT_JSON}" ) # Always pass --forced_text; empty string => use the base model's per-image caption. ARGS+=(--forced_text "${FORCED_TEXT}") python -m mechanistic_interp.compare_baselines "${ARGS[@]}" "$@"