#!/usr/bin/env bash # L15 "make it work" sweep: mirror the L10 winning recipe, then vary data and capacity. # GPUs 2,3 are left alone (live L10 ce095 / ce099 runs). set -uo pipefail cd /egr/research-slim/ghoshavr/reasoning-by-superposition-main export PATH=/egr/research-slim/ghoshavr/conda-envs/superposition/bin:$PATH export WANDB_MODE=offline TORCH=/egr/research-slim/ghoshavr/conda-envs/superposition/bin/torchrun mkdir -p logs launch () { # gpu port config CUDA_VISIBLE_DEVICES=$1 setsid nohup "$TORCH" --standalone --nnodes 1 --nproc_per_node 1 \ --master_port "$2" run.py "args/$3.yaml" > "logs/$3.log" 2>&1 < /dev/null & echo "launched $3 on GPU $1" } launch 1 29701 L15_push_2L_ce95_50k launch 4 29702 L15_push_2L_ce95_100k launch 5 29703 L15_push_4L_ce95_50k launch 6 29704 L15_push_4L_ce90_50k launch 7 29705 L15_push_s0_4L sleep 90 echo echo '=== procs ===' pgrep -af '[r]un.py' | grep -o 'args/L15_push[^ ]*' | sort | uniq -c echo echo '=== GPUs ===' nvidia-smi --query-gpu=index,memory.used,utilization.gpu --format=csv,noheader echo for n in L15_push_2L_ce95_50k L15_push_2L_ce95_100k L15_push_4L_ce95_50k L15_push_4L_ce90_50k L15_push_s0_4L; do echo "--- $n ---" grep -E 'Traceback|Error|error|missing_keys|unexpected_keys|_IncompatibleKeys|train epoch' "logs/$n.log" 2>/dev/null | head -3 done