#!/usr/bin/env bash # Definitive final chain (uli decision): 25 configs + conditional q5-dflash. # Replaces F1/F2/F3 (legacy INTACT in scripts/old/, experiments/runs/old/). # Greedy T=0, non-thinking (--reasoning off), persistent llama-server via # bench_accept.py, idempotent resume per config, "FINAL RUN DONE" marker. # # Usage: # SMOKE=1 bash scripts/run_final_chain.sh # smoke: --max-prompts 2 per config # bash scripts/run_final_chain.sh # full chain (~67-75 h ≈ 3 nights) # DRY_RUN=1 bash scripts/run_final_chain.sh # prints the matrix (26) and exits # # Exit: 0 with no FAILED configs; 1 if any FAILED (check queue-final.log and errors.jsonl). set -uo pipefail REPO="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" cd "$REPO" source scripts/spec-env.sh set +e # spec-env.sh enables set -e; the chain handles errors via exit codes LOG="experiments/runs/queue-final.log" log() { echo "[$(date '+%F %T')] $*" | tee -a "$LOG"; } SAMPLES="experiments/prompts/f1-sample.jsonl" NTOK=256 Q="models/Qwen3-8B" G="models/gemma-4-12b-it" DK="models/drafts" BENCH="python scripts/bench_accept.py" # SMOKE=1 → only 2 prompts per config (smoke gate of the final-run-matrix spec). MAXP=() [ "${SMOKE:-0}" = "1" ] && MAXP=(--max-prompts 2) OK_COUNT=0 FAILED_COUNT=0 SKIP_COUNT=0 OOM_OK=1 G8_CHECKED=0 Q5_DFLASH_OK=0 # Matrix of 25 configs (spec final-run-matrix): name|target|draft|spec_type|p_min|out # (draft/spec_type/p_min empty in solo). out = experiments/runs/final-. MATRIX=( # Qwen3-8B Q4_K_M (8) "final-qwen-q4-solo|$Q-Q4_K_M.gguf||||experiments/runs/final-qwen-q4-solo" "final-qwen-q4-vanilla17b|$Q-Q4_K_M.gguf|$DK/Qwen3-1.7B-Q4_K_M.gguf|draft-simple||experiments/runs/final-qwen-q4-vanilla17b" "final-qwen-q4-eagle3|$Q-Q4_K_M.gguf|$DK/Qwen3-8B-speculator.eagle3-F16.gguf|draft-eagle3||experiments/runs/final-qwen-q4-eagle3" "final-qwen-q4-dflash|$Q-Q4_K_M.gguf|$DK/dflash_qwen3_8b_block7.gguf|draft-dflash||experiments/runs/final-qwen-q4-dflash" "final-qwen-q4-dspark-p0|$Q-Q4_K_M.gguf|$DK/dspark_qwen3_8b_block7.gguf|draft-dspark|0.0|experiments/runs/final-qwen-q4-dspark-p0" "final-qwen-q4-dspark-p2|$Q-Q4_K_M.gguf|$DK/dspark_qwen3_8b_block7.gguf|draft-dspark|0.2|experiments/runs/final-qwen-q4-dspark-p2" "final-qwen-q4-dspark-p4|$Q-Q4_K_M.gguf|$DK/dspark_qwen3_8b_block7.gguf|draft-dspark|0.4|experiments/runs/final-qwen-q4-dspark-p4" "final-qwen-q4-dspark-p6|$Q-Q4_K_M.gguf|$DK/dspark_qwen3_8b_block7.gguf|draft-dspark|0.6|experiments/runs/final-qwen-q4-dspark-p6" # Qwen3-8B Q5_K_M (3) "final-qwen-q5-solo|$Q-Q5_K_M.gguf||||experiments/runs/final-qwen-q5-solo" "final-qwen-q5-eagle3|$Q-Q5_K_M.gguf|$DK/Qwen3-8B-speculator.eagle3-F16.gguf|draft-eagle3||experiments/runs/final-qwen-q5-eagle3" "final-qwen-q5-dspark-p0|$Q-Q5_K_M.gguf|$DK/dspark_qwen3_8b_block7.gguf|draft-dspark|0.0|experiments/runs/final-qwen-q5-dspark-p0" # Qwen3-8B Q8_0 (3) "final-qwen-q8-solo|$Q-Q8_0.gguf||||experiments/runs/final-qwen-q8-solo" "final-qwen-q8-eagle3|$Q-Q8_0.gguf|$DK/Qwen3-8B-speculator.eagle3-F16.gguf|draft-eagle3||experiments/runs/final-qwen-q8-eagle3" "final-qwen-q8-dspark-p0|$Q-Q8_0.gguf|$DK/dspark_qwen3_8b_block7.gguf|draft-dspark|0.0|experiments/runs/final-qwen-q8-dspark-p0" # Gemma 4 12B Q4_K_M (5) "final-gemma-q4-solo|$G-Q4_K_M.gguf||||experiments/runs/final-gemma-q4-solo" "final-gemma-q4-mtp|$G-Q4_K_M.gguf|$DK/mtp-gemma-4-12b-it-Q8_0.gguf|draft-mtp||experiments/runs/final-gemma-q4-mtp" "final-gemma-q4-dflash-f16|$G-Q4_K_M.gguf|$DK/gemma-4-12B-it-DFlash-F16.gguf|draft-dflash||experiments/runs/final-gemma-q4-dflash-f16" "final-gemma-q4-dflash-q4|$G-Q4_K_M.gguf|$DK/gemma-4-12B-it-DFlash-Q4_K_M.gguf|draft-dflash||experiments/runs/final-gemma-q4-dflash-q4" "final-gemma-q4-dflash-q8|$G-Q4_K_M.gguf|$DK/gemma-4-12B-it-DFlash-Q8_0.gguf|draft-dflash||experiments/runs/final-gemma-q4-dflash-q8" # Gemma 4 12B Q5_K_M (3) "final-gemma-q5-solo|$G-Q5_K_M.gguf||||experiments/runs/final-gemma-q5-solo" "final-gemma-q5-mtp|$G-Q5_K_M.gguf|$DK/mtp-gemma-4-12b-it-Q8_0.gguf|draft-mtp||experiments/runs/final-gemma-q5-mtp" "final-gemma-q5-dflash-f16|$G-Q5_K_M.gguf|$DK/gemma-4-12B-it-DFlash-F16.gguf|draft-dflash||experiments/runs/final-gemma-q5-dflash-f16" # Gemma 4 12B Q8_0 (3) — block after OOM-CHECK "final-gemma-q8-solo|$G-Q8_0.gguf||||experiments/runs/final-gemma-q8-solo" "final-gemma-q8-mtp|$G-Q8_0.gguf|$DK/mtp-gemma-4-12b-it-Q8_0.gguf|draft-mtp||experiments/runs/final-gemma-q8-mtp" "final-gemma-q8-dflash-f16|$G-Q8_0.gguf|$DK/gemma-4-12B-it-DFlash-F16.gguf|draft-dflash||experiments/runs/final-gemma-q8-dflash-f16|-ub 512" ) # Files present and ≥ 100 MB (the quantized DFlash drafts 422M/752M fit in; # lesson F3 #4: the 1 GB threshold stalled downloads for 2 h). have_file() { local f="$1" [ -f "$f" ] && [ "$(stat -c%s "$f" 2>/dev/null || echo 0)" -gt 100000000 ] } run_job() { local name="$1" out="$2" shift 2 local rc log "JOB START: $name" $BENCH --config-name "$name" --prompts "$SAMPLES" --n-tokens "$NTOK" \ --out "$out" --resume "${MAXP[@]}" "$@" rc=$? if [ "$rc" -eq 3 ]; then log "JOB LOCK: $name (exit 3 = another runner on this --out) — wait 300s and retry once" sleep 300 $BENCH --config-name "$name" --prompts "$SAMPLES" --n-tokens "$NTOK" \ --out "$out" --resume "${MAXP[@]}" "$@" rc=$? if [ "$rc" -eq 3 ]; then log "CHAIN ABORT: $name still locked (another live chain) — aborting to avoid overwriting results" exit 3 fi fi case "$rc" in 0) log "JOB DONE: $name (exit 0)"; OK_COUNT=$((OK_COUNT + 1)) ;; 2) log "JOB DONE WITH FAILURES: $name (exit 2) — prompts in errors.jsonl"; FAILED_COUNT=$((FAILED_COUNT + 1)) ;; *) log "JOB FAILED: $name (exit $rc)"; FAILED_COUNT=$((FAILED_COUNT + 1)) ;; esac return "$rc" } run_config() { local entry="$1" local name target draft stype pmin out extra IFS='|' read -r name target draft stype pmin out extra <<< "$entry" [ -n "$out" ] || out="experiments/runs/final-$name" if ! have_file "$target"; then log "SKIP: $name (target not available: $target)" SKIP_COUNT=$((SKIP_COUNT + 1)) return fi if [ -n "$draft" ] && ! have_file "$draft"; then log "SKIP: $name (draft not available: $draft)" SKIP_COUNT=$((SKIP_COUNT + 1)) return fi local args=() if [ -n "$draft" ]; then args+=(--draft "$draft" --spec-type "$stype") [ -n "$pmin" ] && args+=(--spec-draft-p-min "$pmin") fi [ -n "$extra" ] && args+=(--extra "$extra") run_job "$name" "$out" --model "$target" "${args[@]}" } # OOM-CHECK before the Gemma Q8 block (legacy pattern): llama-cli with the # worst case g8+dflash-f16 (superset of g8-solo). Failure → SKIP the 3 g8 configs. oom_check_g8() { log "OOM-CHECK: gemma q8 + dflash f16 (llama-cli -ub 512, 1 prompt n=8) — worst case of the g8 block" if "$LLAMA_CPP_BIN/llama-cli" -m "$G-Q8_0.gguf" -md "$DK/gemma-4-12B-it-DFlash-F16.gguf" \ --spec-type draft-dflash -ngl 99 -ngld 99 -n 8 -p "Hello." -ub 512 \ --no-conversation --single-turn --reasoning off -c 2048 -t 8 --seed 42 \ < /dev/null > /tmp/opencode/oom-check-final.log 2>&1; then log "OOM-CHECK: OK (gemma q8 usable with -ub 512)" OOM_OK=1 else log "OOM-CHECK: FAILED → SKIP the 3 g8 configs (see /tmp/opencode/oom-check-final.log; fallback: ctx 1024 or draft Q4/Q8)" OOM_OK=0 fi } # Conditional q5-dflash (spec): decision with REAL sizes + KV ctx 2048 + # overhead vs free VRAM; the decision (numbers + reason) stays in queue-final.log. q5_dflash_decide() { local target="$Q-Q5_K_M.gguf" draft="$DK/dflash_qwen3_8b_block7.gguf" local tsize dsize free_mib need tsize=$(( $(stat -c%s "$target" 2>/dev/null || echo 0) / 1048576 )) dsize=$(( $(stat -c%s "$draft" 2>/dev/null || echo 0) / 1048576 )) if ! have_file "$target" || ! have_file "$draft"; then log "q5-dflash: excluded (missing files: $target / $draft)" return 1 fi free_mib=$(nvidia-smi --query-gpu=memory.free --format=csv,noheader,nounits 2>/dev/null | head -n1 | tr -d ' ') local kv_mib=1024 overhead_mib=512 need=$((tsize + dsize + kv_mib + overhead_mib)) if [ -n "$free_mib" ] && [ "$free_mib" -ge "$need" ]; then log "q5-dflash: FITS — target ${tsize}MiB + draft ${dsize}MiB + KV ${kv_mib}MiB + overhead ${overhead_mib}MiB = ${need}MiB ≤ free VRAM ${free_mib}MiB" return 0 fi log "q5-dflash: DOES NOT FIT — target ${tsize}MiB + draft ${dsize}MiB + KV ${kv_mib}MiB + overhead ${overhead_mib}MiB = ${need}MiB > free VRAM ${free_mib:-unknown}MiB → excluded" return 1 } log "FINAL CHAIN START (SMOKE=${SMOKE:-0}) — ${#MATRIX[@]} configs + conditional q5-dflash" # DRY_RUN: prints the matrix (26 configs) without touching the GPU and exits. if [ "${DRY_RUN:-0}" = "1" ]; then echo "DRY-RUN: final-run matrix (${#MATRIX[@]} + 1 conditional):" for entry in "${MATRIX[@]}"; do IFS='|' read -r dname dtarget ddraft dstype dpmin dout <<< "$entry" echo " $dname | $dtarget | ${ddraft:-—} | ${dstype:-—} | ${dpmin:-—}" done echo " final-qwen-q5-dflash (conditional — decided by real VRAM at runtime)" echo "DRY-RUN: end ($((${#MATRIX[@]} + 1)) configs)" exit 0 fi if q5_dflash_decide; then Q5_DFLASH_OK=1 fi T0=$(date +%s) for entry in "${MATRIX[@]}"; do name="${entry%%|*}" case "$name" in final-qwen-q5-dspark-p0) # Inserts the conditional config at the end of the Q5 block. if [ "$Q5_DFLASH_OK" = "1" ]; then run_config "final-qwen-q5-dflash|$Q-Q5_K_M.gguf|$DK/dflash_qwen3_8b_block7.gguf|draft-dflash||experiments/runs/final-qwen-q5-dflash" fi ;; final-gemma-q8-*) if [ "$G8_CHECKED" -eq 0 ]; then oom_check_g8 G8_CHECKED=1 fi if [ "$OOM_OK" != "1" ]; then log "SKIP: $name (OOM-CHECK g8 failed; fallback: ctx 1024 or draft Q4/Q8)" SKIP_COUNT=$((SKIP_COUNT + 1)) continue fi ;; esac run_config "$entry" done DUR=$(( $(date +%s) - T0 )) log "FINAL RUN DONE (duration ${DUR}s ≈ $((DUR / 3600))h $((DUR % 3600 / 60))m)" log "FINAL SUMMARY: configs OK=$OK_COUNT FAILED=$FAILED_COUNT SKIP=$SKIP_COUNT" if [ "$FAILED_COUNT" -eq 0 ] && [ "$SKIP_COUNT" -eq 0 ]; then log "FINAL RUN: exit 0 (no FAILED or SKIP)" exit 0 fi log "FINAL RUN: exit 1 (FAILED=$FAILED_COUNT SKIP=$SKIP_COUNT — check queue-final.log and errors.jsonl)" exit 1