File size: 2,752 Bytes
b2ebc95
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
#!/bin/bash
# topup_loop.sh <target_graded> <run_dir> [run_dir...]
# Repeatedly resume each run until it has <target_graded> graded episodes.
#
# Why a loop rather than one long run: the shared sandbox pool alternates between bursts and long
# stalls (finding 22). A single pass leaves episodes waiting out their readiness budget, they
# expire as SandboxError, and the run "finishes" incomplete. Re-resuming on a cadence turns those
# into fresh sandbox requests, which is what actually gets served when a burst arrives.
#
# TWO HARD CONSTRAINTS, both learned the expensive way (finding 24):
#
#  1. **Never SIGKILL an eval run.** Each in-flight episode holds a host tunnel, and the account is
#     capped at 32 concurrent tunnels. `PrimeTunnel.expose` releases its tunnel in a shielded
#     `finally`, so SIGTERM cleans up — SIGKILL does not. Kill -9 a run with 24 episodes in flight
#     and you leak 24 tunnels; a few passes and all 32 slots are gone, after which every episode
#     dies instantly with `TunnelError: Maximum number of tunnels (32) reached`. That looks exactly
#     like a sandbox-pool problem and is not one — and it is invisible in `/resources`.
#  2. **Total concurrency across ALL running eval processes must stay under 32.** Not per run —
#     per API token. Two arms at 24 each is 48 and cannot work however healthy the pool looks.
#
# After stopping the runs each pass, the loop reclaims any stray tunnels. That is safe only
# because it has just stopped every run it started; do not run it alongside an unrelated eval.
WS=/mnt/pvc/users/simon/agentptb/runs/a-opus-max/workspace
cd "$WS" || exit 1
export PRIME_API_KEY="$(cat "$AGENTPTB_PRIME_KEY_FILE")"
PY=/root/work/a/prime-rl/.venv/bin/python
TARGET=${1:?target graded}; shift

stop_runs() {
  local pids
  pids=$(ps -eo pid=,args= | awk '/eval --resume/ && !/awk/ {print $1}')
  if [ -n "$pids" ]; then
    kill -TERM $pids 2>/dev/null          # SIGTERM so each episode releases its tunnel
    for _ in $(seq 1 20); do
      sleep 3
      ps -eo pid=,args= | grep -q "[e]val --resume" || break
    done
    pids=$(ps -eo pid=,args= | awk '/eval --resume/ && !/awk/ {print $1}')
    [ -n "$pids" ] && kill -9 $pids 2>/dev/null
    sleep 5
  fi
  "$PY" scripts/tunnels.py --delete 2>&1 | tail -1
}

for pass in $(seq 1 200); do
  alldone=1
  for d in "$@"; do
    g=$("$PY" scripts/graded.py -n "$d")
    echo "$(date -Is) pass=$pass $(basename "$d") graded=$g/$TARGET"
    if [ "${g:-0}" -lt "$TARGET" ]; then
      alldone=0
      setsid nohup bash scripts/resume_eval.sh "$WS/$d" > "logs/$(basename "$d")_resume.log" 2>&1 < /dev/null &
    fi
  done
  [ "$alldone" = "1" ] && { echo "$(date -Is) ALL RUNS AT TARGET"; exit 0; }
  sleep 900
  stop_runs
done