File size: 3,415 Bytes
d6e1c8a
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
#!/usr/bin/env bash
# 32B run2 (α=0.8 τ=2.0 IWA, ckpt iwa-0505_1840) — full eval, sequential on gpu3.
# Apply post-train fixes (chat_template + enable_iwa=False) BEFORE launching.
# Runs alongside run1's main + extended evals.
set -u

REPO=/opt/tiger/thothvl_pretrain
M=/mnt/bn/leonworkspace/terry/model
R=/mnt/bn/leonworkspace/terry/results
NAS_LOGS=/mnt/bn/leonworkspace/terry/logs
LOG_DIR=$NAS_LOGS/eval_32b_run2_$(date +%Y%m%d_%H%M)
mkdir -p "$LOG_DIR"
SUMMARY="$LOG_DIR/summary.log"
: > "$SUMMARY"

cd "$REPO/lmms-eval"
export PYTHONPATH=$REPO/QWENVL-PRIVATE:$PYTHONPATH
export HF_TOKEN=<HF_TOKEN>
export HF_HOME=/mnt/bn/leonworkspace/HF_HOME
export HF_DATASETS_CACHE=$HF_HOME/datasets
export OPENAI_API_KEY=<OPENAI_API_KEY>
export OPENAI_API_URL=https://search-va.byteintl.net/gpt/openapi/online/v2/crawl
export MODEL_VERSION=gpt-4o-2024-11-20
unset http_proxy https_proxy HTTP_PROXY HTTPS_PROXY

CKPT=$M/qwen3vl-32b-dense-roi-K49T3-150k-confluent-a0.8-t2.0-iwa-0505_1840
ATTN=flash_attention_2
TAG=publish-32b-run2-a0.8-576
GPU=${GPU:-3}

hybrid_args() { echo "pretrained=$CKPT,device_map=auto,two_stage_roi=True,roi_baseline=True,roi_conf_thresh=0.15,high_res_thresh=0.1,attn_implementation=$ATTN"; }
simple_args() { echo "pretrained=$CKPT,device_map=auto,attn_implementation=$ATTN"; }

run_one() {
    local cli=$1 args=$2 task=$3
    local out_dir="$R/$TAG/$task"
    if find "$out_dir" -name "*results.json" 2>/dev/null | grep -q .; then
        echo "[$(date '+%F %T')] SKIP gpu$GPU $task (done)" | tee -a "$SUMMARY"
        return
    fi
    mkdir -p "$out_dir"
    local log="$LOG_DIR/${task}.log"
    echo "[$(date '+%F %T')] START gpu$GPU $task ($cli)" | tee -a "$SUMMARY"
    CUDA_VISIBLE_DEVICES=$GPU python3 -m lmms_eval \
        --model "$cli" --model_args "$args" \
        --tasks "$task" --batch_size 1 \
        --output_path "$out_dir" \
        --log_samples --log_samples_suffix "$TAG" \
        > "$log" 2>&1
    local rc=$?
    if [ $rc -eq 0 ] && find "$out_dir" -name "*results.json" 2>/dev/null | grep -q .; then
        echo "[$(date '+%F %T')] DONE  gpu$GPU $task" | tee -a "$SUMMARY"
    else
        echo "[$(date '+%F %T')] ERR   gpu$GPU $task rc=$rc — $log" | tee -a "$SUMMARY"
    fi
}

# Order: small fast tasks first for early sanity check, then the larger ones.
run_one qwen3_vl_hybrid "$(hybrid_args)" vstar_bench    # 191
run_one qwen3_vl_hybrid "$(hybrid_args)" hrbench4k      # 800
run_one qwen3_vl_hybrid "$(hybrid_args)" hrbench8k      # 800
run_one qwen3_vl_hybrid "$(hybrid_args)" realworldqa    # 765
run_one qwen3_vl_hybrid "$(hybrid_args)" ocrbench       # 1000
run_one qwen3_vl_hybrid "$(hybrid_args)" mme            # 2374
run_one qwen3_vl_hybrid "$(hybrid_args)" chartqa        # 2500
run_one qwen3_vl_hybrid "$(hybrid_args)" infovqa_val    # 2801
run_one qwen3_vl_hybrid "$(hybrid_args)" pope           # ~9000
run_one qwen3_vl_hybrid "$(hybrid_args)" scienceqa      # ~4000-21000
run_one qwen3_vl_hybrid "$(hybrid_args)" docvqa_val     # 5349
run_one qwen3_vl_hybrid "$(hybrid_args)" textvqa_val    # 5000
run_one qwen3_vl        "$(simple_args)" gqa            # 12578 — simple (multi-image ROI bug)
run_one qwen3_vl        "$(simple_args)" seedbench      # ~17000 — simple
run_one qwen3_vl_hybrid "$(hybrid_args)" mmerealworld   # 23609 — last, longest

echo "[$(date '+%F %T')] ALL RUN2 EVALS DONE" | tee -a "$SUMMARY"
echo "Logs: $LOG_DIR"