infosec-v1 / code /training /scripts /run_base_baseline.sh
adhikjoshi's picture
Super-squash branch 'main' using huggingface_hub
994182c
Raw
History Blame Contribute Delete
3.27 kB
#!/usr/bin/env bash
# One-command secondary-eval baseline for the BASE model (GPU host).
#
# Serves base Qwen with vLLM, builds the held-out eval sets, and scores them so
# you have base numbers before training. Re-run eval_endpoint.py with a different
# --label/--model against each checkpoint later to track progress.
#
# This covers the vLLM-only secondary evals (vuln-detection accuracy + knowledge MCQ).
# The PRIMARY CyberGym agentic baseline needs Docker and is a separate flow:
# training/recipes/pretraining_cybergym_baseline.md
#
# Usage:
# bash training/scripts/run_base_baseline.sh
# Env overrides: MODEL_NAME, SERVED_MODEL_NAME, PORT, EVAL_LABEL, VLLM_READY_TIMEOUT
set -euo pipefail
MODEL_NAME="${MODEL_NAME:-Qwen/Qwen3.6-27B}"
SERVED_MODEL_NAME="${SERVED_MODEL_NAME:-qwen36-base}"
PORT="${PORT:-8000}"
BASE_URL="http://127.0.0.1:${PORT}/v1"
EVAL_LABEL="${EVAL_LABEL:-base}"
VLLM_READY_TIMEOUT="${VLLM_READY_TIMEOUT:-1800}"
REPORT_DIR="${REPORT_DIR:-reports/eval}"
VLLM_LOG="${VLLM_LOG:-/workspace/tmp/vllm_base.log}"
here="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
echo "[1/4] Building held-out eval sets ..."
python3 "$here/hf_download.py" --key primevul --eval || echo " (primevul eval download failed; continuing)"
python3 "$here/hf_download.py" --key megavul --eval || echo " (megavul eval download failed; continuing)"
python3 "$here/build_eval_sets.py" --mode vuln_detection --out data/eval/vuln_detection_test.jsonl
python3 "$here/hf_download.py" --hf-id theelderemo/pentesting-explanations \
--config mitre_attack --split train --out data/download/pentest_mcq_eval/raw.jsonl || \
echo " (mcq download failed; skipping mcq eval)"
if [[ -f data/download/pentest_mcq_eval/raw.jsonl ]]; then
python3 "$here/build_eval_sets.py" --mode mcq \
--mcq-input data/download/pentest_mcq_eval/raw.jsonl --out data/eval/knowledge_mcq.jsonl
fi
echo "[2/4] Starting base vLLM server (log: $VLLM_LOG) ..."
mkdir -p "$(dirname "$VLLM_LOG")"
MODEL_NAME="$MODEL_NAME" SERVED_MODEL_NAME="$SERVED_MODEL_NAME" PORT="$PORT" \
bash "$here/serve_base_vllm.sh" >"$VLLM_LOG" 2>&1 &
VLLM_PID=$!
trap 'echo "Stopping vLLM ($VLLM_PID)"; kill $VLLM_PID 2>/dev/null || true' EXIT
echo "[3/4] Waiting for vLLM to become ready (timeout ${VLLM_READY_TIMEOUT}s) ..."
elapsed=0
until curl -sf "http://127.0.0.1:${PORT}/v1/models" >/dev/null 2>&1; do
if ! kill -0 "$VLLM_PID" 2>/dev/null; then
echo "vLLM exited early. Last log lines:"; tail -n 40 "$VLLM_LOG"; exit 1
fi
if (( elapsed >= VLLM_READY_TIMEOUT )); then
echo "vLLM did not become ready in time. Last log lines:"; tail -n 40 "$VLLM_LOG"; exit 1
fi
sleep 10; elapsed=$((elapsed + 10))
done
echo " vLLM ready after ${elapsed}s."
echo "[4/4] Scoring base model ..."
EVAL_ARGS=(--label "$EVAL_LABEL" --base-url "$BASE_URL" --model "$SERVED_MODEL_NAME"
--report-dir "$REPORT_DIR" --eval data/eval/vuln_detection_test.jsonl)
[[ -f data/eval/knowledge_mcq.jsonl ]] && EVAL_ARGS+=(--eval data/eval/knowledge_mcq.jsonl)
python3 "$here/eval_endpoint.py" "${EVAL_ARGS[@]}"
echo
echo "Base secondary-eval reports written to ${REPORT_DIR}/${EVAL_LABEL}_eval.{md,json}"
echo "Next: capture the CyberGym agentic baseline (Docker) per training/recipes/pretraining_cybergym_baseline.md"