| #!/usr/bin/env bash |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| set -euo pipefail |
|
|
| MODEL_NAME="${MODEL_NAME:-Qwen/Qwen3.6-27B}" |
| SERVED_MODEL_NAME="${SERVED_MODEL_NAME:-qwen36-base}" |
| PORT="${PORT:-8000}" |
| BASE_URL="http://127.0.0.1:${PORT}/v1" |
| EVAL_LABEL="${EVAL_LABEL:-base}" |
| VLLM_READY_TIMEOUT="${VLLM_READY_TIMEOUT:-1800}" |
| REPORT_DIR="${REPORT_DIR:-reports/eval}" |
| VLLM_LOG="${VLLM_LOG:-/workspace/tmp/vllm_base.log}" |
|
|
| here="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" |
|
|
| echo "[1/4] Building held-out eval sets ..." |
| python3 "$here/hf_download.py" --key primevul --eval || echo " (primevul eval download failed; continuing)" |
| python3 "$here/hf_download.py" --key megavul --eval || echo " (megavul eval download failed; continuing)" |
| python3 "$here/build_eval_sets.py" --mode vuln_detection --out data/eval/vuln_detection_test.jsonl |
|
|
| python3 "$here/hf_download.py" --hf-id theelderemo/pentesting-explanations \ |
| --config mitre_attack --split train --out data/download/pentest_mcq_eval/raw.jsonl || \ |
| echo " (mcq download failed; skipping mcq eval)" |
| if [[ -f data/download/pentest_mcq_eval/raw.jsonl ]]; then |
| python3 "$here/build_eval_sets.py" --mode mcq \ |
| --mcq-input data/download/pentest_mcq_eval/raw.jsonl --out data/eval/knowledge_mcq.jsonl |
| fi |
|
|
| echo "[2/4] Starting base vLLM server (log: $VLLM_LOG) ..." |
| mkdir -p "$(dirname "$VLLM_LOG")" |
| MODEL_NAME="$MODEL_NAME" SERVED_MODEL_NAME="$SERVED_MODEL_NAME" PORT="$PORT" \ |
| bash "$here/serve_base_vllm.sh" >"$VLLM_LOG" 2>&1 & |
| VLLM_PID=$! |
| trap 'echo "Stopping vLLM ($VLLM_PID)"; kill $VLLM_PID 2>/dev/null || true' EXIT |
|
|
| echo "[3/4] Waiting for vLLM to become ready (timeout ${VLLM_READY_TIMEOUT}s) ..." |
| elapsed=0 |
| until curl -sf "http://127.0.0.1:${PORT}/v1/models" >/dev/null 2>&1; do |
| if ! kill -0 "$VLLM_PID" 2>/dev/null; then |
| echo "vLLM exited early. Last log lines:"; tail -n 40 "$VLLM_LOG"; exit 1 |
| fi |
| if (( elapsed >= VLLM_READY_TIMEOUT )); then |
| echo "vLLM did not become ready in time. Last log lines:"; tail -n 40 "$VLLM_LOG"; exit 1 |
| fi |
| sleep 10; elapsed=$((elapsed + 10)) |
| done |
| echo " vLLM ready after ${elapsed}s." |
|
|
| echo "[4/4] Scoring base model ..." |
| EVAL_ARGS=(--label "$EVAL_LABEL" --base-url "$BASE_URL" --model "$SERVED_MODEL_NAME" |
| --report-dir "$REPORT_DIR" --eval data/eval/vuln_detection_test.jsonl) |
| [[ -f data/eval/knowledge_mcq.jsonl ]] && EVAL_ARGS+=(--eval data/eval/knowledge_mcq.jsonl) |
| python3 "$here/eval_endpoint.py" "${EVAL_ARGS[@]}" |
|
|
| echo |
| echo "Base secondary-eval reports written to ${REPORT_DIR}/${EVAL_LABEL}_eval.{md,json}" |
| echo "Next: capture the CyberGym agentic baseline (Docker) per training/recipes/pretraining_cybergym_baseline.md" |
|
|