diff --git "a/evals/protocolF_queue.log" "b/evals/protocolF_queue.log" new file mode 100644--- /dev/null +++ "b/evals/protocolF_queue.log" @@ -0,0 +1,4489 @@ +=== 2026-07-23T23:28:24-07:00 ROW: keep25_unhealed +2026-07-23T23:28:24-07:00 serving outputs/qwen35_pruned_keep25 on GPU 0,1 port 8399 (pp=2 think=false) +2026-07-23T23:28:24-07:00 waiting for server /health ... +2026-07-23T23:29:10-07:00 server up; chat pass [mmlu_pro,gpqa_diamond_cot_zeroshot,minerva_math500,ifeval,gsm8k_cot_zeroshot] +2026-07-23:23:29:17 INFO [_cli.run:388] Selected Tasks: ['mmlu_pro', 'gpqa_diamond_cot_zeroshot', 'minerva_math500', 'ifeval', 'gsm8k_cot_zeroshot'] +2026-07-23:23:29:18 INFO [evaluator:214] Setting random seed to 0 | Setting numpy seed to 1234 | Setting torch manual seed to 1234 | Setting fewshot manual seed to 1234 +2026-07-23:23:29:18 WARNING [evaluator:226] generation_kwargs: {'max_gen_toks': 1280} specified through cli, these settings will update set parameters in yaml tasks. Ensure 'do_sample=True' for non-greedy decoding! +2026-07-23:23:29:18 INFO [evaluator:239] Initializing local-chat-completions model, with arguments: {'model': 'student', 'base_url': 'http://127.0.0.1:8399/v1/chat/completions', 'num_concurrent': 64, 'tokenized_requests': False, 'max_retries': 3} +2026-07-23:23:29:18 INFO [models.api_models:179] Using max length 2048 - 1 +2026-07-23:23:29:18 INFO [models.api_models:200] Using tokenizer None +2026-07-23:23:29:33 INFO [evaluator_utils:446] Selected tasks: +2026-07-23:23:29:33 INFO [evaluator_utils:462] Group: mmlu_pro +2026-07-23:23:29:33 INFO [evaluator_utils:470] Task: mmlu_pro_biology (mmlu_pro/mmlu_pro_biology.yaml) +2026-07-23:23:29:33 INFO [evaluator_utils:470] Task: mmlu_pro_business (mmlu_pro/mmlu_pro_business.yaml) +2026-07-23:23:29:33 INFO [evaluator_utils:470] Task: mmlu_pro_chemistry (mmlu_pro/mmlu_pro_chemistry.yaml) +2026-07-23:23:29:33 INFO [evaluator_utils:470] Task: mmlu_pro_computer_science (mmlu_pro/mmlu_pro_computer_science.yaml) +2026-07-23:23:29:33 INFO [evaluator_utils:470] Task: mmlu_pro_economics (mmlu_pro/mmlu_pro_economics.yaml) +2026-07-23:23:29:33 INFO [evaluator_utils:470] Task: mmlu_pro_engineering (mmlu_pro/mmlu_pro_engineering.yaml) +2026-07-23:23:29:33 INFO [evaluator_utils:470] Task: mmlu_pro_health (mmlu_pro/mmlu_pro_health.yaml) +2026-07-23:23:29:33 INFO [evaluator_utils:470] Task: mmlu_pro_history (mmlu_pro/mmlu_pro_history.yaml) +2026-07-23:23:29:33 INFO [evaluator_utils:470] Task: mmlu_pro_law (mmlu_pro/mmlu_pro_law.yaml) +2026-07-23:23:29:33 INFO [evaluator_utils:470] Task: mmlu_pro_math (mmlu_pro/mmlu_pro_math.yaml) +2026-07-23:23:29:33 INFO [evaluator_utils:470] Task: mmlu_pro_other (mmlu_pro/mmlu_pro_other.yaml) +2026-07-23:23:29:33 INFO [evaluator_utils:470] Task: mmlu_pro_philosophy (mmlu_pro/mmlu_pro_philosophy.yaml) +2026-07-23:23:29:33 INFO [evaluator_utils:470] Task: mmlu_pro_physics (mmlu_pro/mmlu_pro_physics.yaml) +2026-07-23:23:29:33 INFO [evaluator_utils:470] Task: mmlu_pro_psychology (mmlu_pro/mmlu_pro_psychology.yaml) +2026-07-23:23:29:33 INFO [evaluator_utils:480] Task: gpqa_diamond_cot_zeroshot (gpqa/cot_zeroshot/gpqa_diamond_cot_zeroshot.yaml) +2026-07-23:23:29:33 INFO [evaluator_utils:480] Task: gsm8k_cot_zeroshot (gsm8k/gsm8k-cot-zeroshot.yaml) +2026-07-23:23:29:33 INFO [evaluator_utils:480] Task: ifeval (ifeval/ifeval.yaml) +2026-07-23:23:29:33 INFO [evaluator_utils:480] Task: minerva_math500 (minerva_math/minerva_math500.yaml) +2026-07-23:23:29:33 INFO [evaluator:314] mmlu_pro_biology: Using gen_kwargs: {'until': ['Question:'], 'max_gen_toks': 1280, 'do_sample': False, 'temperature': 0.0} +2026-07-23:23:29:33 INFO [evaluator:314] mmlu_pro_business: Using gen_kwargs: {'until': ['Question:'], 'max_gen_toks': 1280, 'do_sample': False, 'temperature': 0.0} +2026-07-23:23:29:33 INFO [evaluator:314] mmlu_pro_chemistry: Using gen_kwargs: {'until': ['Question:'], 'max_gen_toks': 1280, 'do_sample': False, 'temperature': 0.0} +2026-07-23:23:29:33 INFO [evaluator:314] mmlu_pro_computer_science: Using gen_kwargs: {'until': ['Question:'], 'max_gen_toks': 1280, 'do_sample': False, 'temperature': 0.0} +2026-07-23:23:29:33 INFO [evaluator:314] mmlu_pro_economics: Using gen_kwargs: {'until': ['Question:'], 'max_gen_toks': 1280, 'do_sample': False, 'temperature': 0.0} +2026-07-23:23:29:33 INFO [evaluator:314] mmlu_pro_engineering: Using gen_kwargs: {'until': ['Question:'], 'max_gen_toks': 1280, 'do_sample': False, 'temperature': 0.0} +2026-07-23:23:29:33 INFO [evaluator:314] mmlu_pro_health: Using gen_kwargs: {'until': ['Question:'], 'max_gen_toks': 1280, 'do_sample': False, 'temperature': 0.0} +2026-07-23:23:29:33 INFO [evaluator:314] mmlu_pro_history: Using gen_kwargs: {'until': ['Question:'], 'max_gen_toks': 1280, 'do_sample': False, 'temperature': 0.0} +2026-07-23:23:29:33 INFO [evaluator:314] mmlu_pro_law: Using gen_kwargs: {'until': ['Question:'], 'max_gen_toks': 1280, 'do_sample': False, 'temperature': 0.0} +2026-07-23:23:29:33 INFO [evaluator:314] mmlu_pro_math: Using gen_kwargs: {'until': ['Question:'], 'max_gen_toks': 1280, 'do_sample': False, 'temperature': 0.0} +2026-07-23:23:29:33 INFO [evaluator:314] mmlu_pro_other: Using gen_kwargs: {'until': ['Question:'], 'max_gen_toks': 1280, 'do_sample': False, 'temperature': 0.0} +2026-07-23:23:29:33 INFO [evaluator:314] mmlu_pro_philosophy: Using gen_kwargs: {'until': ['Question:'], 'max_gen_toks': 1280, 'do_sample': False, 'temperature': 0.0} +2026-07-23:23:29:33 INFO [evaluator:314] mmlu_pro_physics: Using gen_kwargs: {'until': ['Question:'], 'max_gen_toks': 1280, 'do_sample': False, 'temperature': 0.0} +2026-07-23:23:29:33 INFO [evaluator:314] mmlu_pro_psychology: Using gen_kwargs: {'until': ['Question:'], 'max_gen_toks': 1280, 'do_sample': False, 'temperature': 0.0} +2026-07-23:23:29:33 INFO [evaluator:314] gpqa_diamond_cot_zeroshot: Using gen_kwargs: {'until': [''], 'do_sample': False, 'temperature': 0.0, 'max_gen_toks': 1280} +2026-07-23:23:29:33 INFO [evaluator:314] minerva_math500: Using gen_kwargs: {'until': ['Problem:'], 'do_sample': False, 'temperature': 0.0, 'max_gen_toks': 1280} +2026-07-23:23:29:33 INFO [evaluator:314] ifeval: Using gen_kwargs: {'until': [], 'do_sample': False, 'temperature': 0.0, 'max_gen_toks': 1280} +2026-07-23:23:29:33 INFO [evaluator:314] gsm8k_cot_zeroshot: Using gen_kwargs: {'until': ['Q:', '', '<|im_end|>'], 'do_sample': False, 'max_gen_toks': 1280} +2026-07-23:23:29:33 INFO [api.task:312] Building contexts for mmlu_pro_biology on rank 0... + 0%| | 0/717 [00:00 outputs/evals/protocolF/keep25_unhealed +=== 2026-07-24T00:45:01-07:00 ROW DONE: keep25_unhealed (exit 0) +=== 2026-07-24T00:45:01-07:00 ROW: keep25_step0200 +2026-07-24T00:45:01-07:00 serving outputs/healed_dl/keep25_step0200 on GPU 0,1 port 8399 (pp=2 think=false) +2026-07-24T00:45:01-07:00 waiting for server /health ... +2026-07-24T00:45:46-07:00 server up; chat pass [mmlu_pro,gpqa_diamond_cot_zeroshot,minerva_math500,ifeval,gsm8k_cot_zeroshot] +2026-07-24:00:45:53 INFO [_cli.run:388] Selected Tasks: ['mmlu_pro', 'gpqa_diamond_cot_zeroshot', 'minerva_math500', 'ifeval', 'gsm8k_cot_zeroshot'] +2026-07-24:00:45:54 INFO [evaluator:214] Setting random seed to 0 | Setting numpy seed to 1234 | Setting torch manual seed to 1234 | Setting fewshot manual seed to 1234 +2026-07-24:00:45:54 WARNING [evaluator:226] generation_kwargs: {'max_gen_toks': 1280} specified through cli, these settings will update set parameters in yaml tasks. Ensure 'do_sample=True' for non-greedy decoding! +2026-07-24:00:45:54 INFO [evaluator:239] Initializing local-chat-completions model, with arguments: {'model': 'student', 'base_url': 'http://127.0.0.1:8399/v1/chat/completions', 'num_concurrent': 64, 'tokenized_requests': False, 'max_retries': 3} +2026-07-24:00:45:54 INFO [models.api_models:179] Using max length 2048 - 1 +2026-07-24:00:45:54 INFO [models.api_models:200] Using tokenizer None +2026-07-24:00:46:08 INFO [evaluator_utils:446] Selected tasks: +2026-07-24:00:46:08 INFO [evaluator_utils:462] Group: mmlu_pro +2026-07-24:00:46:08 INFO [evaluator_utils:470] Task: mmlu_pro_biology (mmlu_pro/mmlu_pro_biology.yaml) +2026-07-24:00:46:08 INFO [evaluator_utils:470] Task: mmlu_pro_business (mmlu_pro/mmlu_pro_business.yaml) +2026-07-24:00:46:08 INFO [evaluator_utils:470] Task: mmlu_pro_chemistry (mmlu_pro/mmlu_pro_chemistry.yaml) +2026-07-24:00:46:08 INFO [evaluator_utils:470] Task: mmlu_pro_computer_science (mmlu_pro/mmlu_pro_computer_science.yaml) +2026-07-24:00:46:08 INFO [evaluator_utils:470] Task: mmlu_pro_economics (mmlu_pro/mmlu_pro_economics.yaml) +2026-07-24:00:46:08 INFO [evaluator_utils:470] Task: mmlu_pro_engineering (mmlu_pro/mmlu_pro_engineering.yaml) +2026-07-24:00:46:08 INFO [evaluator_utils:470] Task: mmlu_pro_health (mmlu_pro/mmlu_pro_health.yaml) +2026-07-24:00:46:08 INFO [evaluator_utils:470] Task: mmlu_pro_history (mmlu_pro/mmlu_pro_history.yaml) +2026-07-24:00:46:08 INFO [evaluator_utils:470] Task: mmlu_pro_law (mmlu_pro/mmlu_pro_law.yaml) +2026-07-24:00:46:08 INFO [evaluator_utils:470] Task: mmlu_pro_math (mmlu_pro/mmlu_pro_math.yaml) +2026-07-24:00:46:08 INFO [evaluator_utils:470] Task: mmlu_pro_other (mmlu_pro/mmlu_pro_other.yaml) +2026-07-24:00:46:08 INFO [evaluator_utils:470] Task: mmlu_pro_philosophy (mmlu_pro/mmlu_pro_philosophy.yaml) +2026-07-24:00:46:08 INFO [evaluator_utils:470] Task: mmlu_pro_physics (mmlu_pro/mmlu_pro_physics.yaml) +2026-07-24:00:46:08 INFO [evaluator_utils:470] Task: mmlu_pro_psychology (mmlu_pro/mmlu_pro_psychology.yaml) +2026-07-24:00:46:08 INFO [evaluator_utils:480] Task: gpqa_diamond_cot_zeroshot (gpqa/cot_zeroshot/gpqa_diamond_cot_zeroshot.yaml) +2026-07-24:00:46:08 INFO [evaluator_utils:480] Task: gsm8k_cot_zeroshot (gsm8k/gsm8k-cot-zeroshot.yaml) +2026-07-24:00:46:08 INFO [evaluator_utils:480] Task: ifeval (ifeval/ifeval.yaml) +2026-07-24:00:46:08 INFO [evaluator_utils:480] Task: minerva_math500 (minerva_math/minerva_math500.yaml) +2026-07-24:00:46:08 INFO [evaluator:314] mmlu_pro_biology: Using gen_kwargs: {'until': ['Question:'], 'max_gen_toks': 1280, 'do_sample': False, 'temperature': 0.0} +2026-07-24:00:46:08 INFO [evaluator:314] mmlu_pro_business: Using gen_kwargs: {'until': ['Question:'], 'max_gen_toks': 1280, 'do_sample': False, 'temperature': 0.0} +2026-07-24:00:46:08 INFO [evaluator:314] mmlu_pro_chemistry: Using gen_kwargs: {'until': ['Question:'], 'max_gen_toks': 1280, 'do_sample': False, 'temperature': 0.0} +2026-07-24:00:46:08 INFO [evaluator:314] mmlu_pro_computer_science: Using gen_kwargs: {'until': ['Question:'], 'max_gen_toks': 1280, 'do_sample': False, 'temperature': 0.0} +2026-07-24:00:46:08 INFO [evaluator:314] mmlu_pro_economics: Using gen_kwargs: {'until': ['Question:'], 'max_gen_toks': 1280, 'do_sample': False, 'temperature': 0.0} +2026-07-24:00:46:08 INFO [evaluator:314] mmlu_pro_engineering: Using gen_kwargs: {'until': ['Question:'], 'max_gen_toks': 1280, 'do_sample': False, 'temperature': 0.0} +2026-07-24:00:46:08 INFO [evaluator:314] mmlu_pro_health: Using gen_kwargs: {'until': ['Question:'], 'max_gen_toks': 1280, 'do_sample': False, 'temperature': 0.0} +2026-07-24:00:46:08 INFO [evaluator:314] mmlu_pro_history: Using gen_kwargs: {'until': ['Question:'], 'max_gen_toks': 1280, 'do_sample': False, 'temperature': 0.0} +2026-07-24:00:46:08 INFO [evaluator:314] mmlu_pro_law: Using gen_kwargs: {'until': ['Question:'], 'max_gen_toks': 1280, 'do_sample': False, 'temperature': 0.0} +2026-07-24:00:46:08 INFO [evaluator:314] mmlu_pro_math: Using gen_kwargs: {'until': ['Question:'], 'max_gen_toks': 1280, 'do_sample': False, 'temperature': 0.0} +2026-07-24:00:46:08 INFO [evaluator:314] mmlu_pro_other: Using gen_kwargs: {'until': ['Question:'], 'max_gen_toks': 1280, 'do_sample': False, 'temperature': 0.0} +2026-07-24:00:46:08 INFO [evaluator:314] mmlu_pro_philosophy: Using gen_kwargs: {'until': ['Question:'], 'max_gen_toks': 1280, 'do_sample': False, 'temperature': 0.0} +2026-07-24:00:46:08 INFO [evaluator:314] mmlu_pro_physics: Using gen_kwargs: {'until': ['Question:'], 'max_gen_toks': 1280, 'do_sample': False, 'temperature': 0.0} +2026-07-24:00:46:08 INFO [evaluator:314] mmlu_pro_psychology: Using gen_kwargs: {'until': ['Question:'], 'max_gen_toks': 1280, 'do_sample': False, 'temperature': 0.0} +2026-07-24:00:46:08 INFO [evaluator:314] gpqa_diamond_cot_zeroshot: Using gen_kwargs: {'until': [''], 'do_sample': False, 'temperature': 0.0, 'max_gen_toks': 1280} +2026-07-24:00:46:08 INFO [evaluator:314] minerva_math500: Using gen_kwargs: {'until': ['Problem:'], 'do_sample': False, 'temperature': 0.0, 'max_gen_toks': 1280} +2026-07-24:00:46:08 INFO [evaluator:314] ifeval: Using gen_kwargs: {'until': [], 'do_sample': False, 'temperature': 0.0, 'max_gen_toks': 1280} +2026-07-24:00:46:08 INFO [evaluator:314] gsm8k_cot_zeroshot: Using gen_kwargs: {'until': ['Q:', '', '<|im_end|>'], 'do_sample': False, 'max_gen_toks': 1280} +2026-07-24:00:46:08 INFO [api.task:312] Building contexts for mmlu_pro_biology on rank 0... + 0%| | 0/717 [00:00 outputs/evals/protocolF/keep25_step0200 +=== 2026-07-24T04:28:04-07:00 ROW DONE: keep25_step0200 (exit 0) +=== 2026-07-24T04:28:04-07:00 ROW: keep50_unhealed +2026-07-24T04:28:04-07:00 serving outputs/qwen35_pruned_keep50 on GPU 0,1 port 8399 (pp=2 think=false) +2026-07-24T04:28:04-07:00 waiting for server /health ... +2026-07-24T04:28:59-07:00 server up; chat pass [mmlu_pro,gpqa_diamond_cot_zeroshot,minerva_math500,ifeval,gsm8k_cot_zeroshot] +2026-07-24:04:29:06 INFO [_cli.run:388] Selected Tasks: ['mmlu_pro', 'gpqa_diamond_cot_zeroshot', 'minerva_math500', 'ifeval', 'gsm8k_cot_zeroshot'] +2026-07-24:04:29:08 INFO [evaluator:214] Setting random seed to 0 | Setting numpy seed to 1234 | Setting torch manual seed to 1234 | Setting fewshot manual seed to 1234 +2026-07-24:04:29:08 WARNING [evaluator:226] generation_kwargs: {'max_gen_toks': 1280} specified through cli, these settings will update set parameters in yaml tasks. Ensure 'do_sample=True' for non-greedy decoding! +2026-07-24:04:29:08 INFO [evaluator:239] Initializing local-chat-completions model, with arguments: {'model': 'student', 'base_url': 'http://127.0.0.1:8399/v1/chat/completions', 'num_concurrent': 64, 'tokenized_requests': False, 'max_retries': 3} +2026-07-24:04:29:08 INFO [models.api_models:179] Using max length 2048 - 1 +2026-07-24:04:29:08 INFO [models.api_models:200] Using tokenizer None +2026-07-24:04:29:25 INFO [evaluator_utils:446] Selected tasks: +2026-07-24:04:29:25 INFO [evaluator_utils:462] Group: mmlu_pro +2026-07-24:04:29:25 INFO [evaluator_utils:470] Task: mmlu_pro_biology (mmlu_pro/mmlu_pro_biology.yaml) +2026-07-24:04:29:25 INFO [evaluator_utils:470] Task: mmlu_pro_business (mmlu_pro/mmlu_pro_business.yaml) +2026-07-24:04:29:25 INFO [evaluator_utils:470] Task: mmlu_pro_chemistry (mmlu_pro/mmlu_pro_chemistry.yaml) +2026-07-24:04:29:25 INFO [evaluator_utils:470] Task: mmlu_pro_computer_science (mmlu_pro/mmlu_pro_computer_science.yaml) +2026-07-24:04:29:25 INFO [evaluator_utils:470] Task: mmlu_pro_economics (mmlu_pro/mmlu_pro_economics.yaml) +2026-07-24:04:29:25 INFO [evaluator_utils:470] Task: mmlu_pro_engineering (mmlu_pro/mmlu_pro_engineering.yaml) +2026-07-24:04:29:25 INFO [evaluator_utils:470] Task: mmlu_pro_health (mmlu_pro/mmlu_pro_health.yaml) +2026-07-24:04:29:25 INFO [evaluator_utils:470] Task: mmlu_pro_history (mmlu_pro/mmlu_pro_history.yaml) +2026-07-24:04:29:25 INFO [evaluator_utils:470] Task: mmlu_pro_law (mmlu_pro/mmlu_pro_law.yaml) +2026-07-24:04:29:25 INFO [evaluator_utils:470] Task: mmlu_pro_math (mmlu_pro/mmlu_pro_math.yaml) +2026-07-24:04:29:25 INFO [evaluator_utils:470] Task: mmlu_pro_other (mmlu_pro/mmlu_pro_other.yaml) +2026-07-24:04:29:25 INFO [evaluator_utils:470] Task: mmlu_pro_philosophy (mmlu_pro/mmlu_pro_philosophy.yaml) +2026-07-24:04:29:25 INFO [evaluator_utils:470] Task: mmlu_pro_physics (mmlu_pro/mmlu_pro_physics.yaml) +2026-07-24:04:29:25 INFO [evaluator_utils:470] Task: mmlu_pro_psychology (mmlu_pro/mmlu_pro_psychology.yaml) +2026-07-24:04:29:25 INFO [evaluator_utils:480] Task: gpqa_diamond_cot_zeroshot (gpqa/cot_zeroshot/gpqa_diamond_cot_zeroshot.yaml) +2026-07-24:04:29:25 INFO [evaluator_utils:480] Task: gsm8k_cot_zeroshot (gsm8k/gsm8k-cot-zeroshot.yaml) +2026-07-24:04:29:25 INFO [evaluator_utils:480] Task: ifeval (ifeval/ifeval.yaml) +2026-07-24:04:29:25 INFO [evaluator_utils:480] Task: minerva_math500 (minerva_math/minerva_math500.yaml) +2026-07-24:04:29:25 INFO [evaluator:314] mmlu_pro_biology: Using gen_kwargs: {'until': ['Question:'], 'max_gen_toks': 1280, 'do_sample': False, 'temperature': 0.0} +2026-07-24:04:29:25 INFO [evaluator:314] mmlu_pro_business: Using gen_kwargs: {'until': ['Question:'], 'max_gen_toks': 1280, 'do_sample': False, 'temperature': 0.0} +2026-07-24:04:29:25 INFO [evaluator:314] mmlu_pro_chemistry: Using gen_kwargs: {'until': ['Question:'], 'max_gen_toks': 1280, 'do_sample': False, 'temperature': 0.0} +2026-07-24:04:29:25 INFO [evaluator:314] mmlu_pro_computer_science: Using gen_kwargs: {'until': ['Question:'], 'max_gen_toks': 1280, 'do_sample': False, 'temperature': 0.0} +2026-07-24:04:29:25 INFO [evaluator:314] mmlu_pro_economics: Using gen_kwargs: {'until': ['Question:'], 'max_gen_toks': 1280, 'do_sample': False, 'temperature': 0.0} +2026-07-24:04:29:25 INFO [evaluator:314] mmlu_pro_engineering: Using gen_kwargs: {'until': ['Question:'], 'max_gen_toks': 1280, 'do_sample': False, 'temperature': 0.0} +2026-07-24:04:29:25 INFO [evaluator:314] mmlu_pro_health: Using gen_kwargs: {'until': ['Question:'], 'max_gen_toks': 1280, 'do_sample': False, 'temperature': 0.0} +2026-07-24:04:29:25 INFO [evaluator:314] mmlu_pro_history: Using gen_kwargs: {'until': ['Question:'], 'max_gen_toks': 1280, 'do_sample': False, 'temperature': 0.0} +2026-07-24:04:29:25 INFO [evaluator:314] mmlu_pro_law: Using gen_kwargs: {'until': ['Question:'], 'max_gen_toks': 1280, 'do_sample': False, 'temperature': 0.0} +2026-07-24:04:29:25 INFO [evaluator:314] mmlu_pro_math: Using gen_kwargs: {'until': ['Question:'], 'max_gen_toks': 1280, 'do_sample': False, 'temperature': 0.0} +2026-07-24:04:29:25 INFO [evaluator:314] mmlu_pro_other: Using gen_kwargs: {'until': ['Question:'], 'max_gen_toks': 1280, 'do_sample': False, 'temperature': 0.0} +2026-07-24:04:29:25 INFO [evaluator:314] mmlu_pro_philosophy: Using gen_kwargs: {'until': ['Question:'], 'max_gen_toks': 1280, 'do_sample': False, 'temperature': 0.0} +2026-07-24:04:29:25 INFO [evaluator:314] mmlu_pro_physics: Using gen_kwargs: {'until': ['Question:'], 'max_gen_toks': 1280, 'do_sample': False, 'temperature': 0.0} +2026-07-24:04:29:25 INFO [evaluator:314] mmlu_pro_psychology: Using gen_kwargs: {'until': ['Question:'], 'max_gen_toks': 1280, 'do_sample': False, 'temperature': 0.0} +2026-07-24:04:29:25 INFO [evaluator:314] gpqa_diamond_cot_zeroshot: Using gen_kwargs: {'until': [''], 'do_sample': False, 'temperature': 0.0, 'max_gen_toks': 1280} +2026-07-24:04:29:25 INFO [evaluator:314] minerva_math500: Using gen_kwargs: {'until': ['Problem:'], 'do_sample': False, 'temperature': 0.0, 'max_gen_toks': 1280} +2026-07-24:04:29:25 INFO [evaluator:314] ifeval: Using gen_kwargs: {'until': [], 'do_sample': False, 'temperature': 0.0, 'max_gen_toks': 1280} +2026-07-24:04:29:25 INFO [evaluator:314] gsm8k_cot_zeroshot: Using gen_kwargs: {'until': ['Q:', '', '<|im_end|>'], 'do_sample': False, 'max_gen_toks': 1280} +2026-07-24:04:29:25 INFO [api.task:312] Building contexts for mmlu_pro_biology on rank 0... + 0%| | 0/717 [00:00'], 'do_sample': False, 'temperature': 0.0, 'max_gen_toks': 1280} +2026-07-24:06:37:18 INFO [evaluator:314] minerva_math500: Using gen_kwargs: {'until': ['Problem:'], 'do_sample': False, 'temperature': 0.0, 'max_gen_toks': 1280} +2026-07-24:06:37:18 INFO [evaluator:314] ifeval: Using gen_kwargs: {'until': [], 'do_sample': False, 'temperature': 0.0, 'max_gen_toks': 1280} +2026-07-24:06:37:18 INFO [evaluator:314] gsm8k_cot_zeroshot: Using gen_kwargs: {'until': ['Q:', '', '<|im_end|>'], 'do_sample': False, 'max_gen_toks': 1280} +2026-07-24:06:37:18 INFO [api.task:312] Building contexts for mmlu_pro_biology on rank 0... + 0%| | 0/717 [00:00 outputs/evals/protocolF/keep50_step0200 +=== 2026-07-24T12:12:46-07:00 ROW DONE: keep50_step0200 (exit 0) +=== 2026-07-24T12:12:46-07:00 ROW: keep50_step0200_think +2026-07-24T12:12:46-07:00 serving outputs/healed_dl/keep50_step0200 on GPU 0,1 port 8399 (pp=2 think=true) +2026-07-24T12:12:46-07:00 waiting for server /health ...