diff --git a/run/capability/capability_summary.csv b/run/capability/capability_summary.csv new file mode 100644 index 0000000000000000000000000000000000000000..aafb71cb3a096afc2acc390b5a62d45e1271b0de --- /dev/null +++ b/run/capability/capability_summary.csv @@ -0,0 +1,3 @@ +benchmark,metric,m0,m1,delta_pp +mmlu,"acc,none",67.72,68.42,+0.70 +gsm8k_cot,"exact_match,flexible-extract",61.00,68.00,+7.00 diff --git a/run/capability/capability_summary.md b/run/capability/capability_summary.md new file mode 100644 index 0000000000000000000000000000000000000000..260b06d961937f0edb696c29ba48f3ad22e0716c --- /dev/null +++ b/run/capability/capability_summary.md @@ -0,0 +1,8 @@ +# Capability retention: M0 (base) vs M1 (resist SFT) + +arXiv:2511.21399 App. E.4 methodology — lm-eval-harness, MMLU 5-shot MC, GSM8K 8-shot CoT, greedy, accuracy on the test split. + +| benchmark | metric | M0 | M1 | delta (pp) | +|---|---|---|---|---| +| mmlu | acc,none | 67.7% | 68.4% | +0.7 | +| gsm8k_cot | exact_match,flexible-extract | 61.0% | 68.0% | +7.0 | diff --git a/run/capability/lmeval_run.log b/run/capability/lmeval_run.log new file mode 100644 index 0000000000000000000000000000000000000000..e9992848fbb08868246d00427ba6e11663c8f899 --- /dev/null +++ b/run/capability/lmeval_run.log @@ -0,0 +1,1457 @@ + + +########## m0 / mmlu ########## +/root/steering-resistance/.venv-lmeval/bin/python -m lm_eval --model vllm --model_args pretrained=Qwen/Qwen2.5-3B-Instruct,dtype=bfloat16,gpu_memory_utilization=0.85,max_model_len=4096,seed=0 --tasks mmlu --num_fewshot 5 --batch_size auto --seed 0 --output_path /root/steering-resistance/results/full_3b/capability/m0/mmlu --apply_chat_template --fewshot_as_multiturn --limit 15 +2026-07-23:18:01:15 WARNING [config.evaluate_config:287] --limit SHOULD ONLY BE USED FOR TESTING. REAL METRICS SHOULD NOT BE COMPUTED USING LIMIT. +2026-07-23:18:01:23 INFO [_cli.run:388] Selected Tasks: ['mmlu'] +2026-07-23:18:01:24 INFO [evaluator:214] Setting random seed to 0 | Setting numpy seed to 0 | Setting torch manual seed to 0 | Setting fewshot manual seed to 0 +2026-07-23:18:01:24 INFO [evaluator:239] Initializing vllm model, with arguments: {'pretrained': 'Qwen/Qwen2.5-3B-Instruct', 'dtype': 'bfloat16', 'gpu_memory_utilization': 0.85, 'max_model_len': 4096, 'seed': 0} +INFO 07-23 18:01:31 [api_utils.py:273] non-default args: {'dtype': 'bfloat16', 'max_model_len': 4096, 'gpu_memory_utilization': 0.85, 'disable_log_stats': True, 'model': 'Qwen/Qwen2.5-3B-Instruct'} +INFO 07-23 18:01:42 [model.py:619] Resolved architecture: Qwen2ForCausalLM +INFO 07-23 18:01:42 [model.py:1776] Using max model len 4096 +INFO 07-23 18:01:42 [scheduler.py:252] Chunked prefill is enabled with max_num_batched_tokens=8192. +INFO 07-23 18:01:42 [vllm.py:1042] Asynchronous scheduling is enabled. +INFO 07-23 18:01:42 [kernel.py:292] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['native'], fused_add_rms_norm=['native']) +(EngineCore pid=1893) INFO 07-23 18:01:46 [core.py:114] Initializing a V1 LLM engine (v0.25.1) with config: model='Qwen/Qwen2.5-3B-Instruct', speculative_config=None, tokenizer='Qwen/Qwen2.5-3B-Instruct', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=False, dtype=torch.bfloat16, max_seq_len=4096, download_dir=None, load_format=auto, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, decode_context_parallel_size=1, dcp_comm_backend=ag_rs, disable_custom_all_reduce=False, quantization=None, quantization_config=None, enforce_eager=False, enable_return_routed_experts=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False, enable_mfu_metrics=False, enable_mm_processor_stats=False, enable_logging_iteration_details=False, jit_monitor_mode='warn', jit_monitor_verbose=False), seed=0, served_model_name=Qwen/Qwen2.5-3B-Instruct, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['none'], 'ir_enable_torch_wrap': True, 'splitting_ops': ['vllm::unified_attention_with_output', 'vllm::unified_mla_attention_with_output', 'vllm::mamba_mixer2', 'vllm::mamba_mixer', 'vllm::short_conv', 'vllm::linear_attention', 'vllm::plamo2_mamba_mixer', 'vllm::qwen_gdn_attention_core', 'vllm::gdn_attention_core_xpu', 'vllm::olmo_hybrid_gdn_full_forward', 'vllm::kda_attention', 'vllm::sparse_attn_indexer', 'vllm::rocm_aiter_sparse_attn_indexer', 'vllm::deepseek_v4_attention', 'vllm::hpc_rope_norm_forward', 'vllm::unified_kv_cache_update', 'vllm::unified_mla_kv_cache_update'], 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_vision_items_per_batch': 0, 'encoder_cudagraph_max_frames_per_batch': None, 'compile_sizes': [], 'compile_ranges_endpoints': [8192], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'size_asserts': False, 'alignment_asserts': False, 'scalar_asserts': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128, 136, 144, 152, 160, 168, 176, 184, 192, 200, 208, 216, 224, 232, 240, 248, 256, 272, 288, 304, 320, 336, 352, 368, 384, 400, 416, 432, 448, 464, 480, 496, 512], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': False, 'fuse_act_quant': False, 'fuse_attn_quant': False, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False, 'fuse_rope_kvcache_cat_mla': False, 'fuse_act_padding': False}, 'max_cudagraph_capture_size': 512, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': False, 'static_all_moe_layers': []}, kernel_config=KernelConfig(ir_op_priority=IrOpPriorityConfig(rms_norm=['native'], fused_add_rms_norm=['native']), enable_flashinfer_autotune=True, enable_cutedsl_warmup=True, moe_backend='auto', linear_backend='auto') +(EngineCore pid=1893) ERROR 07-23 18:01:46 [core.py:1231] EngineCore failed to start. +(EngineCore pid=1893) ERROR 07-23 18:01:46 [core.py:1231] Traceback (most recent call last): +(EngineCore pid=1893) ERROR 07-23 18:01:46 [core.py:1231] File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/engine/core.py", line 1200, in run_engine_core +(EngineCore pid=1893) ERROR 07-23 18:01:46 [core.py:1231] engine_core = EngineCoreProc(*args, engine_index=dp_rank, **kwargs) +(EngineCore pid=1893) ERROR 07-23 18:01:46 [core.py:1231] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ +(EngineCore pid=1893) ERROR 07-23 18:01:46 [core.py:1231] File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/tracing/otel.py", line 178, in sync_wrapper +(EngineCore pid=1893) ERROR 07-23 18:01:46 [core.py:1231] return func(*args, **kwargs) +(EngineCore pid=1893) ERROR 07-23 18:01:46 [core.py:1231] ^^^^^^^^^^^^^^^^^^^^^ +(EngineCore pid=1893) ERROR 07-23 18:01:46 [core.py:1231] File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/engine/core.py", line 966, in __init__ +(EngineCore pid=1893) ERROR 07-23 18:01:46 [core.py:1231] super().__init__( +(EngineCore pid=1893) ERROR 07-23 18:01:46 [core.py:1231] File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/engine/core.py", line 123, in __init__ +(EngineCore pid=1893) ERROR 07-23 18:01:46 [core.py:1231] self.model_executor = executor_class(vllm_config) +(EngineCore pid=1893) ERROR 07-23 18:01:46 [core.py:1231] ^^^^^^^^^^^^^^^^^^^^^^^^^^^ +(EngineCore pid=1893) ERROR 07-23 18:01:46 [core.py:1231] File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/tracing/otel.py", line 178, in sync_wrapper +(EngineCore pid=1893) ERROR 07-23 18:01:46 [core.py:1231] return func(*args, **kwargs) +(EngineCore pid=1893) ERROR 07-23 18:01:46 [core.py:1231] ^^^^^^^^^^^^^^^^^^^^^ +(EngineCore pid=1893) ERROR 07-23 18:01:46 [core.py:1231] File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/executor/abstract.py", line 109, in __init__ +(EngineCore pid=1893) ERROR 07-23 18:01:46 [core.py:1231] self._init_executor() +(EngineCore pid=1893) ERROR 07-23 18:01:46 [core.py:1231] File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/executor/uniproc_executor.py", line 63, in _init_executor +(EngineCore pid=1893) ERROR 07-23 18:01:46 [core.py:1231] self.driver_worker.init_device() +(EngineCore pid=1893) ERROR 07-23 18:01:46 [core.py:1231] File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/worker/worker_base.py", line 331, in init_device +(EngineCore pid=1893) ERROR 07-23 18:01:46 [core.py:1231] self.worker.init_device() # type: ignore +(EngineCore pid=1893) ERROR 07-23 18:01:46 [core.py:1231] ^^^^^^^^^^^^^^^^^^^^^^^^^ +(EngineCore pid=1893) ERROR 07-23 18:01:46 [core.py:1231] File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/tracing/otel.py", line 178, in sync_wrapper +(EngineCore pid=1893) ERROR 07-23 18:01:46 [core.py:1231] return func(*args, **kwargs) +(EngineCore pid=1893) ERROR 07-23 18:01:46 [core.py:1231] ^^^^^^^^^^^^^^^^^^^^^ +(EngineCore pid=1893) ERROR 07-23 18:01:46 [core.py:1231] File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/worker/gpu_worker.py", line 343, in init_device +(EngineCore pid=1893) ERROR 07-23 18:01:46 [core.py:1231] torch.accelerator.set_device_index(self.device) +(EngineCore pid=1893) ERROR 07-23 18:01:46 [core.py:1231] File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/torch/accelerator/__init__.py", line 191, in set_device_index +(EngineCore pid=1893) ERROR 07-23 18:01:46 [core.py:1231] torch._C._accelerator_setDeviceIndex(device_index) +(EngineCore pid=1893) ERROR 07-23 18:01:46 [core.py:1231] File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/torch/cuda/__init__.py", line 478, in _lazy_init +(EngineCore pid=1893) ERROR 07-23 18:01:46 [core.py:1231] torch._C._cuda_init() +(EngineCore pid=1893) ERROR 07-23 18:01:46 [core.py:1231] RuntimeError: The NVIDIA driver on your system is too old (found version 12040). Please update your GPU driver by downloading and installing a new version from the URL: http://www.nvidia.com/Download/index.aspx Alternatively, go to: https://pytorch.org to install a PyTorch version that has been compiled with your version of the CUDA driver. +(EngineCore pid=1893) Process EngineCore: +(EngineCore pid=1893) Traceback (most recent call last): +(EngineCore pid=1893) File "/usr/lib/python3.11/multiprocessing/process.py", line 314, in _bootstrap +(EngineCore pid=1893) self.run() +(EngineCore pid=1893) File "/usr/lib/python3.11/multiprocessing/process.py", line 108, in run +(EngineCore pid=1893) self._target(*self._args, **self._kwargs) +(EngineCore pid=1893) File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/engine/core.py", line 1235, in run_engine_core +(EngineCore pid=1893) raise e +(EngineCore pid=1893) File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/engine/core.py", line 1200, in run_engine_core +(EngineCore pid=1893) engine_core = EngineCoreProc(*args, engine_index=dp_rank, **kwargs) +(EngineCore pid=1893) ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ +(EngineCore pid=1893) File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/tracing/otel.py", line 178, in sync_wrapper +(EngineCore pid=1893) return func(*args, **kwargs) +(EngineCore pid=1893) ^^^^^^^^^^^^^^^^^^^^^ +(EngineCore pid=1893) File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/engine/core.py", line 966, in __init__ +(EngineCore pid=1893) super().__init__( +(EngineCore pid=1893) File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/engine/core.py", line 123, in __init__ +(EngineCore pid=1893) self.model_executor = executor_class(vllm_config) +(EngineCore pid=1893) ^^^^^^^^^^^^^^^^^^^^^^^^^^^ +(EngineCore pid=1893) File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/tracing/otel.py", line 178, in sync_wrapper +(EngineCore pid=1893) return func(*args, **kwargs) +(EngineCore pid=1893) ^^^^^^^^^^^^^^^^^^^^^ +(EngineCore pid=1893) File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/executor/abstract.py", line 109, in __init__ +(EngineCore pid=1893) self._init_executor() +(EngineCore pid=1893) File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/executor/uniproc_executor.py", line 63, in _init_executor +(EngineCore pid=1893) self.driver_worker.init_device() +(EngineCore pid=1893) File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/worker/worker_base.py", line 331, in init_device +(EngineCore pid=1893) self.worker.init_device() # type: ignore +(EngineCore pid=1893) ^^^^^^^^^^^^^^^^^^^^^^^^^ +(EngineCore pid=1893) File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/tracing/otel.py", line 178, in sync_wrapper +(EngineCore pid=1893) return func(*args, **kwargs) +(EngineCore pid=1893) ^^^^^^^^^^^^^^^^^^^^^ +(EngineCore pid=1893) File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/worker/gpu_worker.py", line 343, in init_device +(EngineCore pid=1893) torch.accelerator.set_device_index(self.device) +(EngineCore pid=1893) File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/torch/accelerator/__init__.py", line 191, in set_device_index +(EngineCore pid=1893) torch._C._accelerator_setDeviceIndex(device_index) +(EngineCore pid=1893) File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/torch/cuda/__init__.py", line 478, in _lazy_init +(EngineCore pid=1893) torch._C._cuda_init() +(EngineCore pid=1893) RuntimeError: The NVIDIA driver on your system is too old (found version 12040). Please update your GPU driver by downloading and installing a new version from the URL: http://www.nvidia.com/Download/index.aspx Alternatively, go to: https://pytorch.org to install a PyTorch version that has been compiled with your version of the CUDA driver. +Traceback (most recent call last): + File "", line 198, in _run_module_as_main + File "", line 88, in _run_code + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/lm_eval/__main__.py", line 14, in + cli_evaluate() + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/lm_eval/__main__.py", line 10, in cli_evaluate + parser.execute(args) + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/lm_eval/_cli/harness.py", line 60, in execute + args.func(args) + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/lm_eval/_cli/run.py", line 391, in _execute + results = simple_evaluate( + ^^^^^^^^^^^^^^^^ + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/lm_eval/utils.py", line 575, in _wrapper + return fn(*args, **kwargs) + ^^^^^^^^^^^^^^^^^^^ + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/lm_eval/evaluator.py", line 242, in simple_evaluate + lm = lm_eval.api.registry.get_model(model).create_from_arg_obj( + ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/lm_eval/api/model.py", line 169, in create_from_arg_obj + return cls(**arg_dict, **additional_config) + ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/lm_eval/models/vllm_causallms.py", line 146, in __init__ + self.model = LLM(**self.model_args) # type: ignore[invalid-argument-type] + ^^^^^^^^^^^^^^^^^^^^^^ + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/entrypoints/llm.py", line 349, in __init__ + self.llm_engine = LLMEngine.from_engine_args( + ^^^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/engine/llm_engine.py", line 179, in from_engine_args + return cls( + ^^^^ + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/engine/llm_engine.py", line 105, in __init__ + self.engine_core = EngineCoreClient.make_client( + ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/engine/core_client.py", line 103, in make_client + return SyncMPClient(vllm_config, executor_class, log_stats) + ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/tracing/otel.py", line 178, in sync_wrapper + return func(*args, **kwargs) + ^^^^^^^^^^^^^^^^^^^^^ + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/engine/core_client.py", line 786, in __init__ + super().__init__( + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/engine/core_client.py", line 573, in __init__ + with launch_core_engines( + File "/usr/lib/python3.11/contextlib.py", line 144, in __exit__ + next(self.gen) + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/engine/utils.py", line 1213, in launch_core_engines + wait_for_engine_startup( + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/engine/utils.py", line 1272, in wait_for_engine_startup + raise RuntimeError( +RuntimeError: Engine core initialization failed. See root cause above. Failed core proc(s): {'EngineCore': 1} + + +########## m0 / gsm8k_cot ########## +/root/steering-resistance/.venv-lmeval/bin/python -m lm_eval --model vllm --model_args pretrained=Qwen/Qwen2.5-3B-Instruct,dtype=bfloat16,gpu_memory_utilization=0.85,max_model_len=4096,seed=0 --tasks gsm8k_cot --num_fewshot 8 --batch_size auto --seed 0 --output_path /root/steering-resistance/results/full_3b/capability/m0/gsm8k_cot --gen_kwargs do_sample=False --apply_chat_template --fewshot_as_multiturn --limit 200 +2026-07-23:18:01:53 WARNING [config.evaluate_config:287] --limit SHOULD ONLY BE USED FOR TESTING. REAL METRICS SHOULD NOT BE COMPUTED USING LIMIT. +2026-07-23:18:02:01 INFO [_cli.run:388] Selected Tasks: ['gsm8k_cot'] +2026-07-23:18:02:02 INFO [evaluator:214] Setting random seed to 0 | Setting numpy seed to 0 | Setting torch manual seed to 0 | Setting fewshot manual seed to 0 +2026-07-23:18:02:02 WARNING [evaluator:226] generation_kwargs: {'do_sample': False} specified through cli, these settings will update set parameters in yaml tasks. Ensure 'do_sample=True' for non-greedy decoding! +2026-07-23:18:02:02 INFO [evaluator:239] Initializing vllm model, with arguments: {'pretrained': 'Qwen/Qwen2.5-3B-Instruct', 'dtype': 'bfloat16', 'gpu_memory_utilization': 0.85, 'max_model_len': 4096, 'seed': 0} +INFO 07-23 18:02:09 [api_utils.py:273] non-default args: {'dtype': 'bfloat16', 'max_model_len': 4096, 'gpu_memory_utilization': 0.85, 'disable_log_stats': True, 'model': 'Qwen/Qwen2.5-3B-Instruct'} +INFO 07-23 18:02:10 [model.py:619] Resolved architecture: Qwen2ForCausalLM +INFO 07-23 18:02:10 [model.py:1776] Using max model len 4096 +INFO 07-23 18:02:10 [scheduler.py:252] Chunked prefill is enabled with max_num_batched_tokens=8192. +INFO 07-23 18:02:10 [vllm.py:1042] Asynchronous scheduling is enabled. +INFO 07-23 18:02:10 [kernel.py:292] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['native'], fused_add_rms_norm=['native']) +(EngineCore pid=2359) INFO 07-23 18:02:14 [core.py:114] Initializing a V1 LLM engine (v0.25.1) with config: model='Qwen/Qwen2.5-3B-Instruct', speculative_config=None, tokenizer='Qwen/Qwen2.5-3B-Instruct', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=False, dtype=torch.bfloat16, max_seq_len=4096, download_dir=None, load_format=auto, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, decode_context_parallel_size=1, dcp_comm_backend=ag_rs, disable_custom_all_reduce=False, quantization=None, quantization_config=None, enforce_eager=False, enable_return_routed_experts=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False, enable_mfu_metrics=False, enable_mm_processor_stats=False, enable_logging_iteration_details=False, jit_monitor_mode='warn', jit_monitor_verbose=False), seed=0, served_model_name=Qwen/Qwen2.5-3B-Instruct, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['none'], 'ir_enable_torch_wrap': True, 'splitting_ops': ['vllm::unified_attention_with_output', 'vllm::unified_mla_attention_with_output', 'vllm::mamba_mixer2', 'vllm::mamba_mixer', 'vllm::short_conv', 'vllm::linear_attention', 'vllm::plamo2_mamba_mixer', 'vllm::qwen_gdn_attention_core', 'vllm::gdn_attention_core_xpu', 'vllm::olmo_hybrid_gdn_full_forward', 'vllm::kda_attention', 'vllm::sparse_attn_indexer', 'vllm::rocm_aiter_sparse_attn_indexer', 'vllm::deepseek_v4_attention', 'vllm::hpc_rope_norm_forward', 'vllm::unified_kv_cache_update', 'vllm::unified_mla_kv_cache_update'], 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_vision_items_per_batch': 0, 'encoder_cudagraph_max_frames_per_batch': None, 'compile_sizes': [], 'compile_ranges_endpoints': [8192], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'size_asserts': False, 'alignment_asserts': False, 'scalar_asserts': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128, 136, 144, 152, 160, 168, 176, 184, 192, 200, 208, 216, 224, 232, 240, 248, 256, 272, 288, 304, 320, 336, 352, 368, 384, 400, 416, 432, 448, 464, 480, 496, 512], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': False, 'fuse_act_quant': False, 'fuse_attn_quant': False, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False, 'fuse_rope_kvcache_cat_mla': False, 'fuse_act_padding': False}, 'max_cudagraph_capture_size': 512, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': False, 'static_all_moe_layers': []}, kernel_config=KernelConfig(ir_op_priority=IrOpPriorityConfig(rms_norm=['native'], fused_add_rms_norm=['native']), enable_flashinfer_autotune=True, enable_cutedsl_warmup=True, moe_backend='auto', linear_backend='auto') +(EngineCore pid=2359) ERROR 07-23 18:02:15 [core.py:1231] EngineCore failed to start. +(EngineCore pid=2359) ERROR 07-23 18:02:15 [core.py:1231] Traceback (most recent call last): +(EngineCore pid=2359) ERROR 07-23 18:02:15 [core.py:1231] File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/engine/core.py", line 1200, in run_engine_core +(EngineCore pid=2359) ERROR 07-23 18:02:15 [core.py:1231] engine_core = EngineCoreProc(*args, engine_index=dp_rank, **kwargs) +(EngineCore pid=2359) ERROR 07-23 18:02:15 [core.py:1231] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ +(EngineCore pid=2359) ERROR 07-23 18:02:15 [core.py:1231] File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/tracing/otel.py", line 178, in sync_wrapper +(EngineCore pid=2359) ERROR 07-23 18:02:15 [core.py:1231] return func(*args, **kwargs) +(EngineCore pid=2359) ERROR 07-23 18:02:15 [core.py:1231] ^^^^^^^^^^^^^^^^^^^^^ +(EngineCore pid=2359) ERROR 07-23 18:02:15 [core.py:1231] File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/engine/core.py", line 966, in __init__ +(EngineCore pid=2359) ERROR 07-23 18:02:15 [core.py:1231] super().__init__( +(EngineCore pid=2359) ERROR 07-23 18:02:15 [core.py:1231] File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/engine/core.py", line 123, in __init__ +(EngineCore pid=2359) ERROR 07-23 18:02:15 [core.py:1231] self.model_executor = executor_class(vllm_config) +(EngineCore pid=2359) ERROR 07-23 18:02:15 [core.py:1231] ^^^^^^^^^^^^^^^^^^^^^^^^^^^ +(EngineCore pid=2359) ERROR 07-23 18:02:15 [core.py:1231] File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/tracing/otel.py", line 178, in sync_wrapper +(EngineCore pid=2359) ERROR 07-23 18:02:15 [core.py:1231] return func(*args, **kwargs) +(EngineCore pid=2359) ERROR 07-23 18:02:15 [core.py:1231] ^^^^^^^^^^^^^^^^^^^^^ +(EngineCore pid=2359) ERROR 07-23 18:02:15 [core.py:1231] File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/executor/abstract.py", line 109, in __init__ +(EngineCore pid=2359) ERROR 07-23 18:02:15 [core.py:1231] self._init_executor() +(EngineCore pid=2359) ERROR 07-23 18:02:15 [core.py:1231] File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/executor/uniproc_executor.py", line 63, in _init_executor +(EngineCore pid=2359) ERROR 07-23 18:02:15 [core.py:1231] self.driver_worker.init_device() +(EngineCore pid=2359) ERROR 07-23 18:02:15 [core.py:1231] File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/worker/worker_base.py", line 331, in init_device +(EngineCore pid=2359) ERROR 07-23 18:02:15 [core.py:1231] self.worker.init_device() # type: ignore +(EngineCore pid=2359) ERROR 07-23 18:02:15 [core.py:1231] ^^^^^^^^^^^^^^^^^^^^^^^^^ +(EngineCore pid=2359) ERROR 07-23 18:02:15 [core.py:1231] File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/tracing/otel.py", line 178, in sync_wrapper +(EngineCore pid=2359) ERROR 07-23 18:02:15 [core.py:1231] return func(*args, **kwargs) +(EngineCore pid=2359) ERROR 07-23 18:02:15 [core.py:1231] ^^^^^^^^^^^^^^^^^^^^^ +(EngineCore pid=2359) ERROR 07-23 18:02:15 [core.py:1231] File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/worker/gpu_worker.py", line 343, in init_device +(EngineCore pid=2359) ERROR 07-23 18:02:15 [core.py:1231] torch.accelerator.set_device_index(self.device) +(EngineCore pid=2359) ERROR 07-23 18:02:15 [core.py:1231] File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/torch/accelerator/__init__.py", line 191, in set_device_index +(EngineCore pid=2359) ERROR 07-23 18:02:15 [core.py:1231] torch._C._accelerator_setDeviceIndex(device_index) +(EngineCore pid=2359) ERROR 07-23 18:02:15 [core.py:1231] File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/torch/cuda/__init__.py", line 478, in _lazy_init +(EngineCore pid=2359) ERROR 07-23 18:02:15 [core.py:1231] torch._C._cuda_init() +(EngineCore pid=2359) ERROR 07-23 18:02:15 [core.py:1231] RuntimeError: The NVIDIA driver on your system is too old (found version 12040). Please update your GPU driver by downloading and installing a new version from the URL: http://www.nvidia.com/Download/index.aspx Alternatively, go to: https://pytorch.org to install a PyTorch version that has been compiled with your version of the CUDA driver. +(EngineCore pid=2359) Process EngineCore: +(EngineCore pid=2359) Traceback (most recent call last): +(EngineCore pid=2359) File "/usr/lib/python3.11/multiprocessing/process.py", line 314, in _bootstrap +(EngineCore pid=2359) self.run() +(EngineCore pid=2359) File "/usr/lib/python3.11/multiprocessing/process.py", line 108, in run +(EngineCore pid=2359) self._target(*self._args, **self._kwargs) +(EngineCore pid=2359) File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/engine/core.py", line 1235, in run_engine_core +(EngineCore pid=2359) raise e +(EngineCore pid=2359) File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/engine/core.py", line 1200, in run_engine_core +(EngineCore pid=2359) engine_core = EngineCoreProc(*args, engine_index=dp_rank, **kwargs) +(EngineCore pid=2359) ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ +(EngineCore pid=2359) File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/tracing/otel.py", line 178, in sync_wrapper +(EngineCore pid=2359) return func(*args, **kwargs) +(EngineCore pid=2359) ^^^^^^^^^^^^^^^^^^^^^ +(EngineCore pid=2359) File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/engine/core.py", line 966, in __init__ +(EngineCore pid=2359) super().__init__( +(EngineCore pid=2359) File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/engine/core.py", line 123, in __init__ +(EngineCore pid=2359) self.model_executor = executor_class(vllm_config) +(EngineCore pid=2359) ^^^^^^^^^^^^^^^^^^^^^^^^^^^ +(EngineCore pid=2359) File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/tracing/otel.py", line 178, in sync_wrapper +(EngineCore pid=2359) return func(*args, **kwargs) +(EngineCore pid=2359) ^^^^^^^^^^^^^^^^^^^^^ +(EngineCore pid=2359) File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/executor/abstract.py", line 109, in __init__ +(EngineCore pid=2359) self._init_executor() +(EngineCore pid=2359) File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/executor/uniproc_executor.py", line 63, in _init_executor +(EngineCore pid=2359) self.driver_worker.init_device() +(EngineCore pid=2359) File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/worker/worker_base.py", line 331, in init_device +(EngineCore pid=2359) self.worker.init_device() # type: ignore +(EngineCore pid=2359) ^^^^^^^^^^^^^^^^^^^^^^^^^ +(EngineCore pid=2359) File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/tracing/otel.py", line 178, in sync_wrapper +(EngineCore pid=2359) return func(*args, **kwargs) +(EngineCore pid=2359) ^^^^^^^^^^^^^^^^^^^^^ +(EngineCore pid=2359) File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/worker/gpu_worker.py", line 343, in init_device +(EngineCore pid=2359) torch.accelerator.set_device_index(self.device) +(EngineCore pid=2359) File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/torch/accelerator/__init__.py", line 191, in set_device_index +(EngineCore pid=2359) torch._C._accelerator_setDeviceIndex(device_index) +(EngineCore pid=2359) File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/torch/cuda/__init__.py", line 478, in _lazy_init +(EngineCore pid=2359) torch._C._cuda_init() +(EngineCore pid=2359) RuntimeError: The NVIDIA driver on your system is too old (found version 12040). Please update your GPU driver by downloading and installing a new version from the URL: http://www.nvidia.com/Download/index.aspx Alternatively, go to: https://pytorch.org to install a PyTorch version that has been compiled with your version of the CUDA driver. +Traceback (most recent call last): + File "", line 198, in _run_module_as_main + File "", line 88, in _run_code + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/lm_eval/__main__.py", line 14, in + cli_evaluate() + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/lm_eval/__main__.py", line 10, in cli_evaluate + parser.execute(args) + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/lm_eval/_cli/harness.py", line 60, in execute + args.func(args) + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/lm_eval/_cli/run.py", line 391, in _execute + results = simple_evaluate( + ^^^^^^^^^^^^^^^^ + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/lm_eval/utils.py", line 575, in _wrapper + return fn(*args, **kwargs) + ^^^^^^^^^^^^^^^^^^^ + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/lm_eval/evaluator.py", line 242, in simple_evaluate + lm = lm_eval.api.registry.get_model(model).create_from_arg_obj( + ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/lm_eval/api/model.py", line 169, in create_from_arg_obj + return cls(**arg_dict, **additional_config) + ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/lm_eval/models/vllm_causallms.py", line 146, in __init__ + self.model = LLM(**self.model_args) # type: ignore[invalid-argument-type] + ^^^^^^^^^^^^^^^^^^^^^^ + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/entrypoints/llm.py", line 349, in __init__ + self.llm_engine = LLMEngine.from_engine_args( + ^^^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/engine/llm_engine.py", line 179, in from_engine_args + return cls( + ^^^^ + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/engine/llm_engine.py", line 105, in __init__ + self.engine_core = EngineCoreClient.make_client( + ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/engine/core_client.py", line 103, in make_client + return SyncMPClient(vllm_config, executor_class, log_stats) + ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/tracing/otel.py", line 178, in sync_wrapper + return func(*args, **kwargs) + ^^^^^^^^^^^^^^^^^^^^^ + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/engine/core_client.py", line 786, in __init__ + super().__init__( + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/engine/core_client.py", line 573, in __init__ + with launch_core_engines( + File "/usr/lib/python3.11/contextlib.py", line 144, in __exit__ + next(self.gen) + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/engine/utils.py", line 1213, in launch_core_engines + wait_for_engine_startup( + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/engine/utils.py", line 1272, in wait_for_engine_startup + raise RuntimeError( +RuntimeError: Engine core initialization failed. See root cause above. Failed core proc(s): {'EngineCore': 1} + + +########## m1 / mmlu ########## +/root/steering-resistance/.venv-lmeval/bin/python -m lm_eval --model vllm --model_args pretrained=Qwen/Qwen2.5-3B-Instruct,dtype=bfloat16,gpu_memory_utilization=0.85,max_model_len=4096,seed=0,lora_local_path=/root/steering-resistance/results/full_3b/m1_resist_adapter --tasks mmlu --num_fewshot 5 --batch_size auto --seed 0 --output_path /root/steering-resistance/results/full_3b/capability/m1/mmlu --apply_chat_template --fewshot_as_multiturn --limit 15 +2026-07-23:18:02:21 WARNING [config.evaluate_config:287] --limit SHOULD ONLY BE USED FOR TESTING. REAL METRICS SHOULD NOT BE COMPUTED USING LIMIT. +2026-07-23:18:02:29 INFO [_cli.run:388] Selected Tasks: ['mmlu'] +2026-07-23:18:02:30 INFO [evaluator:214] Setting random seed to 0 | Setting numpy seed to 0 | Setting torch manual seed to 0 | Setting fewshot manual seed to 0 +2026-07-23:18:02:30 INFO [evaluator:239] Initializing vllm model, with arguments: {'pretrained': 'Qwen/Qwen2.5-3B-Instruct', 'dtype': 'bfloat16', 'gpu_memory_utilization': 0.85, 'max_model_len': 4096, 'seed': 0, 'lora_local_path': '/root/steering-resistance/results/full_3b/m1_resist_adapter'} +INFO 07-23 18:02:37 [api_utils.py:273] non-default args: {'dtype': 'bfloat16', 'max_model_len': 4096, 'gpu_memory_utilization': 0.85, 'disable_log_stats': True, 'enable_lora': True, 'model': 'Qwen/Qwen2.5-3B-Instruct'} +INFO 07-23 18:02:38 [model.py:619] Resolved architecture: Qwen2ForCausalLM +INFO 07-23 18:02:38 [model.py:1776] Using max model len 4096 +INFO 07-23 18:02:38 [scheduler.py:252] Chunked prefill is enabled with max_num_batched_tokens=8192. +INFO 07-23 18:02:38 [vllm.py:1042] Asynchronous scheduling is enabled. +INFO 07-23 18:02:38 [kernel.py:292] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['native'], fused_add_rms_norm=['native']) +(EngineCore pid=2840) INFO 07-23 18:02:42 [core.py:114] Initializing a V1 LLM engine (v0.25.1) with config: model='Qwen/Qwen2.5-3B-Instruct', speculative_config=None, tokenizer='Qwen/Qwen2.5-3B-Instruct', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=False, dtype=torch.bfloat16, max_seq_len=4096, download_dir=None, load_format=auto, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, decode_context_parallel_size=1, dcp_comm_backend=ag_rs, disable_custom_all_reduce=False, quantization=None, quantization_config=None, enforce_eager=False, enable_return_routed_experts=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False, enable_mfu_metrics=False, enable_mm_processor_stats=False, enable_logging_iteration_details=False, jit_monitor_mode='warn', jit_monitor_verbose=False), seed=0, served_model_name=Qwen/Qwen2.5-3B-Instruct, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['none'], 'ir_enable_torch_wrap': True, 'splitting_ops': ['vllm::unified_attention_with_output', 'vllm::unified_mla_attention_with_output', 'vllm::mamba_mixer2', 'vllm::mamba_mixer', 'vllm::short_conv', 'vllm::linear_attention', 'vllm::plamo2_mamba_mixer', 'vllm::qwen_gdn_attention_core', 'vllm::gdn_attention_core_xpu', 'vllm::olmo_hybrid_gdn_full_forward', 'vllm::kda_attention', 'vllm::sparse_attn_indexer', 'vllm::rocm_aiter_sparse_attn_indexer', 'vllm::deepseek_v4_attention', 'vllm::hpc_rope_norm_forward', 'vllm::unified_kv_cache_update', 'vllm::unified_mla_kv_cache_update'], 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_vision_items_per_batch': 0, 'encoder_cudagraph_max_frames_per_batch': None, 'compile_sizes': [], 'compile_ranges_endpoints': [8192], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'size_asserts': False, 'alignment_asserts': False, 'scalar_asserts': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128, 136, 144, 152, 160, 168, 176, 184, 192, 200, 208, 216, 224, 232, 240, 248, 256, 272, 288, 304, 320, 336, 352, 368, 384, 400, 416, 432, 448, 464, 480, 496, 512], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': False, 'fuse_act_quant': False, 'fuse_attn_quant': False, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False, 'fuse_rope_kvcache_cat_mla': False, 'fuse_act_padding': False}, 'max_cudagraph_capture_size': 512, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': False, 'static_all_moe_layers': []}, kernel_config=KernelConfig(ir_op_priority=IrOpPriorityConfig(rms_norm=['native'], fused_add_rms_norm=['native']), enable_flashinfer_autotune=True, enable_cutedsl_warmup=True, moe_backend='auto', linear_backend='auto') +(EngineCore pid=2840) ERROR 07-23 18:02:43 [core.py:1231] EngineCore failed to start. +(EngineCore pid=2840) ERROR 07-23 18:02:43 [core.py:1231] Traceback (most recent call last): +(EngineCore pid=2840) ERROR 07-23 18:02:43 [core.py:1231] File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/engine/core.py", line 1200, in run_engine_core +(EngineCore pid=2840) ERROR 07-23 18:02:43 [core.py:1231] engine_core = EngineCoreProc(*args, engine_index=dp_rank, **kwargs) +(EngineCore pid=2840) ERROR 07-23 18:02:43 [core.py:1231] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ +(EngineCore pid=2840) ERROR 07-23 18:02:43 [core.py:1231] File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/tracing/otel.py", line 178, in sync_wrapper +(EngineCore pid=2840) ERROR 07-23 18:02:43 [core.py:1231] return func(*args, **kwargs) +(EngineCore pid=2840) ERROR 07-23 18:02:43 [core.py:1231] ^^^^^^^^^^^^^^^^^^^^^ +(EngineCore pid=2840) ERROR 07-23 18:02:43 [core.py:1231] File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/engine/core.py", line 966, in __init__ +(EngineCore pid=2840) ERROR 07-23 18:02:43 [core.py:1231] super().__init__( +(EngineCore pid=2840) ERROR 07-23 18:02:43 [core.py:1231] File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/engine/core.py", line 123, in __init__ +(EngineCore pid=2840) ERROR 07-23 18:02:43 [core.py:1231] self.model_executor = executor_class(vllm_config) +(EngineCore pid=2840) ERROR 07-23 18:02:43 [core.py:1231] ^^^^^^^^^^^^^^^^^^^^^^^^^^^ +(EngineCore pid=2840) ERROR 07-23 18:02:43 [core.py:1231] File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/tracing/otel.py", line 178, in sync_wrapper +(EngineCore pid=2840) ERROR 07-23 18:02:43 [core.py:1231] return func(*args, **kwargs) +(EngineCore pid=2840) ERROR 07-23 18:02:43 [core.py:1231] ^^^^^^^^^^^^^^^^^^^^^ +(EngineCore pid=2840) ERROR 07-23 18:02:43 [core.py:1231] File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/executor/abstract.py", line 109, in __init__ +(EngineCore pid=2840) ERROR 07-23 18:02:43 [core.py:1231] self._init_executor() +(EngineCore pid=2840) ERROR 07-23 18:02:43 [core.py:1231] File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/executor/uniproc_executor.py", line 63, in _init_executor +(EngineCore pid=2840) ERROR 07-23 18:02:43 [core.py:1231] self.driver_worker.init_device() +(EngineCore pid=2840) ERROR 07-23 18:02:43 [core.py:1231] File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/worker/worker_base.py", line 331, in init_device +(EngineCore pid=2840) ERROR 07-23 18:02:43 [core.py:1231] self.worker.init_device() # type: ignore +(EngineCore pid=2840) ERROR 07-23 18:02:43 [core.py:1231] ^^^^^^^^^^^^^^^^^^^^^^^^^ +(EngineCore pid=2840) ERROR 07-23 18:02:43 [core.py:1231] File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/tracing/otel.py", line 178, in sync_wrapper +(EngineCore pid=2840) ERROR 07-23 18:02:43 [core.py:1231] return func(*args, **kwargs) +(EngineCore pid=2840) ERROR 07-23 18:02:43 [core.py:1231] ^^^^^^^^^^^^^^^^^^^^^ +(EngineCore pid=2840) ERROR 07-23 18:02:43 [core.py:1231] File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/worker/gpu_worker.py", line 343, in init_device +(EngineCore pid=2840) ERROR 07-23 18:02:43 [core.py:1231] torch.accelerator.set_device_index(self.device) +(EngineCore pid=2840) ERROR 07-23 18:02:43 [core.py:1231] File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/torch/accelerator/__init__.py", line 191, in set_device_index +(EngineCore pid=2840) ERROR 07-23 18:02:43 [core.py:1231] torch._C._accelerator_setDeviceIndex(device_index) +(EngineCore pid=2840) ERROR 07-23 18:02:43 [core.py:1231] File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/torch/cuda/__init__.py", line 478, in _lazy_init +(EngineCore pid=2840) ERROR 07-23 18:02:43 [core.py:1231] torch._C._cuda_init() +(EngineCore pid=2840) ERROR 07-23 18:02:43 [core.py:1231] RuntimeError: The NVIDIA driver on your system is too old (found version 12040). Please update your GPU driver by downloading and installing a new version from the URL: http://www.nvidia.com/Download/index.aspx Alternatively, go to: https://pytorch.org to install a PyTorch version that has been compiled with your version of the CUDA driver. +(EngineCore pid=2840) Process EngineCore: +(EngineCore pid=2840) Traceback (most recent call last): +(EngineCore pid=2840) File "/usr/lib/python3.11/multiprocessing/process.py", line 314, in _bootstrap +(EngineCore pid=2840) self.run() +(EngineCore pid=2840) File "/usr/lib/python3.11/multiprocessing/process.py", line 108, in run +(EngineCore pid=2840) self._target(*self._args, **self._kwargs) +(EngineCore pid=2840) File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/engine/core.py", line 1235, in run_engine_core +(EngineCore pid=2840) raise e +(EngineCore pid=2840) File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/engine/core.py", line 1200, in run_engine_core +(EngineCore pid=2840) engine_core = EngineCoreProc(*args, engine_index=dp_rank, **kwargs) +(EngineCore pid=2840) ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ +(EngineCore pid=2840) File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/tracing/otel.py", line 178, in sync_wrapper +(EngineCore pid=2840) return func(*args, **kwargs) +(EngineCore pid=2840) ^^^^^^^^^^^^^^^^^^^^^ +(EngineCore pid=2840) File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/engine/core.py", line 966, in __init__ +(EngineCore pid=2840) super().__init__( +(EngineCore pid=2840) File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/engine/core.py", line 123, in __init__ +(EngineCore pid=2840) self.model_executor = executor_class(vllm_config) +(EngineCore pid=2840) ^^^^^^^^^^^^^^^^^^^^^^^^^^^ +(EngineCore pid=2840) File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/tracing/otel.py", line 178, in sync_wrapper +(EngineCore pid=2840) return func(*args, **kwargs) +(EngineCore pid=2840) ^^^^^^^^^^^^^^^^^^^^^ +(EngineCore pid=2840) File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/executor/abstract.py", line 109, in __init__ +(EngineCore pid=2840) self._init_executor() +(EngineCore pid=2840) File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/executor/uniproc_executor.py", line 63, in _init_executor +(EngineCore pid=2840) self.driver_worker.init_device() +(EngineCore pid=2840) File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/worker/worker_base.py", line 331, in init_device +(EngineCore pid=2840) self.worker.init_device() # type: ignore +(EngineCore pid=2840) ^^^^^^^^^^^^^^^^^^^^^^^^^ +(EngineCore pid=2840) File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/tracing/otel.py", line 178, in sync_wrapper +(EngineCore pid=2840) return func(*args, **kwargs) +(EngineCore pid=2840) ^^^^^^^^^^^^^^^^^^^^^ +(EngineCore pid=2840) File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/worker/gpu_worker.py", line 343, in init_device +(EngineCore pid=2840) torch.accelerator.set_device_index(self.device) +(EngineCore pid=2840) File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/torch/accelerator/__init__.py", line 191, in set_device_index +(EngineCore pid=2840) torch._C._accelerator_setDeviceIndex(device_index) +(EngineCore pid=2840) File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/torch/cuda/__init__.py", line 478, in _lazy_init +(EngineCore pid=2840) torch._C._cuda_init() +(EngineCore pid=2840) RuntimeError: The NVIDIA driver on your system is too old (found version 12040). Please update your GPU driver by downloading and installing a new version from the URL: http://www.nvidia.com/Download/index.aspx Alternatively, go to: https://pytorch.org to install a PyTorch version that has been compiled with your version of the CUDA driver. +Traceback (most recent call last): + File "", line 198, in _run_module_as_main + File "", line 88, in _run_code + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/lm_eval/__main__.py", line 14, in + cli_evaluate() + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/lm_eval/__main__.py", line 10, in cli_evaluate + parser.execute(args) + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/lm_eval/_cli/harness.py", line 60, in execute + args.func(args) + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/lm_eval/_cli/run.py", line 391, in _execute + results = simple_evaluate( + ^^^^^^^^^^^^^^^^ + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/lm_eval/utils.py", line 575, in _wrapper + return fn(*args, **kwargs) + ^^^^^^^^^^^^^^^^^^^ + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/lm_eval/evaluator.py", line 242, in simple_evaluate + lm = lm_eval.api.registry.get_model(model).create_from_arg_obj( + ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/lm_eval/api/model.py", line 169, in create_from_arg_obj + return cls(**arg_dict, **additional_config) + ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/lm_eval/models/vllm_causallms.py", line 146, in __init__ + self.model = LLM(**self.model_args) # type: ignore[invalid-argument-type] + ^^^^^^^^^^^^^^^^^^^^^^ + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/entrypoints/llm.py", line 349, in __init__ + self.llm_engine = LLMEngine.from_engine_args( + ^^^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/engine/llm_engine.py", line 179, in from_engine_args + return cls( + ^^^^ + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/engine/llm_engine.py", line 105, in __init__ + self.engine_core = EngineCoreClient.make_client( + ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/engine/core_client.py", line 103, in make_client + return SyncMPClient(vllm_config, executor_class, log_stats) + ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/tracing/otel.py", line 178, in sync_wrapper + return func(*args, **kwargs) + ^^^^^^^^^^^^^^^^^^^^^ + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/engine/core_client.py", line 786, in __init__ + super().__init__( + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/engine/core_client.py", line 573, in __init__ + with launch_core_engines( + File "/usr/lib/python3.11/contextlib.py", line 144, in __exit__ + next(self.gen) + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/engine/utils.py", line 1213, in launch_core_engines + wait_for_engine_startup( + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/engine/utils.py", line 1272, in wait_for_engine_startup + raise RuntimeError( +RuntimeError: Engine core initialization failed. See root cause above. Failed core proc(s): {'EngineCore': 1} + + +########## m1 / gsm8k_cot ########## +/root/steering-resistance/.venv-lmeval/bin/python -m lm_eval --model vllm --model_args pretrained=Qwen/Qwen2.5-3B-Instruct,dtype=bfloat16,gpu_memory_utilization=0.85,max_model_len=4096,seed=0,lora_local_path=/root/steering-resistance/results/full_3b/m1_resist_adapter --tasks gsm8k_cot --num_fewshot 8 --batch_size auto --seed 0 --output_path /root/steering-resistance/results/full_3b/capability/m1/gsm8k_cot --gen_kwargs do_sample=False --apply_chat_template --fewshot_as_multiturn --limit 200 +2026-07-23:18:02:49 WARNING [config.evaluate_config:287] --limit SHOULD ONLY BE USED FOR TESTING. REAL METRICS SHOULD NOT BE COMPUTED USING LIMIT. +2026-07-23:18:02:57 INFO [_cli.run:388] Selected Tasks: ['gsm8k_cot'] +2026-07-23:18:02:58 INFO [evaluator:214] Setting random seed to 0 | Setting numpy seed to 0 | Setting torch manual seed to 0 | Setting fewshot manual seed to 0 +2026-07-23:18:02:58 WARNING [evaluator:226] generation_kwargs: {'do_sample': False} specified through cli, these settings will update set parameters in yaml tasks. Ensure 'do_sample=True' for non-greedy decoding! +2026-07-23:18:02:58 INFO [evaluator:239] Initializing vllm model, with arguments: {'pretrained': 'Qwen/Qwen2.5-3B-Instruct', 'dtype': 'bfloat16', 'gpu_memory_utilization': 0.85, 'max_model_len': 4096, 'seed': 0, 'lora_local_path': '/root/steering-resistance/results/full_3b/m1_resist_adapter'} +INFO 07-23 18:03:05 [api_utils.py:273] non-default args: {'dtype': 'bfloat16', 'max_model_len': 4096, 'gpu_memory_utilization': 0.85, 'disable_log_stats': True, 'enable_lora': True, 'model': 'Qwen/Qwen2.5-3B-Instruct'} +INFO 07-23 18:03:06 [model.py:619] Resolved architecture: Qwen2ForCausalLM +INFO 07-23 18:03:06 [model.py:1776] Using max model len 4096 +INFO 07-23 18:03:06 [scheduler.py:252] Chunked prefill is enabled with max_num_batched_tokens=8192. +INFO 07-23 18:03:06 [vllm.py:1042] Asynchronous scheduling is enabled. +INFO 07-23 18:03:06 [kernel.py:292] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['native'], fused_add_rms_norm=['native']) +(EngineCore pid=3293) INFO 07-23 18:03:10 [core.py:114] Initializing a V1 LLM engine (v0.25.1) with config: model='Qwen/Qwen2.5-3B-Instruct', speculative_config=None, tokenizer='Qwen/Qwen2.5-3B-Instruct', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=False, dtype=torch.bfloat16, max_seq_len=4096, download_dir=None, load_format=auto, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, decode_context_parallel_size=1, dcp_comm_backend=ag_rs, disable_custom_all_reduce=False, quantization=None, quantization_config=None, enforce_eager=False, enable_return_routed_experts=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False, enable_mfu_metrics=False, enable_mm_processor_stats=False, enable_logging_iteration_details=False, jit_monitor_mode='warn', jit_monitor_verbose=False), seed=0, served_model_name=Qwen/Qwen2.5-3B-Instruct, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['none'], 'ir_enable_torch_wrap': True, 'splitting_ops': ['vllm::unified_attention_with_output', 'vllm::unified_mla_attention_with_output', 'vllm::mamba_mixer2', 'vllm::mamba_mixer', 'vllm::short_conv', 'vllm::linear_attention', 'vllm::plamo2_mamba_mixer', 'vllm::qwen_gdn_attention_core', 'vllm::gdn_attention_core_xpu', 'vllm::olmo_hybrid_gdn_full_forward', 'vllm::kda_attention', 'vllm::sparse_attn_indexer', 'vllm::rocm_aiter_sparse_attn_indexer', 'vllm::deepseek_v4_attention', 'vllm::hpc_rope_norm_forward', 'vllm::unified_kv_cache_update', 'vllm::unified_mla_kv_cache_update'], 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_vision_items_per_batch': 0, 'encoder_cudagraph_max_frames_per_batch': None, 'compile_sizes': [], 'compile_ranges_endpoints': [8192], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'size_asserts': False, 'alignment_asserts': False, 'scalar_asserts': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128, 136, 144, 152, 160, 168, 176, 184, 192, 200, 208, 216, 224, 232, 240, 248, 256, 272, 288, 304, 320, 336, 352, 368, 384, 400, 416, 432, 448, 464, 480, 496, 512], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': False, 'fuse_act_quant': False, 'fuse_attn_quant': False, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False, 'fuse_rope_kvcache_cat_mla': False, 'fuse_act_padding': False}, 'max_cudagraph_capture_size': 512, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': False, 'static_all_moe_layers': []}, kernel_config=KernelConfig(ir_op_priority=IrOpPriorityConfig(rms_norm=['native'], fused_add_rms_norm=['native']), enable_flashinfer_autotune=True, enable_cutedsl_warmup=True, moe_backend='auto', linear_backend='auto') +(EngineCore pid=3293) ERROR 07-23 18:03:11 [core.py:1231] EngineCore failed to start. +(EngineCore pid=3293) ERROR 07-23 18:03:11 [core.py:1231] Traceback (most recent call last): +(EngineCore pid=3293) ERROR 07-23 18:03:11 [core.py:1231] File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/engine/core.py", line 1200, in run_engine_core +(EngineCore pid=3293) ERROR 07-23 18:03:11 [core.py:1231] engine_core = EngineCoreProc(*args, engine_index=dp_rank, **kwargs) +(EngineCore pid=3293) ERROR 07-23 18:03:11 [core.py:1231] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ +(EngineCore pid=3293) ERROR 07-23 18:03:11 [core.py:1231] File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/tracing/otel.py", line 178, in sync_wrapper +(EngineCore pid=3293) ERROR 07-23 18:03:11 [core.py:1231] return func(*args, **kwargs) +(EngineCore pid=3293) ERROR 07-23 18:03:11 [core.py:1231] ^^^^^^^^^^^^^^^^^^^^^ +(EngineCore pid=3293) ERROR 07-23 18:03:11 [core.py:1231] File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/engine/core.py", line 966, in __init__ +(EngineCore pid=3293) ERROR 07-23 18:03:11 [core.py:1231] super().__init__( +(EngineCore pid=3293) ERROR 07-23 18:03:11 [core.py:1231] File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/engine/core.py", line 123, in __init__ +(EngineCore pid=3293) ERROR 07-23 18:03:11 [core.py:1231] self.model_executor = executor_class(vllm_config) +(EngineCore pid=3293) ERROR 07-23 18:03:11 [core.py:1231] ^^^^^^^^^^^^^^^^^^^^^^^^^^^ +(EngineCore pid=3293) ERROR 07-23 18:03:11 [core.py:1231] File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/tracing/otel.py", line 178, in sync_wrapper +(EngineCore pid=3293) ERROR 07-23 18:03:11 [core.py:1231] return func(*args, **kwargs) +(EngineCore pid=3293) ERROR 07-23 18:03:11 [core.py:1231] ^^^^^^^^^^^^^^^^^^^^^ +(EngineCore pid=3293) ERROR 07-23 18:03:11 [core.py:1231] File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/executor/abstract.py", line 109, in __init__ +(EngineCore pid=3293) ERROR 07-23 18:03:11 [core.py:1231] self._init_executor() +(EngineCore pid=3293) ERROR 07-23 18:03:11 [core.py:1231] File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/executor/uniproc_executor.py", line 63, in _init_executor +(EngineCore pid=3293) ERROR 07-23 18:03:11 [core.py:1231] self.driver_worker.init_device() +(EngineCore pid=3293) ERROR 07-23 18:03:11 [core.py:1231] File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/worker/worker_base.py", line 331, in init_device +(EngineCore pid=3293) ERROR 07-23 18:03:11 [core.py:1231] self.worker.init_device() # type: ignore +(EngineCore pid=3293) ERROR 07-23 18:03:11 [core.py:1231] ^^^^^^^^^^^^^^^^^^^^^^^^^ +(EngineCore pid=3293) ERROR 07-23 18:03:11 [core.py:1231] File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/tracing/otel.py", line 178, in sync_wrapper +(EngineCore pid=3293) ERROR 07-23 18:03:11 [core.py:1231] return func(*args, **kwargs) +(EngineCore pid=3293) ERROR 07-23 18:03:11 [core.py:1231] ^^^^^^^^^^^^^^^^^^^^^ +(EngineCore pid=3293) ERROR 07-23 18:03:11 [core.py:1231] File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/worker/gpu_worker.py", line 343, in init_device +(EngineCore pid=3293) ERROR 07-23 18:03:11 [core.py:1231] torch.accelerator.set_device_index(self.device) +(EngineCore pid=3293) ERROR 07-23 18:03:11 [core.py:1231] File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/torch/accelerator/__init__.py", line 191, in set_device_index +(EngineCore pid=3293) ERROR 07-23 18:03:11 [core.py:1231] torch._C._accelerator_setDeviceIndex(device_index) +(EngineCore pid=3293) ERROR 07-23 18:03:11 [core.py:1231] File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/torch/cuda/__init__.py", line 478, in _lazy_init +(EngineCore pid=3293) ERROR 07-23 18:03:11 [core.py:1231] torch._C._cuda_init() +(EngineCore pid=3293) ERROR 07-23 18:03:11 [core.py:1231] RuntimeError: The NVIDIA driver on your system is too old (found version 12040). Please update your GPU driver by downloading and installing a new version from the URL: http://www.nvidia.com/Download/index.aspx Alternatively, go to: https://pytorch.org to install a PyTorch version that has been compiled with your version of the CUDA driver. +(EngineCore pid=3293) Process EngineCore: +(EngineCore pid=3293) Traceback (most recent call last): +(EngineCore pid=3293) File "/usr/lib/python3.11/multiprocessing/process.py", line 314, in _bootstrap +(EngineCore pid=3293) self.run() +(EngineCore pid=3293) File "/usr/lib/python3.11/multiprocessing/process.py", line 108, in run +(EngineCore pid=3293) self._target(*self._args, **self._kwargs) +(EngineCore pid=3293) File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/engine/core.py", line 1235, in run_engine_core +(EngineCore pid=3293) raise e +(EngineCore pid=3293) File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/engine/core.py", line 1200, in run_engine_core +(EngineCore pid=3293) engine_core = EngineCoreProc(*args, engine_index=dp_rank, **kwargs) +(EngineCore pid=3293) ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ +(EngineCore pid=3293) File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/tracing/otel.py", line 178, in sync_wrapper +(EngineCore pid=3293) return func(*args, **kwargs) +(EngineCore pid=3293) ^^^^^^^^^^^^^^^^^^^^^ +(EngineCore pid=3293) File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/engine/core.py", line 966, in __init__ +(EngineCore pid=3293) super().__init__( +(EngineCore pid=3293) File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/engine/core.py", line 123, in __init__ +(EngineCore pid=3293) self.model_executor = executor_class(vllm_config) +(EngineCore pid=3293) ^^^^^^^^^^^^^^^^^^^^^^^^^^^ +(EngineCore pid=3293) File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/tracing/otel.py", line 178, in sync_wrapper +(EngineCore pid=3293) return func(*args, **kwargs) +(EngineCore pid=3293) ^^^^^^^^^^^^^^^^^^^^^ +(EngineCore pid=3293) File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/executor/abstract.py", line 109, in __init__ +(EngineCore pid=3293) self._init_executor() +(EngineCore pid=3293) File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/executor/uniproc_executor.py", line 63, in _init_executor +(EngineCore pid=3293) self.driver_worker.init_device() +(EngineCore pid=3293) File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/worker/worker_base.py", line 331, in init_device +(EngineCore pid=3293) self.worker.init_device() # type: ignore +(EngineCore pid=3293) ^^^^^^^^^^^^^^^^^^^^^^^^^ +(EngineCore pid=3293) File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/tracing/otel.py", line 178, in sync_wrapper +(EngineCore pid=3293) return func(*args, **kwargs) +(EngineCore pid=3293) ^^^^^^^^^^^^^^^^^^^^^ +(EngineCore pid=3293) File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/worker/gpu_worker.py", line 343, in init_device +(EngineCore pid=3293) torch.accelerator.set_device_index(self.device) +(EngineCore pid=3293) File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/torch/accelerator/__init__.py", line 191, in set_device_index +(EngineCore pid=3293) torch._C._accelerator_setDeviceIndex(device_index) +(EngineCore pid=3293) File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/torch/cuda/__init__.py", line 478, in _lazy_init +(EngineCore pid=3293) torch._C._cuda_init() +(EngineCore pid=3293) RuntimeError: The NVIDIA driver on your system is too old (found version 12040). Please update your GPU driver by downloading and installing a new version from the URL: http://www.nvidia.com/Download/index.aspx Alternatively, go to: https://pytorch.org to install a PyTorch version that has been compiled with your version of the CUDA driver. +Traceback (most recent call last): + File "", line 198, in _run_module_as_main + File "", line 88, in _run_code + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/lm_eval/__main__.py", line 14, in + cli_evaluate() + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/lm_eval/__main__.py", line 10, in cli_evaluate + parser.execute(args) + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/lm_eval/_cli/harness.py", line 60, in execute + args.func(args) + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/lm_eval/_cli/run.py", line 391, in _execute + results = simple_evaluate( + ^^^^^^^^^^^^^^^^ + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/lm_eval/utils.py", line 575, in _wrapper + return fn(*args, **kwargs) + ^^^^^^^^^^^^^^^^^^^ + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/lm_eval/evaluator.py", line 242, in simple_evaluate + lm = lm_eval.api.registry.get_model(model).create_from_arg_obj( + ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/lm_eval/api/model.py", line 169, in create_from_arg_obj + return cls(**arg_dict, **additional_config) + ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/lm_eval/models/vllm_causallms.py", line 146, in __init__ + self.model = LLM(**self.model_args) # type: ignore[invalid-argument-type] + ^^^^^^^^^^^^^^^^^^^^^^ + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/entrypoints/llm.py", line 349, in __init__ + self.llm_engine = LLMEngine.from_engine_args( + ^^^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/engine/llm_engine.py", line 179, in from_engine_args + return cls( + ^^^^ + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/engine/llm_engine.py", line 105, in __init__ + self.engine_core = EngineCoreClient.make_client( + ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/engine/core_client.py", line 103, in make_client + return SyncMPClient(vllm_config, executor_class, log_stats) + ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/tracing/otel.py", line 178, in sync_wrapper + return func(*args, **kwargs) + ^^^^^^^^^^^^^^^^^^^^^ + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/engine/core_client.py", line 786, in __init__ + super().__init__( + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/engine/core_client.py", line 573, in __init__ + with launch_core_engines( + File "/usr/lib/python3.11/contextlib.py", line 144, in __exit__ + next(self.gen) + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/engine/utils.py", line 1213, in launch_core_engines + wait_for_engine_startup( + File "/root/steering-resistance/.venv-lmeval/lib/python3.11/site-packages/vllm/v1/engine/utils.py", line 1272, in wait_for_engine_startup + raise RuntimeError( +RuntimeError: Engine core initialization failed. See root cause above. Failed core proc(s): {'EngineCore': 1} + + +########## m0 / mmlu ########## +/usr/bin/python -m lm_eval --model hf --model_args pretrained=Qwen/Qwen2.5-3B-Instruct,dtype=bfloat16 --tasks mmlu --num_fewshot 5 --batch_size auto --seed 0 --output_path /root/steering-resistance/results/full_3b/capability/m0/mmlu --apply_chat_template --fewshot_as_multiturn --limit 15 +/usr/local/lib/python3.11/dist-packages/requests/__init__.py:113: RequestsDependencyWarning: urllib3 (2.2.3) or chardet (6.0.0.post1)/charset_normalizer (3.3.2) doesn't match a supported version! + warnings.warn( +2026-07-23:18:08:36 WARNING [config.evaluate_config:287] --limit SHOULD ONLY BE USED FOR TESTING. REAL METRICS SHOULD NOT BE COMPUTED USING LIMIT. +2026-07-23:18:08:44 INFO [_cli.run:388] Selected Tasks: ['mmlu'] +2026-07-23:18:08:45 INFO [evaluator:214] Setting random seed to 0 | Setting numpy seed to 0 | Setting torch manual seed to 0 | Setting fewshot manual seed to 0 +2026-07-23:18:08:45 INFO [evaluator:239] Initializing hf model, with arguments: {'pretrained': 'Qwen/Qwen2.5-3B-Instruct', 'dtype': 'bfloat16'} +2026-07-23:18:08:47 INFO [models.huggingface:286] Using device 'cuda:0' +2026-07-23:18:08:49 INFO [models.huggingface:579] Model parallel was set to False, max memory was not set, and device map was set to {'': 'cuda:0'} + Loading weights: 0%| | 0/434 [00:00', '<|im_end|>']} +2026-07-23:18:16:18 WARNING [evaluator:333] Overwriting default num_fewshot of gsm8k_cot from 8 to 8 +2026-07-23:18:16:18 INFO [api.task:312] Building contexts for gsm8k_cot on rank 0... + 0%| | 0/200 [00:00', '<|im_end|>']} +2026-07-23:18:42:07 WARNING [evaluator:333] Overwriting default num_fewshot of gsm8k_cot from 8 to 8 +2026-07-23:18:42:07 INFO [api.task:312] Building contexts for gsm8k_cot on rank 0... + 0%| | 0/200 [00:00", + "<|im_end|>" + ] + }, + "repeats": 1, + "filter_list": [ + { + "filter": [ + { + "function": "regex", + "regex_pattern": "The answer is (\\-?[0-9\\.\\,]+)." + }, + { + "function": "take_first" + } + ], + "name": "strict-match" + }, + { + "filter": [ + { + "function": "regex", + "group_select": -1, + "regex_pattern": "(-?[$0-9.,]{2,})|(-?[0-9]+)" + }, + { + "function": "take_first" + } + ], + "name": "flexible-extract" + } + ], + "should_decontaminate": false, + "metadata": { + "version": 3.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/gsm8k/gsm8k-cot.yaml" + } + } + }, + "versions": { + "gsm8k_cot": 3.0 + }, + "n-shot": { + "gsm8k_cot": 8 + }, + "higher_is_better": { + "gsm8k_cot": { + "exact_match": true + } + }, + "n-samples": { + "gsm8k_cot": { + "original": 1319, + "effective": 200 + } + }, + "config": { + "model": "hf", + "model_args": { + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16" + }, + "model_num_parameters": 3085938688, + "model_dtype": "torch.bfloat16", + "model_revision": "main", + "model_sha": "aa8e72537993ba99e69dfaafa59ed015b17504d1", + "batch_size": "auto", + "batch_sizes": [], + "device": "cuda:0", + "use_cache": null, + "limit": 200.0, + "bootstrap_iters": 100000, + "gen_kwargs": { + "do_sample": false + }, + "random_seed": 0, + "numpy_seed": 0, + "torch_seed": 0, + "fewshot_seed": 0 + }, + "git_hash": "eb4f2be22f7baf6d268c3dd5e46d49d6bd2e74ae", + "date": 1784830566.9120035, + "pretty_env_info": "PyTorch version: 2.6.0+cu124\nIs debug build: False\nCUDA used to build PyTorch: 12.4\nROCM used to build PyTorch: N/A\n\nOS: Ubuntu 22.04.5 LTS (x86_64)\nGCC version: (Ubuntu 11.4.0-1ubuntu1~22.04) 11.4.0\nClang version: Could not collect\nCMake version: Could not collect\nLibc version: glibc-2.35\n\nPython version: 3.11.10 (main, Sep 7 2024, 18:35:41) [GCC 11.4.0] (64-bit runtime)\nPython platform: Linux-6.8.0-52-generic-x86_64-with-glibc2.35\nIs CUDA available: True\nCUDA runtime version: Could not collect\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: GPU 0: NVIDIA RTX A4000\nNvidia driver version: 550.144.03\ncuDNN version: Could not collect\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 48 bits physical, 48 bits virtual\nByte Order: Little Endian\nCPU(s): 112\nOn-line CPU(s) list: 0-111\nVendor ID: AuthenticAMD\nModel name: AMD EPYC 7453 28-Core Processor\nCPU family: 25\nModel: 1\nThread(s) per core: 2\nCore(s) per socket: 28\nSocket(s): 2\nStepping: 1\nFrequency boost: enabled\nCPU max MHz: 3488.5249\nCPU min MHz: 1500.0000\nBogoMIPS: 5489.75\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush mmx fxsr sse sse2 ht syscall nx mmxext fxsr_opt pdpe1gb rdtscp lm constant_tsc rep_good nopl nonstop_tsc cpuid extd_apicid aperfmperf rapl pni pclmulqdq monitor ssse3 fma cx16 pcid sse4_1 sse4_2 movbe popcnt aes xsave avx f16c rdrand lahf_lm cmp_legacy svm extapic cr8_legacy abm sse4a misalignsse 3dnowprefetch osvw ibs skinit wdt tce topoext perfctr_core perfctr_nb bpext perfctr_llc mwaitx cpb cat_l3 cdp_l3 hw_pstate ssbd mba ibrs ibpb stibp vmmcall fsgsbase bmi1 avx2 smep bmi2 erms invpcid cqm rdt_a rdseed adx smap clflushopt clwb sha_ni xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local user_shstk clzero irperf xsaveerptr rdpru wbnoinvd amd_ppin brs arat npt lbrv svm_lock nrip_save tsc_scale vmcb_clean flushbyasid decodeassists pausefilter pfthreshold v_vmsave_vmload vgif v_spec_ctrl umip pku ospke vaes vpclmulqdq rdpid overflow_recov succor smca fsrm debug_swap\nVirtualization: AMD-V\nL1d cache: 1.8 MiB (56 instances)\nL1i cache: 1.8 MiB (56 instances)\nL2 cache: 28 MiB (56 instances)\nL3 cache: 128 MiB (8 instances)\nNUMA node(s): 2\nNUMA node0 CPU(s): 0-27,56-83\nNUMA node1 CPU(s): 28-55,84-111\nVulnerability Gather data sampling: Not affected\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Mmio stale data: Not affected\nVulnerability Reg file data sampling: Not affected\nVulnerability Retbleed: Not affected\nVulnerability Spec rstack overflow: Mitigation; Safe RET\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Retpolines; IBPB conditional; IBRS_FW; STIBP always-on; RSB filling; PBRSB-eIBRS Not affected; BHI Not affected\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] numpy==2.4.6\n[pip3] nvidia-cublas-cu12==12.4.5.8\n[pip3] nvidia-cuda-cupti-cu12==12.4.127\n[pip3] nvidia-cuda-nvrtc-cu12==12.4.127\n[pip3] nvidia-cuda-runtime-cu12==12.4.127\n[pip3] nvidia-cudnn-cu12==9.1.0.70\n[pip3] nvidia-cufft-cu12==11.2.1.3\n[pip3] nvidia-curand-cu12==10.3.5.147\n[pip3] nvidia-cusolver-cu12==11.6.1.9\n[pip3] nvidia-cusparse-cu12==12.3.1.170\n[pip3] nvidia-cusparselt-cu12==0.6.2\n[pip3] nvidia-nccl-cu12==2.21.5\n[pip3] nvidia-nvjitlink-cu12==12.4.127\n[pip3] nvidia-nvtx-cu12==12.4.127\n[pip3] torch==2.6.0+cu124\n[pip3] triton==3.2.0\n[conda] Could not collect", + "transformers_version": "5.14.1", + "lm_eval_version": "0.4.12", + "upper_git_hash": null, + "tokenizer_pad_token": [ + "<|endoftext|>", + "151643" + ], + "tokenizer_eos_token": [ + "<|im_end|>", + "151645" + ], + "tokenizer_bos_token": [ + null, + "None" + ], + "eot_token_id": 151645, + "max_length": 32768, + "task_hashes": {}, + "model_source": "hf", + "model_name": "Qwen/Qwen2.5-3B-Instruct", + "model_name_sanitized": "Qwen__Qwen2.5-3B-Instruct", + "system_instruction": null, + "system_instruction_sha": null, + "fewshot_as_multiturn": true, + "chat_template": "{%- if tools %}\n {{- '<|im_start|>system\\n' }}\n {%- if messages[0]['role'] == 'system' %}\n {{- messages[0]['content'] }}\n {%- else %}\n {{- 'You are Qwen, created by Alibaba Cloud. You are a helpful assistant.' }}\n {%- endif %}\n {{- \"\\n\\n# Tools\\n\\nYou may call one or more functions to assist with the user query.\\n\\nYou are provided with function signatures within XML tags:\\n\" }}\n {%- for tool in tools %}\n {{- \"\\n\" }}\n {{- tool | tojson }}\n {%- endfor %}\n {{- \"\\n\\n\\nFor each function call, return a json object with function name and arguments within XML tags:\\n\\n{\\\"name\\\": , \\\"arguments\\\": }\\n<|im_end|>\\n\" }}\n{%- else %}\n {%- if messages[0]['role'] == 'system' %}\n {{- '<|im_start|>system\\n' + messages[0]['content'] + '<|im_end|>\\n' }}\n {%- else %}\n {{- '<|im_start|>system\\nYou are Qwen, created by Alibaba Cloud. You are a helpful assistant.<|im_end|>\\n' }}\n {%- endif %}\n{%- endif %}\n{%- for message in messages %}\n {%- if (message.role == \"user\") or (message.role == \"system\" and not loop.first) or (message.role == \"assistant\" and not message.tool_calls) %}\n {{- '<|im_start|>' + message.role + '\\n' + message.content + '<|im_end|>' + '\\n' }}\n {%- elif message.role == \"assistant\" %}\n {{- '<|im_start|>' + message.role }}\n {%- if message.content %}\n {{- '\\n' + message.content }}\n {%- endif %}\n {%- for tool_call in message.tool_calls %}\n {%- if tool_call.function is defined %}\n {%- set tool_call = tool_call.function %}\n {%- endif %}\n {{- '\\n\\n{\"name\": \"' }}\n {{- tool_call.name }}\n {{- '\", \"arguments\": ' }}\n {{- tool_call.arguments | tojson }}\n {{- '}\\n' }}\n {%- endfor %}\n {{- '<|im_end|>\\n' }}\n {%- elif message.role == \"tool\" %}\n {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != \"tool\") %}\n {{- '<|im_start|>user' }}\n {%- endif %}\n {{- '\\n\\n' }}\n {{- message.content }}\n {{- '\\n' }}\n {%- if loop.last or (messages[loop.index0 + 1].role != \"tool\") %}\n {{- '<|im_end|>\\n' }}\n {%- endif %}\n {%- endif %}\n{%- endfor %}\n{%- if add_generation_prompt %}\n {{- '<|im_start|>assistant\\n' }}\n{%- endif %}\n", + "chat_template_sha": "cd8e9439f0570856fd70470bf8889ebd8b5d1107207f67a5efb46e342330527f", + "total_evaluation_time_seconds": "1229.2357175983489" +} \ No newline at end of file diff --git a/run/capability/m0/mmlu/Qwen__Qwen2.5-3B-Instruct/results_2026-07-23T18-15-57.670538.json b/run/capability/m0/mmlu/Qwen__Qwen2.5-3B-Instruct/results_2026-07-23T18-15-57.670538.json new file mode 100644 index 0000000000000000000000000000000000000000..36ae203ea6022f7fb263f8ad08516f2d6567a224 --- /dev/null +++ b/run/capability/m0/mmlu/Qwen__Qwen2.5-3B-Instruct/results_2026-07-23T18-15-57.670538.json @@ -0,0 +1,4311 @@ +{ + "results": { + "mmlu_abstract_algebra": { + "name": "mmlu_abstract_algebra", + "alias": "abstract_algebra", + "sample_len": 15, + "acc,none": 0.4, + "acc_stderr,none": 0.13093073414159542 + }, + "mmlu_anatomy": { + "name": "mmlu_anatomy", + "alias": "anatomy", + "sample_len": 15, + "acc,none": 0.5333333333333333, + "acc_stderr,none": 0.1333333333333333 + }, + "mmlu_astronomy": { + "name": "mmlu_astronomy", + "alias": "astronomy", + "sample_len": 15, + "acc,none": 0.8, + "acc_stderr,none": 0.10690449676496976 + }, + "mmlu_college_biology": { + "name": "mmlu_college_biology", + "alias": "college_biology", + "sample_len": 15, + "acc,none": 0.8666666666666667, + "acc_stderr,none": 0.09085135251589957 + }, + "mmlu_college_chemistry": { + "name": "mmlu_college_chemistry", + "alias": "college_chemistry", + "sample_len": 15, + "acc,none": 0.26666666666666666, + "acc_stderr,none": 0.11818736805705578 + }, + "mmlu_college_computer_science": { + "name": "mmlu_college_computer_science", + "alias": "college_computer_science", + "sample_len": 15, + "acc,none": 0.4666666666666667, + "acc_stderr,none": 0.1333333333333333 + }, + "mmlu_college_mathematics": { + "name": "mmlu_college_mathematics", + "alias": "college_mathematics", + "sample_len": 15, + "acc,none": 0.3333333333333333, + "acc_stderr,none": 0.12598815766974242 + }, + "mmlu_college_physics": { + "name": "mmlu_college_physics", + "alias": "college_physics", + "sample_len": 15, + "acc,none": 0.5333333333333333, + "acc_stderr,none": 0.1333333333333333 + }, + "mmlu_computer_security": { + "name": "mmlu_computer_security", + "alias": "computer_security", + "sample_len": 15, + "acc,none": 0.6, + "acc_stderr,none": 0.13093073414159542 + }, + "mmlu_conceptual_physics": { + "name": "mmlu_conceptual_physics", + "alias": "conceptual_physics", + "sample_len": 15, + "acc,none": 0.9333333333333333, + "acc_stderr,none": 0.06666666666666667 + }, + "mmlu_electrical_engineering": { + "name": "mmlu_electrical_engineering", + "alias": "electrical_engineering", + "sample_len": 15, + "acc,none": 0.6666666666666666, + "acc_stderr,none": 0.12598815766974242 + }, + "mmlu_elementary_mathematics": { + "name": "mmlu_elementary_mathematics", + "alias": "elementary_mathematics", + "sample_len": 15, + "acc,none": 0.7333333333333333, + "acc_stderr,none": 0.11818736805705575 + }, + "mmlu_high_school_biology": { + "name": "mmlu_high_school_biology", + "alias": "high_school_biology", + "sample_len": 15, + "acc,none": 0.8666666666666667, + "acc_stderr,none": 0.09085135251589956 + }, + "mmlu_high_school_chemistry": { + "name": "mmlu_high_school_chemistry", + "alias": "high_school_chemistry", + "sample_len": 15, + "acc,none": 0.8, + "acc_stderr,none": 0.10690449676496977 + }, + "mmlu_high_school_computer_science": { + "name": "mmlu_high_school_computer_science", + "alias": "high_school_computer_science", + "sample_len": 15, + "acc,none": 0.7333333333333333, + "acc_stderr,none": 0.11818736805705575 + }, + "mmlu_high_school_mathematics": { + "name": "mmlu_high_school_mathematics", + "alias": "high_school_mathematics", + "sample_len": 15, + "acc,none": 0.4, + "acc_stderr,none": 0.13093073414159542 + }, + "mmlu_high_school_physics": { + "name": "mmlu_high_school_physics", + "alias": "high_school_physics", + "sample_len": 15, + "acc,none": 0.4, + "acc_stderr,none": 0.13093073414159542 + }, + "mmlu_high_school_statistics": { + "name": "mmlu_high_school_statistics", + "alias": "high_school_statistics", + "sample_len": 15, + "acc,none": 0.7333333333333333, + "acc_stderr,none": 0.11818736805705575 + }, + "mmlu_machine_learning": { + "name": "mmlu_machine_learning", + "alias": "machine_learning", + "sample_len": 15, + "acc,none": 0.3333333333333333, + "acc_stderr,none": 0.12598815766974242 + }, + "mmlu_business_ethics": { + "name": "mmlu_business_ethics", + "alias": "business_ethics", + "sample_len": 15, + "acc,none": 0.6666666666666666, + "acc_stderr,none": 0.12598815766974242 + }, + "mmlu_clinical_knowledge": { + "name": "mmlu_clinical_knowledge", + "alias": "clinical_knowledge", + "sample_len": 15, + "acc,none": 0.6, + "acc_stderr,none": 0.13093073414159542 + }, + "mmlu_college_medicine": { + "name": "mmlu_college_medicine", + "alias": "college_medicine", + "sample_len": 15, + "acc,none": 0.6666666666666666, + "acc_stderr,none": 0.12598815766974242 + }, + "mmlu_global_facts": { + "name": "mmlu_global_facts", + "alias": "global_facts", + "sample_len": 15, + "acc,none": 0.26666666666666666, + "acc_stderr,none": 0.11818736805705578 + }, + "mmlu_human_aging": { + "name": "mmlu_human_aging", + "alias": "human_aging", + "sample_len": 15, + "acc,none": 0.7333333333333333, + "acc_stderr,none": 0.11818736805705576 + }, + "mmlu_management": { + "name": "mmlu_management", + "alias": "management", + "sample_len": 15, + "acc,none": 0.6666666666666666, + "acc_stderr,none": 0.12598815766974242 + }, + "mmlu_marketing": { + "name": "mmlu_marketing", + "alias": "marketing", + "sample_len": 15, + "acc,none": 0.6666666666666666, + "acc_stderr,none": 0.12598815766974242 + }, + "mmlu_medical_genetics": { + "name": "mmlu_medical_genetics", + "alias": "medical_genetics", + "sample_len": 15, + "acc,none": 1.0, + "acc_stderr,none": 0.0 + }, + "mmlu_miscellaneous": { + "name": "mmlu_miscellaneous", + "alias": "miscellaneous", + "sample_len": 15, + "acc,none": 0.8666666666666667, + "acc_stderr,none": 0.09085135251589958 + }, + "mmlu_nutrition": { + "name": "mmlu_nutrition", + "alias": "nutrition", + "sample_len": 15, + "acc,none": 0.8666666666666667, + "acc_stderr,none": 0.09085135251589957 + }, + "mmlu_professional_accounting": { + "name": "mmlu_professional_accounting", + "alias": "professional_accounting", + "sample_len": 15, + "acc,none": 0.5333333333333333, + "acc_stderr,none": 0.1333333333333333 + }, + "mmlu_professional_medicine": { + "name": "mmlu_professional_medicine", + "alias": "professional_medicine", + "sample_len": 15, + "acc,none": 0.6666666666666666, + "acc_stderr,none": 0.12598815766974242 + }, + "mmlu_virology": { + "name": "mmlu_virology", + "alias": "virology", + "sample_len": 15, + "acc,none": 0.6, + "acc_stderr,none": 0.13093073414159542 + }, + "mmlu_econometrics": { + "name": "mmlu_econometrics", + "alias": "econometrics", + "sample_len": 15, + "acc,none": 0.5333333333333333, + "acc_stderr,none": 0.1333333333333333 + }, + "mmlu_high_school_geography": { + "name": "mmlu_high_school_geography", + "alias": "high_school_geography", + "sample_len": 15, + "acc,none": 0.7333333333333333, + "acc_stderr,none": 0.11818736805705576 + }, + "mmlu_high_school_government_and_politics": { + "name": "mmlu_high_school_government_and_politics", + "alias": "high_school_government_and_politics", + "sample_len": 15, + "acc,none": 0.8666666666666667, + "acc_stderr,none": 0.09085135251589957 + }, + "mmlu_high_school_macroeconomics": { + "name": "mmlu_high_school_macroeconomics", + "alias": "high_school_macroeconomics", + "sample_len": 15, + "acc,none": 0.7333333333333333, + "acc_stderr,none": 0.11818736805705576 + }, + "mmlu_high_school_microeconomics": { + "name": "mmlu_high_school_microeconomics", + "alias": "high_school_microeconomics", + "sample_len": 15, + "acc,none": 0.7333333333333333, + "acc_stderr,none": 0.11818736805705576 + }, + "mmlu_high_school_psychology": { + "name": "mmlu_high_school_psychology", + "alias": "high_school_psychology", + "sample_len": 15, + "acc,none": 0.9333333333333333, + "acc_stderr,none": 0.06666666666666667 + }, + "mmlu_human_sexuality": { + "name": "mmlu_human_sexuality", + "alias": "human_sexuality", + "sample_len": 15, + "acc,none": 0.7333333333333333, + "acc_stderr,none": 0.11818736805705576 + }, + "mmlu_professional_psychology": { + "name": "mmlu_professional_psychology", + "alias": "professional_psychology", + "sample_len": 15, + "acc,none": 0.9333333333333333, + "acc_stderr,none": 0.06666666666666667 + }, + "mmlu_public_relations": { + "name": "mmlu_public_relations", + "alias": "public_relations", + "sample_len": 15, + "acc,none": 0.5333333333333333, + "acc_stderr,none": 0.1333333333333333 + }, + "mmlu_security_studies": { + "name": "mmlu_security_studies", + "alias": "security_studies", + "sample_len": 15, + "acc,none": 0.9333333333333333, + "acc_stderr,none": 0.06666666666666667 + }, + "mmlu_sociology": { + "name": "mmlu_sociology", + "alias": "sociology", + "sample_len": 15, + "acc,none": 0.8, + "acc_stderr,none": 0.10690449676496976 + }, + "mmlu_us_foreign_policy": { + "name": "mmlu_us_foreign_policy", + "alias": "us_foreign_policy", + "sample_len": 15, + "acc,none": 0.8666666666666667, + "acc_stderr,none": 0.09085135251589958 + }, + "mmlu_formal_logic": { + "name": "mmlu_formal_logic", + "alias": "formal_logic", + "sample_len": 15, + "acc,none": 0.3333333333333333, + "acc_stderr,none": 0.12598815766974242 + }, + "mmlu_high_school_european_history": { + "name": "mmlu_high_school_european_history", + "alias": "high_school_european_history", + "sample_len": 15, + "acc,none": 0.7333333333333333, + "acc_stderr,none": 0.11818736805705575 + }, + "mmlu_high_school_us_history": { + "name": "mmlu_high_school_us_history", + "alias": "high_school_us_history", + "sample_len": 15, + "acc,none": 0.8, + "acc_stderr,none": 0.10690449676496976 + }, + "mmlu_high_school_world_history": { + "name": "mmlu_high_school_world_history", + "alias": "high_school_world_history", + "sample_len": 15, + "acc,none": 0.9333333333333333, + "acc_stderr,none": 0.06666666666666667 + }, + "mmlu_international_law": { + "name": "mmlu_international_law", + "alias": "international_law", + "sample_len": 15, + "acc,none": 0.8666666666666667, + "acc_stderr,none": 0.09085135251589958 + }, + "mmlu_jurisprudence": { + "name": "mmlu_jurisprudence", + "alias": "jurisprudence", + "sample_len": 15, + "acc,none": 0.8, + "acc_stderr,none": 0.10690449676496977 + }, + "mmlu_logical_fallacies": { + "name": "mmlu_logical_fallacies", + "alias": "logical_fallacies", + "sample_len": 15, + "acc,none": 0.9333333333333333, + "acc_stderr,none": 0.06666666666666667 + }, + "mmlu_moral_disputes": { + "name": "mmlu_moral_disputes", + "alias": "moral_disputes", + "sample_len": 15, + "acc,none": 0.4666666666666667, + "acc_stderr,none": 0.1333333333333333 + }, + "mmlu_moral_scenarios": { + "name": "mmlu_moral_scenarios", + "alias": "moral_scenarios", + "sample_len": 15, + "acc,none": 0.3333333333333333, + "acc_stderr,none": 0.12598815766974242 + }, + "mmlu_philosophy": { + "name": "mmlu_philosophy", + "alias": "philosophy", + "sample_len": 15, + "acc,none": 0.7333333333333333, + "acc_stderr,none": 0.11818736805705576 + }, + "mmlu_prehistory": { + "name": "mmlu_prehistory", + "alias": "prehistory", + "sample_len": 15, + "acc,none": 0.6666666666666666, + "acc_stderr,none": 0.12598815766974242 + }, + "mmlu_professional_law": { + "name": "mmlu_professional_law", + "alias": "professional_law", + "sample_len": 15, + "acc,none": 0.6, + "acc_stderr,none": 0.13093073414159542 + }, + "mmlu_world_religions": { + "name": "mmlu_world_religions", + "alias": "world_religions", + "sample_len": 15, + "acc,none": 0.8666666666666667, + "acc_stderr,none": 0.09085135251589957 + }, + "mmlu_stem": { + "alias": "stem", + "name": "mmlu_stem", + "sample_len": 285, + "acc,none": 0.6, + "acc_stderr,none": 0.027307946810779415, + "sample_count": { + "acc,none": 285 + } + }, + "mmlu_other": { + "alias": "other", + "name": "mmlu_other", + "sample_len": 195, + "acc,none": 0.676923076923077, + "acc_stderr,none": 0.03225939759165421, + "sample_count": { + "acc,none": 195 + } + }, + "mmlu_social_sciences": { + "alias": "social sciences", + "name": "mmlu_social_sciences", + "sample_len": 180, + "acc,none": 0.7777777777777778, + "acc_stderr,none": 0.030356494442706786, + "sample_count": { + "acc,none": 180 + } + }, + "mmlu_humanities": { + "alias": "humanities", + "name": "mmlu_humanities", + "sample_len": 195, + "acc,none": 0.6974358974358974, + "acc_stderr,none": 0.030646887418830607, + "sample_count": { + "acc,none": 195 + } + }, + "mmlu": { + "alias": "mmlu", + "name": "mmlu", + "sample_len": 855, + "acc,none": 0.6771929824561403, + "acc_stderr,none": 0.01505614747010033, + "sample_count": { + "acc,none": 855 + } + } + }, + "groups": { + "mmlu_stem": { + "alias": "stem", + "name": "mmlu_stem", + "sample_len": 285, + "acc,none": 0.6, + "acc_stderr,none": 0.027307946810779415, + "sample_count": { + "acc,none": 285 + } + }, + "mmlu_other": { + "alias": "other", + "name": "mmlu_other", + "sample_len": 195, + "acc,none": 0.676923076923077, + "acc_stderr,none": 0.03225939759165421, + "sample_count": { + "acc,none": 195 + } + }, + "mmlu_social_sciences": { + "alias": "social sciences", + "name": "mmlu_social_sciences", + "sample_len": 180, + "acc,none": 0.7777777777777778, + "acc_stderr,none": 0.030356494442706786, + "sample_count": { + "acc,none": 180 + } + }, + "mmlu_humanities": { + "alias": "humanities", + "name": "mmlu_humanities", + "sample_len": 195, + "acc,none": 0.6974358974358974, + "acc_stderr,none": 0.030646887418830607, + "sample_count": { + "acc,none": 195 + } + }, + "mmlu": { + "alias": "mmlu", + "name": "mmlu", + "sample_len": 855, + "acc,none": 0.6771929824561403, + "acc_stderr,none": 0.01505614747010033, + "sample_count": { + "acc,none": 855 + } + } + }, + "group_subtasks": { + "mmlu": [ + "mmlu_stem", + "mmlu_other", + "mmlu_social_sciences", + "mmlu_humanities" + ], + "mmlu_stem": [ + "mmlu_abstract_algebra", + "mmlu_anatomy", + "mmlu_astronomy", + "mmlu_college_biology", + "mmlu_college_chemistry", + "mmlu_college_computer_science", + "mmlu_college_mathematics", + "mmlu_college_physics", + "mmlu_computer_security", + "mmlu_conceptual_physics", + "mmlu_electrical_engineering", + "mmlu_elementary_mathematics", + "mmlu_high_school_biology", + "mmlu_high_school_chemistry", + "mmlu_high_school_computer_science", + "mmlu_high_school_mathematics", + "mmlu_high_school_physics", + "mmlu_high_school_statistics", + "mmlu_machine_learning" + ], + "mmlu_other": [ + "mmlu_business_ethics", + "mmlu_clinical_knowledge", + "mmlu_college_medicine", + "mmlu_global_facts", + "mmlu_human_aging", + "mmlu_management", + "mmlu_marketing", + "mmlu_medical_genetics", + "mmlu_miscellaneous", + "mmlu_nutrition", + "mmlu_professional_accounting", + "mmlu_professional_medicine", + "mmlu_virology" + ], + "mmlu_social_sciences": [ + "mmlu_econometrics", + "mmlu_high_school_geography", + "mmlu_high_school_government_and_politics", + "mmlu_high_school_macroeconomics", + "mmlu_high_school_microeconomics", + "mmlu_high_school_psychology", + "mmlu_human_sexuality", + "mmlu_professional_psychology", + "mmlu_public_relations", + "mmlu_security_studies", + "mmlu_sociology", + "mmlu_us_foreign_policy" + ], + "mmlu_humanities": [ + "mmlu_formal_logic", + "mmlu_high_school_european_history", + "mmlu_high_school_us_history", + "mmlu_high_school_world_history", + "mmlu_international_law", + "mmlu_jurisprudence", + "mmlu_logical_fallacies", + "mmlu_moral_disputes", + "mmlu_moral_scenarios", + "mmlu_philosophy", + "mmlu_prehistory", + "mmlu_professional_law", + "mmlu_world_religions" + ] + }, + "configs": { + "mmlu_abstract_algebra": { + "task": "mmlu_abstract_algebra", + "task_alias": "abstract_algebra", + "dataset_path": "cais/mmlu", + "dataset_name": "abstract_algebra", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about abstract algebra.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_abstract_algebra.yaml" + } + }, + "mmlu_anatomy": { + "task": "mmlu_anatomy", + "task_alias": "anatomy", + "dataset_path": "cais/mmlu", + "dataset_name": "anatomy", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about anatomy.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_anatomy.yaml" + } + }, + "mmlu_astronomy": { + "task": "mmlu_astronomy", + "task_alias": "astronomy", + "dataset_path": "cais/mmlu", + "dataset_name": "astronomy", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about astronomy.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_astronomy.yaml" + } + }, + "mmlu_business_ethics": { + "task": "mmlu_business_ethics", + "task_alias": "business_ethics", + "dataset_path": "cais/mmlu", + "dataset_name": "business_ethics", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about business ethics.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_business_ethics.yaml" + } + }, + "mmlu_clinical_knowledge": { + "task": "mmlu_clinical_knowledge", + "task_alias": "clinical_knowledge", + "dataset_path": "cais/mmlu", + "dataset_name": "clinical_knowledge", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about clinical knowledge.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_clinical_knowledge.yaml" + } + }, + "mmlu_college_biology": { + "task": "mmlu_college_biology", + "task_alias": "college_biology", + "dataset_path": "cais/mmlu", + "dataset_name": "college_biology", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about college biology.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_college_biology.yaml" + } + }, + "mmlu_college_chemistry": { + "task": "mmlu_college_chemistry", + "task_alias": "college_chemistry", + "dataset_path": "cais/mmlu", + "dataset_name": "college_chemistry", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about college chemistry.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_college_chemistry.yaml" + } + }, + "mmlu_college_computer_science": { + "task": "mmlu_college_computer_science", + "task_alias": "college_computer_science", + "dataset_path": "cais/mmlu", + "dataset_name": "college_computer_science", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about college computer science.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_college_computer_science.yaml" + } + }, + "mmlu_college_mathematics": { + "task": "mmlu_college_mathematics", + "task_alias": "college_mathematics", + "dataset_path": "cais/mmlu", + "dataset_name": "college_mathematics", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about college mathematics.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_college_mathematics.yaml" + } + }, + "mmlu_college_medicine": { + "task": "mmlu_college_medicine", + "task_alias": "college_medicine", + "dataset_path": "cais/mmlu", + "dataset_name": "college_medicine", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about college medicine.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_college_medicine.yaml" + } + }, + "mmlu_college_physics": { + "task": "mmlu_college_physics", + "task_alias": "college_physics", + "dataset_path": "cais/mmlu", + "dataset_name": "college_physics", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about college physics.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_college_physics.yaml" + } + }, + "mmlu_computer_security": { + "task": "mmlu_computer_security", + "task_alias": "computer_security", + "dataset_path": "cais/mmlu", + "dataset_name": "computer_security", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about computer security.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_computer_security.yaml" + } + }, + "mmlu_conceptual_physics": { + "task": "mmlu_conceptual_physics", + "task_alias": "conceptual_physics", + "dataset_path": "cais/mmlu", + "dataset_name": "conceptual_physics", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about conceptual physics.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_conceptual_physics.yaml" + } + }, + "mmlu_econometrics": { + "task": "mmlu_econometrics", + "task_alias": "econometrics", + "dataset_path": "cais/mmlu", + "dataset_name": "econometrics", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about econometrics.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_econometrics.yaml" + } + }, + "mmlu_electrical_engineering": { + "task": "mmlu_electrical_engineering", + "task_alias": "electrical_engineering", + "dataset_path": "cais/mmlu", + "dataset_name": "electrical_engineering", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about electrical engineering.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_electrical_engineering.yaml" + } + }, + "mmlu_elementary_mathematics": { + "task": "mmlu_elementary_mathematics", + "task_alias": "elementary_mathematics", + "dataset_path": "cais/mmlu", + "dataset_name": "elementary_mathematics", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about elementary mathematics.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_elementary_mathematics.yaml" + } + }, + "mmlu_formal_logic": { + "task": "mmlu_formal_logic", + "task_alias": "formal_logic", + "dataset_path": "cais/mmlu", + "dataset_name": "formal_logic", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about formal logic.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_formal_logic.yaml" + } + }, + "mmlu_global_facts": { + "task": "mmlu_global_facts", + "task_alias": "global_facts", + "dataset_path": "cais/mmlu", + "dataset_name": "global_facts", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about global facts.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_global_facts.yaml" + } + }, + "mmlu_high_school_biology": { + "task": "mmlu_high_school_biology", + "task_alias": "high_school_biology", + "dataset_path": "cais/mmlu", + "dataset_name": "high_school_biology", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about high school biology.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_high_school_biology.yaml" + } + }, + "mmlu_high_school_chemistry": { + "task": "mmlu_high_school_chemistry", + "task_alias": "high_school_chemistry", + "dataset_path": "cais/mmlu", + "dataset_name": "high_school_chemistry", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about high school chemistry.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_high_school_chemistry.yaml" + } + }, + "mmlu_high_school_computer_science": { + "task": "mmlu_high_school_computer_science", + "task_alias": "high_school_computer_science", + "dataset_path": "cais/mmlu", + "dataset_name": "high_school_computer_science", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about high school computer science.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_high_school_computer_science.yaml" + } + }, + "mmlu_high_school_european_history": { + "task": "mmlu_high_school_european_history", + "task_alias": "high_school_european_history", + "dataset_path": "cais/mmlu", + "dataset_name": "high_school_european_history", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about high school european history.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_high_school_european_history.yaml" + } + }, + "mmlu_high_school_geography": { + "task": "mmlu_high_school_geography", + "task_alias": "high_school_geography", + "dataset_path": "cais/mmlu", + "dataset_name": "high_school_geography", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about high school geography.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_high_school_geography.yaml" + } + }, + "mmlu_high_school_government_and_politics": { + "task": "mmlu_high_school_government_and_politics", + "task_alias": "high_school_government_and_politics", + "dataset_path": "cais/mmlu", + "dataset_name": "high_school_government_and_politics", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about high school government and politics.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_high_school_government_and_politics.yaml" + } + }, + "mmlu_high_school_macroeconomics": { + "task": "mmlu_high_school_macroeconomics", + "task_alias": "high_school_macroeconomics", + "dataset_path": "cais/mmlu", + "dataset_name": "high_school_macroeconomics", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about high school macroeconomics.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_high_school_macroeconomics.yaml" + } + }, + "mmlu_high_school_mathematics": { + "task": "mmlu_high_school_mathematics", + "task_alias": "high_school_mathematics", + "dataset_path": "cais/mmlu", + "dataset_name": "high_school_mathematics", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about high school mathematics.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_high_school_mathematics.yaml" + } + }, + "mmlu_high_school_microeconomics": { + "task": "mmlu_high_school_microeconomics", + "task_alias": "high_school_microeconomics", + "dataset_path": "cais/mmlu", + "dataset_name": "high_school_microeconomics", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about high school microeconomics.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_high_school_microeconomics.yaml" + } + }, + "mmlu_high_school_physics": { + "task": "mmlu_high_school_physics", + "task_alias": "high_school_physics", + "dataset_path": "cais/mmlu", + "dataset_name": "high_school_physics", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about high school physics.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_high_school_physics.yaml" + } + }, + "mmlu_high_school_psychology": { + "task": "mmlu_high_school_psychology", + "task_alias": "high_school_psychology", + "dataset_path": "cais/mmlu", + "dataset_name": "high_school_psychology", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about high school psychology.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_high_school_psychology.yaml" + } + }, + "mmlu_high_school_statistics": { + "task": "mmlu_high_school_statistics", + "task_alias": "high_school_statistics", + "dataset_path": "cais/mmlu", + "dataset_name": "high_school_statistics", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about high school statistics.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_high_school_statistics.yaml" + } + }, + "mmlu_high_school_us_history": { + "task": "mmlu_high_school_us_history", + "task_alias": "high_school_us_history", + "dataset_path": "cais/mmlu", + "dataset_name": "high_school_us_history", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about high school us history.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_high_school_us_history.yaml" + } + }, + "mmlu_high_school_world_history": { + "task": "mmlu_high_school_world_history", + "task_alias": "high_school_world_history", + "dataset_path": "cais/mmlu", + "dataset_name": "high_school_world_history", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about high school world history.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_high_school_world_history.yaml" + } + }, + "mmlu_human_aging": { + "task": "mmlu_human_aging", + "task_alias": "human_aging", + "dataset_path": "cais/mmlu", + "dataset_name": "human_aging", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about human aging.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_human_aging.yaml" + } + }, + "mmlu_human_sexuality": { + "task": "mmlu_human_sexuality", + "task_alias": "human_sexuality", + "dataset_path": "cais/mmlu", + "dataset_name": "human_sexuality", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about human sexuality.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_human_sexuality.yaml" + } + }, + "mmlu_international_law": { + "task": "mmlu_international_law", + "task_alias": "international_law", + "dataset_path": "cais/mmlu", + "dataset_name": "international_law", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about international law.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_international_law.yaml" + } + }, + "mmlu_jurisprudence": { + "task": "mmlu_jurisprudence", + "task_alias": "jurisprudence", + "dataset_path": "cais/mmlu", + "dataset_name": "jurisprudence", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about jurisprudence.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_jurisprudence.yaml" + } + }, + "mmlu_logical_fallacies": { + "task": "mmlu_logical_fallacies", + "task_alias": "logical_fallacies", + "dataset_path": "cais/mmlu", + "dataset_name": "logical_fallacies", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about logical fallacies.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_logical_fallacies.yaml" + } + }, + "mmlu_machine_learning": { + "task": "mmlu_machine_learning", + "task_alias": "machine_learning", + "dataset_path": "cais/mmlu", + "dataset_name": "machine_learning", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about machine learning.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_machine_learning.yaml" + } + }, + "mmlu_management": { + "task": "mmlu_management", + "task_alias": "management", + "dataset_path": "cais/mmlu", + "dataset_name": "management", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about management.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_management.yaml" + } + }, + "mmlu_marketing": { + "task": "mmlu_marketing", + "task_alias": "marketing", + "dataset_path": "cais/mmlu", + "dataset_name": "marketing", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about marketing.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_marketing.yaml" + } + }, + "mmlu_medical_genetics": { + "task": "mmlu_medical_genetics", + "task_alias": "medical_genetics", + "dataset_path": "cais/mmlu", + "dataset_name": "medical_genetics", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about medical genetics.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_medical_genetics.yaml" + } + }, + "mmlu_miscellaneous": { + "task": "mmlu_miscellaneous", + "task_alias": "miscellaneous", + "dataset_path": "cais/mmlu", + "dataset_name": "miscellaneous", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about miscellaneous.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_miscellaneous.yaml" + } + }, + "mmlu_moral_disputes": { + "task": "mmlu_moral_disputes", + "task_alias": "moral_disputes", + "dataset_path": "cais/mmlu", + "dataset_name": "moral_disputes", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about moral disputes.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_moral_disputes.yaml" + } + }, + "mmlu_moral_scenarios": { + "task": "mmlu_moral_scenarios", + "task_alias": "moral_scenarios", + "dataset_path": "cais/mmlu", + "dataset_name": "moral_scenarios", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about moral scenarios.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_moral_scenarios.yaml" + } + }, + "mmlu_nutrition": { + "task": "mmlu_nutrition", + "task_alias": "nutrition", + "dataset_path": "cais/mmlu", + "dataset_name": "nutrition", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about nutrition.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_nutrition.yaml" + } + }, + "mmlu_philosophy": { + "task": "mmlu_philosophy", + "task_alias": "philosophy", + "dataset_path": "cais/mmlu", + "dataset_name": "philosophy", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about philosophy.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_philosophy.yaml" + } + }, + "mmlu_prehistory": { + "task": "mmlu_prehistory", + "task_alias": "prehistory", + "dataset_path": "cais/mmlu", + "dataset_name": "prehistory", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about prehistory.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_prehistory.yaml" + } + }, + "mmlu_professional_accounting": { + "task": "mmlu_professional_accounting", + "task_alias": "professional_accounting", + "dataset_path": "cais/mmlu", + "dataset_name": "professional_accounting", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about professional accounting.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_professional_accounting.yaml" + } + }, + "mmlu_professional_law": { + "task": "mmlu_professional_law", + "task_alias": "professional_law", + "dataset_path": "cais/mmlu", + "dataset_name": "professional_law", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about professional law.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_professional_law.yaml" + } + }, + "mmlu_professional_medicine": { + "task": "mmlu_professional_medicine", + "task_alias": "professional_medicine", + "dataset_path": "cais/mmlu", + "dataset_name": "professional_medicine", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about professional medicine.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_professional_medicine.yaml" + } + }, + "mmlu_professional_psychology": { + "task": "mmlu_professional_psychology", + "task_alias": "professional_psychology", + "dataset_path": "cais/mmlu", + "dataset_name": "professional_psychology", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about professional psychology.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_professional_psychology.yaml" + } + }, + "mmlu_public_relations": { + "task": "mmlu_public_relations", + "task_alias": "public_relations", + "dataset_path": "cais/mmlu", + "dataset_name": "public_relations", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about public relations.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_public_relations.yaml" + } + }, + "mmlu_security_studies": { + "task": "mmlu_security_studies", + "task_alias": "security_studies", + "dataset_path": "cais/mmlu", + "dataset_name": "security_studies", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about security studies.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_security_studies.yaml" + } + }, + "mmlu_sociology": { + "task": "mmlu_sociology", + "task_alias": "sociology", + "dataset_path": "cais/mmlu", + "dataset_name": "sociology", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about sociology.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_sociology.yaml" + } + }, + "mmlu_us_foreign_policy": { + "task": "mmlu_us_foreign_policy", + "task_alias": "us_foreign_policy", + "dataset_path": "cais/mmlu", + "dataset_name": "us_foreign_policy", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about us foreign policy.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_us_foreign_policy.yaml" + } + }, + "mmlu_virology": { + "task": "mmlu_virology", + "task_alias": "virology", + "dataset_path": "cais/mmlu", + "dataset_name": "virology", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about virology.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_virology.yaml" + } + }, + "mmlu_world_religions": { + "task": "mmlu_world_religions", + "task_alias": "world_religions", + "dataset_path": "cais/mmlu", + "dataset_name": "world_religions", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about world religions.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_world_religions.yaml" + } + } + }, + "versions": { + "mmlu": "2", + "mmlu_abstract_algebra": 1.0, + "mmlu_anatomy": 1.0, + "mmlu_astronomy": 1.0, + "mmlu_business_ethics": 1.0, + "mmlu_clinical_knowledge": 1.0, + "mmlu_college_biology": 1.0, + "mmlu_college_chemistry": 1.0, + "mmlu_college_computer_science": 1.0, + "mmlu_college_mathematics": 1.0, + "mmlu_college_medicine": 1.0, + "mmlu_college_physics": 1.0, + "mmlu_computer_security": 1.0, + "mmlu_conceptual_physics": 1.0, + "mmlu_econometrics": 1.0, + "mmlu_electrical_engineering": 1.0, + "mmlu_elementary_mathematics": 1.0, + "mmlu_formal_logic": 1.0, + "mmlu_global_facts": 1.0, + "mmlu_high_school_biology": 1.0, + "mmlu_high_school_chemistry": 1.0, + "mmlu_high_school_computer_science": 1.0, + "mmlu_high_school_european_history": 1.0, + "mmlu_high_school_geography": 1.0, + "mmlu_high_school_government_and_politics": 1.0, + "mmlu_high_school_macroeconomics": 1.0, + "mmlu_high_school_mathematics": 1.0, + "mmlu_high_school_microeconomics": 1.0, + "mmlu_high_school_physics": 1.0, + "mmlu_high_school_psychology": 1.0, + "mmlu_high_school_statistics": 1.0, + "mmlu_high_school_us_history": 1.0, + "mmlu_high_school_world_history": 1.0, + "mmlu_human_aging": 1.0, + "mmlu_human_sexuality": 1.0, + "mmlu_humanities": "2", + "mmlu_international_law": 1.0, + "mmlu_jurisprudence": 1.0, + "mmlu_logical_fallacies": 1.0, + "mmlu_machine_learning": 1.0, + "mmlu_management": 1.0, + "mmlu_marketing": 1.0, + "mmlu_medical_genetics": 1.0, + "mmlu_miscellaneous": 1.0, + "mmlu_moral_disputes": 1.0, + "mmlu_moral_scenarios": 1.0, + "mmlu_nutrition": 1.0, + "mmlu_other": "2", + "mmlu_philosophy": 1.0, + "mmlu_prehistory": 1.0, + "mmlu_professional_accounting": 1.0, + "mmlu_professional_law": 1.0, + "mmlu_professional_medicine": 1.0, + "mmlu_professional_psychology": 1.0, + "mmlu_public_relations": 1.0, + "mmlu_security_studies": 1.0, + "mmlu_social_sciences": "2", + "mmlu_sociology": 1.0, + "mmlu_stem": "2", + "mmlu_us_foreign_policy": 1.0, + "mmlu_virology": 1.0, + "mmlu_world_religions": 1.0 + }, + "n-shot": { + "mmlu_abstract_algebra": 5, + "mmlu_anatomy": 5, + "mmlu_astronomy": 5, + "mmlu_business_ethics": 5, + "mmlu_clinical_knowledge": 5, + "mmlu_college_biology": 5, + "mmlu_college_chemistry": 5, + "mmlu_college_computer_science": 5, + "mmlu_college_mathematics": 5, + "mmlu_college_medicine": 5, + "mmlu_college_physics": 5, + "mmlu_computer_security": 5, + "mmlu_conceptual_physics": 5, + "mmlu_econometrics": 5, + "mmlu_electrical_engineering": 5, + "mmlu_elementary_mathematics": 5, + "mmlu_formal_logic": 5, + "mmlu_global_facts": 5, + "mmlu_high_school_biology": 5, + "mmlu_high_school_chemistry": 5, + "mmlu_high_school_computer_science": 5, + "mmlu_high_school_european_history": 5, + "mmlu_high_school_geography": 5, + "mmlu_high_school_government_and_politics": 5, + "mmlu_high_school_macroeconomics": 5, + "mmlu_high_school_mathematics": 5, + "mmlu_high_school_microeconomics": 5, + "mmlu_high_school_physics": 5, + "mmlu_high_school_psychology": 5, + "mmlu_high_school_statistics": 5, + "mmlu_high_school_us_history": 5, + "mmlu_high_school_world_history": 5, + "mmlu_human_aging": 5, + "mmlu_human_sexuality": 5, + "mmlu_humanities": 5, + "mmlu_international_law": 5, + "mmlu_jurisprudence": 5, + "mmlu_logical_fallacies": 5, + "mmlu_machine_learning": 5, + "mmlu_management": 5, + "mmlu_marketing": 5, + "mmlu_medical_genetics": 5, + "mmlu_miscellaneous": 5, + "mmlu_moral_disputes": 5, + "mmlu_moral_scenarios": 5, + "mmlu_nutrition": 5, + "mmlu_other": 5, + "mmlu_philosophy": 5, + "mmlu_prehistory": 5, + "mmlu_professional_accounting": 5, + "mmlu_professional_law": 5, + "mmlu_professional_medicine": 5, + "mmlu_professional_psychology": 5, + "mmlu_public_relations": 5, + "mmlu_security_studies": 5, + "mmlu_social_sciences": 5, + "mmlu_sociology": 5, + "mmlu_stem": 5, + "mmlu_us_foreign_policy": 5, + "mmlu_virology": 5, + "mmlu_world_religions": 5 + }, + "higher_is_better": { + "mmlu_abstract_algebra": { + "acc": true + }, + "mmlu_anatomy": { + "acc": true + }, + "mmlu_astronomy": { + "acc": true + }, + "mmlu_business_ethics": { + "acc": true + }, + "mmlu_clinical_knowledge": { + "acc": true + }, + "mmlu_college_biology": { + "acc": true + }, + "mmlu_college_chemistry": { + "acc": true + }, + "mmlu_college_computer_science": { + "acc": true + }, + "mmlu_college_mathematics": { + "acc": true + }, + "mmlu_college_medicine": { + "acc": true + }, + "mmlu_college_physics": { + "acc": true + }, + "mmlu_computer_security": { + "acc": true + }, + "mmlu_conceptual_physics": { + "acc": true + }, + "mmlu_econometrics": { + "acc": true + }, + "mmlu_electrical_engineering": { + "acc": true + }, + "mmlu_elementary_mathematics": { + "acc": true + }, + "mmlu_formal_logic": { + "acc": true + }, + "mmlu_global_facts": { + "acc": true + }, + "mmlu_high_school_biology": { + "acc": true + }, + "mmlu_high_school_chemistry": { + "acc": true + }, + "mmlu_high_school_computer_science": { + "acc": true + }, + "mmlu_high_school_european_history": { + "acc": true + }, + "mmlu_high_school_geography": { + "acc": true + }, + "mmlu_high_school_government_and_politics": { + "acc": true + }, + "mmlu_high_school_macroeconomics": { + "acc": true + }, + "mmlu_high_school_mathematics": { + "acc": true + }, + "mmlu_high_school_microeconomics": { + "acc": true + }, + "mmlu_high_school_physics": { + "acc": true + }, + "mmlu_high_school_psychology": { + "acc": true + }, + "mmlu_high_school_statistics": { + "acc": true + }, + "mmlu_high_school_us_history": { + "acc": true + }, + "mmlu_high_school_world_history": { + "acc": true + }, + "mmlu_human_aging": { + "acc": true + }, + "mmlu_human_sexuality": { + "acc": true + }, + "mmlu_humanities": { + "acc": true + }, + "mmlu_international_law": { + "acc": true + }, + "mmlu_jurisprudence": { + "acc": true + }, + "mmlu_logical_fallacies": { + "acc": true + }, + "mmlu_machine_learning": { + "acc": true + }, + "mmlu_management": { + "acc": true + }, + "mmlu_marketing": { + "acc": true + }, + "mmlu_medical_genetics": { + "acc": true + }, + "mmlu_miscellaneous": { + "acc": true + }, + "mmlu_moral_disputes": { + "acc": true + }, + "mmlu_moral_scenarios": { + "acc": true + }, + "mmlu_nutrition": { + "acc": true + }, + "mmlu_other": { + "acc": true + }, + "mmlu_philosophy": { + "acc": true + }, + "mmlu_prehistory": { + "acc": true + }, + "mmlu_professional_accounting": { + "acc": true + }, + "mmlu_professional_law": { + "acc": true + }, + "mmlu_professional_medicine": { + "acc": true + }, + "mmlu_professional_psychology": { + "acc": true + }, + "mmlu_public_relations": { + "acc": true + }, + "mmlu_security_studies": { + "acc": true + }, + "mmlu_social_sciences": { + "acc": true + }, + "mmlu_sociology": { + "acc": true + }, + "mmlu_stem": { + "acc": true + }, + "mmlu_us_foreign_policy": { + "acc": true + }, + "mmlu_virology": { + "acc": true + }, + "mmlu_world_religions": { + "acc": true + } + }, + "n-samples": { + "mmlu_abstract_algebra": { + "original": 100, + "effective": 15 + }, + "mmlu_anatomy": { + "original": 135, + "effective": 15 + }, + "mmlu_astronomy": { + "original": 152, + "effective": 15 + }, + "mmlu_college_biology": { + "original": 144, + "effective": 15 + }, + "mmlu_college_chemistry": { + "original": 100, + "effective": 15 + }, + "mmlu_college_computer_science": { + "original": 100, + "effective": 15 + }, + "mmlu_college_mathematics": { + "original": 100, + "effective": 15 + }, + "mmlu_college_physics": { + "original": 102, + "effective": 15 + }, + "mmlu_computer_security": { + "original": 100, + "effective": 15 + }, + "mmlu_conceptual_physics": { + "original": 235, + "effective": 15 + }, + "mmlu_electrical_engineering": { + "original": 145, + "effective": 15 + }, + "mmlu_elementary_mathematics": { + "original": 378, + "effective": 15 + }, + "mmlu_high_school_biology": { + "original": 310, + "effective": 15 + }, + "mmlu_high_school_chemistry": { + "original": 203, + "effective": 15 + }, + "mmlu_high_school_computer_science": { + "original": 100, + "effective": 15 + }, + "mmlu_high_school_mathematics": { + "original": 270, + "effective": 15 + }, + "mmlu_high_school_physics": { + "original": 151, + "effective": 15 + }, + "mmlu_high_school_statistics": { + "original": 216, + "effective": 15 + }, + "mmlu_machine_learning": { + "original": 112, + "effective": 15 + }, + "mmlu_business_ethics": { + "original": 100, + "effective": 15 + }, + "mmlu_clinical_knowledge": { + "original": 265, + "effective": 15 + }, + "mmlu_college_medicine": { + "original": 173, + "effective": 15 + }, + "mmlu_global_facts": { + "original": 100, + "effective": 15 + }, + "mmlu_human_aging": { + "original": 223, + "effective": 15 + }, + "mmlu_management": { + "original": 103, + "effective": 15 + }, + "mmlu_marketing": { + "original": 234, + "effective": 15 + }, + "mmlu_medical_genetics": { + "original": 100, + "effective": 15 + }, + "mmlu_miscellaneous": { + "original": 783, + "effective": 15 + }, + "mmlu_nutrition": { + "original": 306, + "effective": 15 + }, + "mmlu_professional_accounting": { + "original": 282, + "effective": 15 + }, + "mmlu_professional_medicine": { + "original": 272, + "effective": 15 + }, + "mmlu_virology": { + "original": 166, + "effective": 15 + }, + "mmlu_econometrics": { + "original": 114, + "effective": 15 + }, + "mmlu_high_school_geography": { + "original": 198, + "effective": 15 + }, + "mmlu_high_school_government_and_politics": { + "original": 193, + "effective": 15 + }, + "mmlu_high_school_macroeconomics": { + "original": 390, + "effective": 15 + }, + "mmlu_high_school_microeconomics": { + "original": 238, + "effective": 15 + }, + "mmlu_high_school_psychology": { + "original": 545, + "effective": 15 + }, + "mmlu_human_sexuality": { + "original": 131, + "effective": 15 + }, + "mmlu_professional_psychology": { + "original": 612, + "effective": 15 + }, + "mmlu_public_relations": { + "original": 110, + "effective": 15 + }, + "mmlu_security_studies": { + "original": 245, + "effective": 15 + }, + "mmlu_sociology": { + "original": 201, + "effective": 15 + }, + "mmlu_us_foreign_policy": { + "original": 100, + "effective": 15 + }, + "mmlu_formal_logic": { + "original": 126, + "effective": 15 + }, + "mmlu_high_school_european_history": { + "original": 165, + "effective": 15 + }, + "mmlu_high_school_us_history": { + "original": 204, + "effective": 15 + }, + "mmlu_high_school_world_history": { + "original": 237, + "effective": 15 + }, + "mmlu_international_law": { + "original": 121, + "effective": 15 + }, + "mmlu_jurisprudence": { + "original": 108, + "effective": 15 + }, + "mmlu_logical_fallacies": { + "original": 163, + "effective": 15 + }, + "mmlu_moral_disputes": { + "original": 346, + "effective": 15 + }, + "mmlu_moral_scenarios": { + "original": 895, + "effective": 15 + }, + "mmlu_philosophy": { + "original": 311, + "effective": 15 + }, + "mmlu_prehistory": { + "original": 324, + "effective": 15 + }, + "mmlu_professional_law": { + "original": 1534, + "effective": 15 + }, + "mmlu_world_religions": { + "original": 171, + "effective": 15 + } + }, + "config": { + "model": "hf", + "model_args": { + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16" + }, + "model_num_parameters": 3085938688, + "model_dtype": "torch.bfloat16", + "model_revision": "main", + "model_sha": "aa8e72537993ba99e69dfaafa59ed015b17504d1", + "batch_size": "auto", + "batch_sizes": [ + 2 + ], + "device": "cuda:0", + "use_cache": null, + "limit": 15.0, + "bootstrap_iters": 100000, + "gen_kwargs": {}, + "random_seed": 0, + "numpy_seed": 0, + "torch_seed": 0, + "fewshot_seed": 0 + }, + "git_hash": "eb4f2be22f7baf6d268c3dd5e46d49d6bd2e74ae", + "date": 1784830124.1132693, + "pretty_env_info": "PyTorch version: 2.6.0+cu124\nIs debug build: False\nCUDA used to build PyTorch: 12.4\nROCM used to build PyTorch: N/A\n\nOS: Ubuntu 22.04.5 LTS (x86_64)\nGCC version: (Ubuntu 11.4.0-1ubuntu1~22.04) 11.4.0\nClang version: Could not collect\nCMake version: Could not collect\nLibc version: glibc-2.35\n\nPython version: 3.11.10 (main, Sep 7 2024, 18:35:41) [GCC 11.4.0] (64-bit runtime)\nPython platform: Linux-6.8.0-52-generic-x86_64-with-glibc2.35\nIs CUDA available: True\nCUDA runtime version: Could not collect\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: GPU 0: NVIDIA RTX A4000\nNvidia driver version: 550.144.03\ncuDNN version: Could not collect\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 48 bits physical, 48 bits virtual\nByte Order: Little Endian\nCPU(s): 112\nOn-line CPU(s) list: 0-111\nVendor ID: AuthenticAMD\nModel name: AMD EPYC 7453 28-Core Processor\nCPU family: 25\nModel: 1\nThread(s) per core: 2\nCore(s) per socket: 28\nSocket(s): 2\nStepping: 1\nFrequency boost: enabled\nCPU max MHz: 3488.5249\nCPU min MHz: 1500.0000\nBogoMIPS: 5489.75\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush mmx fxsr sse sse2 ht syscall nx mmxext fxsr_opt pdpe1gb rdtscp lm constant_tsc rep_good nopl nonstop_tsc cpuid extd_apicid aperfmperf rapl pni pclmulqdq monitor ssse3 fma cx16 pcid sse4_1 sse4_2 movbe popcnt aes xsave avx f16c rdrand lahf_lm cmp_legacy svm extapic cr8_legacy abm sse4a misalignsse 3dnowprefetch osvw ibs skinit wdt tce topoext perfctr_core perfctr_nb bpext perfctr_llc mwaitx cpb cat_l3 cdp_l3 hw_pstate ssbd mba ibrs ibpb stibp vmmcall fsgsbase bmi1 avx2 smep bmi2 erms invpcid cqm rdt_a rdseed adx smap clflushopt clwb sha_ni xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local user_shstk clzero irperf xsaveerptr rdpru wbnoinvd amd_ppin brs arat npt lbrv svm_lock nrip_save tsc_scale vmcb_clean flushbyasid decodeassists pausefilter pfthreshold v_vmsave_vmload vgif v_spec_ctrl umip pku ospke vaes vpclmulqdq rdpid overflow_recov succor smca fsrm debug_swap\nVirtualization: AMD-V\nL1d cache: 1.8 MiB (56 instances)\nL1i cache: 1.8 MiB (56 instances)\nL2 cache: 28 MiB (56 instances)\nL3 cache: 128 MiB (8 instances)\nNUMA node(s): 2\nNUMA node0 CPU(s): 0-27,56-83\nNUMA node1 CPU(s): 28-55,84-111\nVulnerability Gather data sampling: Not affected\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Mmio stale data: Not affected\nVulnerability Reg file data sampling: Not affected\nVulnerability Retbleed: Not affected\nVulnerability Spec rstack overflow: Mitigation; Safe RET\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Retpolines; IBPB conditional; IBRS_FW; STIBP always-on; RSB filling; PBRSB-eIBRS Not affected; BHI Not affected\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] numpy==2.4.6\n[pip3] nvidia-cublas-cu12==12.4.5.8\n[pip3] nvidia-cuda-cupti-cu12==12.4.127\n[pip3] nvidia-cuda-nvrtc-cu12==12.4.127\n[pip3] nvidia-cuda-runtime-cu12==12.4.127\n[pip3] nvidia-cudnn-cu12==9.1.0.70\n[pip3] nvidia-cufft-cu12==11.2.1.3\n[pip3] nvidia-curand-cu12==10.3.5.147\n[pip3] nvidia-cusolver-cu12==11.6.1.9\n[pip3] nvidia-cusparse-cu12==12.3.1.170\n[pip3] nvidia-cusparselt-cu12==0.6.2\n[pip3] nvidia-nccl-cu12==2.21.5\n[pip3] nvidia-nvjitlink-cu12==12.4.127\n[pip3] nvidia-nvtx-cu12==12.4.127\n[pip3] torch==2.6.0+cu124\n[pip3] triton==3.2.0\n[conda] Could not collect", + "transformers_version": "5.14.1", + "lm_eval_version": "0.4.12", + "upper_git_hash": null, + "tokenizer_pad_token": [ + "<|endoftext|>", + "151643" + ], + "tokenizer_eos_token": [ + "<|im_end|>", + "151645" + ], + "tokenizer_bos_token": [ + null, + "None" + ], + "eot_token_id": 151645, + "max_length": 32768, + "task_hashes": {}, + "model_source": "hf", + "model_name": "Qwen/Qwen2.5-3B-Instruct", + "model_name_sanitized": "Qwen__Qwen2.5-3B-Instruct", + "system_instruction": null, + "system_instruction_sha": null, + "fewshot_as_multiturn": true, + "chat_template": "{%- if tools %}\n {{- '<|im_start|>system\\n' }}\n {%- if messages[0]['role'] == 'system' %}\n {{- messages[0]['content'] }}\n {%- else %}\n {{- 'You are Qwen, created by Alibaba Cloud. You are a helpful assistant.' }}\n {%- endif %}\n {{- \"\\n\\n# Tools\\n\\nYou may call one or more functions to assist with the user query.\\n\\nYou are provided with function signatures within XML tags:\\n\" }}\n {%- for tool in tools %}\n {{- \"\\n\" }}\n {{- tool | tojson }}\n {%- endfor %}\n {{- \"\\n\\n\\nFor each function call, return a json object with function name and arguments within XML tags:\\n\\n{\\\"name\\\": , \\\"arguments\\\": }\\n<|im_end|>\\n\" }}\n{%- else %}\n {%- if messages[0]['role'] == 'system' %}\n {{- '<|im_start|>system\\n' + messages[0]['content'] + '<|im_end|>\\n' }}\n {%- else %}\n {{- '<|im_start|>system\\nYou are Qwen, created by Alibaba Cloud. You are a helpful assistant.<|im_end|>\\n' }}\n {%- endif %}\n{%- endif %}\n{%- for message in messages %}\n {%- if (message.role == \"user\") or (message.role == \"system\" and not loop.first) or (message.role == \"assistant\" and not message.tool_calls) %}\n {{- '<|im_start|>' + message.role + '\\n' + message.content + '<|im_end|>' + '\\n' }}\n {%- elif message.role == \"assistant\" %}\n {{- '<|im_start|>' + message.role }}\n {%- if message.content %}\n {{- '\\n' + message.content }}\n {%- endif %}\n {%- for tool_call in message.tool_calls %}\n {%- if tool_call.function is defined %}\n {%- set tool_call = tool_call.function %}\n {%- endif %}\n {{- '\\n\\n{\"name\": \"' }}\n {{- tool_call.name }}\n {{- '\", \"arguments\": ' }}\n {{- tool_call.arguments | tojson }}\n {{- '}\\n' }}\n {%- endfor %}\n {{- '<|im_end|>\\n' }}\n {%- elif message.role == \"tool\" %}\n {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != \"tool\") %}\n {{- '<|im_start|>user' }}\n {%- endif %}\n {{- '\\n\\n' }}\n {{- message.content }}\n {{- '\\n' }}\n {%- if loop.last or (messages[loop.index0 + 1].role != \"tool\") %}\n {{- '<|im_end|>\\n' }}\n {%- endif %}\n {%- endif %}\n{%- endfor %}\n{%- if add_generation_prompt %}\n {{- '<|im_start|>assistant\\n' }}\n{%- endif %}\n", + "chat_template_sha": "cd8e9439f0570856fd70470bf8889ebd8b5d1107207f67a5efb46e342330527f", + "total_evaluation_time_seconds": "440.5475653670728" +} \ No newline at end of file diff --git a/run/capability/m1/gsm8k_cot/__root__steering-resistance__results__full_3b__m1_resist_adapter/results_2026-07-23T19-07-18.393003.json b/run/capability/m1/gsm8k_cot/__root__steering-resistance__results__full_3b__m1_resist_adapter/results_2026-07-23T19-07-18.393003.json new file mode 100644 index 0000000000000000000000000000000000000000..ec7a3f858c625e7d16472d2bb9b8b0d494ad8116 --- /dev/null +++ b/run/capability/m1/gsm8k_cot/__root__steering-resistance__results__full_3b__m1_resist_adapter/results_2026-07-23T19-07-18.393003.json @@ -0,0 +1,208 @@ +{ + "results": { + "gsm8k_cot": { + "name": "gsm8k_cot", + "alias": "gsm8k_cot", + "sample_len": 200, + "exact_match,strict-match": 0.41, + "exact_match_stderr,strict-match": 0.034865138597849274, + "exact_match,flexible-extract": 0.68, + "exact_match_stderr,flexible-extract": 0.033067617644508635 + } + }, + "group_subtasks": {}, + "configs": { + "gsm8k_cot": { + "task": "gsm8k_cot", + "dataset_path": "openai/gsm8k", + "dataset_name": "main", + "test_split": "test", + "doc_to_text": "Q: {{question}}\nA:", + "doc_to_target": "{{answer.split('####')[-1].strip() if answer is defined else target}}", + "unsafe_code": false, + "description": "", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": null, + "process_docs": null, + "fewshot_indices": null, + "samples": [ + { + "question": "There are 15 trees in the grove. Grove workers will plant trees in the grove today. After they are done, there will be 21 trees. How many trees did the grove workers plant today?", + "target": "There are 15 trees originally. Then there were 21 trees after some more were planted. So there must have been 21 - 15 = 6. The answer is 6." + }, + { + "question": "If there are 3 cars in the parking lot and 2 more cars arrive, how many cars are in the parking lot?", + "target": "There are originally 3 cars. 2 more cars arrive. 3 + 2 = 5. The answer is 5." + }, + { + "question": "Leah had 32 chocolates and her sister had 42. If they ate 35, how many pieces do they have left in total?", + "target": "Originally, Leah had 32 chocolates. Her sister had 42. So in total they had 32 + 42 = 74. After eating 35, they had 74 - 35 = 39. The answer is 39." + }, + { + "question": "Jason had 20 lollipops. He gave Denny some lollipops. Now Jason has 12 lollipops. How many lollipops did Jason give to Denny?", + "target": "Jason started with 20 lollipops. Then he had 12 after giving some to Denny. So he gave Denny 20 - 12 = 8. The answer is 8." + }, + { + "question": "Shawn has five toys. For Christmas, he got two toys each from his mom and dad. How many toys does he have now?", + "target": "Shawn started with 5 toys. If he got 2 toys each from his mom and dad, then that is 4 more toys. 5 + 4 = 9. The answer is 9." + }, + { + "question": "There were nine computers in the server room. Five more computers were installed each day, from monday to thursday. How many computers are now in the server room?", + "target": "There were originally 9 computers. For each of 4 days, 5 more computers were added. So 5 * 4 = 20 computers were added. 9 + 20 is 29. The answer is 29." + }, + { + "question": "Michael had 58 golf balls. On tuesday, he lost 23 golf balls. On wednesday, he lost 2 more. How many golf balls did he have at the end of wednesday?", + "target": "Michael started with 58 golf balls. After losing 23 on tuesday, he had 58 - 23 = 35. After losing 2 more, he had 35 - 2 = 33 golf balls. The answer is 33." + }, + { + "question": "Olivia has $23. She bought five bagels for $3 each. How much money does she have left?", + "target": "Olivia had 23 dollars. 5 bagels for 3 dollars each will be 5 x 3 = 15 dollars. So she has 23 - 15 dollars left. 23 - 15 is 8. The answer is 8." + } + ], + "doc_to_text": "Q: {{question}}\nA:", + "doc_to_choice": null, + "doc_to_target": "{{answer.split('####')[-1].strip() if answer is defined else target}}", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 8, + "metric_list": [ + { + "aggregation": "mean", + "higher_is_better": true, + "ignore_case": true, + "ignore_punctuation": false, + "metric": "exact_match", + "regexes_to_ignore": [ + ",", + "\\$", + "(?s).*#### ", + "\\.$" + ] + } + ], + "output_type": "generate_until", + "generation_kwargs": { + "do_sample": false, + "until": [ + "Q:", + "", + "<|im_end|>" + ] + }, + "repeats": 1, + "filter_list": [ + { + "filter": [ + { + "function": "regex", + "regex_pattern": "The answer is (\\-?[0-9\\.\\,]+)." + }, + { + "function": "take_first" + } + ], + "name": "strict-match" + }, + { + "filter": [ + { + "function": "regex", + "group_select": -1, + "regex_pattern": "(-?[$0-9.,]{2,})|(-?[0-9]+)" + }, + { + "function": "take_first" + } + ], + "name": "flexible-extract" + } + ], + "should_decontaminate": false, + "metadata": { + "version": 3.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/gsm8k/gsm8k-cot.yaml" + } + } + }, + "versions": { + "gsm8k_cot": 3.0 + }, + "n-shot": { + "gsm8k_cot": 8 + }, + "higher_is_better": { + "gsm8k_cot": { + "exact_match": true + } + }, + "n-samples": { + "gsm8k_cot": { + "original": 1319, + "effective": 200 + } + }, + "config": { + "model": "hf", + "model_args": { + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter" + }, + "model_num_parameters": 3115317248, + "model_dtype": "torch.bfloat16", + "model_revision": "main", + "model_sha": "aa8e72537993ba99e69dfaafa59ed015b17504d1", + "peft_sha": "", + "batch_size": "auto", + "batch_sizes": [], + "device": "cuda:0", + "use_cache": null, + "limit": 200.0, + "bootstrap_iters": 100000, + "gen_kwargs": { + "do_sample": false + }, + "random_seed": 0, + "numpy_seed": 0, + "torch_seed": 0, + "fewshot_seed": 0 + }, + "git_hash": "eb4f2be22f7baf6d268c3dd5e46d49d6bd2e74ae", + "date": 1784832116.7343817, + "pretty_env_info": "PyTorch version: 2.6.0+cu124\nIs debug build: False\nCUDA used to build PyTorch: 12.4\nROCM used to build PyTorch: N/A\n\nOS: Ubuntu 22.04.5 LTS (x86_64)\nGCC version: (Ubuntu 11.4.0-1ubuntu1~22.04) 11.4.0\nClang version: Could not collect\nCMake version: Could not collect\nLibc version: glibc-2.35\n\nPython version: 3.11.10 (main, Sep 7 2024, 18:35:41) [GCC 11.4.0] (64-bit runtime)\nPython platform: Linux-6.8.0-52-generic-x86_64-with-glibc2.35\nIs CUDA available: True\nCUDA runtime version: Could not collect\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: GPU 0: NVIDIA RTX A4000\nNvidia driver version: 550.144.03\ncuDNN version: Could not collect\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 48 bits physical, 48 bits virtual\nByte Order: Little Endian\nCPU(s): 112\nOn-line CPU(s) list: 0-111\nVendor ID: AuthenticAMD\nModel name: AMD EPYC 7453 28-Core Processor\nCPU family: 25\nModel: 1\nThread(s) per core: 2\nCore(s) per socket: 28\nSocket(s): 2\nStepping: 1\nFrequency boost: enabled\nCPU max MHz: 3488.5249\nCPU min MHz: 1500.0000\nBogoMIPS: 5489.75\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush mmx fxsr sse sse2 ht syscall nx mmxext fxsr_opt pdpe1gb rdtscp lm constant_tsc rep_good nopl nonstop_tsc cpuid extd_apicid aperfmperf rapl pni pclmulqdq monitor ssse3 fma cx16 pcid sse4_1 sse4_2 movbe popcnt aes xsave avx f16c rdrand lahf_lm cmp_legacy svm extapic cr8_legacy abm sse4a misalignsse 3dnowprefetch osvw ibs skinit wdt tce topoext perfctr_core perfctr_nb bpext perfctr_llc mwaitx cpb cat_l3 cdp_l3 hw_pstate ssbd mba ibrs ibpb stibp vmmcall fsgsbase bmi1 avx2 smep bmi2 erms invpcid cqm rdt_a rdseed adx smap clflushopt clwb sha_ni xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local user_shstk clzero irperf xsaveerptr rdpru wbnoinvd amd_ppin brs arat npt lbrv svm_lock nrip_save tsc_scale vmcb_clean flushbyasid decodeassists pausefilter pfthreshold v_vmsave_vmload vgif v_spec_ctrl umip pku ospke vaes vpclmulqdq rdpid overflow_recov succor smca fsrm debug_swap\nVirtualization: AMD-V\nL1d cache: 1.8 MiB (56 instances)\nL1i cache: 1.8 MiB (56 instances)\nL2 cache: 28 MiB (56 instances)\nL3 cache: 128 MiB (8 instances)\nNUMA node(s): 2\nNUMA node0 CPU(s): 0-27,56-83\nNUMA node1 CPU(s): 28-55,84-111\nVulnerability Gather data sampling: Not affected\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Mmio stale data: Not affected\nVulnerability Reg file data sampling: Not affected\nVulnerability Retbleed: Not affected\nVulnerability Spec rstack overflow: Mitigation; Safe RET\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Retpolines; IBPB conditional; IBRS_FW; STIBP always-on; RSB filling; PBRSB-eIBRS Not affected; BHI Not affected\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] numpy==2.4.6\n[pip3] nvidia-cublas-cu12==12.4.5.8\n[pip3] nvidia-cuda-cupti-cu12==12.4.127\n[pip3] nvidia-cuda-nvrtc-cu12==12.4.127\n[pip3] nvidia-cuda-runtime-cu12==12.4.127\n[pip3] nvidia-cudnn-cu12==9.1.0.70\n[pip3] nvidia-cufft-cu12==11.2.1.3\n[pip3] nvidia-curand-cu12==10.3.5.147\n[pip3] nvidia-cusolver-cu12==11.6.1.9\n[pip3] nvidia-cusparse-cu12==12.3.1.170\n[pip3] nvidia-cusparselt-cu12==0.6.2\n[pip3] nvidia-nccl-cu12==2.21.5\n[pip3] nvidia-nvjitlink-cu12==12.4.127\n[pip3] nvidia-nvtx-cu12==12.4.127\n[pip3] torch==2.6.0+cu124\n[pip3] triton==3.2.0\n[conda] Could not collect", + "transformers_version": "5.14.1", + "lm_eval_version": "0.4.12", + "upper_git_hash": null, + "tokenizer_pad_token": [ + "<|endoftext|>", + "151643" + ], + "tokenizer_eos_token": [ + "<|im_end|>", + "151645" + ], + "tokenizer_bos_token": [ + null, + "None" + ], + "eot_token_id": 151645, + "max_length": 32768, + "task_hashes": {}, + "model_source": "hf", + "model_name": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "model_name_sanitized": "__root__steering-resistance__results__full_3b__m1_resist_adapter", + "system_instruction": null, + "system_instruction_sha": null, + "fewshot_as_multiturn": true, + "chat_template": "{%- if tools %}\n {{- '<|im_start|>system\\n' }}\n {%- if messages[0]['role'] == 'system' %}\n {{- messages[0]['content'] }}\n {%- else %}\n {{- 'You are Qwen, created by Alibaba Cloud. You are a helpful assistant.' }}\n {%- endif %}\n {{- \"\\n\\n# Tools\\n\\nYou may call one or more functions to assist with the user query.\\n\\nYou are provided with function signatures within XML tags:\\n\" }}\n {%- for tool in tools %}\n {{- \"\\n\" }}\n {{- tool | tojson }}\n {%- endfor %}\n {{- \"\\n\\n\\nFor each function call, return a json object with function name and arguments within XML tags:\\n\\n{\\\"name\\\": , \\\"arguments\\\": }\\n<|im_end|>\\n\" }}\n{%- else %}\n {%- if messages[0]['role'] == 'system' %}\n {{- '<|im_start|>system\\n' + messages[0]['content'] + '<|im_end|>\\n' }}\n {%- else %}\n {{- '<|im_start|>system\\nYou are Qwen, created by Alibaba Cloud. You are a helpful assistant.<|im_end|>\\n' }}\n {%- endif %}\n{%- endif %}\n{%- for message in messages %}\n {%- if (message.role == \"user\") or (message.role == \"system\" and not loop.first) or (message.role == \"assistant\" and not message.tool_calls) %}\n {{- '<|im_start|>' + message.role + '\\n' + message.content + '<|im_end|>' + '\\n' }}\n {%- elif message.role == \"assistant\" %}\n {{- '<|im_start|>' + message.role }}\n {%- if message.content %}\n {{- '\\n' + message.content }}\n {%- endif %}\n {%- for tool_call in message.tool_calls %}\n {%- if tool_call.function is defined %}\n {%- set tool_call = tool_call.function %}\n {%- endif %}\n {{- '\\n\\n{\"name\": \"' }}\n {{- tool_call.name }}\n {{- '\", \"arguments\": ' }}\n {{- tool_call.arguments | tojson }}\n {{- '}\\n' }}\n {%- endfor %}\n {{- '<|im_end|>\\n' }}\n {%- elif message.role == \"tool\" %}\n {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != \"tool\") %}\n {{- '<|im_start|>user' }}\n {%- endif %}\n {{- '\\n\\n' }}\n {{- message.content }}\n {{- '\\n' }}\n {%- if loop.last or (messages[loop.index0 + 1].role != \"tool\") %}\n {{- '<|im_end|>\\n' }}\n {%- endif %}\n {%- endif %}\n{%- endfor %}\n{%- if add_generation_prompt %}\n {{- '<|im_start|>assistant\\n' }}\n{%- endif %}\n", + "chat_template_sha": "cd8e9439f0570856fd70470bf8889ebd8b5d1107207f67a5efb46e342330527f", + "total_evaluation_time_seconds": "1528.508546601981" +} \ No newline at end of file diff --git a/run/capability/m1/mmlu/__root__steering-resistance__results__full_3b__m1_resist_adapter/results_2026-07-23T18-41-47.488437.json b/run/capability/m1/mmlu/__root__steering-resistance__results__full_3b__m1_resist_adapter/results_2026-07-23T18-41-47.488437.json new file mode 100644 index 0000000000000000000000000000000000000000..8b28010047e95d58ac27b9685164cab1420c3049 --- /dev/null +++ b/run/capability/m1/mmlu/__root__steering-resistance__results__full_3b__m1_resist_adapter/results_2026-07-23T18-41-47.488437.json @@ -0,0 +1,4370 @@ +{ + "results": { + "mmlu_abstract_algebra": { + "name": "mmlu_abstract_algebra", + "alias": "abstract_algebra", + "sample_len": 15, + "acc,none": 0.4, + "acc_stderr,none": 0.13093073414159542 + }, + "mmlu_anatomy": { + "name": "mmlu_anatomy", + "alias": "anatomy", + "sample_len": 15, + "acc,none": 0.5333333333333333, + "acc_stderr,none": 0.1333333333333333 + }, + "mmlu_astronomy": { + "name": "mmlu_astronomy", + "alias": "astronomy", + "sample_len": 15, + "acc,none": 0.8, + "acc_stderr,none": 0.10690449676496976 + }, + "mmlu_college_biology": { + "name": "mmlu_college_biology", + "alias": "college_biology", + "sample_len": 15, + "acc,none": 0.8666666666666667, + "acc_stderr,none": 0.09085135251589957 + }, + "mmlu_college_chemistry": { + "name": "mmlu_college_chemistry", + "alias": "college_chemistry", + "sample_len": 15, + "acc,none": 0.4, + "acc_stderr,none": 0.13093073414159542 + }, + "mmlu_college_computer_science": { + "name": "mmlu_college_computer_science", + "alias": "college_computer_science", + "sample_len": 15, + "acc,none": 0.5333333333333333, + "acc_stderr,none": 0.1333333333333333 + }, + "mmlu_college_mathematics": { + "name": "mmlu_college_mathematics", + "alias": "college_mathematics", + "sample_len": 15, + "acc,none": 0.4, + "acc_stderr,none": 0.13093073414159542 + }, + "mmlu_college_physics": { + "name": "mmlu_college_physics", + "alias": "college_physics", + "sample_len": 15, + "acc,none": 0.4666666666666667, + "acc_stderr,none": 0.1333333333333333 + }, + "mmlu_computer_security": { + "name": "mmlu_computer_security", + "alias": "computer_security", + "sample_len": 15, + "acc,none": 0.7333333333333333, + "acc_stderr,none": 0.11818736805705576 + }, + "mmlu_conceptual_physics": { + "name": "mmlu_conceptual_physics", + "alias": "conceptual_physics", + "sample_len": 15, + "acc,none": 0.9333333333333333, + "acc_stderr,none": 0.06666666666666667 + }, + "mmlu_electrical_engineering": { + "name": "mmlu_electrical_engineering", + "alias": "electrical_engineering", + "sample_len": 15, + "acc,none": 0.6, + "acc_stderr,none": 0.13093073414159542 + }, + "mmlu_elementary_mathematics": { + "name": "mmlu_elementary_mathematics", + "alias": "elementary_mathematics", + "sample_len": 15, + "acc,none": 0.8, + "acc_stderr,none": 0.10690449676496977 + }, + "mmlu_high_school_biology": { + "name": "mmlu_high_school_biology", + "alias": "high_school_biology", + "sample_len": 15, + "acc,none": 0.7333333333333333, + "acc_stderr,none": 0.11818736805705576 + }, + "mmlu_high_school_chemistry": { + "name": "mmlu_high_school_chemistry", + "alias": "high_school_chemistry", + "sample_len": 15, + "acc,none": 0.8666666666666667, + "acc_stderr,none": 0.09085135251589957 + }, + "mmlu_high_school_computer_science": { + "name": "mmlu_high_school_computer_science", + "alias": "high_school_computer_science", + "sample_len": 15, + "acc,none": 0.8, + "acc_stderr,none": 0.10690449676496976 + }, + "mmlu_high_school_mathematics": { + "name": "mmlu_high_school_mathematics", + "alias": "high_school_mathematics", + "sample_len": 15, + "acc,none": 0.2, + "acc_stderr,none": 0.10690449676496977 + }, + "mmlu_high_school_physics": { + "name": "mmlu_high_school_physics", + "alias": "high_school_physics", + "sample_len": 15, + "acc,none": 0.4666666666666667, + "acc_stderr,none": 0.1333333333333333 + }, + "mmlu_high_school_statistics": { + "name": "mmlu_high_school_statistics", + "alias": "high_school_statistics", + "sample_len": 15, + "acc,none": 0.7333333333333333, + "acc_stderr,none": 0.11818736805705575 + }, + "mmlu_machine_learning": { + "name": "mmlu_machine_learning", + "alias": "machine_learning", + "sample_len": 15, + "acc,none": 0.4, + "acc_stderr,none": 0.13093073414159542 + }, + "mmlu_business_ethics": { + "name": "mmlu_business_ethics", + "alias": "business_ethics", + "sample_len": 15, + "acc,none": 0.6666666666666666, + "acc_stderr,none": 0.12598815766974242 + }, + "mmlu_clinical_knowledge": { + "name": "mmlu_clinical_knowledge", + "alias": "clinical_knowledge", + "sample_len": 15, + "acc,none": 0.6, + "acc_stderr,none": 0.13093073414159542 + }, + "mmlu_college_medicine": { + "name": "mmlu_college_medicine", + "alias": "college_medicine", + "sample_len": 15, + "acc,none": 0.7333333333333333, + "acc_stderr,none": 0.11818736805705576 + }, + "mmlu_global_facts": { + "name": "mmlu_global_facts", + "alias": "global_facts", + "sample_len": 15, + "acc,none": 0.26666666666666666, + "acc_stderr,none": 0.11818736805705578 + }, + "mmlu_human_aging": { + "name": "mmlu_human_aging", + "alias": "human_aging", + "sample_len": 15, + "acc,none": 0.8, + "acc_stderr,none": 0.10690449676496976 + }, + "mmlu_management": { + "name": "mmlu_management", + "alias": "management", + "sample_len": 15, + "acc,none": 0.7333333333333333, + "acc_stderr,none": 0.11818736805705576 + }, + "mmlu_marketing": { + "name": "mmlu_marketing", + "alias": "marketing", + "sample_len": 15, + "acc,none": 0.8, + "acc_stderr,none": 0.10690449676496977 + }, + "mmlu_medical_genetics": { + "name": "mmlu_medical_genetics", + "alias": "medical_genetics", + "sample_len": 15, + "acc,none": 1.0, + "acc_stderr,none": 0.0 + }, + "mmlu_miscellaneous": { + "name": "mmlu_miscellaneous", + "alias": "miscellaneous", + "sample_len": 15, + "acc,none": 0.8666666666666667, + "acc_stderr,none": 0.09085135251589958 + }, + "mmlu_nutrition": { + "name": "mmlu_nutrition", + "alias": "nutrition", + "sample_len": 15, + "acc,none": 0.8, + "acc_stderr,none": 0.10690449676496977 + }, + "mmlu_professional_accounting": { + "name": "mmlu_professional_accounting", + "alias": "professional_accounting", + "sample_len": 15, + "acc,none": 0.4, + "acc_stderr,none": 0.13093073414159545 + }, + "mmlu_professional_medicine": { + "name": "mmlu_professional_medicine", + "alias": "professional_medicine", + "sample_len": 15, + "acc,none": 0.6666666666666666, + "acc_stderr,none": 0.12598815766974242 + }, + "mmlu_virology": { + "name": "mmlu_virology", + "alias": "virology", + "sample_len": 15, + "acc,none": 0.6, + "acc_stderr,none": 0.13093073414159542 + }, + "mmlu_econometrics": { + "name": "mmlu_econometrics", + "alias": "econometrics", + "sample_len": 15, + "acc,none": 0.6, + "acc_stderr,none": 0.13093073414159542 + }, + "mmlu_high_school_geography": { + "name": "mmlu_high_school_geography", + "alias": "high_school_geography", + "sample_len": 15, + "acc,none": 0.8, + "acc_stderr,none": 0.10690449676496977 + }, + "mmlu_high_school_government_and_politics": { + "name": "mmlu_high_school_government_and_politics", + "alias": "high_school_government_and_politics", + "sample_len": 15, + "acc,none": 0.8666666666666667, + "acc_stderr,none": 0.09085135251589957 + }, + "mmlu_high_school_macroeconomics": { + "name": "mmlu_high_school_macroeconomics", + "alias": "high_school_macroeconomics", + "sample_len": 15, + "acc,none": 0.6, + "acc_stderr,none": 0.13093073414159542 + }, + "mmlu_high_school_microeconomics": { + "name": "mmlu_high_school_microeconomics", + "alias": "high_school_microeconomics", + "sample_len": 15, + "acc,none": 0.7333333333333333, + "acc_stderr,none": 0.11818736805705576 + }, + "mmlu_high_school_psychology": { + "name": "mmlu_high_school_psychology", + "alias": "high_school_psychology", + "sample_len": 15, + "acc,none": 0.9333333333333333, + "acc_stderr,none": 0.06666666666666667 + }, + "mmlu_human_sexuality": { + "name": "mmlu_human_sexuality", + "alias": "human_sexuality", + "sample_len": 15, + "acc,none": 0.7333333333333333, + "acc_stderr,none": 0.11818736805705576 + }, + "mmlu_professional_psychology": { + "name": "mmlu_professional_psychology", + "alias": "professional_psychology", + "sample_len": 15, + "acc,none": 0.9333333333333333, + "acc_stderr,none": 0.06666666666666667 + }, + "mmlu_public_relations": { + "name": "mmlu_public_relations", + "alias": "public_relations", + "sample_len": 15, + "acc,none": 0.5333333333333333, + "acc_stderr,none": 0.1333333333333333 + }, + "mmlu_security_studies": { + "name": "mmlu_security_studies", + "alias": "security_studies", + "sample_len": 15, + "acc,none": 0.8, + "acc_stderr,none": 0.10690449676496977 + }, + "mmlu_sociology": { + "name": "mmlu_sociology", + "alias": "sociology", + "sample_len": 15, + "acc,none": 0.8, + "acc_stderr,none": 0.10690449676496976 + }, + "mmlu_us_foreign_policy": { + "name": "mmlu_us_foreign_policy", + "alias": "us_foreign_policy", + "sample_len": 15, + "acc,none": 0.8666666666666667, + "acc_stderr,none": 0.09085135251589958 + }, + "mmlu_formal_logic": { + "name": "mmlu_formal_logic", + "alias": "formal_logic", + "sample_len": 15, + "acc,none": 0.3333333333333333, + "acc_stderr,none": 0.12598815766974242 + }, + "mmlu_high_school_european_history": { + "name": "mmlu_high_school_european_history", + "alias": "high_school_european_history", + "sample_len": 15, + "acc,none": 0.7333333333333333, + "acc_stderr,none": 0.11818736805705575 + }, + "mmlu_high_school_us_history": { + "name": "mmlu_high_school_us_history", + "alias": "high_school_us_history", + "sample_len": 15, + "acc,none": 0.8, + "acc_stderr,none": 0.10690449676496976 + }, + "mmlu_high_school_world_history": { + "name": "mmlu_high_school_world_history", + "alias": "high_school_world_history", + "sample_len": 15, + "acc,none": 0.8, + "acc_stderr,none": 0.10690449676496976 + }, + "mmlu_international_law": { + "name": "mmlu_international_law", + "alias": "international_law", + "sample_len": 15, + "acc,none": 0.9333333333333333, + "acc_stderr,none": 0.06666666666666665 + }, + "mmlu_jurisprudence": { + "name": "mmlu_jurisprudence", + "alias": "jurisprudence", + "sample_len": 15, + "acc,none": 0.8, + "acc_stderr,none": 0.10690449676496977 + }, + "mmlu_logical_fallacies": { + "name": "mmlu_logical_fallacies", + "alias": "logical_fallacies", + "sample_len": 15, + "acc,none": 0.9333333333333333, + "acc_stderr,none": 0.06666666666666667 + }, + "mmlu_moral_disputes": { + "name": "mmlu_moral_disputes", + "alias": "moral_disputes", + "sample_len": 15, + "acc,none": 0.4, + "acc_stderr,none": 0.13093073414159542 + }, + "mmlu_moral_scenarios": { + "name": "mmlu_moral_scenarios", + "alias": "moral_scenarios", + "sample_len": 15, + "acc,none": 0.4, + "acc_stderr,none": 0.13093073414159542 + }, + "mmlu_philosophy": { + "name": "mmlu_philosophy", + "alias": "philosophy", + "sample_len": 15, + "acc,none": 0.8, + "acc_stderr,none": 0.10690449676496976 + }, + "mmlu_prehistory": { + "name": "mmlu_prehistory", + "alias": "prehistory", + "sample_len": 15, + "acc,none": 0.7333333333333333, + "acc_stderr,none": 0.11818736805705576 + }, + "mmlu_professional_law": { + "name": "mmlu_professional_law", + "alias": "professional_law", + "sample_len": 15, + "acc,none": 0.6666666666666666, + "acc_stderr,none": 0.12598815766974242 + }, + "mmlu_world_religions": { + "name": "mmlu_world_religions", + "alias": "world_religions", + "sample_len": 15, + "acc,none": 0.8666666666666667, + "acc_stderr,none": 0.09085135251589957 + }, + "mmlu_stem": { + "alias": "stem", + "name": "mmlu_stem", + "sample_len": 285, + "acc,none": 0.6140350877192983, + "acc_stderr,none": 0.02711404177703964, + "sample_count": { + "acc,none": 285 + } + }, + "mmlu_other": { + "alias": "other", + "name": "mmlu_other", + "sample_len": 195, + "acc,none": 0.6871794871794872, + "acc_stderr,none": 0.03149331173656363, + "sample_count": { + "acc,none": 195 + } + }, + "mmlu_social_sciences": { + "alias": "social sciences", + "name": "mmlu_social_sciences", + "sample_len": 180, + "acc,none": 0.7666666666666667, + "acc_stderr,none": 0.03114510614341045, + "sample_count": { + "acc,none": 180 + } + }, + "mmlu_humanities": { + "alias": "humanities", + "name": "mmlu_humanities", + "sample_len": 195, + "acc,none": 0.7076923076923077, + "acc_stderr,none": 0.03046245111544585, + "sample_count": { + "acc,none": 195 + } + }, + "mmlu": { + "alias": "mmlu", + "name": "mmlu", + "sample_len": 855, + "acc,none": 0.6842105263157895, + "acc_stderr,none": 0.014984590520591782, + "sample_count": { + "acc,none": 855 + } + } + }, + "groups": { + "mmlu_stem": { + "alias": "stem", + "name": "mmlu_stem", + "sample_len": 285, + "acc,none": 0.6140350877192983, + "acc_stderr,none": 0.02711404177703964, + "sample_count": { + "acc,none": 285 + } + }, + "mmlu_other": { + "alias": "other", + "name": "mmlu_other", + "sample_len": 195, + "acc,none": 0.6871794871794872, + "acc_stderr,none": 0.03149331173656363, + "sample_count": { + "acc,none": 195 + } + }, + "mmlu_social_sciences": { + "alias": "social sciences", + "name": "mmlu_social_sciences", + "sample_len": 180, + "acc,none": 0.7666666666666667, + "acc_stderr,none": 0.03114510614341045, + "sample_count": { + "acc,none": 180 + } + }, + "mmlu_humanities": { + "alias": "humanities", + "name": "mmlu_humanities", + "sample_len": 195, + "acc,none": 0.7076923076923077, + "acc_stderr,none": 0.03046245111544585, + "sample_count": { + "acc,none": 195 + } + }, + "mmlu": { + "alias": "mmlu", + "name": "mmlu", + "sample_len": 855, + "acc,none": 0.6842105263157895, + "acc_stderr,none": 0.014984590520591782, + "sample_count": { + "acc,none": 855 + } + } + }, + "group_subtasks": { + "mmlu": [ + "mmlu_stem", + "mmlu_other", + "mmlu_social_sciences", + "mmlu_humanities" + ], + "mmlu_stem": [ + "mmlu_abstract_algebra", + "mmlu_anatomy", + "mmlu_astronomy", + "mmlu_college_biology", + "mmlu_college_chemistry", + "mmlu_college_computer_science", + "mmlu_college_mathematics", + "mmlu_college_physics", + "mmlu_computer_security", + "mmlu_conceptual_physics", + "mmlu_electrical_engineering", + "mmlu_elementary_mathematics", + "mmlu_high_school_biology", + "mmlu_high_school_chemistry", + "mmlu_high_school_computer_science", + "mmlu_high_school_mathematics", + "mmlu_high_school_physics", + "mmlu_high_school_statistics", + "mmlu_machine_learning" + ], + "mmlu_other": [ + "mmlu_business_ethics", + "mmlu_clinical_knowledge", + "mmlu_college_medicine", + "mmlu_global_facts", + "mmlu_human_aging", + "mmlu_management", + "mmlu_marketing", + "mmlu_medical_genetics", + "mmlu_miscellaneous", + "mmlu_nutrition", + "mmlu_professional_accounting", + "mmlu_professional_medicine", + "mmlu_virology" + ], + "mmlu_social_sciences": [ + "mmlu_econometrics", + "mmlu_high_school_geography", + "mmlu_high_school_government_and_politics", + "mmlu_high_school_macroeconomics", + "mmlu_high_school_microeconomics", + "mmlu_high_school_psychology", + "mmlu_human_sexuality", + "mmlu_professional_psychology", + "mmlu_public_relations", + "mmlu_security_studies", + "mmlu_sociology", + "mmlu_us_foreign_policy" + ], + "mmlu_humanities": [ + "mmlu_formal_logic", + "mmlu_high_school_european_history", + "mmlu_high_school_us_history", + "mmlu_high_school_world_history", + "mmlu_international_law", + "mmlu_jurisprudence", + "mmlu_logical_fallacies", + "mmlu_moral_disputes", + "mmlu_moral_scenarios", + "mmlu_philosophy", + "mmlu_prehistory", + "mmlu_professional_law", + "mmlu_world_religions" + ] + }, + "configs": { + "mmlu_abstract_algebra": { + "task": "mmlu_abstract_algebra", + "task_alias": "abstract_algebra", + "dataset_path": "cais/mmlu", + "dataset_name": "abstract_algebra", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about abstract algebra.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_abstract_algebra.yaml" + } + }, + "mmlu_anatomy": { + "task": "mmlu_anatomy", + "task_alias": "anatomy", + "dataset_path": "cais/mmlu", + "dataset_name": "anatomy", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about anatomy.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_anatomy.yaml" + } + }, + "mmlu_astronomy": { + "task": "mmlu_astronomy", + "task_alias": "astronomy", + "dataset_path": "cais/mmlu", + "dataset_name": "astronomy", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about astronomy.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_astronomy.yaml" + } + }, + "mmlu_business_ethics": { + "task": "mmlu_business_ethics", + "task_alias": "business_ethics", + "dataset_path": "cais/mmlu", + "dataset_name": "business_ethics", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about business ethics.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_business_ethics.yaml" + } + }, + "mmlu_clinical_knowledge": { + "task": "mmlu_clinical_knowledge", + "task_alias": "clinical_knowledge", + "dataset_path": "cais/mmlu", + "dataset_name": "clinical_knowledge", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about clinical knowledge.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_clinical_knowledge.yaml" + } + }, + "mmlu_college_biology": { + "task": "mmlu_college_biology", + "task_alias": "college_biology", + "dataset_path": "cais/mmlu", + "dataset_name": "college_biology", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about college biology.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_college_biology.yaml" + } + }, + "mmlu_college_chemistry": { + "task": "mmlu_college_chemistry", + "task_alias": "college_chemistry", + "dataset_path": "cais/mmlu", + "dataset_name": "college_chemistry", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about college chemistry.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_college_chemistry.yaml" + } + }, + "mmlu_college_computer_science": { + "task": "mmlu_college_computer_science", + "task_alias": "college_computer_science", + "dataset_path": "cais/mmlu", + "dataset_name": "college_computer_science", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about college computer science.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_college_computer_science.yaml" + } + }, + "mmlu_college_mathematics": { + "task": "mmlu_college_mathematics", + "task_alias": "college_mathematics", + "dataset_path": "cais/mmlu", + "dataset_name": "college_mathematics", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about college mathematics.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_college_mathematics.yaml" + } + }, + "mmlu_college_medicine": { + "task": "mmlu_college_medicine", + "task_alias": "college_medicine", + "dataset_path": "cais/mmlu", + "dataset_name": "college_medicine", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about college medicine.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_college_medicine.yaml" + } + }, + "mmlu_college_physics": { + "task": "mmlu_college_physics", + "task_alias": "college_physics", + "dataset_path": "cais/mmlu", + "dataset_name": "college_physics", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about college physics.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_college_physics.yaml" + } + }, + "mmlu_computer_security": { + "task": "mmlu_computer_security", + "task_alias": "computer_security", + "dataset_path": "cais/mmlu", + "dataset_name": "computer_security", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about computer security.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_computer_security.yaml" + } + }, + "mmlu_conceptual_physics": { + "task": "mmlu_conceptual_physics", + "task_alias": "conceptual_physics", + "dataset_path": "cais/mmlu", + "dataset_name": "conceptual_physics", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about conceptual physics.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_conceptual_physics.yaml" + } + }, + "mmlu_econometrics": { + "task": "mmlu_econometrics", + "task_alias": "econometrics", + "dataset_path": "cais/mmlu", + "dataset_name": "econometrics", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about econometrics.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_econometrics.yaml" + } + }, + "mmlu_electrical_engineering": { + "task": "mmlu_electrical_engineering", + "task_alias": "electrical_engineering", + "dataset_path": "cais/mmlu", + "dataset_name": "electrical_engineering", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about electrical engineering.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_electrical_engineering.yaml" + } + }, + "mmlu_elementary_mathematics": { + "task": "mmlu_elementary_mathematics", + "task_alias": "elementary_mathematics", + "dataset_path": "cais/mmlu", + "dataset_name": "elementary_mathematics", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about elementary mathematics.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_elementary_mathematics.yaml" + } + }, + "mmlu_formal_logic": { + "task": "mmlu_formal_logic", + "task_alias": "formal_logic", + "dataset_path": "cais/mmlu", + "dataset_name": "formal_logic", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about formal logic.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_formal_logic.yaml" + } + }, + "mmlu_global_facts": { + "task": "mmlu_global_facts", + "task_alias": "global_facts", + "dataset_path": "cais/mmlu", + "dataset_name": "global_facts", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about global facts.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_global_facts.yaml" + } + }, + "mmlu_high_school_biology": { + "task": "mmlu_high_school_biology", + "task_alias": "high_school_biology", + "dataset_path": "cais/mmlu", + "dataset_name": "high_school_biology", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about high school biology.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_high_school_biology.yaml" + } + }, + "mmlu_high_school_chemistry": { + "task": "mmlu_high_school_chemistry", + "task_alias": "high_school_chemistry", + "dataset_path": "cais/mmlu", + "dataset_name": "high_school_chemistry", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about high school chemistry.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_high_school_chemistry.yaml" + } + }, + "mmlu_high_school_computer_science": { + "task": "mmlu_high_school_computer_science", + "task_alias": "high_school_computer_science", + "dataset_path": "cais/mmlu", + "dataset_name": "high_school_computer_science", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about high school computer science.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_high_school_computer_science.yaml" + } + }, + "mmlu_high_school_european_history": { + "task": "mmlu_high_school_european_history", + "task_alias": "high_school_european_history", + "dataset_path": "cais/mmlu", + "dataset_name": "high_school_european_history", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about high school european history.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_high_school_european_history.yaml" + } + }, + "mmlu_high_school_geography": { + "task": "mmlu_high_school_geography", + "task_alias": "high_school_geography", + "dataset_path": "cais/mmlu", + "dataset_name": "high_school_geography", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about high school geography.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_high_school_geography.yaml" + } + }, + "mmlu_high_school_government_and_politics": { + "task": "mmlu_high_school_government_and_politics", + "task_alias": "high_school_government_and_politics", + "dataset_path": "cais/mmlu", + "dataset_name": "high_school_government_and_politics", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about high school government and politics.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_high_school_government_and_politics.yaml" + } + }, + "mmlu_high_school_macroeconomics": { + "task": "mmlu_high_school_macroeconomics", + "task_alias": "high_school_macroeconomics", + "dataset_path": "cais/mmlu", + "dataset_name": "high_school_macroeconomics", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about high school macroeconomics.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_high_school_macroeconomics.yaml" + } + }, + "mmlu_high_school_mathematics": { + "task": "mmlu_high_school_mathematics", + "task_alias": "high_school_mathematics", + "dataset_path": "cais/mmlu", + "dataset_name": "high_school_mathematics", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about high school mathematics.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_high_school_mathematics.yaml" + } + }, + "mmlu_high_school_microeconomics": { + "task": "mmlu_high_school_microeconomics", + "task_alias": "high_school_microeconomics", + "dataset_path": "cais/mmlu", + "dataset_name": "high_school_microeconomics", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about high school microeconomics.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_high_school_microeconomics.yaml" + } + }, + "mmlu_high_school_physics": { + "task": "mmlu_high_school_physics", + "task_alias": "high_school_physics", + "dataset_path": "cais/mmlu", + "dataset_name": "high_school_physics", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about high school physics.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_high_school_physics.yaml" + } + }, + "mmlu_high_school_psychology": { + "task": "mmlu_high_school_psychology", + "task_alias": "high_school_psychology", + "dataset_path": "cais/mmlu", + "dataset_name": "high_school_psychology", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about high school psychology.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_high_school_psychology.yaml" + } + }, + "mmlu_high_school_statistics": { + "task": "mmlu_high_school_statistics", + "task_alias": "high_school_statistics", + "dataset_path": "cais/mmlu", + "dataset_name": "high_school_statistics", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about high school statistics.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_high_school_statistics.yaml" + } + }, + "mmlu_high_school_us_history": { + "task": "mmlu_high_school_us_history", + "task_alias": "high_school_us_history", + "dataset_path": "cais/mmlu", + "dataset_name": "high_school_us_history", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about high school us history.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_high_school_us_history.yaml" + } + }, + "mmlu_high_school_world_history": { + "task": "mmlu_high_school_world_history", + "task_alias": "high_school_world_history", + "dataset_path": "cais/mmlu", + "dataset_name": "high_school_world_history", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about high school world history.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_high_school_world_history.yaml" + } + }, + "mmlu_human_aging": { + "task": "mmlu_human_aging", + "task_alias": "human_aging", + "dataset_path": "cais/mmlu", + "dataset_name": "human_aging", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about human aging.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_human_aging.yaml" + } + }, + "mmlu_human_sexuality": { + "task": "mmlu_human_sexuality", + "task_alias": "human_sexuality", + "dataset_path": "cais/mmlu", + "dataset_name": "human_sexuality", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about human sexuality.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_human_sexuality.yaml" + } + }, + "mmlu_international_law": { + "task": "mmlu_international_law", + "task_alias": "international_law", + "dataset_path": "cais/mmlu", + "dataset_name": "international_law", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about international law.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_international_law.yaml" + } + }, + "mmlu_jurisprudence": { + "task": "mmlu_jurisprudence", + "task_alias": "jurisprudence", + "dataset_path": "cais/mmlu", + "dataset_name": "jurisprudence", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about jurisprudence.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_jurisprudence.yaml" + } + }, + "mmlu_logical_fallacies": { + "task": "mmlu_logical_fallacies", + "task_alias": "logical_fallacies", + "dataset_path": "cais/mmlu", + "dataset_name": "logical_fallacies", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about logical fallacies.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_logical_fallacies.yaml" + } + }, + "mmlu_machine_learning": { + "task": "mmlu_machine_learning", + "task_alias": "machine_learning", + "dataset_path": "cais/mmlu", + "dataset_name": "machine_learning", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about machine learning.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_machine_learning.yaml" + } + }, + "mmlu_management": { + "task": "mmlu_management", + "task_alias": "management", + "dataset_path": "cais/mmlu", + "dataset_name": "management", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about management.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_management.yaml" + } + }, + "mmlu_marketing": { + "task": "mmlu_marketing", + "task_alias": "marketing", + "dataset_path": "cais/mmlu", + "dataset_name": "marketing", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about marketing.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_marketing.yaml" + } + }, + "mmlu_medical_genetics": { + "task": "mmlu_medical_genetics", + "task_alias": "medical_genetics", + "dataset_path": "cais/mmlu", + "dataset_name": "medical_genetics", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about medical genetics.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_medical_genetics.yaml" + } + }, + "mmlu_miscellaneous": { + "task": "mmlu_miscellaneous", + "task_alias": "miscellaneous", + "dataset_path": "cais/mmlu", + "dataset_name": "miscellaneous", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about miscellaneous.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_miscellaneous.yaml" + } + }, + "mmlu_moral_disputes": { + "task": "mmlu_moral_disputes", + "task_alias": "moral_disputes", + "dataset_path": "cais/mmlu", + "dataset_name": "moral_disputes", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about moral disputes.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_moral_disputes.yaml" + } + }, + "mmlu_moral_scenarios": { + "task": "mmlu_moral_scenarios", + "task_alias": "moral_scenarios", + "dataset_path": "cais/mmlu", + "dataset_name": "moral_scenarios", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about moral scenarios.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_moral_scenarios.yaml" + } + }, + "mmlu_nutrition": { + "task": "mmlu_nutrition", + "task_alias": "nutrition", + "dataset_path": "cais/mmlu", + "dataset_name": "nutrition", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about nutrition.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_nutrition.yaml" + } + }, + "mmlu_philosophy": { + "task": "mmlu_philosophy", + "task_alias": "philosophy", + "dataset_path": "cais/mmlu", + "dataset_name": "philosophy", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about philosophy.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_philosophy.yaml" + } + }, + "mmlu_prehistory": { + "task": "mmlu_prehistory", + "task_alias": "prehistory", + "dataset_path": "cais/mmlu", + "dataset_name": "prehistory", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about prehistory.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_prehistory.yaml" + } + }, + "mmlu_professional_accounting": { + "task": "mmlu_professional_accounting", + "task_alias": "professional_accounting", + "dataset_path": "cais/mmlu", + "dataset_name": "professional_accounting", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about professional accounting.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_professional_accounting.yaml" + } + }, + "mmlu_professional_law": { + "task": "mmlu_professional_law", + "task_alias": "professional_law", + "dataset_path": "cais/mmlu", + "dataset_name": "professional_law", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about professional law.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_professional_law.yaml" + } + }, + "mmlu_professional_medicine": { + "task": "mmlu_professional_medicine", + "task_alias": "professional_medicine", + "dataset_path": "cais/mmlu", + "dataset_name": "professional_medicine", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about professional medicine.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_professional_medicine.yaml" + } + }, + "mmlu_professional_psychology": { + "task": "mmlu_professional_psychology", + "task_alias": "professional_psychology", + "dataset_path": "cais/mmlu", + "dataset_name": "professional_psychology", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about professional psychology.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_professional_psychology.yaml" + } + }, + "mmlu_public_relations": { + "task": "mmlu_public_relations", + "task_alias": "public_relations", + "dataset_path": "cais/mmlu", + "dataset_name": "public_relations", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about public relations.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_public_relations.yaml" + } + }, + "mmlu_security_studies": { + "task": "mmlu_security_studies", + "task_alias": "security_studies", + "dataset_path": "cais/mmlu", + "dataset_name": "security_studies", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about security studies.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_security_studies.yaml" + } + }, + "mmlu_sociology": { + "task": "mmlu_sociology", + "task_alias": "sociology", + "dataset_path": "cais/mmlu", + "dataset_name": "sociology", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about sociology.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_sociology.yaml" + } + }, + "mmlu_us_foreign_policy": { + "task": "mmlu_us_foreign_policy", + "task_alias": "us_foreign_policy", + "dataset_path": "cais/mmlu", + "dataset_name": "us_foreign_policy", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about us foreign policy.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_us_foreign_policy.yaml" + } + }, + "mmlu_virology": { + "task": "mmlu_virology", + "task_alias": "virology", + "dataset_path": "cais/mmlu", + "dataset_name": "virology", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about virology.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_virology.yaml" + } + }, + "mmlu_world_religions": { + "task": "mmlu_world_religions", + "task_alias": "world_religions", + "dataset_path": "cais/mmlu", + "dataset_name": "world_religions", + "test_split": "test", + "fewshot_split": "dev", + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_target": "answer", + "unsafe_code": false, + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "description": "The following are multiple choice questions (with answers) about world religions.\n\n", + "target_delimiter": " ", + "fewshot_delimiter": "\n\n", + "fewshot_config": { + "sampler": "first_n", + "split": "dev", + "process_docs": null, + "fewshot_indices": null, + "samples": null, + "doc_to_text": "{{question.strip()}}\nA. {{choices[0]}}\nB. {{choices[1]}}\nC. {{choices[2]}}\nD. {{choices[3]}}\nAnswer:", + "doc_to_choice": [ + "A", + "B", + "C", + "D" + ], + "doc_to_target": "answer", + "gen_prefix": null, + "fewshot_delimiter": "\n\n", + "target_delimiter": " " + }, + "num_fewshot": 5, + "metric_list": [ + { + "metric": "acc", + "aggregation": "mean", + "higher_is_better": true + } + ], + "output_type": "multiple_choice", + "repeats": 1, + "should_decontaminate": false, + "metadata": { + "version": 1.0, + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "config_source": "/usr/local/lib/python3.11/dist-packages/lm_eval/tasks/mmlu/default/mmlu_world_religions.yaml" + } + } + }, + "versions": { + "mmlu": "2", + "mmlu_abstract_algebra": 1.0, + "mmlu_anatomy": 1.0, + "mmlu_astronomy": 1.0, + "mmlu_business_ethics": 1.0, + "mmlu_clinical_knowledge": 1.0, + "mmlu_college_biology": 1.0, + "mmlu_college_chemistry": 1.0, + "mmlu_college_computer_science": 1.0, + "mmlu_college_mathematics": 1.0, + "mmlu_college_medicine": 1.0, + "mmlu_college_physics": 1.0, + "mmlu_computer_security": 1.0, + "mmlu_conceptual_physics": 1.0, + "mmlu_econometrics": 1.0, + "mmlu_electrical_engineering": 1.0, + "mmlu_elementary_mathematics": 1.0, + "mmlu_formal_logic": 1.0, + "mmlu_global_facts": 1.0, + "mmlu_high_school_biology": 1.0, + "mmlu_high_school_chemistry": 1.0, + "mmlu_high_school_computer_science": 1.0, + "mmlu_high_school_european_history": 1.0, + "mmlu_high_school_geography": 1.0, + "mmlu_high_school_government_and_politics": 1.0, + "mmlu_high_school_macroeconomics": 1.0, + "mmlu_high_school_mathematics": 1.0, + "mmlu_high_school_microeconomics": 1.0, + "mmlu_high_school_physics": 1.0, + "mmlu_high_school_psychology": 1.0, + "mmlu_high_school_statistics": 1.0, + "mmlu_high_school_us_history": 1.0, + "mmlu_high_school_world_history": 1.0, + "mmlu_human_aging": 1.0, + "mmlu_human_sexuality": 1.0, + "mmlu_humanities": "2", + "mmlu_international_law": 1.0, + "mmlu_jurisprudence": 1.0, + "mmlu_logical_fallacies": 1.0, + "mmlu_machine_learning": 1.0, + "mmlu_management": 1.0, + "mmlu_marketing": 1.0, + "mmlu_medical_genetics": 1.0, + "mmlu_miscellaneous": 1.0, + "mmlu_moral_disputes": 1.0, + "mmlu_moral_scenarios": 1.0, + "mmlu_nutrition": 1.0, + "mmlu_other": "2", + "mmlu_philosophy": 1.0, + "mmlu_prehistory": 1.0, + "mmlu_professional_accounting": 1.0, + "mmlu_professional_law": 1.0, + "mmlu_professional_medicine": 1.0, + "mmlu_professional_psychology": 1.0, + "mmlu_public_relations": 1.0, + "mmlu_security_studies": 1.0, + "mmlu_social_sciences": "2", + "mmlu_sociology": 1.0, + "mmlu_stem": "2", + "mmlu_us_foreign_policy": 1.0, + "mmlu_virology": 1.0, + "mmlu_world_religions": 1.0 + }, + "n-shot": { + "mmlu_abstract_algebra": 5, + "mmlu_anatomy": 5, + "mmlu_astronomy": 5, + "mmlu_business_ethics": 5, + "mmlu_clinical_knowledge": 5, + "mmlu_college_biology": 5, + "mmlu_college_chemistry": 5, + "mmlu_college_computer_science": 5, + "mmlu_college_mathematics": 5, + "mmlu_college_medicine": 5, + "mmlu_college_physics": 5, + "mmlu_computer_security": 5, + "mmlu_conceptual_physics": 5, + "mmlu_econometrics": 5, + "mmlu_electrical_engineering": 5, + "mmlu_elementary_mathematics": 5, + "mmlu_formal_logic": 5, + "mmlu_global_facts": 5, + "mmlu_high_school_biology": 5, + "mmlu_high_school_chemistry": 5, + "mmlu_high_school_computer_science": 5, + "mmlu_high_school_european_history": 5, + "mmlu_high_school_geography": 5, + "mmlu_high_school_government_and_politics": 5, + "mmlu_high_school_macroeconomics": 5, + "mmlu_high_school_mathematics": 5, + "mmlu_high_school_microeconomics": 5, + "mmlu_high_school_physics": 5, + "mmlu_high_school_psychology": 5, + "mmlu_high_school_statistics": 5, + "mmlu_high_school_us_history": 5, + "mmlu_high_school_world_history": 5, + "mmlu_human_aging": 5, + "mmlu_human_sexuality": 5, + "mmlu_humanities": 5, + "mmlu_international_law": 5, + "mmlu_jurisprudence": 5, + "mmlu_logical_fallacies": 5, + "mmlu_machine_learning": 5, + "mmlu_management": 5, + "mmlu_marketing": 5, + "mmlu_medical_genetics": 5, + "mmlu_miscellaneous": 5, + "mmlu_moral_disputes": 5, + "mmlu_moral_scenarios": 5, + "mmlu_nutrition": 5, + "mmlu_other": 5, + "mmlu_philosophy": 5, + "mmlu_prehistory": 5, + "mmlu_professional_accounting": 5, + "mmlu_professional_law": 5, + "mmlu_professional_medicine": 5, + "mmlu_professional_psychology": 5, + "mmlu_public_relations": 5, + "mmlu_security_studies": 5, + "mmlu_social_sciences": 5, + "mmlu_sociology": 5, + "mmlu_stem": 5, + "mmlu_us_foreign_policy": 5, + "mmlu_virology": 5, + "mmlu_world_religions": 5 + }, + "higher_is_better": { + "mmlu_abstract_algebra": { + "acc": true + }, + "mmlu_anatomy": { + "acc": true + }, + "mmlu_astronomy": { + "acc": true + }, + "mmlu_business_ethics": { + "acc": true + }, + "mmlu_clinical_knowledge": { + "acc": true + }, + "mmlu_college_biology": { + "acc": true + }, + "mmlu_college_chemistry": { + "acc": true + }, + "mmlu_college_computer_science": { + "acc": true + }, + "mmlu_college_mathematics": { + "acc": true + }, + "mmlu_college_medicine": { + "acc": true + }, + "mmlu_college_physics": { + "acc": true + }, + "mmlu_computer_security": { + "acc": true + }, + "mmlu_conceptual_physics": { + "acc": true + }, + "mmlu_econometrics": { + "acc": true + }, + "mmlu_electrical_engineering": { + "acc": true + }, + "mmlu_elementary_mathematics": { + "acc": true + }, + "mmlu_formal_logic": { + "acc": true + }, + "mmlu_global_facts": { + "acc": true + }, + "mmlu_high_school_biology": { + "acc": true + }, + "mmlu_high_school_chemistry": { + "acc": true + }, + "mmlu_high_school_computer_science": { + "acc": true + }, + "mmlu_high_school_european_history": { + "acc": true + }, + "mmlu_high_school_geography": { + "acc": true + }, + "mmlu_high_school_government_and_politics": { + "acc": true + }, + "mmlu_high_school_macroeconomics": { + "acc": true + }, + "mmlu_high_school_mathematics": { + "acc": true + }, + "mmlu_high_school_microeconomics": { + "acc": true + }, + "mmlu_high_school_physics": { + "acc": true + }, + "mmlu_high_school_psychology": { + "acc": true + }, + "mmlu_high_school_statistics": { + "acc": true + }, + "mmlu_high_school_us_history": { + "acc": true + }, + "mmlu_high_school_world_history": { + "acc": true + }, + "mmlu_human_aging": { + "acc": true + }, + "mmlu_human_sexuality": { + "acc": true + }, + "mmlu_humanities": { + "acc": true + }, + "mmlu_international_law": { + "acc": true + }, + "mmlu_jurisprudence": { + "acc": true + }, + "mmlu_logical_fallacies": { + "acc": true + }, + "mmlu_machine_learning": { + "acc": true + }, + "mmlu_management": { + "acc": true + }, + "mmlu_marketing": { + "acc": true + }, + "mmlu_medical_genetics": { + "acc": true + }, + "mmlu_miscellaneous": { + "acc": true + }, + "mmlu_moral_disputes": { + "acc": true + }, + "mmlu_moral_scenarios": { + "acc": true + }, + "mmlu_nutrition": { + "acc": true + }, + "mmlu_other": { + "acc": true + }, + "mmlu_philosophy": { + "acc": true + }, + "mmlu_prehistory": { + "acc": true + }, + "mmlu_professional_accounting": { + "acc": true + }, + "mmlu_professional_law": { + "acc": true + }, + "mmlu_professional_medicine": { + "acc": true + }, + "mmlu_professional_psychology": { + "acc": true + }, + "mmlu_public_relations": { + "acc": true + }, + "mmlu_security_studies": { + "acc": true + }, + "mmlu_social_sciences": { + "acc": true + }, + "mmlu_sociology": { + "acc": true + }, + "mmlu_stem": { + "acc": true + }, + "mmlu_us_foreign_policy": { + "acc": true + }, + "mmlu_virology": { + "acc": true + }, + "mmlu_world_religions": { + "acc": true + } + }, + "n-samples": { + "mmlu_abstract_algebra": { + "original": 100, + "effective": 15 + }, + "mmlu_anatomy": { + "original": 135, + "effective": 15 + }, + "mmlu_astronomy": { + "original": 152, + "effective": 15 + }, + "mmlu_college_biology": { + "original": 144, + "effective": 15 + }, + "mmlu_college_chemistry": { + "original": 100, + "effective": 15 + }, + "mmlu_college_computer_science": { + "original": 100, + "effective": 15 + }, + "mmlu_college_mathematics": { + "original": 100, + "effective": 15 + }, + "mmlu_college_physics": { + "original": 102, + "effective": 15 + }, + "mmlu_computer_security": { + "original": 100, + "effective": 15 + }, + "mmlu_conceptual_physics": { + "original": 235, + "effective": 15 + }, + "mmlu_electrical_engineering": { + "original": 145, + "effective": 15 + }, + "mmlu_elementary_mathematics": { + "original": 378, + "effective": 15 + }, + "mmlu_high_school_biology": { + "original": 310, + "effective": 15 + }, + "mmlu_high_school_chemistry": { + "original": 203, + "effective": 15 + }, + "mmlu_high_school_computer_science": { + "original": 100, + "effective": 15 + }, + "mmlu_high_school_mathematics": { + "original": 270, + "effective": 15 + }, + "mmlu_high_school_physics": { + "original": 151, + "effective": 15 + }, + "mmlu_high_school_statistics": { + "original": 216, + "effective": 15 + }, + "mmlu_machine_learning": { + "original": 112, + "effective": 15 + }, + "mmlu_business_ethics": { + "original": 100, + "effective": 15 + }, + "mmlu_clinical_knowledge": { + "original": 265, + "effective": 15 + }, + "mmlu_college_medicine": { + "original": 173, + "effective": 15 + }, + "mmlu_global_facts": { + "original": 100, + "effective": 15 + }, + "mmlu_human_aging": { + "original": 223, + "effective": 15 + }, + "mmlu_management": { + "original": 103, + "effective": 15 + }, + "mmlu_marketing": { + "original": 234, + "effective": 15 + }, + "mmlu_medical_genetics": { + "original": 100, + "effective": 15 + }, + "mmlu_miscellaneous": { + "original": 783, + "effective": 15 + }, + "mmlu_nutrition": { + "original": 306, + "effective": 15 + }, + "mmlu_professional_accounting": { + "original": 282, + "effective": 15 + }, + "mmlu_professional_medicine": { + "original": 272, + "effective": 15 + }, + "mmlu_virology": { + "original": 166, + "effective": 15 + }, + "mmlu_econometrics": { + "original": 114, + "effective": 15 + }, + "mmlu_high_school_geography": { + "original": 198, + "effective": 15 + }, + "mmlu_high_school_government_and_politics": { + "original": 193, + "effective": 15 + }, + "mmlu_high_school_macroeconomics": { + "original": 390, + "effective": 15 + }, + "mmlu_high_school_microeconomics": { + "original": 238, + "effective": 15 + }, + "mmlu_high_school_psychology": { + "original": 545, + "effective": 15 + }, + "mmlu_human_sexuality": { + "original": 131, + "effective": 15 + }, + "mmlu_professional_psychology": { + "original": 612, + "effective": 15 + }, + "mmlu_public_relations": { + "original": 110, + "effective": 15 + }, + "mmlu_security_studies": { + "original": 245, + "effective": 15 + }, + "mmlu_sociology": { + "original": 201, + "effective": 15 + }, + "mmlu_us_foreign_policy": { + "original": 100, + "effective": 15 + }, + "mmlu_formal_logic": { + "original": 126, + "effective": 15 + }, + "mmlu_high_school_european_history": { + "original": 165, + "effective": 15 + }, + "mmlu_high_school_us_history": { + "original": 204, + "effective": 15 + }, + "mmlu_high_school_world_history": { + "original": 237, + "effective": 15 + }, + "mmlu_international_law": { + "original": 121, + "effective": 15 + }, + "mmlu_jurisprudence": { + "original": 108, + "effective": 15 + }, + "mmlu_logical_fallacies": { + "original": 163, + "effective": 15 + }, + "mmlu_moral_disputes": { + "original": 346, + "effective": 15 + }, + "mmlu_moral_scenarios": { + "original": 895, + "effective": 15 + }, + "mmlu_philosophy": { + "original": 311, + "effective": 15 + }, + "mmlu_prehistory": { + "original": 324, + "effective": 15 + }, + "mmlu_professional_law": { + "original": 1534, + "effective": 15 + }, + "mmlu_world_religions": { + "original": 171, + "effective": 15 + } + }, + "config": { + "model": "hf", + "model_args": { + "pretrained": "Qwen/Qwen2.5-3B-Instruct", + "dtype": "bfloat16", + "peft": "/root/steering-resistance/results/full_3b/m1_resist_adapter" + }, + "model_num_parameters": 3115317248, + "model_dtype": "torch.bfloat16", + "model_revision": "main", + "model_sha": "aa8e72537993ba99e69dfaafa59ed015b17504d1", + "peft_sha": "", + "batch_size": "auto", + "batch_sizes": [ + 3 + ], + "device": "cuda:0", + "use_cache": null, + "limit": 15.0, + "bootstrap_iters": 100000, + "gen_kwargs": {}, + "random_seed": 0, + "numpy_seed": 0, + "torch_seed": 0, + "fewshot_seed": 0 + }, + "git_hash": "eb4f2be22f7baf6d268c3dd5e46d49d6bd2e74ae", + "date": 1784831798.5202086, + "pretty_env_info": "PyTorch version: 2.6.0+cu124\nIs debug build: False\nCUDA used to build PyTorch: 12.4\nROCM used to build PyTorch: N/A\n\nOS: Ubuntu 22.04.5 LTS (x86_64)\nGCC version: (Ubuntu 11.4.0-1ubuntu1~22.04) 11.4.0\nClang version: Could not collect\nCMake version: Could not collect\nLibc version: glibc-2.35\n\nPython version: 3.11.10 (main, Sep 7 2024, 18:35:41) [GCC 11.4.0] (64-bit runtime)\nPython platform: Linux-6.8.0-52-generic-x86_64-with-glibc2.35\nIs CUDA available: True\nCUDA runtime version: Could not collect\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: GPU 0: NVIDIA RTX A4000\nNvidia driver version: 550.144.03\ncuDNN version: Could not collect\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 48 bits physical, 48 bits virtual\nByte Order: Little Endian\nCPU(s): 112\nOn-line CPU(s) list: 0-111\nVendor ID: AuthenticAMD\nModel name: AMD EPYC 7453 28-Core Processor\nCPU family: 25\nModel: 1\nThread(s) per core: 2\nCore(s) per socket: 28\nSocket(s): 2\nStepping: 1\nFrequency boost: enabled\nCPU max MHz: 3488.5249\nCPU min MHz: 1500.0000\nBogoMIPS: 5489.75\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush mmx fxsr sse sse2 ht syscall nx mmxext fxsr_opt pdpe1gb rdtscp lm constant_tsc rep_good nopl nonstop_tsc cpuid extd_apicid aperfmperf rapl pni pclmulqdq monitor ssse3 fma cx16 pcid sse4_1 sse4_2 movbe popcnt aes xsave avx f16c rdrand lahf_lm cmp_legacy svm extapic cr8_legacy abm sse4a misalignsse 3dnowprefetch osvw ibs skinit wdt tce topoext perfctr_core perfctr_nb bpext perfctr_llc mwaitx cpb cat_l3 cdp_l3 hw_pstate ssbd mba ibrs ibpb stibp vmmcall fsgsbase bmi1 avx2 smep bmi2 erms invpcid cqm rdt_a rdseed adx smap clflushopt clwb sha_ni xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local user_shstk clzero irperf xsaveerptr rdpru wbnoinvd amd_ppin brs arat npt lbrv svm_lock nrip_save tsc_scale vmcb_clean flushbyasid decodeassists pausefilter pfthreshold v_vmsave_vmload vgif v_spec_ctrl umip pku ospke vaes vpclmulqdq rdpid overflow_recov succor smca fsrm debug_swap\nVirtualization: AMD-V\nL1d cache: 1.8 MiB (56 instances)\nL1i cache: 1.8 MiB (56 instances)\nL2 cache: 28 MiB (56 instances)\nL3 cache: 128 MiB (8 instances)\nNUMA node(s): 2\nNUMA node0 CPU(s): 0-27,56-83\nNUMA node1 CPU(s): 28-55,84-111\nVulnerability Gather data sampling: Not affected\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Mmio stale data: Not affected\nVulnerability Reg file data sampling: Not affected\nVulnerability Retbleed: Not affected\nVulnerability Spec rstack overflow: Mitigation; Safe RET\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Retpolines; IBPB conditional; IBRS_FW; STIBP always-on; RSB filling; PBRSB-eIBRS Not affected; BHI Not affected\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] numpy==2.4.6\n[pip3] nvidia-cublas-cu12==12.4.5.8\n[pip3] nvidia-cuda-cupti-cu12==12.4.127\n[pip3] nvidia-cuda-nvrtc-cu12==12.4.127\n[pip3] nvidia-cuda-runtime-cu12==12.4.127\n[pip3] nvidia-cudnn-cu12==9.1.0.70\n[pip3] nvidia-cufft-cu12==11.2.1.3\n[pip3] nvidia-curand-cu12==10.3.5.147\n[pip3] nvidia-cusolver-cu12==11.6.1.9\n[pip3] nvidia-cusparse-cu12==12.3.1.170\n[pip3] nvidia-cusparselt-cu12==0.6.2\n[pip3] nvidia-nccl-cu12==2.21.5\n[pip3] nvidia-nvjitlink-cu12==12.4.127\n[pip3] nvidia-nvtx-cu12==12.4.127\n[pip3] torch==2.6.0+cu124\n[pip3] triton==3.2.0\n[conda] Could not collect", + "transformers_version": "5.14.1", + "lm_eval_version": "0.4.12", + "upper_git_hash": null, + "tokenizer_pad_token": [ + "<|endoftext|>", + "151643" + ], + "tokenizer_eos_token": [ + "<|im_end|>", + "151645" + ], + "tokenizer_bos_token": [ + null, + "None" + ], + "eot_token_id": 151645, + "max_length": 32768, + "task_hashes": {}, + "model_source": "hf", + "model_name": "/root/steering-resistance/results/full_3b/m1_resist_adapter", + "model_name_sanitized": "__root__steering-resistance__results__full_3b__m1_resist_adapter", + "system_instruction": null, + "system_instruction_sha": null, + "fewshot_as_multiturn": true, + "chat_template": "{%- if tools %}\n {{- '<|im_start|>system\\n' }}\n {%- if messages[0]['role'] == 'system' %}\n {{- messages[0]['content'] }}\n {%- else %}\n {{- 'You are Qwen, created by Alibaba Cloud. You are a helpful assistant.' }}\n {%- endif %}\n {{- \"\\n\\n# Tools\\n\\nYou may call one or more functions to assist with the user query.\\n\\nYou are provided with function signatures within XML tags:\\n\" }}\n {%- for tool in tools %}\n {{- \"\\n\" }}\n {{- tool | tojson }}\n {%- endfor %}\n {{- \"\\n\\n\\nFor each function call, return a json object with function name and arguments within XML tags:\\n\\n{\\\"name\\\": , \\\"arguments\\\": }\\n<|im_end|>\\n\" }}\n{%- else %}\n {%- if messages[0]['role'] == 'system' %}\n {{- '<|im_start|>system\\n' + messages[0]['content'] + '<|im_end|>\\n' }}\n {%- else %}\n {{- '<|im_start|>system\\nYou are Qwen, created by Alibaba Cloud. You are a helpful assistant.<|im_end|>\\n' }}\n {%- endif %}\n{%- endif %}\n{%- for message in messages %}\n {%- if (message.role == \"user\") or (message.role == \"system\" and not loop.first) or (message.role == \"assistant\" and not message.tool_calls) %}\n {{- '<|im_start|>' + message.role + '\\n' + message.content + '<|im_end|>' + '\\n' }}\n {%- elif message.role == \"assistant\" %}\n {{- '<|im_start|>' + message.role }}\n {%- if message.content %}\n {{- '\\n' + message.content }}\n {%- endif %}\n {%- for tool_call in message.tool_calls %}\n {%- if tool_call.function is defined %}\n {%- set tool_call = tool_call.function %}\n {%- endif %}\n {{- '\\n\\n{\"name\": \"' }}\n {{- tool_call.name }}\n {{- '\", \"arguments\": ' }}\n {{- tool_call.arguments | tojson }}\n {{- '}\\n' }}\n {%- endfor %}\n {{- '<|im_end|>\\n' }}\n {%- elif message.role == \"tool\" %}\n {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != \"tool\") %}\n {{- '<|im_start|>user' }}\n {%- endif %}\n {{- '\\n\\n' }}\n {{- message.content }}\n {{- '\\n' }}\n {%- if loop.last or (messages[loop.index0 + 1].role != \"tool\") %}\n {{- '<|im_end|>\\n' }}\n {%- endif %}\n {%- endif %}\n{%- endfor %}\n{%- if add_generation_prompt %}\n {{- '<|im_start|>assistant\\n' }}\n{%- endif %}\n", + "chat_template_sha": "cd8e9439f0570856fd70470bf8889ebd8b5d1107207f67a5efb46e342330527f", + "total_evaluation_time_seconds": "315.86453066021204" +} \ No newline at end of file diff --git a/run/invocations.jsonl b/run/invocations.jsonl new file mode 100644 index 0000000000000000000000000000000000000000..05f83d252fab113ac5be7788a1be5dfb049847c9 --- /dev/null +++ b/run/invocations.jsonl @@ -0,0 +1 @@ +{"time": "2026-07-23T16:19:29+0000", "stages": ["vectors", "data", "eval_m0", "train", "eval_m1"], "commit": "none", "status": "success", "headline": "clean 100%->100% · steer_heldout@1.6 correct 0%->3%", "wandb_url": null} diff --git a/run/run_meta.json b/run/run_meta.json index ee9930148275f220202be2315d1f7c1c10b12bfc..095bfd96ae508a79bb4a060a74fa3bf5d9d7e276 100644 --- a/run/run_meta.json +++ b/run/run_meta.json @@ -110,5 +110,51 @@ "sha256": "fa2356571420fd8c2a444aee6e8c879b865e0cd21f5ade2938a0308effbed8f2" } }, - "status": "running" + "status": "success", + "wandb_url": null, + "headline": "clean 100%->100% · steer_heldout@1.6 correct 0%->3%", + "hub_url": "https://huggingface.co/JacoDuToit/steer-full_3b", + "finished_at": "2026-07-23T17:41:53+0000", + "artifacts": { + "eval_m0.jsonl": { + "sha256": "104de23c0313f90e885f92c12006dca7433350b6725f369622d5d2531210b27c", + "bytes": 4576432 + }, + "eval_m1.jsonl": { + "sha256": "03ae380b0768f511b1bedafce30c516daa4a10b71aa9a1bb0a8113300fca4469", + "bytes": 3877293 + }, + "eval_questions.json": { + "sha256": "8771796f4c91a61a901278e799032623900657aaa8f9e44ce789dd1fadd8bf8f", + "bytes": 27617 + }, + "m1_resist_adapter/README.md": { + "sha256": "3581cc6c99fe8b3ab31a0fc4a131726f5d714856745f1598891a0151c2540256", + "bytes": 5202 + }, + "m1_resist_adapter/adapter_config.json": { + "sha256": "390e661694a2f685b6256b5d38a80aa41c5236a115df9cfb5edd4a96da0e568e", + "bytes": 1103 + }, + "m1_resist_adapter/adapter_model.safetensors": { + "sha256": "bc5cf2921ad4f341fc270e0b53641804b4c725c9031d5d232dd6cb1f8877a23d", + "bytes": 119801528 + }, + "summary.csv": { + "sha256": "8de370b8f4acdc3b46173f4bb76c96185e27f8bffc117c4585e8efb4966e834e", + "bytes": 2607 + }, + "summary.md": { + "sha256": "df82f8ebb8f8c63fae08a18eb0be3ef4805052dff62bc7e812d5953600f2febe", + "bytes": 2617 + }, + "train_examples.json": { + "sha256": "3e2ac73debbb5da8cadb4e114a35f6346f015ca0f22d52b9486a7fd2b2b7e824", + "bytes": 284577 + }, + "vectors.pt": { + "sha256": "5c9783914f6419b376d133a3872349ac5a07b42c8456894a3214ffa6618158c4", + "bytes": 7757862 + } + } } \ No newline at end of file