# flake8: noqa
from mmengine.config import read_base
from opencompass.datasets import GPQADataset, GPQAEvaluator, JsonlDataset, LongBenchRougeEvaluator
from opencompass.models import OpenAISDK
from opencompass.openicl.icl_inferencer import GenInferencer
from opencompass.openicl.icl_prompt_template import PromptTemplate
from opencompass.openicl.icl_retriever import ZeroRetriever
from opencompass.partitioners import NaivePartitioner, NumWorkerPartitioner
from opencompass.partitioners.sub_num_worker import SubjectiveNumWorkerPartitioner
from opencompass.runners import LocalRunner
from opencompass.tasks import OpenICLEvalTask, OpenICLInferTask
from opencompass.utils import first_option_postprocess
from opencompass_local_evaluators import MeetingBankRougeEvaluator
# Dataset Configurations
with read_base():
# Datasets
from opencompass.configs.datasets.aime2024.aime2024_0shot_nocot_genericllmeval_academic_gen import aime2024_datasets
from opencompass.configs.datasets.aime2025.aime2025_llmjudge_academic import aime2025_datasets
from opencompass.configs.datasets.gpqa.gpqa_cascade_eval_academic import gpqa_datasets
from opencompass.configs.datasets.humaneval.humaneval_gen_8e312c import humaneval_datasets
from opencompass.configs.datasets.HLE.hle_llmverify_academic import hle_datasets
from opencompass.configs.datasets.IFEval.IFEval_gen_353ae7 import ifeval_datasets
from opencompass.configs.datasets.leval.levalmeetingsumm.leval_meetingsumm_gen_b03798 import LEval_meetingsumm_datasets
from opencompass.configs.datasets.livecodebench.livecodebench_v6_academic import LCBCodeGeneration_dataset
from opencompass.configs.datasets.longbench.longbenchqmsum.longbench_qmsum_gen_d33331 import LongBench_qmsum_datasets
from opencompass.configs.datasets.longbench.longbenchsamsum.longbench_samsum_gen_f4416d import LongBench_samsum_datasets
from opencompass.configs.datasets.mbpp.mbpp_gen import mbpp_datasets
from opencompass.configs.datasets.mmlu_pro.mmlu_pro_0shot_nocot_genericllmeval_gen_08c1de import mmlu_pro_datasets
from opencompass.configs.datasets.subjective.arena_hard.arena_hard_compare_new import arenahard_datasets
# Summary Groups
from opencompass.configs.summarizers.groups.mmlu_pro import mmlu_pro_summary_groups
gpqa_llada_five_shot_prompt = """
Here are some example questions from experts. Answer the final question yourself, following the format of the previous questions exactly.
Question: In a given population, 1 out of every 400 people has a cancer caused by a completely recessive allele, b. Assuming the population is in Hardy-Weinberg equilibrium, which of the following is the expected proportion of individuals who carry the b allele but are not expected to develop the cancer? Choices: (A) 1/400 (B) 19/400 (C) 20/400 (D) 38/400 The correct answer is (D)
Question: A Fe pellet of 0.056 g is first dissolved in 10 mL of hydrobromic acid HBr (0.1 M). The resulting solution is then titrated by KMnO4 (0.02 M). How many equivalence points are there? choices: (A) Two points, 25 ml and 35 ml (B) One point, 25 mL (C) One point, 10 ml (D) Two points, 25 ml and 30 ml The correct answer is (A)
Question: Consider a quantum mechanical system containing a particle of mass $m$ moving in an istropic three dimensional potential of the form $V(r) = 1/2 m \\omega^2 r^2$ corresponding to the acted force obeying Hooke's law. Here, $\\omega$ is the angular frequency of oscillation and $r$ is the radial distance of the particle from the origin in spherical polar coordinate. What is the value of energy of the third excited state, and how many linearly independent eigenfunctions are possible for the same energy eigenvalue? Choices: (A) 11 \\pi^2 \\hbar^2 / (2m r^2), 3 (B) (9/2) \\hbar \\omega , 10 (C) 11 \\pi^2 \\hbar^2 / (2m r^2), 10 (D) (9/2) \\hbar \\omega, 3 The correct answer is (B)
Question: Your overhear two chemists talking to each other as they leave a synthetic organic chemistry lab. One asks the other "So, how did it go?" The second chemist replies, "Not well - my compounds are on top of each other." What is the second chemist most likely referring to? Choices: (A) The compounds they are working with have similar polarities. (B) The compounds they are working with have similar boiling points. (C) The compounds they are working with are bonding to each other through non-covalent/van der Waals interactions. (D) The compounds they are working with have similar optical rotations. The correct answer is (A)
Question: Mitochondria are semi-autonomous cellular organelles in charge of energy production. They encode for a part of their own translational machinery and respiratory complexes. Mitochondrial function is governed by over a thousand proteins imported from the cell, contributing to processes like the transport of proteins, ribosome biogenesis and translation regulation, respiratory oxidation, metabolism, and apoptotic signaling cascade. Mutations in the code for mitochondrial protein networks can cause numerous diseases in humans that are inherited through generations. Mutations of which of the mitochondrial proteins listed below are least likely to be genetically transmitted from a father to his children? Choices: (A) Translocase of inner mitochondrial membrane 17B (B) ATP binding cassette subfamily B member 8 (C) NADH dehydrogenase 2 (D) Tu translation elongation factor, mitochondrial The correct answer is (C)
""".strip()
gpqa_llada_datasets = [
dict(
abbr="GPQA_diamond",
type=GPQADataset,
path="./data/gpqa/",
name="gpqa_diamond.csv",
reader_cfg=dict(input_columns=["question", "A", "B", "C", "D"], output_column="answer"),
infer_cfg=dict(
prompt_template=dict(
type=PromptTemplate,
template=dict(
round=[
dict(
role="HUMAN",
prompt=(
f"{gpqa_llada_five_shot_prompt}\n"
"Question:\n{question}\n"
"Choices:\n"
"A. {A}\n"
"B. {B}\n"
"C. {C}\n"
"D. {D}\n"
"When you're ready to answer, please use the format "
"'The correct answer is (X)' where X is the correct letter choice from 'ABCD'."
),
),
],
),
),
retriever=dict(type=ZeroRetriever),
inferencer=dict(type=GenInferencer),
),
eval_cfg=dict(
evaluator=dict(type=GPQAEvaluator),
pred_postprocessor=dict(type=first_option_postprocess, options="ABCD"),
),
)
]
meetingbank_10k_datasets = [
dict(
abbr="MeetingBank_10k",
type=JsonlDataset,
path="data/meetingbank/test_10k.jsonl",
reader_cfg=dict(
input_columns=["transcript"],
output_column="summary",
train_split="test",
test_split="test",
),
infer_cfg=dict(
prompt_template=dict(
type=PromptTemplate,
template=dict(
round=[
dict(
role="HUMAN",
prompt=(
"Summarize the following city council meeting transcript excerpt "
"in one concise paragraph. Focus on the concrete decisions, "
"requests, actions, and outcomes.\n\n"
"Transcript:\n{transcript}\n\nSummary:"
),
),
],
),
),
retriever=dict(type=ZeroRetriever),
inferencer=dict(type=GenInferencer),
),
eval_cfg=dict(
evaluator=dict(type=MeetingBankRougeEvaluator),
pred_role="BOT",
),
)
]
#
datasets = sum((v for k, v in locals().items() if k.endswith('_datasets')), [])
datasets = sum((v for k, v in locals().items() if k.endswith('_datasets')), []) + [LCBCodeGeneration_dataset]
#
TASK_TAG = ''
API_SERVER_ADDR = 'http://'
SERVED_MODEL_PATH = ''
TARGET_TEMPERATURE = 0.6
TARGET_MAX_OUT_LEN = 64000
TARGET_QPS = 10
TARGET_BATCH_SIZE = 32
INFER_NUM_WORKER = 8
INFER_MIN_TASK_SIZE = 16
INFER_MAX_WORKERS = 16
EVAL_MAX_WORKERS = 16
models = [
dict(abbr=TASK_TAG,
key='dummy',
openai_api_base=f'{API_SERVER_ADDR}/v1',
type=OpenAISDK,
path=SERVED_MODEL_PATH,
temperature=TARGET_TEMPERATURE,
meta_template=dict(round=[
dict(role='HUMAN', api_role='HUMAN'),
dict(role='BOT', api_role='BOT', generate=True),
], ),
query_per_second=TARGET_QPS,
max_out_len=TARGET_MAX_OUT_LEN,
max_seq_len=65536,
batch_size=TARGET_BATCH_SIZE,
retry=10,
pred_postprocessor=dict(type="extract-non-reasoning-content"),
verbose=False)
]
JUDGER_ADDR = 'http://'
JUDGER_MODEL_PATH = ''
judge_cfg = dict(
abbr='CompassVerifier',
type=OpenAISDK,
path=JUDGER_MODEL_PATH,
key='YOUR_API_KEY',
openai_api_base=f'{JUDGER_ADDR}/v1',
meta_template=dict(round=[
dict(role='HUMAN', api_role='HUMAN'),
dict(role='BOT', api_role='BOT', generate=True),
]),
query_per_second=8,
batch_size=32,
temperature=0.001,
max_out_len=8192,
max_seq_len=65536,
mode='mid',
)
judge_models = [judge_cfg]
OPENCOMPASS_SUBJECTIVE = False
try:
dataset_items_for_judge = list(datasets)
except TypeError:
dataset_items_for_judge = []
for item in dataset_items_for_judge:
if 'judge_cfg' in item['eval_cfg']['evaluator']:
item['eval_cfg']['evaluator']['judge_cfg'] = judge_cfg
if 'llm_evaluator' in item['eval_cfg']['evaluator'].keys(
) and 'judge_cfg' in item['eval_cfg']['evaluator']['llm_evaluator']:
item['eval_cfg']['evaluator']['llm_evaluator']['judge_cfg'] = judge_cfg
#######################################################################
# Dataset Summarizer #
#######################################################################
core_summary_groups = [
{
'name':
'core_average',
'subsets': [
['IFEval', 'Prompt-level-strict-accuracy'],
['hle_llmjudge', 'accuracy'],
['aime2025_repeat_32', 'accuracy (32 runs average)'],
['GPQA_diamond_repeat_4', 'accuracy (4 runs average)'],
['mmlu_pro', 'naive_average'],
['lcb_code_generation_repeat_6', 'pass@1 (6 runs average)'],
],
},
]
summarizer = dict(
dataset_abbrs=[
['core_average', 'naive_average'],
'',
'Instruction Following',
['IFEval', 'Prompt-level-strict-accuracy'],
'',
'General Reasoning',
['hle_llmjudge', 'accuracy'],
['GPQA_diamond_repeat_4', 'accuracy (4 runs average)'],
['GPQA_diamond', 'accuracy'],
'',
'Math Calculation',
['aime2024', 'accuracy'],
['aime2025_repeat_32', 'accuracy (32 runs average)'],
'',
'Knowledge',
['mmlu_pro', 'naive_average'],
'',
'Code',
['openai_humaneval', 'humaneval_1'],
['mbpp', 'score'],
['lcb_code_generation_repeat_6', 'pass@1 (6 runs average)'],
'',
'Language / Summarization',
['LongBench_qmsum', 'score'],
['LongBench_samsum', 'score'],
['LEval_meeting_summ', 'rougeL'],
['MeetingBank_10k', 'score'],
],
summary_groups=sum([v for k, v in locals().items() if k.endswith('_summary_groups')], []),
)
#######################################################################
# Inference/Evaluation Configuration #
#######################################################################
# infer with local runner
infer = dict(
partitioner=dict(type=NumWorkerPartitioner, num_worker=INFER_NUM_WORKER, min_task_size=INFER_MIN_TASK_SIZE),
runner=dict(
type=LocalRunner,
max_num_workers=INFER_MAX_WORKERS,
retry=0, # Modify if needed
task=dict(type=OpenICLInferTask),
),
)
# eval with local runner
if OPENCOMPASS_SUBJECTIVE:
eval = dict(
partitioner=dict(
type=SubjectiveNumWorkerPartitioner,
models=models,
judge_models=judge_models,
num_worker=EVAL_MAX_WORKERS,
),
runner=dict(type=LocalRunner, max_num_workers=EVAL_MAX_WORKERS, task=dict(type=OpenICLEvalTask)),
)
else:
eval = dict(
partitioner=dict(type=NaivePartitioner, n=10),
runner=dict(type=LocalRunner, max_num_workers=EVAL_MAX_WORKERS, task=dict(type=OpenICLEvalTask)),
)