initial upload from H20 cluster: eval/20260527_000425
Browse files- .gitattributes +11 -0
- eval/20260527_000425/20260527_000425/configs/task_config.yaml +381 -0
- eval/20260527_000425/20260527_000425/logs/eval_log.log +3 -0
- eval/20260527_000425/20260527_000425/predictions/dir-ppo-llama3.1-8b/humaneval_openai_humaneval.jsonl +0 -0
- eval/20260527_000425/20260527_000425/predictions/dir-ppo-llama3.1-8b/ifeval_default.jsonl +0 -0
- eval/20260527_000425/20260527_000425/predictions/dir-ppo-llama3.1-8b/race_high.jsonl +3 -0
- eval/20260527_000425/20260527_000425/predictions/dir-ppo-llama3.1-8b/race_middle.jsonl +3 -0
- eval/20260527_000425/20260527_000425/predictions/dir-ppo-llama3.1-8b/trivia_qa_rc.wikipedia.jsonl +3 -0
- eval/20260527_000425/20260527_000425/reports/dir-ppo-llama3.1-8b/humaneval.json +116 -0
- eval/20260527_000425/20260527_000425/reports/dir-ppo-llama3.1-8b/ifeval.json +162 -0
- eval/20260527_000425/20260527_000425/reports/dir-ppo-llama3.1-8b/race.json +98 -0
- eval/20260527_000425/20260527_000425/reports/dir-ppo-llama3.1-8b/trivia_qa.json +93 -0
- eval/20260527_000425/20260527_000425/reports/report.html +0 -0
- eval/20260527_000425/20260527_000425/reviews/dir-ppo-llama3.1-8b/humaneval_openai_humaneval.jsonl +0 -0
- eval/20260527_000425/20260527_000425/reviews/dir-ppo-llama3.1-8b/ifeval_default.jsonl +0 -0
- eval/20260527_000425/20260527_000425/reviews/dir-ppo-llama3.1-8b/race_high.jsonl +3 -0
- eval/20260527_000425/20260527_000425/reviews/dir-ppo-llama3.1-8b/race_middle.jsonl +0 -0
- eval/20260527_000425/20260527_000425/reviews/dir-ppo-llama3.1-8b/trivia_qa_rc.wikipedia.jsonl +3 -0
- eval/20260527_000425/configs/task_config.yaml +381 -0
- eval/20260527_000425/logs/eval_log.log +0 -0
- eval/20260527_000425/predictions/dir-ppo-llama3.1-8b/ifeval_default.jsonl +0 -0
- eval/20260527_000425/predictions/dir-ppo-llama3.1-8b/race_high.jsonl +3 -0
- eval/20260527_000425/predictions/dir-ppo-llama3.1-8b/race_middle.jsonl +3 -0
- eval/20260527_000425/predictions/dir-ppo-llama3.1-8b/trivia_qa_rc.wikipedia.jsonl +3 -0
- eval/20260527_000425/reports/dir-ppo-llama3.1-8b/ifeval.json +162 -0
- eval/20260527_000425/reports/dir-ppo-llama3.1-8b/race.json +98 -0
- eval/20260527_000425/reviews/dir-ppo-llama3.1-8b/ifeval_default.jsonl +0 -0
- eval/20260527_000425/reviews/dir-ppo-llama3.1-8b/race_high.jsonl +3 -0
- eval/20260527_000425/reviews/dir-ppo-llama3.1-8b/race_middle.jsonl +0 -0
- eval/20260527_000425/reviews/dir-ppo-llama3.1-8b/trivia_qa_rc.wikipedia.jsonl +3 -0
.gitattributes
CHANGED
|
@@ -34,3 +34,14 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
| 36 |
ppo_dir_v2_merged/tokenizer.json filter=lfs diff=lfs merge=lfs -text
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
| 36 |
ppo_dir_v2_merged/tokenizer.json filter=lfs diff=lfs merge=lfs -text
|
| 37 |
+
eval/20260527_000425/20260527_000425/logs/eval_log.log filter=lfs diff=lfs merge=lfs -text
|
| 38 |
+
eval/20260527_000425/20260527_000425/predictions/dir-ppo-llama3.1-8b/race_high.jsonl filter=lfs diff=lfs merge=lfs -text
|
| 39 |
+
eval/20260527_000425/20260527_000425/predictions/dir-ppo-llama3.1-8b/race_middle.jsonl filter=lfs diff=lfs merge=lfs -text
|
| 40 |
+
eval/20260527_000425/20260527_000425/predictions/dir-ppo-llama3.1-8b/trivia_qa_rc.wikipedia.jsonl filter=lfs diff=lfs merge=lfs -text
|
| 41 |
+
eval/20260527_000425/20260527_000425/reviews/dir-ppo-llama3.1-8b/race_high.jsonl filter=lfs diff=lfs merge=lfs -text
|
| 42 |
+
eval/20260527_000425/20260527_000425/reviews/dir-ppo-llama3.1-8b/trivia_qa_rc.wikipedia.jsonl filter=lfs diff=lfs merge=lfs -text
|
| 43 |
+
eval/20260527_000425/predictions/dir-ppo-llama3.1-8b/race_high.jsonl filter=lfs diff=lfs merge=lfs -text
|
| 44 |
+
eval/20260527_000425/predictions/dir-ppo-llama3.1-8b/race_middle.jsonl filter=lfs diff=lfs merge=lfs -text
|
| 45 |
+
eval/20260527_000425/predictions/dir-ppo-llama3.1-8b/trivia_qa_rc.wikipedia.jsonl filter=lfs diff=lfs merge=lfs -text
|
| 46 |
+
eval/20260527_000425/reviews/dir-ppo-llama3.1-8b/race_high.jsonl filter=lfs diff=lfs merge=lfs -text
|
| 47 |
+
eval/20260527_000425/reviews/dir-ppo-llama3.1-8b/trivia_qa_rc.wikipedia.jsonl filter=lfs diff=lfs merge=lfs -text
|
eval/20260527_000425/20260527_000425/configs/task_config.yaml
ADDED
|
@@ -0,0 +1,381 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
agent_config: null
|
| 2 |
+
analysis_report: false
|
| 3 |
+
api_url: http://127.0.0.1:8801/v1
|
| 4 |
+
chat_template: null
|
| 5 |
+
collect_perf: true
|
| 6 |
+
dataset_args:
|
| 7 |
+
humaneval:
|
| 8 |
+
aggregation: mean_and_pass_at_k
|
| 9 |
+
data_statistics: null
|
| 10 |
+
dataset_id: opencompass/humaneval
|
| 11 |
+
default_subset: default
|
| 12 |
+
description: '
|
| 13 |
+
|
| 14 |
+
## Overview
|
| 15 |
+
|
| 16 |
+
|
| 17 |
+
HumanEval is a benchmark for evaluating the code generation capabilities of
|
| 18 |
+
language models. It consists of 164 hand-written Python programming problems
|
| 19 |
+
with function signatures, docstrings, and comprehensive test cases.
|
| 20 |
+
|
| 21 |
+
|
| 22 |
+
## Task Description
|
| 23 |
+
|
| 24 |
+
|
| 25 |
+
- **Task Type**: Code Generation (Python)
|
| 26 |
+
|
| 27 |
+
- **Input**: Function signature with docstring describing the expected behavior
|
| 28 |
+
|
| 29 |
+
- **Output**: Complete Python function implementation
|
| 30 |
+
|
| 31 |
+
- **Languages**: Python only
|
| 32 |
+
|
| 33 |
+
|
| 34 |
+
## Key Features
|
| 35 |
+
|
| 36 |
+
|
| 37 |
+
- 164 hand-crafted programming problems
|
| 38 |
+
|
| 39 |
+
- Each problem includes a function signature, docstring, and test cases
|
| 40 |
+
|
| 41 |
+
- Problems range from simple string manipulation to complex algorithms
|
| 42 |
+
|
| 43 |
+
- Canonical solutions provided for reference
|
| 44 |
+
|
| 45 |
+
- Automatic correctness verification through test execution
|
| 46 |
+
|
| 47 |
+
|
| 48 |
+
## Evaluation Notes
|
| 49 |
+
|
| 50 |
+
|
| 51 |
+
- **Security Warning**: By default, code is executed in the local environment.
|
| 52 |
+
We strongly recommend using sandbox execution for safety. See the [sandbox documentation](https://evalscope.readthedocs.io/en/latest/user_guides/sandbox.html)
|
| 53 |
+
for details.
|
| 54 |
+
|
| 55 |
+
- Supports `pass@k` metric calculation for measuring generation quality
|
| 56 |
+
|
| 57 |
+
- Default timeout is 4 seconds per problem
|
| 58 |
+
|
| 59 |
+
- Code is extracted from markdown code blocks if present
|
| 60 |
+
|
| 61 |
+
'
|
| 62 |
+
eval_split: test
|
| 63 |
+
extra_params: {}
|
| 64 |
+
few_shot_num: 0
|
| 65 |
+
few_shot_prompt_template: null
|
| 66 |
+
few_shot_random: false
|
| 67 |
+
filters: null
|
| 68 |
+
force_redownload: false
|
| 69 |
+
metric_list:
|
| 70 |
+
- acc
|
| 71 |
+
name: humaneval
|
| 72 |
+
output_types:
|
| 73 |
+
- generation
|
| 74 |
+
paper_url: null
|
| 75 |
+
pretty_name: HumanEval
|
| 76 |
+
prompt_template: 'Read the following function signature and docstring, and fully
|
| 77 |
+
implement the function described. Your response should only contain the code
|
| 78 |
+
for this function.
|
| 79 |
+
|
| 80 |
+
{question}'
|
| 81 |
+
query_template: null
|
| 82 |
+
review_timeout: 4
|
| 83 |
+
sample_example: null
|
| 84 |
+
sandbox_config:
|
| 85 |
+
image: python:3.11-slim
|
| 86 |
+
tools_config:
|
| 87 |
+
python_executor: {}
|
| 88 |
+
shell_executor: {}
|
| 89 |
+
shuffle: false
|
| 90 |
+
shuffle_choices: false
|
| 91 |
+
subset_list:
|
| 92 |
+
- openai_humaneval
|
| 93 |
+
system_prompt: null
|
| 94 |
+
tags:
|
| 95 |
+
- Coding
|
| 96 |
+
train_split: null
|
| 97 |
+
ifeval:
|
| 98 |
+
aggregation: mean
|
| 99 |
+
data_statistics: null
|
| 100 |
+
dataset_id: opencompass/ifeval
|
| 101 |
+
default_subset: default
|
| 102 |
+
description: "\n## Overview\n\nIFEval (Instruction-Following Eval) is a benchmark\
|
| 103 |
+
\ for evaluating how well language models follow explicit, verifiable instructions.\
|
| 104 |
+
\ It contains prompts with specific formatting, content, or structural requirements\
|
| 105 |
+
\ that can be objectively verified.\n\n## Task Description\n\n- **Task Type**:\
|
| 106 |
+
\ Instruction Following Evaluation\n- **Input**: Prompts with explicit, verifiable\
|
| 107 |
+
\ constraints\n- **Output**: Response that follows all specified instructions\n\
|
| 108 |
+
- **Constraint Types**: Format, length, keywords, structure, etc.\n\n## Key\
|
| 109 |
+
\ Features\n\n- ~500 prompts with 25 types of verifiable instructions\n- Instructions\
|
| 110 |
+
\ are objectively checkable (not subjective)\n- Examples: \"write exactly 3\
|
| 111 |
+
\ paragraphs\", \"include the word X\", \"use bullet points\"\n- Tests instruction\
|
| 112 |
+
\ comprehension and compliance\n- No ambiguity in evaluation criteria\n\n##\
|
| 113 |
+
\ Evaluation Notes\n\n- Default configuration uses **0-shot** evaluation\n-\
|
| 114 |
+
\ Four metrics available:\n - `prompt_level_strict`: All instructions in prompt\
|
| 115 |
+
\ must be followed\n - `prompt_level_loose`: Some tolerance for minor deviations\n\
|
| 116 |
+
\ - `inst_level_strict`: Per-instruction accuracy (strict)\n - `inst_level_loose`:\
|
| 117 |
+
\ Per-instruction accuracy (loose)\n- `prompt_level_strict` is the primary metric\n\
|
| 118 |
+
- Automatic verification of instruction compliance\n"
|
| 119 |
+
eval_split: train
|
| 120 |
+
extra_params: {}
|
| 121 |
+
few_shot_num: 0
|
| 122 |
+
few_shot_prompt_template: null
|
| 123 |
+
few_shot_random: false
|
| 124 |
+
filters: null
|
| 125 |
+
force_redownload: false
|
| 126 |
+
metric_list:
|
| 127 |
+
- prompt_level_strict
|
| 128 |
+
- inst_level_strict
|
| 129 |
+
- prompt_level_loose
|
| 130 |
+
- inst_level_loose
|
| 131 |
+
name: ifeval
|
| 132 |
+
output_types:
|
| 133 |
+
- generation
|
| 134 |
+
paper_url: null
|
| 135 |
+
pretty_name: IFEval
|
| 136 |
+
prompt_template: ''
|
| 137 |
+
query_template: null
|
| 138 |
+
review_timeout: null
|
| 139 |
+
sample_example: null
|
| 140 |
+
sandbox_config: {}
|
| 141 |
+
shuffle: false
|
| 142 |
+
shuffle_choices: false
|
| 143 |
+
subset_list:
|
| 144 |
+
- default
|
| 145 |
+
system_prompt: null
|
| 146 |
+
tags:
|
| 147 |
+
- InstructionFollowing
|
| 148 |
+
train_split: null
|
| 149 |
+
race:
|
| 150 |
+
aggregation: mean
|
| 151 |
+
data_statistics: null
|
| 152 |
+
dataset_id: evalscope/race
|
| 153 |
+
default_subset: default
|
| 154 |
+
description: '
|
| 155 |
+
|
| 156 |
+
## Overview
|
| 157 |
+
|
| 158 |
+
|
| 159 |
+
RACE (ReAding Comprehension from Examinations) is a large-scale reading comprehension
|
| 160 |
+
benchmark collected from Chinese middle school and high school English examinations.
|
| 161 |
+
It tests comprehensive reading comprehension abilities.
|
| 162 |
+
|
| 163 |
+
|
| 164 |
+
## Task Description
|
| 165 |
+
|
| 166 |
+
|
| 167 |
+
- **Task Type**: Reading Comprehension (Multiple-Choice)
|
| 168 |
+
|
| 169 |
+
- **Input**: Article passage with question and 4 answer choices
|
| 170 |
+
|
| 171 |
+
- **Output**: Correct answer letter (A, B, C, or D)
|
| 172 |
+
|
| 173 |
+
- **Difficulty Levels**: Middle school and High school
|
| 174 |
+
|
| 175 |
+
|
| 176 |
+
## Key Features
|
| 177 |
+
|
| 178 |
+
|
| 179 |
+
- 28,000+ passages with 100,000 questions
|
| 180 |
+
|
| 181 |
+
- Real examination questions for authentic difficulty
|
| 182 |
+
|
| 183 |
+
- Two subsets: middle (easier) and high (harder)
|
| 184 |
+
|
| 185 |
+
- Tests various comprehension skills (inference, vocabulary, main idea, etc.)
|
| 186 |
+
|
| 187 |
+
- Diverse article topics and question types
|
| 188 |
+
|
| 189 |
+
|
| 190 |
+
## Evaluation Notes
|
| 191 |
+
|
| 192 |
+
|
| 193 |
+
- Default configuration uses **3-shot** examples
|
| 194 |
+
|
| 195 |
+
- Maximum few-shot number is 3 (context length consideration)
|
| 196 |
+
|
| 197 |
+
- Uses Chain-of-Thought (CoT) prompting
|
| 198 |
+
|
| 199 |
+
- Two subsets available: `high` and `middle`
|
| 200 |
+
|
| 201 |
+
- Evaluates on test split
|
| 202 |
+
|
| 203 |
+
'
|
| 204 |
+
eval_split: test
|
| 205 |
+
extra_params: {}
|
| 206 |
+
few_shot_num: 3
|
| 207 |
+
few_shot_prompt_template: null
|
| 208 |
+
few_shot_random: false
|
| 209 |
+
filters: null
|
| 210 |
+
force_redownload: false
|
| 211 |
+
metric_list:
|
| 212 |
+
- acc
|
| 213 |
+
name: race
|
| 214 |
+
output_types:
|
| 215 |
+
- generation
|
| 216 |
+
paper_url: null
|
| 217 |
+
pretty_name: RACE
|
| 218 |
+
prompt_template: 'Answer the following multiple choice question. The last line
|
| 219 |
+
of your response should be of the following format: ''ANSWER: [LETTER]'' (without
|
| 220 |
+
quotes) where [LETTER] is one of {letters}. Think step by step before answering.
|
| 221 |
+
|
| 222 |
+
|
| 223 |
+
{question}
|
| 224 |
+
|
| 225 |
+
|
| 226 |
+
{choices}'
|
| 227 |
+
query_template: null
|
| 228 |
+
review_timeout: null
|
| 229 |
+
sample_example: null
|
| 230 |
+
sandbox_config: {}
|
| 231 |
+
shuffle: false
|
| 232 |
+
shuffle_choices: false
|
| 233 |
+
subset_list:
|
| 234 |
+
- high
|
| 235 |
+
- middle
|
| 236 |
+
system_prompt: null
|
| 237 |
+
tags:
|
| 238 |
+
- Reasoning
|
| 239 |
+
- MCQ
|
| 240 |
+
train_split: train
|
| 241 |
+
trivia_qa:
|
| 242 |
+
aggregation: mean
|
| 243 |
+
data_statistics: null
|
| 244 |
+
dataset_id: evalscope/trivia_qa
|
| 245 |
+
default_subset: default
|
| 246 |
+
description: '
|
| 247 |
+
|
| 248 |
+
## Overview
|
| 249 |
+
|
| 250 |
+
|
| 251 |
+
TriviaQA is a large-scale reading comprehension dataset containing over 650K
|
| 252 |
+
question-answer-evidence triples. Questions are collected from trivia enthusiast
|
| 253 |
+
websites and paired with Wikipedia articles as evidence documents.
|
| 254 |
+
|
| 255 |
+
|
| 256 |
+
## Task Description
|
| 257 |
+
|
| 258 |
+
|
| 259 |
+
- **Task Type**: Reading Comprehension / Question Answering
|
| 260 |
+
|
| 261 |
+
- **Input**: Question with Wikipedia context passage
|
| 262 |
+
|
| 263 |
+
- **Output**: Answer extracted or generated from context
|
| 264 |
+
|
| 265 |
+
- **Domain**: General knowledge trivia questions
|
| 266 |
+
|
| 267 |
+
|
| 268 |
+
## Key Features
|
| 269 |
+
|
| 270 |
+
|
| 271 |
+
- 650K+ question-answer-evidence triples
|
| 272 |
+
|
| 273 |
+
- Questions written by trivia enthusiasts (naturally challenging)
|
| 274 |
+
|
| 275 |
+
- Multiple valid answer aliases for flexible evaluation
|
| 276 |
+
|
| 277 |
+
- Wikipedia articles provide evidence passages
|
| 278 |
+
|
| 279 |
+
- Tests both reading comprehension and knowledge retrieval
|
| 280 |
+
|
| 281 |
+
|
| 282 |
+
## Evaluation Notes
|
| 283 |
+
|
| 284 |
+
|
| 285 |
+
- Default configuration uses **0-shot** evaluation
|
| 286 |
+
|
| 287 |
+
- Uses the Wikipedia reading comprehension subset (rc.wikipedia)
|
| 288 |
+
|
| 289 |
+
- Answers should follow the format: "ANSWER: [ANSWER]"
|
| 290 |
+
|
| 291 |
+
- Supports inclusion-based matching for answer comparison
|
| 292 |
+
|
| 293 |
+
- Evaluates on validation split
|
| 294 |
+
|
| 295 |
+
'
|
| 296 |
+
eval_split: validation
|
| 297 |
+
extra_params: {}
|
| 298 |
+
few_shot_num: 0
|
| 299 |
+
few_shot_prompt_template: null
|
| 300 |
+
few_shot_random: false
|
| 301 |
+
filters: null
|
| 302 |
+
force_redownload: false
|
| 303 |
+
metric_list:
|
| 304 |
+
- acc:
|
| 305 |
+
allow_inclusion: true
|
| 306 |
+
name: trivia_qa
|
| 307 |
+
output_types:
|
| 308 |
+
- generation
|
| 309 |
+
paper_url: null
|
| 310 |
+
pretty_name: TriviaQA
|
| 311 |
+
prompt_template: 'Read the content and answer the following question.
|
| 312 |
+
|
| 313 |
+
|
| 314 |
+
Content: {content}
|
| 315 |
+
|
| 316 |
+
|
| 317 |
+
Question: {question}
|
| 318 |
+
|
| 319 |
+
|
| 320 |
+
Keep your The last line of your response should be of the form "ANSWER: [ANSWER]"
|
| 321 |
+
(without quotes) where [ANSWER] is the answer to the problem.
|
| 322 |
+
|
| 323 |
+
'
|
| 324 |
+
query_template: null
|
| 325 |
+
review_timeout: null
|
| 326 |
+
sample_example: null
|
| 327 |
+
sandbox_config: {}
|
| 328 |
+
shuffle: false
|
| 329 |
+
shuffle_choices: false
|
| 330 |
+
subset_list:
|
| 331 |
+
- rc.wikipedia
|
| 332 |
+
system_prompt: null
|
| 333 |
+
tags:
|
| 334 |
+
- QA
|
| 335 |
+
- ReadingComprehension
|
| 336 |
+
train_split: null
|
| 337 |
+
dataset_dir: /root/.cache/modelscope/hub/datasets
|
| 338 |
+
dataset_hub: modelscope
|
| 339 |
+
datasets:
|
| 340 |
+
- ifeval
|
| 341 |
+
- race
|
| 342 |
+
- trivia_qa
|
| 343 |
+
- humaneval
|
| 344 |
+
debug: false
|
| 345 |
+
enable_progress_tracker: false
|
| 346 |
+
eval_backend: Native
|
| 347 |
+
eval_batch_size: 16
|
| 348 |
+
eval_config: null
|
| 349 |
+
eval_type: server
|
| 350 |
+
evalscope_version: 1.7.1
|
| 351 |
+
generation_config:
|
| 352 |
+
batch_size: 16
|
| 353 |
+
do_sample: true
|
| 354 |
+
max_new_tokens: 2048
|
| 355 |
+
temperature: 0.7
|
| 356 |
+
ignore_errors: true
|
| 357 |
+
judge_model_args: {}
|
| 358 |
+
judge_strategy: auto
|
| 359 |
+
judge_worker_num: 1
|
| 360 |
+
limit: null
|
| 361 |
+
model: dir-ppo-llama3.1-8b
|
| 362 |
+
model_args: {}
|
| 363 |
+
model_id: dir-ppo-llama3.1-8b
|
| 364 |
+
model_task: text_generation
|
| 365 |
+
no_timestamp: false
|
| 366 |
+
repeats: 1
|
| 367 |
+
rerun_review: false
|
| 368 |
+
sandbox:
|
| 369 |
+
default_config: {}
|
| 370 |
+
enabled: false
|
| 371 |
+
engine: docker
|
| 372 |
+
manager_config: {}
|
| 373 |
+
pool_size: null
|
| 374 |
+
sandbox_manager_config: {}
|
| 375 |
+
sandbox_type: docker
|
| 376 |
+
seed: 42
|
| 377 |
+
stream: null
|
| 378 |
+
timeout: null
|
| 379 |
+
use_cache: null
|
| 380 |
+
use_sandbox: false
|
| 381 |
+
work_dir: ./outputs/20260527_000425
|
eval/20260527_000425/20260527_000425/logs/eval_log.log
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:b4e24adb27f4e9824f5bd4b5f70d41c417786bbf23c8b36174a21befca1c0252
|
| 3 |
+
size 26273843
|
eval/20260527_000425/20260527_000425/predictions/dir-ppo-llama3.1-8b/humaneval_openai_humaneval.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
eval/20260527_000425/20260527_000425/predictions/dir-ppo-llama3.1-8b/ifeval_default.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
eval/20260527_000425/20260527_000425/predictions/dir-ppo-llama3.1-8b/race_high.jsonl
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:d595bed6f072c3ef9ed0e19010e34b124818146eb724a04531238384b16fe08e
|
| 3 |
+
size 40497772
|
eval/20260527_000425/20260527_000425/predictions/dir-ppo-llama3.1-8b/race_middle.jsonl
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:b0ea7aff3ae009236736a2c2ca8c02a74eedf3b6f87b4091de2918a67bca719c
|
| 3 |
+
size 10885732
|
eval/20260527_000425/20260527_000425/predictions/dir-ppo-llama3.1-8b/trivia_qa_rc.wikipedia.jsonl
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:a0decfb092206f175111431550b702003521ce5ce21f4555ef86198a500d5f83
|
| 3 |
+
size 125325681
|
eval/20260527_000425/20260527_000425/reports/dir-ppo-llama3.1-8b/humaneval.json
ADDED
|
@@ -0,0 +1,116 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"name": "dir-ppo-llama3.1-8b@humaneval",
|
| 3 |
+
"dataset_name": "humaneval",
|
| 4 |
+
"dataset_pretty_name": "HumanEval",
|
| 5 |
+
"dataset_description": "\n## Overview\n\nHumanEval is a benchmark for evaluating the code generation capabilities of language models. It consists of 164 hand-written Python programming problems with function signatures, docstrings, and comprehensive test cases.\n\n## Task Description\n\n- **Task Type**: Code Generation (Python)\n- **Input**: Function signature with docstring describing the expected behavior\n- **Output**: Complete Python function implementation\n- **Languages**: Python only\n\n## Key Features\n\n- 164 hand-crafted programming problems\n- Each problem includes a function signature, docstring, and test cases\n- Problems range from simple string manipulation to complex algorithms\n- Canonical solutions provided for reference\n- Automatic correctness verification through test execution\n\n## Evaluation Notes\n\n- **Security Warning**: By default, code is executed in the local environment. We strongly recommend using sandbox execution for safety. See the [sandbox documentation](https://evalscope.readthedocs.io/en/latest/user_guides/sandbox.html) for details.\n- Supports `pass@k` metric calculation for measuring generation quality\n- Default timeout is 4 seconds per problem\n- Code is extracted from markdown code blocks if present\n",
|
| 6 |
+
"model_name": "dir-ppo-llama3.1-8b",
|
| 7 |
+
"score": 0.6768,
|
| 8 |
+
"metrics": [
|
| 9 |
+
{
|
| 10 |
+
"name": "mean_acc",
|
| 11 |
+
"num": 164,
|
| 12 |
+
"score": 0.6768,
|
| 13 |
+
"macro_score": 0.6768,
|
| 14 |
+
"categories": [
|
| 15 |
+
{
|
| 16 |
+
"name": [
|
| 17 |
+
"default"
|
| 18 |
+
],
|
| 19 |
+
"num": 164,
|
| 20 |
+
"score": 0.6768,
|
| 21 |
+
"macro_score": 0.6768,
|
| 22 |
+
"subsets": [
|
| 23 |
+
{
|
| 24 |
+
"name": "openai_humaneval",
|
| 25 |
+
"score": 0.6768,
|
| 26 |
+
"num": 164
|
| 27 |
+
}
|
| 28 |
+
]
|
| 29 |
+
}
|
| 30 |
+
]
|
| 31 |
+
},
|
| 32 |
+
{
|
| 33 |
+
"name": "mean_acc_pass@1",
|
| 34 |
+
"num": 164,
|
| 35 |
+
"score": 0.6768,
|
| 36 |
+
"macro_score": 0.6768,
|
| 37 |
+
"categories": [
|
| 38 |
+
{
|
| 39 |
+
"name": [
|
| 40 |
+
"default"
|
| 41 |
+
],
|
| 42 |
+
"num": 164,
|
| 43 |
+
"score": 0.6768,
|
| 44 |
+
"macro_score": 0.6768,
|
| 45 |
+
"subsets": [
|
| 46 |
+
{
|
| 47 |
+
"name": "openai_humaneval",
|
| 48 |
+
"score": 0.6768,
|
| 49 |
+
"num": 164
|
| 50 |
+
}
|
| 51 |
+
]
|
| 52 |
+
}
|
| 53 |
+
]
|
| 54 |
+
}
|
| 55 |
+
],
|
| 56 |
+
"analysis": "N/A",
|
| 57 |
+
"perf_metrics": {
|
| 58 |
+
"summary": {
|
| 59 |
+
"n_samples": 164,
|
| 60 |
+
"latency": {
|
| 61 |
+
"mean": 2.326436,
|
| 62 |
+
"std": 1.460745,
|
| 63 |
+
"min": 0.40422,
|
| 64 |
+
"25%": 1.377736,
|
| 65 |
+
"50%": 2.034455,
|
| 66 |
+
"75%": 2.972601,
|
| 67 |
+
"90%": 3.901483,
|
| 68 |
+
"99%": 7.0451,
|
| 69 |
+
"max": 10.605742
|
| 70 |
+
},
|
| 71 |
+
"throughput": {
|
| 72 |
+
"avg_output_tps": 56.83,
|
| 73 |
+
"avg_req_ps": 0.4298
|
| 74 |
+
},
|
| 75 |
+
"usage": {
|
| 76 |
+
"input_tokens": {
|
| 77 |
+
"mean": 192.426829,
|
| 78 |
+
"std": 64.30068,
|
| 79 |
+
"min": 99.0,
|
| 80 |
+
"25%": 148.0,
|
| 81 |
+
"50%": 178.0,
|
| 82 |
+
"75%": 215.75,
|
| 83 |
+
"90%": 286.7,
|
| 84 |
+
"99%": 372.07,
|
| 85 |
+
"max": 452.0
|
| 86 |
+
},
|
| 87 |
+
"output_tokens": {
|
| 88 |
+
"mean": 132.207317,
|
| 89 |
+
"std": 87.969283,
|
| 90 |
+
"min": 26.0,
|
| 91 |
+
"25%": 68.75,
|
| 92 |
+
"50%": 118.0,
|
| 93 |
+
"75%": 169.25,
|
| 94 |
+
"90%": 221.7,
|
| 95 |
+
"99%": 441.16,
|
| 96 |
+
"max": 638.0
|
| 97 |
+
},
|
| 98 |
+
"total_tokens": {
|
| 99 |
+
"mean": 324.634146,
|
| 100 |
+
"std": 133.096811,
|
| 101 |
+
"min": 146.0,
|
| 102 |
+
"25%": 232.75,
|
| 103 |
+
"50%": 298.0,
|
| 104 |
+
"75%": 371.5,
|
| 105 |
+
"90%": 485.8,
|
| 106 |
+
"99%": 760.69,
|
| 107 |
+
"max": 1090.0
|
| 108 |
+
},
|
| 109 |
+
"total_input_tokens": 31558,
|
| 110 |
+
"total_output_tokens": 21682,
|
| 111 |
+
"total_tokens_count": 53240
|
| 112 |
+
}
|
| 113 |
+
}
|
| 114 |
+
},
|
| 115 |
+
"num": 164
|
| 116 |
+
}
|
eval/20260527_000425/20260527_000425/reports/dir-ppo-llama3.1-8b/ifeval.json
ADDED
|
@@ -0,0 +1,162 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"name": "dir-ppo-llama3.1-8b@ifeval",
|
| 3 |
+
"dataset_name": "ifeval",
|
| 4 |
+
"dataset_pretty_name": "IFEval",
|
| 5 |
+
"dataset_description": "\n## Overview\n\nIFEval (Instruction-Following Eval) is a benchmark for evaluating how well language models follow explicit, verifiable instructions. It contains prompts with specific formatting, content, or structural requirements that can be objectively verified.\n\n## Task Description\n\n- **Task Type**: Instruction Following Evaluation\n- **Input**: Prompts with explicit, verifiable constraints\n- **Output**: Response that follows all specified instructions\n- **Constraint Types**: Format, length, keywords, structure, etc.\n\n## Key Features\n\n- ~500 prompts with 25 types of verifiable instructions\n- Instructions are objectively checkable (not subjective)\n- Examples: \"write exactly 3 paragraphs\", \"include the word X\", \"use bullet points\"\n- Tests instruction comprehension and compliance\n- No ambiguity in evaluation criteria\n\n## Evaluation Notes\n\n- Default configuration uses **0-shot** evaluation\n- Four metrics available:\n - `prompt_level_strict`: All instructions in prompt must be followed\n - `prompt_level_loose`: Some tolerance for minor deviations\n - `inst_level_strict`: Per-instruction accuracy (strict)\n - `inst_level_loose`: Per-instruction accuracy (loose)\n- `prompt_level_strict` is the primary metric\n- Automatic verification of instruction compliance\n",
|
| 6 |
+
"model_name": "dir-ppo-llama3.1-8b",
|
| 7 |
+
"score": 0.6981,
|
| 8 |
+
"metrics": [
|
| 9 |
+
{
|
| 10 |
+
"name": "mean_prompt_level_strict",
|
| 11 |
+
"num": 540,
|
| 12 |
+
"score": 0.6981,
|
| 13 |
+
"macro_score": 0.6981,
|
| 14 |
+
"categories": [
|
| 15 |
+
{
|
| 16 |
+
"name": [
|
| 17 |
+
"default"
|
| 18 |
+
],
|
| 19 |
+
"num": 540,
|
| 20 |
+
"score": 0.6981,
|
| 21 |
+
"macro_score": 0.6981,
|
| 22 |
+
"subsets": [
|
| 23 |
+
{
|
| 24 |
+
"name": "default",
|
| 25 |
+
"score": 0.6981,
|
| 26 |
+
"num": 540
|
| 27 |
+
}
|
| 28 |
+
]
|
| 29 |
+
}
|
| 30 |
+
]
|
| 31 |
+
},
|
| 32 |
+
{
|
| 33 |
+
"name": "mean_inst_level_strict",
|
| 34 |
+
"num": 540,
|
| 35 |
+
"score": 0.7901,
|
| 36 |
+
"macro_score": 0.7901,
|
| 37 |
+
"categories": [
|
| 38 |
+
{
|
| 39 |
+
"name": [
|
| 40 |
+
"default"
|
| 41 |
+
],
|
| 42 |
+
"num": 540,
|
| 43 |
+
"score": 0.7901,
|
| 44 |
+
"macro_score": 0.7901,
|
| 45 |
+
"subsets": [
|
| 46 |
+
{
|
| 47 |
+
"name": "default",
|
| 48 |
+
"score": 0.7901,
|
| 49 |
+
"num": 540
|
| 50 |
+
}
|
| 51 |
+
]
|
| 52 |
+
}
|
| 53 |
+
]
|
| 54 |
+
},
|
| 55 |
+
{
|
| 56 |
+
"name": "mean_prompt_level_loose",
|
| 57 |
+
"num": 540,
|
| 58 |
+
"score": 0.787,
|
| 59 |
+
"macro_score": 0.787,
|
| 60 |
+
"categories": [
|
| 61 |
+
{
|
| 62 |
+
"name": [
|
| 63 |
+
"default"
|
| 64 |
+
],
|
| 65 |
+
"num": 540,
|
| 66 |
+
"score": 0.787,
|
| 67 |
+
"macro_score": 0.787,
|
| 68 |
+
"subsets": [
|
| 69 |
+
{
|
| 70 |
+
"name": "default",
|
| 71 |
+
"score": 0.787,
|
| 72 |
+
"num": 540
|
| 73 |
+
}
|
| 74 |
+
]
|
| 75 |
+
}
|
| 76 |
+
]
|
| 77 |
+
},
|
| 78 |
+
{
|
| 79 |
+
"name": "mean_inst_level_loose",
|
| 80 |
+
"num": 540,
|
| 81 |
+
"score": 0.8623,
|
| 82 |
+
"macro_score": 0.8623,
|
| 83 |
+
"categories": [
|
| 84 |
+
{
|
| 85 |
+
"name": [
|
| 86 |
+
"default"
|
| 87 |
+
],
|
| 88 |
+
"num": 540,
|
| 89 |
+
"score": 0.8623,
|
| 90 |
+
"macro_score": 0.8623,
|
| 91 |
+
"subsets": [
|
| 92 |
+
{
|
| 93 |
+
"name": "default",
|
| 94 |
+
"score": 0.8623,
|
| 95 |
+
"num": 540
|
| 96 |
+
}
|
| 97 |
+
]
|
| 98 |
+
}
|
| 99 |
+
]
|
| 100 |
+
}
|
| 101 |
+
],
|
| 102 |
+
"analysis": "N/A",
|
| 103 |
+
"perf_metrics": {
|
| 104 |
+
"summary": {
|
| 105 |
+
"n_samples": 541,
|
| 106 |
+
"latency": {
|
| 107 |
+
"mean": 7.710747,
|
| 108 |
+
"std": 16.563807,
|
| 109 |
+
"min": 0.066271,
|
| 110 |
+
"25%": 2.457928,
|
| 111 |
+
"50%": 5.100689,
|
| 112 |
+
"75%": 8.237214,
|
| 113 |
+
"90%": 12.235768,
|
| 114 |
+
"99%": 107.019331,
|
| 115 |
+
"max": 160.99844
|
| 116 |
+
},
|
| 117 |
+
"throughput": {
|
| 118 |
+
"avg_output_tps": 52.84,
|
| 119 |
+
"avg_req_ps": 0.1297
|
| 120 |
+
},
|
| 121 |
+
"usage": {
|
| 122 |
+
"input_tokens": {
|
| 123 |
+
"mean": 80.924214,
|
| 124 |
+
"std": 23.92573,
|
| 125 |
+
"min": 48.0,
|
| 126 |
+
"25%": 67.0,
|
| 127 |
+
"50%": 76.0,
|
| 128 |
+
"75%": 90.0,
|
| 129 |
+
"90%": 107.0,
|
| 130 |
+
"99%": 135.6,
|
| 131 |
+
"max": 395.0
|
| 132 |
+
},
|
| 133 |
+
"output_tokens": {
|
| 134 |
+
"mean": 407.428835,
|
| 135 |
+
"std": 862.651879,
|
| 136 |
+
"min": 3.0,
|
| 137 |
+
"25%": 127.0,
|
| 138 |
+
"50%": 265.0,
|
| 139 |
+
"75%": 436.0,
|
| 140 |
+
"90%": 642.0,
|
| 141 |
+
"99%": 5846.8,
|
| 142 |
+
"max": 8138.0
|
| 143 |
+
},
|
| 144 |
+
"total_tokens": {
|
| 145 |
+
"mean": 488.35305,
|
| 146 |
+
"std": 861.312936,
|
| 147 |
+
"min": 69.0,
|
| 148 |
+
"25%": 212.0,
|
| 149 |
+
"50%": 346.0,
|
| 150 |
+
"75%": 521.0,
|
| 151 |
+
"90%": 725.0,
|
| 152 |
+
"99%": 5930.8,
|
| 153 |
+
"max": 8192.0
|
| 154 |
+
},
|
| 155 |
+
"total_input_tokens": 43780,
|
| 156 |
+
"total_output_tokens": 220419,
|
| 157 |
+
"total_tokens_count": 264199
|
| 158 |
+
}
|
| 159 |
+
}
|
| 160 |
+
},
|
| 161 |
+
"num": 540
|
| 162 |
+
}
|
eval/20260527_000425/20260527_000425/reports/dir-ppo-llama3.1-8b/race.json
ADDED
|
@@ -0,0 +1,98 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"name": "dir-ppo-llama3.1-8b@race",
|
| 3 |
+
"dataset_name": "race",
|
| 4 |
+
"dataset_pretty_name": "RACE",
|
| 5 |
+
"dataset_description": "\n## Overview\n\nRACE (ReAding Comprehension from Examinations) is a large-scale reading comprehension benchmark collected from Chinese middle school and high school English examinations. It tests comprehensive reading comprehension abilities.\n\n## Task Description\n\n- **Task Type**: Reading Comprehension (Multiple-Choice)\n- **Input**: Article passage with question and 4 answer choices\n- **Output**: Correct answer letter (A, B, C, or D)\n- **Difficulty Levels**: Middle school and High school\n\n## Key Features\n\n- 28,000+ passages with 100,000 questions\n- Real examination questions for authentic difficulty\n- Two subsets: middle (easier) and high (harder)\n- Tests various comprehension skills (inference, vocabulary, main idea, etc.)\n- Diverse article topics and question types\n\n## Evaluation Notes\n\n- Default configuration uses **3-shot** examples\n- Maximum few-shot number is 3 (context length consideration)\n- Uses Chain-of-Thought (CoT) prompting\n- Two subsets available: `high` and `middle`\n- Evaluates on test split\n",
|
| 6 |
+
"model_name": "dir-ppo-llama3.1-8b",
|
| 7 |
+
"score": 0.833,
|
| 8 |
+
"metrics": [
|
| 9 |
+
{
|
| 10 |
+
"name": "mean_acc",
|
| 11 |
+
"num": 4934,
|
| 12 |
+
"score": 0.833,
|
| 13 |
+
"macro_score": 0.833,
|
| 14 |
+
"categories": [
|
| 15 |
+
{
|
| 16 |
+
"name": [
|
| 17 |
+
"default"
|
| 18 |
+
],
|
| 19 |
+
"num": 4934,
|
| 20 |
+
"score": 0.833,
|
| 21 |
+
"macro_score": 0.8433,
|
| 22 |
+
"subsets": [
|
| 23 |
+
{
|
| 24 |
+
"name": "high",
|
| 25 |
+
"score": 0.8188,
|
| 26 |
+
"num": 3498
|
| 27 |
+
},
|
| 28 |
+
{
|
| 29 |
+
"name": "middle",
|
| 30 |
+
"score": 0.8677,
|
| 31 |
+
"num": 1436
|
| 32 |
+
}
|
| 33 |
+
]
|
| 34 |
+
}
|
| 35 |
+
]
|
| 36 |
+
}
|
| 37 |
+
],
|
| 38 |
+
"analysis": "N/A",
|
| 39 |
+
"perf_metrics": {
|
| 40 |
+
"summary": {
|
| 41 |
+
"n_samples": 4934,
|
| 42 |
+
"latency": {
|
| 43 |
+
"mean": 3.518324,
|
| 44 |
+
"std": 4.350995,
|
| 45 |
+
"min": 0.375683,
|
| 46 |
+
"25%": 2.235433,
|
| 47 |
+
"50%": 3.250558,
|
| 48 |
+
"75%": 4.360585,
|
| 49 |
+
"90%": 5.39817,
|
| 50 |
+
"99%": 7.618212,
|
| 51 |
+
"max": 139.063051
|
| 52 |
+
},
|
| 53 |
+
"throughput": {
|
| 54 |
+
"avg_output_tps": 51.33,
|
| 55 |
+
"avg_req_ps": 0.2842
|
| 56 |
+
},
|
| 57 |
+
"usage": {
|
| 58 |
+
"input_tokens": {
|
| 59 |
+
"mean": 1668.64471,
|
| 60 |
+
"std": 334.066106,
|
| 61 |
+
"min": 958.0,
|
| 62 |
+
"25%": 1274.0,
|
| 63 |
+
"50%": 1824.0,
|
| 64 |
+
"75%": 1885.0,
|
| 65 |
+
"90%": 1953.0,
|
| 66 |
+
"99%": 2269.67,
|
| 67 |
+
"max": 2584.0
|
| 68 |
+
},
|
| 69 |
+
"output_tokens": {
|
| 70 |
+
"mean": 180.605188,
|
| 71 |
+
"std": 216.808972,
|
| 72 |
+
"min": 20.0,
|
| 73 |
+
"25%": 114.0,
|
| 74 |
+
"50%": 168.0,
|
| 75 |
+
"75%": 228.0,
|
| 76 |
+
"90%": 274.0,
|
| 77 |
+
"99%": 387.01,
|
| 78 |
+
"max": 6932.0
|
| 79 |
+
},
|
| 80 |
+
"total_tokens": {
|
| 81 |
+
"mean": 1849.249899,
|
| 82 |
+
"std": 410.659869,
|
| 83 |
+
"min": 1028.0,
|
| 84 |
+
"25%": 1479.0,
|
| 85 |
+
"50%": 1975.0,
|
| 86 |
+
"75%": 2084.0,
|
| 87 |
+
"90%": 2180.0,
|
| 88 |
+
"99%": 2527.39,
|
| 89 |
+
"max": 8192.0
|
| 90 |
+
},
|
| 91 |
+
"total_input_tokens": 8233093,
|
| 92 |
+
"total_output_tokens": 891106,
|
| 93 |
+
"total_tokens_count": 9124199
|
| 94 |
+
}
|
| 95 |
+
}
|
| 96 |
+
},
|
| 97 |
+
"num": 4934
|
| 98 |
+
}
|
eval/20260527_000425/20260527_000425/reports/dir-ppo-llama3.1-8b/trivia_qa.json
ADDED
|
@@ -0,0 +1,93 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"name": "dir-ppo-llama3.1-8b@trivia_qa",
|
| 3 |
+
"dataset_name": "trivia_qa",
|
| 4 |
+
"dataset_pretty_name": "TriviaQA",
|
| 5 |
+
"dataset_description": "\n## Overview\n\nTriviaQA is a large-scale reading comprehension dataset containing over 650K question-answer-evidence triples. Questions are collected from trivia enthusiast websites and paired with Wikipedia articles as evidence documents.\n\n## Task Description\n\n- **Task Type**: Reading Comprehension / Question Answering\n- **Input**: Question with Wikipedia context passage\n- **Output**: Answer extracted or generated from context\n- **Domain**: General knowledge trivia questions\n\n## Key Features\n\n- 650K+ question-answer-evidence triples\n- Questions written by trivia enthusiasts (naturally challenging)\n- Multiple valid answer aliases for flexible evaluation\n- Wikipedia articles provide evidence passages\n- Tests both reading comprehension and knowledge retrieval\n\n## Evaluation Notes\n\n- Default configuration uses **0-shot** evaluation\n- Uses the Wikipedia reading comprehension subset (rc.wikipedia)\n- Answers should follow the format: \"ANSWER: [ANSWER]\"\n- Supports inclusion-based matching for answer comparison\n- Evaluates on validation split\n",
|
| 6 |
+
"model_name": "dir-ppo-llama3.1-8b",
|
| 7 |
+
"score": 0.0407,
|
| 8 |
+
"metrics": [
|
| 9 |
+
{
|
| 10 |
+
"name": "mean_acc",
|
| 11 |
+
"num": 3689,
|
| 12 |
+
"score": 0.0407,
|
| 13 |
+
"macro_score": 0.0407,
|
| 14 |
+
"categories": [
|
| 15 |
+
{
|
| 16 |
+
"name": [
|
| 17 |
+
"default"
|
| 18 |
+
],
|
| 19 |
+
"num": 3689,
|
| 20 |
+
"score": 0.0407,
|
| 21 |
+
"macro_score": 0.0407,
|
| 22 |
+
"subsets": [
|
| 23 |
+
{
|
| 24 |
+
"name": "rc.wikipedia",
|
| 25 |
+
"score": 0.0407,
|
| 26 |
+
"num": 3689
|
| 27 |
+
}
|
| 28 |
+
]
|
| 29 |
+
}
|
| 30 |
+
]
|
| 31 |
+
}
|
| 32 |
+
],
|
| 33 |
+
"analysis": "N/A",
|
| 34 |
+
"perf_metrics": {
|
| 35 |
+
"summary": {
|
| 36 |
+
"n_samples": 3689,
|
| 37 |
+
"latency": {
|
| 38 |
+
"mean": 1.375701,
|
| 39 |
+
"std": 4.526989,
|
| 40 |
+
"min": 0.115312,
|
| 41 |
+
"25%": 0.658521,
|
| 42 |
+
"50%": 0.914985,
|
| 43 |
+
"75%": 1.307014,
|
| 44 |
+
"90%": 1.959365,
|
| 45 |
+
"99%": 5.197951,
|
| 46 |
+
"max": 113.092732
|
| 47 |
+
},
|
| 48 |
+
"throughput": {
|
| 49 |
+
"avg_output_tps": 42.44,
|
| 50 |
+
"avg_req_ps": 0.7269
|
| 51 |
+
},
|
| 52 |
+
"usage": {
|
| 53 |
+
"input_tokens": {
|
| 54 |
+
"mean": 3750.501762,
|
| 55 |
+
"std": 2268.520724,
|
| 56 |
+
"min": 116.0,
|
| 57 |
+
"25%": 1732.0,
|
| 58 |
+
"50%": 3504.0,
|
| 59 |
+
"75%": 5585.0,
|
| 60 |
+
"90%": 7181.6,
|
| 61 |
+
"99%": 8087.0,
|
| 62 |
+
"max": 8190.0
|
| 63 |
+
},
|
| 64 |
+
"output_tokens": {
|
| 65 |
+
"mean": 58.387097,
|
| 66 |
+
"std": 291.002579,
|
| 67 |
+
"min": 2.0,
|
| 68 |
+
"25%": 21.0,
|
| 69 |
+
"50%": 30.0,
|
| 70 |
+
"75%": 45.0,
|
| 71 |
+
"90%": 83.0,
|
| 72 |
+
"99%": 286.36,
|
| 73 |
+
"max": 7337.0
|
| 74 |
+
},
|
| 75 |
+
"total_tokens": {
|
| 76 |
+
"mean": 3808.888859,
|
| 77 |
+
"std": 2283.354375,
|
| 78 |
+
"min": 134.0,
|
| 79 |
+
"25%": 1788.0,
|
| 80 |
+
"50%": 3565.0,
|
| 81 |
+
"75%": 5643.0,
|
| 82 |
+
"90%": 7250.4,
|
| 83 |
+
"99%": 8175.48,
|
| 84 |
+
"max": 8192.0
|
| 85 |
+
},
|
| 86 |
+
"total_input_tokens": 13835601,
|
| 87 |
+
"total_output_tokens": 215390,
|
| 88 |
+
"total_tokens_count": 14050991
|
| 89 |
+
}
|
| 90 |
+
}
|
| 91 |
+
},
|
| 92 |
+
"num": 3689
|
| 93 |
+
}
|
eval/20260527_000425/20260527_000425/reports/report.html
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
eval/20260527_000425/20260527_000425/reviews/dir-ppo-llama3.1-8b/humaneval_openai_humaneval.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
eval/20260527_000425/20260527_000425/reviews/dir-ppo-llama3.1-8b/ifeval_default.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
eval/20260527_000425/20260527_000425/reviews/dir-ppo-llama3.1-8b/race_high.jsonl
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:9bca1ce5a3af7d111c3e98c1d4268a70b41e6a5b06d9d43dd98169389cf01a46
|
| 3 |
+
size 38411680
|
eval/20260527_000425/20260527_000425/reviews/dir-ppo-llama3.1-8b/race_middle.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
eval/20260527_000425/20260527_000425/reviews/dir-ppo-llama3.1-8b/trivia_qa_rc.wikipedia.jsonl
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:b7a4814f193d0768bd1440bb673f997cc476dd2ff24d0e31317483a2f69ae12e
|
| 3 |
+
size 124882928
|
eval/20260527_000425/configs/task_config.yaml
ADDED
|
@@ -0,0 +1,381 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
agent_config: null
|
| 2 |
+
analysis_report: false
|
| 3 |
+
api_url: http://127.0.0.1:8801/v1
|
| 4 |
+
chat_template: null
|
| 5 |
+
collect_perf: true
|
| 6 |
+
dataset_args:
|
| 7 |
+
humaneval:
|
| 8 |
+
aggregation: mean_and_pass_at_k
|
| 9 |
+
data_statistics: null
|
| 10 |
+
dataset_id: opencompass/humaneval
|
| 11 |
+
default_subset: default
|
| 12 |
+
description: '
|
| 13 |
+
|
| 14 |
+
## Overview
|
| 15 |
+
|
| 16 |
+
|
| 17 |
+
HumanEval is a benchmark for evaluating the code generation capabilities of
|
| 18 |
+
language models. It consists of 164 hand-written Python programming problems
|
| 19 |
+
with function signatures, docstrings, and comprehensive test cases.
|
| 20 |
+
|
| 21 |
+
|
| 22 |
+
## Task Description
|
| 23 |
+
|
| 24 |
+
|
| 25 |
+
- **Task Type**: Code Generation (Python)
|
| 26 |
+
|
| 27 |
+
- **Input**: Function signature with docstring describing the expected behavior
|
| 28 |
+
|
| 29 |
+
- **Output**: Complete Python function implementation
|
| 30 |
+
|
| 31 |
+
- **Languages**: Python only
|
| 32 |
+
|
| 33 |
+
|
| 34 |
+
## Key Features
|
| 35 |
+
|
| 36 |
+
|
| 37 |
+
- 164 hand-crafted programming problems
|
| 38 |
+
|
| 39 |
+
- Each problem includes a function signature, docstring, and test cases
|
| 40 |
+
|
| 41 |
+
- Problems range from simple string manipulation to complex algorithms
|
| 42 |
+
|
| 43 |
+
- Canonical solutions provided for reference
|
| 44 |
+
|
| 45 |
+
- Automatic correctness verification through test execution
|
| 46 |
+
|
| 47 |
+
|
| 48 |
+
## Evaluation Notes
|
| 49 |
+
|
| 50 |
+
|
| 51 |
+
- **Security Warning**: By default, code is executed in the local environment.
|
| 52 |
+
We strongly recommend using sandbox execution for safety. See the [sandbox documentation](https://evalscope.readthedocs.io/en/latest/user_guides/sandbox.html)
|
| 53 |
+
for details.
|
| 54 |
+
|
| 55 |
+
- Supports `pass@k` metric calculation for measuring generation quality
|
| 56 |
+
|
| 57 |
+
- Default timeout is 4 seconds per problem
|
| 58 |
+
|
| 59 |
+
- Code is extracted from markdown code blocks if present
|
| 60 |
+
|
| 61 |
+
'
|
| 62 |
+
eval_split: test
|
| 63 |
+
extra_params: {}
|
| 64 |
+
few_shot_num: 0
|
| 65 |
+
few_shot_prompt_template: null
|
| 66 |
+
few_shot_random: false
|
| 67 |
+
filters: null
|
| 68 |
+
force_redownload: false
|
| 69 |
+
metric_list:
|
| 70 |
+
- acc
|
| 71 |
+
name: humaneval
|
| 72 |
+
output_types:
|
| 73 |
+
- generation
|
| 74 |
+
paper_url: null
|
| 75 |
+
pretty_name: HumanEval
|
| 76 |
+
prompt_template: 'Read the following function signature and docstring, and fully
|
| 77 |
+
implement the function described. Your response should only contain the code
|
| 78 |
+
for this function.
|
| 79 |
+
|
| 80 |
+
{question}'
|
| 81 |
+
query_template: null
|
| 82 |
+
review_timeout: 4
|
| 83 |
+
sample_example: null
|
| 84 |
+
sandbox_config:
|
| 85 |
+
image: python:3.11-slim
|
| 86 |
+
tools_config:
|
| 87 |
+
python_executor: {}
|
| 88 |
+
shell_executor: {}
|
| 89 |
+
shuffle: false
|
| 90 |
+
shuffle_choices: false
|
| 91 |
+
subset_list:
|
| 92 |
+
- openai_humaneval
|
| 93 |
+
system_prompt: null
|
| 94 |
+
tags:
|
| 95 |
+
- Coding
|
| 96 |
+
train_split: null
|
| 97 |
+
ifeval:
|
| 98 |
+
aggregation: mean
|
| 99 |
+
data_statistics: null
|
| 100 |
+
dataset_id: opencompass/ifeval
|
| 101 |
+
default_subset: default
|
| 102 |
+
description: "\n## Overview\n\nIFEval (Instruction-Following Eval) is a benchmark\
|
| 103 |
+
\ for evaluating how well language models follow explicit, verifiable instructions.\
|
| 104 |
+
\ It contains prompts with specific formatting, content, or structural requirements\
|
| 105 |
+
\ that can be objectively verified.\n\n## Task Description\n\n- **Task Type**:\
|
| 106 |
+
\ Instruction Following Evaluation\n- **Input**: Prompts with explicit, verifiable\
|
| 107 |
+
\ constraints\n- **Output**: Response that follows all specified instructions\n\
|
| 108 |
+
- **Constraint Types**: Format, length, keywords, structure, etc.\n\n## Key\
|
| 109 |
+
\ Features\n\n- ~500 prompts with 25 types of verifiable instructions\n- Instructions\
|
| 110 |
+
\ are objectively checkable (not subjective)\n- Examples: \"write exactly 3\
|
| 111 |
+
\ paragraphs\", \"include the word X\", \"use bullet points\"\n- Tests instruction\
|
| 112 |
+
\ comprehension and compliance\n- No ambiguity in evaluation criteria\n\n##\
|
| 113 |
+
\ Evaluation Notes\n\n- Default configuration uses **0-shot** evaluation\n-\
|
| 114 |
+
\ Four metrics available:\n - `prompt_level_strict`: All instructions in prompt\
|
| 115 |
+
\ must be followed\n - `prompt_level_loose`: Some tolerance for minor deviations\n\
|
| 116 |
+
\ - `inst_level_strict`: Per-instruction accuracy (strict)\n - `inst_level_loose`:\
|
| 117 |
+
\ Per-instruction accuracy (loose)\n- `prompt_level_strict` is the primary metric\n\
|
| 118 |
+
- Automatic verification of instruction compliance\n"
|
| 119 |
+
eval_split: train
|
| 120 |
+
extra_params: {}
|
| 121 |
+
few_shot_num: 0
|
| 122 |
+
few_shot_prompt_template: null
|
| 123 |
+
few_shot_random: false
|
| 124 |
+
filters: null
|
| 125 |
+
force_redownload: false
|
| 126 |
+
metric_list:
|
| 127 |
+
- prompt_level_strict
|
| 128 |
+
- inst_level_strict
|
| 129 |
+
- prompt_level_loose
|
| 130 |
+
- inst_level_loose
|
| 131 |
+
name: ifeval
|
| 132 |
+
output_types:
|
| 133 |
+
- generation
|
| 134 |
+
paper_url: null
|
| 135 |
+
pretty_name: IFEval
|
| 136 |
+
prompt_template: ''
|
| 137 |
+
query_template: null
|
| 138 |
+
review_timeout: null
|
| 139 |
+
sample_example: null
|
| 140 |
+
sandbox_config: {}
|
| 141 |
+
shuffle: false
|
| 142 |
+
shuffle_choices: false
|
| 143 |
+
subset_list:
|
| 144 |
+
- default
|
| 145 |
+
system_prompt: null
|
| 146 |
+
tags:
|
| 147 |
+
- InstructionFollowing
|
| 148 |
+
train_split: null
|
| 149 |
+
race:
|
| 150 |
+
aggregation: mean
|
| 151 |
+
data_statistics: null
|
| 152 |
+
dataset_id: evalscope/race
|
| 153 |
+
default_subset: default
|
| 154 |
+
description: '
|
| 155 |
+
|
| 156 |
+
## Overview
|
| 157 |
+
|
| 158 |
+
|
| 159 |
+
RACE (ReAding Comprehension from Examinations) is a large-scale reading comprehension
|
| 160 |
+
benchmark collected from Chinese middle school and high school English examinations.
|
| 161 |
+
It tests comprehensive reading comprehension abilities.
|
| 162 |
+
|
| 163 |
+
|
| 164 |
+
## Task Description
|
| 165 |
+
|
| 166 |
+
|
| 167 |
+
- **Task Type**: Reading Comprehension (Multiple-Choice)
|
| 168 |
+
|
| 169 |
+
- **Input**: Article passage with question and 4 answer choices
|
| 170 |
+
|
| 171 |
+
- **Output**: Correct answer letter (A, B, C, or D)
|
| 172 |
+
|
| 173 |
+
- **Difficulty Levels**: Middle school and High school
|
| 174 |
+
|
| 175 |
+
|
| 176 |
+
## Key Features
|
| 177 |
+
|
| 178 |
+
|
| 179 |
+
- 28,000+ passages with 100,000 questions
|
| 180 |
+
|
| 181 |
+
- Real examination questions for authentic difficulty
|
| 182 |
+
|
| 183 |
+
- Two subsets: middle (easier) and high (harder)
|
| 184 |
+
|
| 185 |
+
- Tests various comprehension skills (inference, vocabulary, main idea, etc.)
|
| 186 |
+
|
| 187 |
+
- Diverse article topics and question types
|
| 188 |
+
|
| 189 |
+
|
| 190 |
+
## Evaluation Notes
|
| 191 |
+
|
| 192 |
+
|
| 193 |
+
- Default configuration uses **3-shot** examples
|
| 194 |
+
|
| 195 |
+
- Maximum few-shot number is 3 (context length consideration)
|
| 196 |
+
|
| 197 |
+
- Uses Chain-of-Thought (CoT) prompting
|
| 198 |
+
|
| 199 |
+
- Two subsets available: `high` and `middle`
|
| 200 |
+
|
| 201 |
+
- Evaluates on test split
|
| 202 |
+
|
| 203 |
+
'
|
| 204 |
+
eval_split: test
|
| 205 |
+
extra_params: {}
|
| 206 |
+
few_shot_num: 3
|
| 207 |
+
few_shot_prompt_template: null
|
| 208 |
+
few_shot_random: false
|
| 209 |
+
filters: null
|
| 210 |
+
force_redownload: false
|
| 211 |
+
metric_list:
|
| 212 |
+
- acc
|
| 213 |
+
name: race
|
| 214 |
+
output_types:
|
| 215 |
+
- generation
|
| 216 |
+
paper_url: null
|
| 217 |
+
pretty_name: RACE
|
| 218 |
+
prompt_template: 'Answer the following multiple choice question. The last line
|
| 219 |
+
of your response should be of the following format: ''ANSWER: [LETTER]'' (without
|
| 220 |
+
quotes) where [LETTER] is one of {letters}. Think step by step before answering.
|
| 221 |
+
|
| 222 |
+
|
| 223 |
+
{question}
|
| 224 |
+
|
| 225 |
+
|
| 226 |
+
{choices}'
|
| 227 |
+
query_template: null
|
| 228 |
+
review_timeout: null
|
| 229 |
+
sample_example: null
|
| 230 |
+
sandbox_config: {}
|
| 231 |
+
shuffle: false
|
| 232 |
+
shuffle_choices: false
|
| 233 |
+
subset_list:
|
| 234 |
+
- high
|
| 235 |
+
- middle
|
| 236 |
+
system_prompt: null
|
| 237 |
+
tags:
|
| 238 |
+
- Reasoning
|
| 239 |
+
- MCQ
|
| 240 |
+
train_split: train
|
| 241 |
+
trivia_qa:
|
| 242 |
+
aggregation: mean
|
| 243 |
+
data_statistics: null
|
| 244 |
+
dataset_id: evalscope/trivia_qa
|
| 245 |
+
default_subset: default
|
| 246 |
+
description: '
|
| 247 |
+
|
| 248 |
+
## Overview
|
| 249 |
+
|
| 250 |
+
|
| 251 |
+
TriviaQA is a large-scale reading comprehension dataset containing over 650K
|
| 252 |
+
question-answer-evidence triples. Questions are collected from trivia enthusiast
|
| 253 |
+
websites and paired with Wikipedia articles as evidence documents.
|
| 254 |
+
|
| 255 |
+
|
| 256 |
+
## Task Description
|
| 257 |
+
|
| 258 |
+
|
| 259 |
+
- **Task Type**: Reading Comprehension / Question Answering
|
| 260 |
+
|
| 261 |
+
- **Input**: Question with Wikipedia context passage
|
| 262 |
+
|
| 263 |
+
- **Output**: Answer extracted or generated from context
|
| 264 |
+
|
| 265 |
+
- **Domain**: General knowledge trivia questions
|
| 266 |
+
|
| 267 |
+
|
| 268 |
+
## Key Features
|
| 269 |
+
|
| 270 |
+
|
| 271 |
+
- 650K+ question-answer-evidence triples
|
| 272 |
+
|
| 273 |
+
- Questions written by trivia enthusiasts (naturally challenging)
|
| 274 |
+
|
| 275 |
+
- Multiple valid answer aliases for flexible evaluation
|
| 276 |
+
|
| 277 |
+
- Wikipedia articles provide evidence passages
|
| 278 |
+
|
| 279 |
+
- Tests both reading comprehension and knowledge retrieval
|
| 280 |
+
|
| 281 |
+
|
| 282 |
+
## Evaluation Notes
|
| 283 |
+
|
| 284 |
+
|
| 285 |
+
- Default configuration uses **0-shot** evaluation
|
| 286 |
+
|
| 287 |
+
- Uses the Wikipedia reading comprehension subset (rc.wikipedia)
|
| 288 |
+
|
| 289 |
+
- Answers should follow the format: "ANSWER: [ANSWER]"
|
| 290 |
+
|
| 291 |
+
- Supports inclusion-based matching for answer comparison
|
| 292 |
+
|
| 293 |
+
- Evaluates on validation split
|
| 294 |
+
|
| 295 |
+
'
|
| 296 |
+
eval_split: validation
|
| 297 |
+
extra_params: {}
|
| 298 |
+
few_shot_num: 0
|
| 299 |
+
few_shot_prompt_template: null
|
| 300 |
+
few_shot_random: false
|
| 301 |
+
filters: null
|
| 302 |
+
force_redownload: false
|
| 303 |
+
metric_list:
|
| 304 |
+
- acc:
|
| 305 |
+
allow_inclusion: true
|
| 306 |
+
name: trivia_qa
|
| 307 |
+
output_types:
|
| 308 |
+
- generation
|
| 309 |
+
paper_url: null
|
| 310 |
+
pretty_name: TriviaQA
|
| 311 |
+
prompt_template: 'Read the content and answer the following question.
|
| 312 |
+
|
| 313 |
+
|
| 314 |
+
Content: {content}
|
| 315 |
+
|
| 316 |
+
|
| 317 |
+
Question: {question}
|
| 318 |
+
|
| 319 |
+
|
| 320 |
+
Keep your The last line of your response should be of the form "ANSWER: [ANSWER]"
|
| 321 |
+
(without quotes) where [ANSWER] is the answer to the problem.
|
| 322 |
+
|
| 323 |
+
'
|
| 324 |
+
query_template: null
|
| 325 |
+
review_timeout: null
|
| 326 |
+
sample_example: null
|
| 327 |
+
sandbox_config: {}
|
| 328 |
+
shuffle: false
|
| 329 |
+
shuffle_choices: false
|
| 330 |
+
subset_list:
|
| 331 |
+
- rc.wikipedia
|
| 332 |
+
system_prompt: null
|
| 333 |
+
tags:
|
| 334 |
+
- QA
|
| 335 |
+
- ReadingComprehension
|
| 336 |
+
train_split: null
|
| 337 |
+
dataset_dir: /root/.cache/modelscope/hub/datasets
|
| 338 |
+
dataset_hub: modelscope
|
| 339 |
+
datasets:
|
| 340 |
+
- ifeval
|
| 341 |
+
- race
|
| 342 |
+
- trivia_qa
|
| 343 |
+
- humaneval
|
| 344 |
+
debug: false
|
| 345 |
+
enable_progress_tracker: false
|
| 346 |
+
eval_backend: Native
|
| 347 |
+
eval_batch_size: 16
|
| 348 |
+
eval_config: null
|
| 349 |
+
eval_type: server
|
| 350 |
+
evalscope_version: 1.7.1
|
| 351 |
+
generation_config:
|
| 352 |
+
batch_size: 16
|
| 353 |
+
do_sample: true
|
| 354 |
+
max_new_tokens: 2048
|
| 355 |
+
temperature: 0.7
|
| 356 |
+
ignore_errors: true
|
| 357 |
+
judge_model_args: {}
|
| 358 |
+
judge_strategy: auto
|
| 359 |
+
judge_worker_num: 1
|
| 360 |
+
limit: null
|
| 361 |
+
model: dir-ppo-llama3.1-8b
|
| 362 |
+
model_args: {}
|
| 363 |
+
model_id: dir-ppo-llama3.1-8b
|
| 364 |
+
model_task: text_generation
|
| 365 |
+
no_timestamp: false
|
| 366 |
+
repeats: 1
|
| 367 |
+
rerun_review: false
|
| 368 |
+
sandbox:
|
| 369 |
+
default_config: {}
|
| 370 |
+
enabled: false
|
| 371 |
+
engine: docker
|
| 372 |
+
manager_config: {}
|
| 373 |
+
pool_size: null
|
| 374 |
+
sandbox_manager_config: {}
|
| 375 |
+
sandbox_type: docker
|
| 376 |
+
seed: 42
|
| 377 |
+
stream: null
|
| 378 |
+
timeout: null
|
| 379 |
+
use_cache: null
|
| 380 |
+
use_sandbox: false
|
| 381 |
+
work_dir: ./outputs/20260527_000425
|
eval/20260527_000425/logs/eval_log.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
eval/20260527_000425/predictions/dir-ppo-llama3.1-8b/ifeval_default.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
eval/20260527_000425/predictions/dir-ppo-llama3.1-8b/race_high.jsonl
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:d595bed6f072c3ef9ed0e19010e34b124818146eb724a04531238384b16fe08e
|
| 3 |
+
size 40497772
|
eval/20260527_000425/predictions/dir-ppo-llama3.1-8b/race_middle.jsonl
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:b0ea7aff3ae009236736a2c2ca8c02a74eedf3b6f87b4091de2918a67bca719c
|
| 3 |
+
size 10885732
|
eval/20260527_000425/predictions/dir-ppo-llama3.1-8b/trivia_qa_rc.wikipedia.jsonl
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:64d916009fd65f0108bed676b6b0d28785cae1a41b955cf7e466df9454a28f69
|
| 3 |
+
size 46729079
|
eval/20260527_000425/reports/dir-ppo-llama3.1-8b/ifeval.json
ADDED
|
@@ -0,0 +1,162 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"name": "dir-ppo-llama3.1-8b@ifeval",
|
| 3 |
+
"dataset_name": "ifeval",
|
| 4 |
+
"dataset_pretty_name": "IFEval",
|
| 5 |
+
"dataset_description": "\n## Overview\n\nIFEval (Instruction-Following Eval) is a benchmark for evaluating how well language models follow explicit, verifiable instructions. It contains prompts with specific formatting, content, or structural requirements that can be objectively verified.\n\n## Task Description\n\n- **Task Type**: Instruction Following Evaluation\n- **Input**: Prompts with explicit, verifiable constraints\n- **Output**: Response that follows all specified instructions\n- **Constraint Types**: Format, length, keywords, structure, etc.\n\n## Key Features\n\n- ~500 prompts with 25 types of verifiable instructions\n- Instructions are objectively checkable (not subjective)\n- Examples: \"write exactly 3 paragraphs\", \"include the word X\", \"use bullet points\"\n- Tests instruction comprehension and compliance\n- No ambiguity in evaluation criteria\n\n## Evaluation Notes\n\n- Default configuration uses **0-shot** evaluation\n- Four metrics available:\n - `prompt_level_strict`: All instructions in prompt must be followed\n - `prompt_level_loose`: Some tolerance for minor deviations\n - `inst_level_strict`: Per-instruction accuracy (strict)\n - `inst_level_loose`: Per-instruction accuracy (loose)\n- `prompt_level_strict` is the primary metric\n- Automatic verification of instruction compliance\n",
|
| 6 |
+
"model_name": "dir-ppo-llama3.1-8b",
|
| 7 |
+
"score": 0.6981,
|
| 8 |
+
"metrics": [
|
| 9 |
+
{
|
| 10 |
+
"name": "mean_prompt_level_strict",
|
| 11 |
+
"num": 540,
|
| 12 |
+
"score": 0.6981,
|
| 13 |
+
"macro_score": 0.6981,
|
| 14 |
+
"categories": [
|
| 15 |
+
{
|
| 16 |
+
"name": [
|
| 17 |
+
"default"
|
| 18 |
+
],
|
| 19 |
+
"num": 540,
|
| 20 |
+
"score": 0.6981,
|
| 21 |
+
"macro_score": 0.6981,
|
| 22 |
+
"subsets": [
|
| 23 |
+
{
|
| 24 |
+
"name": "default",
|
| 25 |
+
"score": 0.6981,
|
| 26 |
+
"num": 540
|
| 27 |
+
}
|
| 28 |
+
]
|
| 29 |
+
}
|
| 30 |
+
]
|
| 31 |
+
},
|
| 32 |
+
{
|
| 33 |
+
"name": "mean_inst_level_strict",
|
| 34 |
+
"num": 540,
|
| 35 |
+
"score": 0.7901,
|
| 36 |
+
"macro_score": 0.7901,
|
| 37 |
+
"categories": [
|
| 38 |
+
{
|
| 39 |
+
"name": [
|
| 40 |
+
"default"
|
| 41 |
+
],
|
| 42 |
+
"num": 540,
|
| 43 |
+
"score": 0.7901,
|
| 44 |
+
"macro_score": 0.7901,
|
| 45 |
+
"subsets": [
|
| 46 |
+
{
|
| 47 |
+
"name": "default",
|
| 48 |
+
"score": 0.7901,
|
| 49 |
+
"num": 540
|
| 50 |
+
}
|
| 51 |
+
]
|
| 52 |
+
}
|
| 53 |
+
]
|
| 54 |
+
},
|
| 55 |
+
{
|
| 56 |
+
"name": "mean_prompt_level_loose",
|
| 57 |
+
"num": 540,
|
| 58 |
+
"score": 0.787,
|
| 59 |
+
"macro_score": 0.787,
|
| 60 |
+
"categories": [
|
| 61 |
+
{
|
| 62 |
+
"name": [
|
| 63 |
+
"default"
|
| 64 |
+
],
|
| 65 |
+
"num": 540,
|
| 66 |
+
"score": 0.787,
|
| 67 |
+
"macro_score": 0.787,
|
| 68 |
+
"subsets": [
|
| 69 |
+
{
|
| 70 |
+
"name": "default",
|
| 71 |
+
"score": 0.787,
|
| 72 |
+
"num": 540
|
| 73 |
+
}
|
| 74 |
+
]
|
| 75 |
+
}
|
| 76 |
+
]
|
| 77 |
+
},
|
| 78 |
+
{
|
| 79 |
+
"name": "mean_inst_level_loose",
|
| 80 |
+
"num": 540,
|
| 81 |
+
"score": 0.8623,
|
| 82 |
+
"macro_score": 0.8623,
|
| 83 |
+
"categories": [
|
| 84 |
+
{
|
| 85 |
+
"name": [
|
| 86 |
+
"default"
|
| 87 |
+
],
|
| 88 |
+
"num": 540,
|
| 89 |
+
"score": 0.8623,
|
| 90 |
+
"macro_score": 0.8623,
|
| 91 |
+
"subsets": [
|
| 92 |
+
{
|
| 93 |
+
"name": "default",
|
| 94 |
+
"score": 0.8623,
|
| 95 |
+
"num": 540
|
| 96 |
+
}
|
| 97 |
+
]
|
| 98 |
+
}
|
| 99 |
+
]
|
| 100 |
+
}
|
| 101 |
+
],
|
| 102 |
+
"analysis": "N/A",
|
| 103 |
+
"perf_metrics": {
|
| 104 |
+
"summary": {
|
| 105 |
+
"n_samples": 541,
|
| 106 |
+
"latency": {
|
| 107 |
+
"mean": 7.710747,
|
| 108 |
+
"std": 16.563807,
|
| 109 |
+
"min": 0.066271,
|
| 110 |
+
"25%": 2.457928,
|
| 111 |
+
"50%": 5.100689,
|
| 112 |
+
"75%": 8.237214,
|
| 113 |
+
"90%": 12.235768,
|
| 114 |
+
"99%": 107.019331,
|
| 115 |
+
"max": 160.99844
|
| 116 |
+
},
|
| 117 |
+
"throughput": {
|
| 118 |
+
"avg_output_tps": 52.84,
|
| 119 |
+
"avg_req_ps": 0.1297
|
| 120 |
+
},
|
| 121 |
+
"usage": {
|
| 122 |
+
"input_tokens": {
|
| 123 |
+
"mean": 80.924214,
|
| 124 |
+
"std": 23.92573,
|
| 125 |
+
"min": 48.0,
|
| 126 |
+
"25%": 67.0,
|
| 127 |
+
"50%": 76.0,
|
| 128 |
+
"75%": 90.0,
|
| 129 |
+
"90%": 107.0,
|
| 130 |
+
"99%": 135.6,
|
| 131 |
+
"max": 395.0
|
| 132 |
+
},
|
| 133 |
+
"output_tokens": {
|
| 134 |
+
"mean": 407.428835,
|
| 135 |
+
"std": 862.651879,
|
| 136 |
+
"min": 3.0,
|
| 137 |
+
"25%": 127.0,
|
| 138 |
+
"50%": 265.0,
|
| 139 |
+
"75%": 436.0,
|
| 140 |
+
"90%": 642.0,
|
| 141 |
+
"99%": 5846.8,
|
| 142 |
+
"max": 8138.0
|
| 143 |
+
},
|
| 144 |
+
"total_tokens": {
|
| 145 |
+
"mean": 488.35305,
|
| 146 |
+
"std": 861.312936,
|
| 147 |
+
"min": 69.0,
|
| 148 |
+
"25%": 212.0,
|
| 149 |
+
"50%": 346.0,
|
| 150 |
+
"75%": 521.0,
|
| 151 |
+
"90%": 725.0,
|
| 152 |
+
"99%": 5930.8,
|
| 153 |
+
"max": 8192.0
|
| 154 |
+
},
|
| 155 |
+
"total_input_tokens": 43780,
|
| 156 |
+
"total_output_tokens": 220419,
|
| 157 |
+
"total_tokens_count": 264199
|
| 158 |
+
}
|
| 159 |
+
}
|
| 160 |
+
},
|
| 161 |
+
"num": 540
|
| 162 |
+
}
|
eval/20260527_000425/reports/dir-ppo-llama3.1-8b/race.json
ADDED
|
@@ -0,0 +1,98 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"name": "dir-ppo-llama3.1-8b@race",
|
| 3 |
+
"dataset_name": "race",
|
| 4 |
+
"dataset_pretty_name": "RACE",
|
| 5 |
+
"dataset_description": "\n## Overview\n\nRACE (ReAding Comprehension from Examinations) is a large-scale reading comprehension benchmark collected from Chinese middle school and high school English examinations. It tests comprehensive reading comprehension abilities.\n\n## Task Description\n\n- **Task Type**: Reading Comprehension (Multiple-Choice)\n- **Input**: Article passage with question and 4 answer choices\n- **Output**: Correct answer letter (A, B, C, or D)\n- **Difficulty Levels**: Middle school and High school\n\n## Key Features\n\n- 28,000+ passages with 100,000 questions\n- Real examination questions for authentic difficulty\n- Two subsets: middle (easier) and high (harder)\n- Tests various comprehension skills (inference, vocabulary, main idea, etc.)\n- Diverse article topics and question types\n\n## Evaluation Notes\n\n- Default configuration uses **3-shot** examples\n- Maximum few-shot number is 3 (context length consideration)\n- Uses Chain-of-Thought (CoT) prompting\n- Two subsets available: `high` and `middle`\n- Evaluates on test split\n",
|
| 6 |
+
"model_name": "dir-ppo-llama3.1-8b",
|
| 7 |
+
"score": 0.833,
|
| 8 |
+
"metrics": [
|
| 9 |
+
{
|
| 10 |
+
"name": "mean_acc",
|
| 11 |
+
"num": 4934,
|
| 12 |
+
"score": 0.833,
|
| 13 |
+
"macro_score": 0.833,
|
| 14 |
+
"categories": [
|
| 15 |
+
{
|
| 16 |
+
"name": [
|
| 17 |
+
"default"
|
| 18 |
+
],
|
| 19 |
+
"num": 4934,
|
| 20 |
+
"score": 0.833,
|
| 21 |
+
"macro_score": 0.8433,
|
| 22 |
+
"subsets": [
|
| 23 |
+
{
|
| 24 |
+
"name": "high",
|
| 25 |
+
"score": 0.8188,
|
| 26 |
+
"num": 3498
|
| 27 |
+
},
|
| 28 |
+
{
|
| 29 |
+
"name": "middle",
|
| 30 |
+
"score": 0.8677,
|
| 31 |
+
"num": 1436
|
| 32 |
+
}
|
| 33 |
+
]
|
| 34 |
+
}
|
| 35 |
+
]
|
| 36 |
+
}
|
| 37 |
+
],
|
| 38 |
+
"analysis": "N/A",
|
| 39 |
+
"perf_metrics": {
|
| 40 |
+
"summary": {
|
| 41 |
+
"n_samples": 4934,
|
| 42 |
+
"latency": {
|
| 43 |
+
"mean": 3.518324,
|
| 44 |
+
"std": 4.350995,
|
| 45 |
+
"min": 0.375683,
|
| 46 |
+
"25%": 2.235433,
|
| 47 |
+
"50%": 3.250558,
|
| 48 |
+
"75%": 4.360585,
|
| 49 |
+
"90%": 5.39817,
|
| 50 |
+
"99%": 7.618212,
|
| 51 |
+
"max": 139.063051
|
| 52 |
+
},
|
| 53 |
+
"throughput": {
|
| 54 |
+
"avg_output_tps": 51.33,
|
| 55 |
+
"avg_req_ps": 0.2842
|
| 56 |
+
},
|
| 57 |
+
"usage": {
|
| 58 |
+
"input_tokens": {
|
| 59 |
+
"mean": 1668.64471,
|
| 60 |
+
"std": 334.066106,
|
| 61 |
+
"min": 958.0,
|
| 62 |
+
"25%": 1274.0,
|
| 63 |
+
"50%": 1824.0,
|
| 64 |
+
"75%": 1885.0,
|
| 65 |
+
"90%": 1953.0,
|
| 66 |
+
"99%": 2269.67,
|
| 67 |
+
"max": 2584.0
|
| 68 |
+
},
|
| 69 |
+
"output_tokens": {
|
| 70 |
+
"mean": 180.605188,
|
| 71 |
+
"std": 216.808972,
|
| 72 |
+
"min": 20.0,
|
| 73 |
+
"25%": 114.0,
|
| 74 |
+
"50%": 168.0,
|
| 75 |
+
"75%": 228.0,
|
| 76 |
+
"90%": 274.0,
|
| 77 |
+
"99%": 387.01,
|
| 78 |
+
"max": 6932.0
|
| 79 |
+
},
|
| 80 |
+
"total_tokens": {
|
| 81 |
+
"mean": 1849.249899,
|
| 82 |
+
"std": 410.659869,
|
| 83 |
+
"min": 1028.0,
|
| 84 |
+
"25%": 1479.0,
|
| 85 |
+
"50%": 1975.0,
|
| 86 |
+
"75%": 2084.0,
|
| 87 |
+
"90%": 2180.0,
|
| 88 |
+
"99%": 2527.39,
|
| 89 |
+
"max": 8192.0
|
| 90 |
+
},
|
| 91 |
+
"total_input_tokens": 8233093,
|
| 92 |
+
"total_output_tokens": 891106,
|
| 93 |
+
"total_tokens_count": 9124199
|
| 94 |
+
}
|
| 95 |
+
}
|
| 96 |
+
},
|
| 97 |
+
"num": 4934
|
| 98 |
+
}
|
eval/20260527_000425/reviews/dir-ppo-llama3.1-8b/ifeval_default.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
eval/20260527_000425/reviews/dir-ppo-llama3.1-8b/race_high.jsonl
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:9bca1ce5a3af7d111c3e98c1d4268a70b41e6a5b06d9d43dd98169389cf01a46
|
| 3 |
+
size 38411680
|
eval/20260527_000425/reviews/dir-ppo-llama3.1-8b/race_middle.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
eval/20260527_000425/reviews/dir-ppo-llama3.1-8b/trivia_qa_rc.wikipedia.jsonl
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:fb259e7f4c082c3794ef1c4b4671d1bf7b24a4bee17839a9e079287f132d033d
|
| 3 |
+
size 45028591
|