SilverStRock commited on
Commit
f3fcc0e
·
verified ·
1 Parent(s): daa2382

initial upload from H20 cluster: eval/20260527_000425

Browse files
Files changed (30) hide show
  1. .gitattributes +11 -0
  2. eval/20260527_000425/20260527_000425/configs/task_config.yaml +381 -0
  3. eval/20260527_000425/20260527_000425/logs/eval_log.log +3 -0
  4. eval/20260527_000425/20260527_000425/predictions/dir-ppo-llama3.1-8b/humaneval_openai_humaneval.jsonl +0 -0
  5. eval/20260527_000425/20260527_000425/predictions/dir-ppo-llama3.1-8b/ifeval_default.jsonl +0 -0
  6. eval/20260527_000425/20260527_000425/predictions/dir-ppo-llama3.1-8b/race_high.jsonl +3 -0
  7. eval/20260527_000425/20260527_000425/predictions/dir-ppo-llama3.1-8b/race_middle.jsonl +3 -0
  8. eval/20260527_000425/20260527_000425/predictions/dir-ppo-llama3.1-8b/trivia_qa_rc.wikipedia.jsonl +3 -0
  9. eval/20260527_000425/20260527_000425/reports/dir-ppo-llama3.1-8b/humaneval.json +116 -0
  10. eval/20260527_000425/20260527_000425/reports/dir-ppo-llama3.1-8b/ifeval.json +162 -0
  11. eval/20260527_000425/20260527_000425/reports/dir-ppo-llama3.1-8b/race.json +98 -0
  12. eval/20260527_000425/20260527_000425/reports/dir-ppo-llama3.1-8b/trivia_qa.json +93 -0
  13. eval/20260527_000425/20260527_000425/reports/report.html +0 -0
  14. eval/20260527_000425/20260527_000425/reviews/dir-ppo-llama3.1-8b/humaneval_openai_humaneval.jsonl +0 -0
  15. eval/20260527_000425/20260527_000425/reviews/dir-ppo-llama3.1-8b/ifeval_default.jsonl +0 -0
  16. eval/20260527_000425/20260527_000425/reviews/dir-ppo-llama3.1-8b/race_high.jsonl +3 -0
  17. eval/20260527_000425/20260527_000425/reviews/dir-ppo-llama3.1-8b/race_middle.jsonl +0 -0
  18. eval/20260527_000425/20260527_000425/reviews/dir-ppo-llama3.1-8b/trivia_qa_rc.wikipedia.jsonl +3 -0
  19. eval/20260527_000425/configs/task_config.yaml +381 -0
  20. eval/20260527_000425/logs/eval_log.log +0 -0
  21. eval/20260527_000425/predictions/dir-ppo-llama3.1-8b/ifeval_default.jsonl +0 -0
  22. eval/20260527_000425/predictions/dir-ppo-llama3.1-8b/race_high.jsonl +3 -0
  23. eval/20260527_000425/predictions/dir-ppo-llama3.1-8b/race_middle.jsonl +3 -0
  24. eval/20260527_000425/predictions/dir-ppo-llama3.1-8b/trivia_qa_rc.wikipedia.jsonl +3 -0
  25. eval/20260527_000425/reports/dir-ppo-llama3.1-8b/ifeval.json +162 -0
  26. eval/20260527_000425/reports/dir-ppo-llama3.1-8b/race.json +98 -0
  27. eval/20260527_000425/reviews/dir-ppo-llama3.1-8b/ifeval_default.jsonl +0 -0
  28. eval/20260527_000425/reviews/dir-ppo-llama3.1-8b/race_high.jsonl +3 -0
  29. eval/20260527_000425/reviews/dir-ppo-llama3.1-8b/race_middle.jsonl +0 -0
  30. eval/20260527_000425/reviews/dir-ppo-llama3.1-8b/trivia_qa_rc.wikipedia.jsonl +3 -0
.gitattributes CHANGED
@@ -34,3 +34,14 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
36
  ppo_dir_v2_merged/tokenizer.json filter=lfs diff=lfs merge=lfs -text
 
 
 
 
 
 
 
 
 
 
 
 
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
36
  ppo_dir_v2_merged/tokenizer.json filter=lfs diff=lfs merge=lfs -text
37
+ eval/20260527_000425/20260527_000425/logs/eval_log.log filter=lfs diff=lfs merge=lfs -text
38
+ eval/20260527_000425/20260527_000425/predictions/dir-ppo-llama3.1-8b/race_high.jsonl filter=lfs diff=lfs merge=lfs -text
39
+ eval/20260527_000425/20260527_000425/predictions/dir-ppo-llama3.1-8b/race_middle.jsonl filter=lfs diff=lfs merge=lfs -text
40
+ eval/20260527_000425/20260527_000425/predictions/dir-ppo-llama3.1-8b/trivia_qa_rc.wikipedia.jsonl filter=lfs diff=lfs merge=lfs -text
41
+ eval/20260527_000425/20260527_000425/reviews/dir-ppo-llama3.1-8b/race_high.jsonl filter=lfs diff=lfs merge=lfs -text
42
+ eval/20260527_000425/20260527_000425/reviews/dir-ppo-llama3.1-8b/trivia_qa_rc.wikipedia.jsonl filter=lfs diff=lfs merge=lfs -text
43
+ eval/20260527_000425/predictions/dir-ppo-llama3.1-8b/race_high.jsonl filter=lfs diff=lfs merge=lfs -text
44
+ eval/20260527_000425/predictions/dir-ppo-llama3.1-8b/race_middle.jsonl filter=lfs diff=lfs merge=lfs -text
45
+ eval/20260527_000425/predictions/dir-ppo-llama3.1-8b/trivia_qa_rc.wikipedia.jsonl filter=lfs diff=lfs merge=lfs -text
46
+ eval/20260527_000425/reviews/dir-ppo-llama3.1-8b/race_high.jsonl filter=lfs diff=lfs merge=lfs -text
47
+ eval/20260527_000425/reviews/dir-ppo-llama3.1-8b/trivia_qa_rc.wikipedia.jsonl filter=lfs diff=lfs merge=lfs -text
eval/20260527_000425/20260527_000425/configs/task_config.yaml ADDED
@@ -0,0 +1,381 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ agent_config: null
2
+ analysis_report: false
3
+ api_url: http://127.0.0.1:8801/v1
4
+ chat_template: null
5
+ collect_perf: true
6
+ dataset_args:
7
+ humaneval:
8
+ aggregation: mean_and_pass_at_k
9
+ data_statistics: null
10
+ dataset_id: opencompass/humaneval
11
+ default_subset: default
12
+ description: '
13
+
14
+ ## Overview
15
+
16
+
17
+ HumanEval is a benchmark for evaluating the code generation capabilities of
18
+ language models. It consists of 164 hand-written Python programming problems
19
+ with function signatures, docstrings, and comprehensive test cases.
20
+
21
+
22
+ ## Task Description
23
+
24
+
25
+ - **Task Type**: Code Generation (Python)
26
+
27
+ - **Input**: Function signature with docstring describing the expected behavior
28
+
29
+ - **Output**: Complete Python function implementation
30
+
31
+ - **Languages**: Python only
32
+
33
+
34
+ ## Key Features
35
+
36
+
37
+ - 164 hand-crafted programming problems
38
+
39
+ - Each problem includes a function signature, docstring, and test cases
40
+
41
+ - Problems range from simple string manipulation to complex algorithms
42
+
43
+ - Canonical solutions provided for reference
44
+
45
+ - Automatic correctness verification through test execution
46
+
47
+
48
+ ## Evaluation Notes
49
+
50
+
51
+ - **Security Warning**: By default, code is executed in the local environment.
52
+ We strongly recommend using sandbox execution for safety. See the [sandbox documentation](https://evalscope.readthedocs.io/en/latest/user_guides/sandbox.html)
53
+ for details.
54
+
55
+ - Supports `pass@k` metric calculation for measuring generation quality
56
+
57
+ - Default timeout is 4 seconds per problem
58
+
59
+ - Code is extracted from markdown code blocks if present
60
+
61
+ '
62
+ eval_split: test
63
+ extra_params: {}
64
+ few_shot_num: 0
65
+ few_shot_prompt_template: null
66
+ few_shot_random: false
67
+ filters: null
68
+ force_redownload: false
69
+ metric_list:
70
+ - acc
71
+ name: humaneval
72
+ output_types:
73
+ - generation
74
+ paper_url: null
75
+ pretty_name: HumanEval
76
+ prompt_template: 'Read the following function signature and docstring, and fully
77
+ implement the function described. Your response should only contain the code
78
+ for this function.
79
+
80
+ {question}'
81
+ query_template: null
82
+ review_timeout: 4
83
+ sample_example: null
84
+ sandbox_config:
85
+ image: python:3.11-slim
86
+ tools_config:
87
+ python_executor: {}
88
+ shell_executor: {}
89
+ shuffle: false
90
+ shuffle_choices: false
91
+ subset_list:
92
+ - openai_humaneval
93
+ system_prompt: null
94
+ tags:
95
+ - Coding
96
+ train_split: null
97
+ ifeval:
98
+ aggregation: mean
99
+ data_statistics: null
100
+ dataset_id: opencompass/ifeval
101
+ default_subset: default
102
+ description: "\n## Overview\n\nIFEval (Instruction-Following Eval) is a benchmark\
103
+ \ for evaluating how well language models follow explicit, verifiable instructions.\
104
+ \ It contains prompts with specific formatting, content, or structural requirements\
105
+ \ that can be objectively verified.\n\n## Task Description\n\n- **Task Type**:\
106
+ \ Instruction Following Evaluation\n- **Input**: Prompts with explicit, verifiable\
107
+ \ constraints\n- **Output**: Response that follows all specified instructions\n\
108
+ - **Constraint Types**: Format, length, keywords, structure, etc.\n\n## Key\
109
+ \ Features\n\n- ~500 prompts with 25 types of verifiable instructions\n- Instructions\
110
+ \ are objectively checkable (not subjective)\n- Examples: \"write exactly 3\
111
+ \ paragraphs\", \"include the word X\", \"use bullet points\"\n- Tests instruction\
112
+ \ comprehension and compliance\n- No ambiguity in evaluation criteria\n\n##\
113
+ \ Evaluation Notes\n\n- Default configuration uses **0-shot** evaluation\n-\
114
+ \ Four metrics available:\n - `prompt_level_strict`: All instructions in prompt\
115
+ \ must be followed\n - `prompt_level_loose`: Some tolerance for minor deviations\n\
116
+ \ - `inst_level_strict`: Per-instruction accuracy (strict)\n - `inst_level_loose`:\
117
+ \ Per-instruction accuracy (loose)\n- `prompt_level_strict` is the primary metric\n\
118
+ - Automatic verification of instruction compliance\n"
119
+ eval_split: train
120
+ extra_params: {}
121
+ few_shot_num: 0
122
+ few_shot_prompt_template: null
123
+ few_shot_random: false
124
+ filters: null
125
+ force_redownload: false
126
+ metric_list:
127
+ - prompt_level_strict
128
+ - inst_level_strict
129
+ - prompt_level_loose
130
+ - inst_level_loose
131
+ name: ifeval
132
+ output_types:
133
+ - generation
134
+ paper_url: null
135
+ pretty_name: IFEval
136
+ prompt_template: ''
137
+ query_template: null
138
+ review_timeout: null
139
+ sample_example: null
140
+ sandbox_config: {}
141
+ shuffle: false
142
+ shuffle_choices: false
143
+ subset_list:
144
+ - default
145
+ system_prompt: null
146
+ tags:
147
+ - InstructionFollowing
148
+ train_split: null
149
+ race:
150
+ aggregation: mean
151
+ data_statistics: null
152
+ dataset_id: evalscope/race
153
+ default_subset: default
154
+ description: '
155
+
156
+ ## Overview
157
+
158
+
159
+ RACE (ReAding Comprehension from Examinations) is a large-scale reading comprehension
160
+ benchmark collected from Chinese middle school and high school English examinations.
161
+ It tests comprehensive reading comprehension abilities.
162
+
163
+
164
+ ## Task Description
165
+
166
+
167
+ - **Task Type**: Reading Comprehension (Multiple-Choice)
168
+
169
+ - **Input**: Article passage with question and 4 answer choices
170
+
171
+ - **Output**: Correct answer letter (A, B, C, or D)
172
+
173
+ - **Difficulty Levels**: Middle school and High school
174
+
175
+
176
+ ## Key Features
177
+
178
+
179
+ - 28,000+ passages with 100,000 questions
180
+
181
+ - Real examination questions for authentic difficulty
182
+
183
+ - Two subsets: middle (easier) and high (harder)
184
+
185
+ - Tests various comprehension skills (inference, vocabulary, main idea, etc.)
186
+
187
+ - Diverse article topics and question types
188
+
189
+
190
+ ## Evaluation Notes
191
+
192
+
193
+ - Default configuration uses **3-shot** examples
194
+
195
+ - Maximum few-shot number is 3 (context length consideration)
196
+
197
+ - Uses Chain-of-Thought (CoT) prompting
198
+
199
+ - Two subsets available: `high` and `middle`
200
+
201
+ - Evaluates on test split
202
+
203
+ '
204
+ eval_split: test
205
+ extra_params: {}
206
+ few_shot_num: 3
207
+ few_shot_prompt_template: null
208
+ few_shot_random: false
209
+ filters: null
210
+ force_redownload: false
211
+ metric_list:
212
+ - acc
213
+ name: race
214
+ output_types:
215
+ - generation
216
+ paper_url: null
217
+ pretty_name: RACE
218
+ prompt_template: 'Answer the following multiple choice question. The last line
219
+ of your response should be of the following format: ''ANSWER: [LETTER]'' (without
220
+ quotes) where [LETTER] is one of {letters}. Think step by step before answering.
221
+
222
+
223
+ {question}
224
+
225
+
226
+ {choices}'
227
+ query_template: null
228
+ review_timeout: null
229
+ sample_example: null
230
+ sandbox_config: {}
231
+ shuffle: false
232
+ shuffle_choices: false
233
+ subset_list:
234
+ - high
235
+ - middle
236
+ system_prompt: null
237
+ tags:
238
+ - Reasoning
239
+ - MCQ
240
+ train_split: train
241
+ trivia_qa:
242
+ aggregation: mean
243
+ data_statistics: null
244
+ dataset_id: evalscope/trivia_qa
245
+ default_subset: default
246
+ description: '
247
+
248
+ ## Overview
249
+
250
+
251
+ TriviaQA is a large-scale reading comprehension dataset containing over 650K
252
+ question-answer-evidence triples. Questions are collected from trivia enthusiast
253
+ websites and paired with Wikipedia articles as evidence documents.
254
+
255
+
256
+ ## Task Description
257
+
258
+
259
+ - **Task Type**: Reading Comprehension / Question Answering
260
+
261
+ - **Input**: Question with Wikipedia context passage
262
+
263
+ - **Output**: Answer extracted or generated from context
264
+
265
+ - **Domain**: General knowledge trivia questions
266
+
267
+
268
+ ## Key Features
269
+
270
+
271
+ - 650K+ question-answer-evidence triples
272
+
273
+ - Questions written by trivia enthusiasts (naturally challenging)
274
+
275
+ - Multiple valid answer aliases for flexible evaluation
276
+
277
+ - Wikipedia articles provide evidence passages
278
+
279
+ - Tests both reading comprehension and knowledge retrieval
280
+
281
+
282
+ ## Evaluation Notes
283
+
284
+
285
+ - Default configuration uses **0-shot** evaluation
286
+
287
+ - Uses the Wikipedia reading comprehension subset (rc.wikipedia)
288
+
289
+ - Answers should follow the format: "ANSWER: [ANSWER]"
290
+
291
+ - Supports inclusion-based matching for answer comparison
292
+
293
+ - Evaluates on validation split
294
+
295
+ '
296
+ eval_split: validation
297
+ extra_params: {}
298
+ few_shot_num: 0
299
+ few_shot_prompt_template: null
300
+ few_shot_random: false
301
+ filters: null
302
+ force_redownload: false
303
+ metric_list:
304
+ - acc:
305
+ allow_inclusion: true
306
+ name: trivia_qa
307
+ output_types:
308
+ - generation
309
+ paper_url: null
310
+ pretty_name: TriviaQA
311
+ prompt_template: 'Read the content and answer the following question.
312
+
313
+
314
+ Content: {content}
315
+
316
+
317
+ Question: {question}
318
+
319
+
320
+ Keep your The last line of your response should be of the form "ANSWER: [ANSWER]"
321
+ (without quotes) where [ANSWER] is the answer to the problem.
322
+
323
+ '
324
+ query_template: null
325
+ review_timeout: null
326
+ sample_example: null
327
+ sandbox_config: {}
328
+ shuffle: false
329
+ shuffle_choices: false
330
+ subset_list:
331
+ - rc.wikipedia
332
+ system_prompt: null
333
+ tags:
334
+ - QA
335
+ - ReadingComprehension
336
+ train_split: null
337
+ dataset_dir: /root/.cache/modelscope/hub/datasets
338
+ dataset_hub: modelscope
339
+ datasets:
340
+ - ifeval
341
+ - race
342
+ - trivia_qa
343
+ - humaneval
344
+ debug: false
345
+ enable_progress_tracker: false
346
+ eval_backend: Native
347
+ eval_batch_size: 16
348
+ eval_config: null
349
+ eval_type: server
350
+ evalscope_version: 1.7.1
351
+ generation_config:
352
+ batch_size: 16
353
+ do_sample: true
354
+ max_new_tokens: 2048
355
+ temperature: 0.7
356
+ ignore_errors: true
357
+ judge_model_args: {}
358
+ judge_strategy: auto
359
+ judge_worker_num: 1
360
+ limit: null
361
+ model: dir-ppo-llama3.1-8b
362
+ model_args: {}
363
+ model_id: dir-ppo-llama3.1-8b
364
+ model_task: text_generation
365
+ no_timestamp: false
366
+ repeats: 1
367
+ rerun_review: false
368
+ sandbox:
369
+ default_config: {}
370
+ enabled: false
371
+ engine: docker
372
+ manager_config: {}
373
+ pool_size: null
374
+ sandbox_manager_config: {}
375
+ sandbox_type: docker
376
+ seed: 42
377
+ stream: null
378
+ timeout: null
379
+ use_cache: null
380
+ use_sandbox: false
381
+ work_dir: ./outputs/20260527_000425
eval/20260527_000425/20260527_000425/logs/eval_log.log ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b4e24adb27f4e9824f5bd4b5f70d41c417786bbf23c8b36174a21befca1c0252
3
+ size 26273843
eval/20260527_000425/20260527_000425/predictions/dir-ppo-llama3.1-8b/humaneval_openai_humaneval.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
eval/20260527_000425/20260527_000425/predictions/dir-ppo-llama3.1-8b/ifeval_default.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
eval/20260527_000425/20260527_000425/predictions/dir-ppo-llama3.1-8b/race_high.jsonl ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d595bed6f072c3ef9ed0e19010e34b124818146eb724a04531238384b16fe08e
3
+ size 40497772
eval/20260527_000425/20260527_000425/predictions/dir-ppo-llama3.1-8b/race_middle.jsonl ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b0ea7aff3ae009236736a2c2ca8c02a74eedf3b6f87b4091de2918a67bca719c
3
+ size 10885732
eval/20260527_000425/20260527_000425/predictions/dir-ppo-llama3.1-8b/trivia_qa_rc.wikipedia.jsonl ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a0decfb092206f175111431550b702003521ce5ce21f4555ef86198a500d5f83
3
+ size 125325681
eval/20260527_000425/20260527_000425/reports/dir-ppo-llama3.1-8b/humaneval.json ADDED
@@ -0,0 +1,116 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "name": "dir-ppo-llama3.1-8b@humaneval",
3
+ "dataset_name": "humaneval",
4
+ "dataset_pretty_name": "HumanEval",
5
+ "dataset_description": "\n## Overview\n\nHumanEval is a benchmark for evaluating the code generation capabilities of language models. It consists of 164 hand-written Python programming problems with function signatures, docstrings, and comprehensive test cases.\n\n## Task Description\n\n- **Task Type**: Code Generation (Python)\n- **Input**: Function signature with docstring describing the expected behavior\n- **Output**: Complete Python function implementation\n- **Languages**: Python only\n\n## Key Features\n\n- 164 hand-crafted programming problems\n- Each problem includes a function signature, docstring, and test cases\n- Problems range from simple string manipulation to complex algorithms\n- Canonical solutions provided for reference\n- Automatic correctness verification through test execution\n\n## Evaluation Notes\n\n- **Security Warning**: By default, code is executed in the local environment. We strongly recommend using sandbox execution for safety. See the [sandbox documentation](https://evalscope.readthedocs.io/en/latest/user_guides/sandbox.html) for details.\n- Supports `pass@k` metric calculation for measuring generation quality\n- Default timeout is 4 seconds per problem\n- Code is extracted from markdown code blocks if present\n",
6
+ "model_name": "dir-ppo-llama3.1-8b",
7
+ "score": 0.6768,
8
+ "metrics": [
9
+ {
10
+ "name": "mean_acc",
11
+ "num": 164,
12
+ "score": 0.6768,
13
+ "macro_score": 0.6768,
14
+ "categories": [
15
+ {
16
+ "name": [
17
+ "default"
18
+ ],
19
+ "num": 164,
20
+ "score": 0.6768,
21
+ "macro_score": 0.6768,
22
+ "subsets": [
23
+ {
24
+ "name": "openai_humaneval",
25
+ "score": 0.6768,
26
+ "num": 164
27
+ }
28
+ ]
29
+ }
30
+ ]
31
+ },
32
+ {
33
+ "name": "mean_acc_pass@1",
34
+ "num": 164,
35
+ "score": 0.6768,
36
+ "macro_score": 0.6768,
37
+ "categories": [
38
+ {
39
+ "name": [
40
+ "default"
41
+ ],
42
+ "num": 164,
43
+ "score": 0.6768,
44
+ "macro_score": 0.6768,
45
+ "subsets": [
46
+ {
47
+ "name": "openai_humaneval",
48
+ "score": 0.6768,
49
+ "num": 164
50
+ }
51
+ ]
52
+ }
53
+ ]
54
+ }
55
+ ],
56
+ "analysis": "N/A",
57
+ "perf_metrics": {
58
+ "summary": {
59
+ "n_samples": 164,
60
+ "latency": {
61
+ "mean": 2.326436,
62
+ "std": 1.460745,
63
+ "min": 0.40422,
64
+ "25%": 1.377736,
65
+ "50%": 2.034455,
66
+ "75%": 2.972601,
67
+ "90%": 3.901483,
68
+ "99%": 7.0451,
69
+ "max": 10.605742
70
+ },
71
+ "throughput": {
72
+ "avg_output_tps": 56.83,
73
+ "avg_req_ps": 0.4298
74
+ },
75
+ "usage": {
76
+ "input_tokens": {
77
+ "mean": 192.426829,
78
+ "std": 64.30068,
79
+ "min": 99.0,
80
+ "25%": 148.0,
81
+ "50%": 178.0,
82
+ "75%": 215.75,
83
+ "90%": 286.7,
84
+ "99%": 372.07,
85
+ "max": 452.0
86
+ },
87
+ "output_tokens": {
88
+ "mean": 132.207317,
89
+ "std": 87.969283,
90
+ "min": 26.0,
91
+ "25%": 68.75,
92
+ "50%": 118.0,
93
+ "75%": 169.25,
94
+ "90%": 221.7,
95
+ "99%": 441.16,
96
+ "max": 638.0
97
+ },
98
+ "total_tokens": {
99
+ "mean": 324.634146,
100
+ "std": 133.096811,
101
+ "min": 146.0,
102
+ "25%": 232.75,
103
+ "50%": 298.0,
104
+ "75%": 371.5,
105
+ "90%": 485.8,
106
+ "99%": 760.69,
107
+ "max": 1090.0
108
+ },
109
+ "total_input_tokens": 31558,
110
+ "total_output_tokens": 21682,
111
+ "total_tokens_count": 53240
112
+ }
113
+ }
114
+ },
115
+ "num": 164
116
+ }
eval/20260527_000425/20260527_000425/reports/dir-ppo-llama3.1-8b/ifeval.json ADDED
@@ -0,0 +1,162 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "name": "dir-ppo-llama3.1-8b@ifeval",
3
+ "dataset_name": "ifeval",
4
+ "dataset_pretty_name": "IFEval",
5
+ "dataset_description": "\n## Overview\n\nIFEval (Instruction-Following Eval) is a benchmark for evaluating how well language models follow explicit, verifiable instructions. It contains prompts with specific formatting, content, or structural requirements that can be objectively verified.\n\n## Task Description\n\n- **Task Type**: Instruction Following Evaluation\n- **Input**: Prompts with explicit, verifiable constraints\n- **Output**: Response that follows all specified instructions\n- **Constraint Types**: Format, length, keywords, structure, etc.\n\n## Key Features\n\n- ~500 prompts with 25 types of verifiable instructions\n- Instructions are objectively checkable (not subjective)\n- Examples: \"write exactly 3 paragraphs\", \"include the word X\", \"use bullet points\"\n- Tests instruction comprehension and compliance\n- No ambiguity in evaluation criteria\n\n## Evaluation Notes\n\n- Default configuration uses **0-shot** evaluation\n- Four metrics available:\n - `prompt_level_strict`: All instructions in prompt must be followed\n - `prompt_level_loose`: Some tolerance for minor deviations\n - `inst_level_strict`: Per-instruction accuracy (strict)\n - `inst_level_loose`: Per-instruction accuracy (loose)\n- `prompt_level_strict` is the primary metric\n- Automatic verification of instruction compliance\n",
6
+ "model_name": "dir-ppo-llama3.1-8b",
7
+ "score": 0.6981,
8
+ "metrics": [
9
+ {
10
+ "name": "mean_prompt_level_strict",
11
+ "num": 540,
12
+ "score": 0.6981,
13
+ "macro_score": 0.6981,
14
+ "categories": [
15
+ {
16
+ "name": [
17
+ "default"
18
+ ],
19
+ "num": 540,
20
+ "score": 0.6981,
21
+ "macro_score": 0.6981,
22
+ "subsets": [
23
+ {
24
+ "name": "default",
25
+ "score": 0.6981,
26
+ "num": 540
27
+ }
28
+ ]
29
+ }
30
+ ]
31
+ },
32
+ {
33
+ "name": "mean_inst_level_strict",
34
+ "num": 540,
35
+ "score": 0.7901,
36
+ "macro_score": 0.7901,
37
+ "categories": [
38
+ {
39
+ "name": [
40
+ "default"
41
+ ],
42
+ "num": 540,
43
+ "score": 0.7901,
44
+ "macro_score": 0.7901,
45
+ "subsets": [
46
+ {
47
+ "name": "default",
48
+ "score": 0.7901,
49
+ "num": 540
50
+ }
51
+ ]
52
+ }
53
+ ]
54
+ },
55
+ {
56
+ "name": "mean_prompt_level_loose",
57
+ "num": 540,
58
+ "score": 0.787,
59
+ "macro_score": 0.787,
60
+ "categories": [
61
+ {
62
+ "name": [
63
+ "default"
64
+ ],
65
+ "num": 540,
66
+ "score": 0.787,
67
+ "macro_score": 0.787,
68
+ "subsets": [
69
+ {
70
+ "name": "default",
71
+ "score": 0.787,
72
+ "num": 540
73
+ }
74
+ ]
75
+ }
76
+ ]
77
+ },
78
+ {
79
+ "name": "mean_inst_level_loose",
80
+ "num": 540,
81
+ "score": 0.8623,
82
+ "macro_score": 0.8623,
83
+ "categories": [
84
+ {
85
+ "name": [
86
+ "default"
87
+ ],
88
+ "num": 540,
89
+ "score": 0.8623,
90
+ "macro_score": 0.8623,
91
+ "subsets": [
92
+ {
93
+ "name": "default",
94
+ "score": 0.8623,
95
+ "num": 540
96
+ }
97
+ ]
98
+ }
99
+ ]
100
+ }
101
+ ],
102
+ "analysis": "N/A",
103
+ "perf_metrics": {
104
+ "summary": {
105
+ "n_samples": 541,
106
+ "latency": {
107
+ "mean": 7.710747,
108
+ "std": 16.563807,
109
+ "min": 0.066271,
110
+ "25%": 2.457928,
111
+ "50%": 5.100689,
112
+ "75%": 8.237214,
113
+ "90%": 12.235768,
114
+ "99%": 107.019331,
115
+ "max": 160.99844
116
+ },
117
+ "throughput": {
118
+ "avg_output_tps": 52.84,
119
+ "avg_req_ps": 0.1297
120
+ },
121
+ "usage": {
122
+ "input_tokens": {
123
+ "mean": 80.924214,
124
+ "std": 23.92573,
125
+ "min": 48.0,
126
+ "25%": 67.0,
127
+ "50%": 76.0,
128
+ "75%": 90.0,
129
+ "90%": 107.0,
130
+ "99%": 135.6,
131
+ "max": 395.0
132
+ },
133
+ "output_tokens": {
134
+ "mean": 407.428835,
135
+ "std": 862.651879,
136
+ "min": 3.0,
137
+ "25%": 127.0,
138
+ "50%": 265.0,
139
+ "75%": 436.0,
140
+ "90%": 642.0,
141
+ "99%": 5846.8,
142
+ "max": 8138.0
143
+ },
144
+ "total_tokens": {
145
+ "mean": 488.35305,
146
+ "std": 861.312936,
147
+ "min": 69.0,
148
+ "25%": 212.0,
149
+ "50%": 346.0,
150
+ "75%": 521.0,
151
+ "90%": 725.0,
152
+ "99%": 5930.8,
153
+ "max": 8192.0
154
+ },
155
+ "total_input_tokens": 43780,
156
+ "total_output_tokens": 220419,
157
+ "total_tokens_count": 264199
158
+ }
159
+ }
160
+ },
161
+ "num": 540
162
+ }
eval/20260527_000425/20260527_000425/reports/dir-ppo-llama3.1-8b/race.json ADDED
@@ -0,0 +1,98 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "name": "dir-ppo-llama3.1-8b@race",
3
+ "dataset_name": "race",
4
+ "dataset_pretty_name": "RACE",
5
+ "dataset_description": "\n## Overview\n\nRACE (ReAding Comprehension from Examinations) is a large-scale reading comprehension benchmark collected from Chinese middle school and high school English examinations. It tests comprehensive reading comprehension abilities.\n\n## Task Description\n\n- **Task Type**: Reading Comprehension (Multiple-Choice)\n- **Input**: Article passage with question and 4 answer choices\n- **Output**: Correct answer letter (A, B, C, or D)\n- **Difficulty Levels**: Middle school and High school\n\n## Key Features\n\n- 28,000+ passages with 100,000 questions\n- Real examination questions for authentic difficulty\n- Two subsets: middle (easier) and high (harder)\n- Tests various comprehension skills (inference, vocabulary, main idea, etc.)\n- Diverse article topics and question types\n\n## Evaluation Notes\n\n- Default configuration uses **3-shot** examples\n- Maximum few-shot number is 3 (context length consideration)\n- Uses Chain-of-Thought (CoT) prompting\n- Two subsets available: `high` and `middle`\n- Evaluates on test split\n",
6
+ "model_name": "dir-ppo-llama3.1-8b",
7
+ "score": 0.833,
8
+ "metrics": [
9
+ {
10
+ "name": "mean_acc",
11
+ "num": 4934,
12
+ "score": 0.833,
13
+ "macro_score": 0.833,
14
+ "categories": [
15
+ {
16
+ "name": [
17
+ "default"
18
+ ],
19
+ "num": 4934,
20
+ "score": 0.833,
21
+ "macro_score": 0.8433,
22
+ "subsets": [
23
+ {
24
+ "name": "high",
25
+ "score": 0.8188,
26
+ "num": 3498
27
+ },
28
+ {
29
+ "name": "middle",
30
+ "score": 0.8677,
31
+ "num": 1436
32
+ }
33
+ ]
34
+ }
35
+ ]
36
+ }
37
+ ],
38
+ "analysis": "N/A",
39
+ "perf_metrics": {
40
+ "summary": {
41
+ "n_samples": 4934,
42
+ "latency": {
43
+ "mean": 3.518324,
44
+ "std": 4.350995,
45
+ "min": 0.375683,
46
+ "25%": 2.235433,
47
+ "50%": 3.250558,
48
+ "75%": 4.360585,
49
+ "90%": 5.39817,
50
+ "99%": 7.618212,
51
+ "max": 139.063051
52
+ },
53
+ "throughput": {
54
+ "avg_output_tps": 51.33,
55
+ "avg_req_ps": 0.2842
56
+ },
57
+ "usage": {
58
+ "input_tokens": {
59
+ "mean": 1668.64471,
60
+ "std": 334.066106,
61
+ "min": 958.0,
62
+ "25%": 1274.0,
63
+ "50%": 1824.0,
64
+ "75%": 1885.0,
65
+ "90%": 1953.0,
66
+ "99%": 2269.67,
67
+ "max": 2584.0
68
+ },
69
+ "output_tokens": {
70
+ "mean": 180.605188,
71
+ "std": 216.808972,
72
+ "min": 20.0,
73
+ "25%": 114.0,
74
+ "50%": 168.0,
75
+ "75%": 228.0,
76
+ "90%": 274.0,
77
+ "99%": 387.01,
78
+ "max": 6932.0
79
+ },
80
+ "total_tokens": {
81
+ "mean": 1849.249899,
82
+ "std": 410.659869,
83
+ "min": 1028.0,
84
+ "25%": 1479.0,
85
+ "50%": 1975.0,
86
+ "75%": 2084.0,
87
+ "90%": 2180.0,
88
+ "99%": 2527.39,
89
+ "max": 8192.0
90
+ },
91
+ "total_input_tokens": 8233093,
92
+ "total_output_tokens": 891106,
93
+ "total_tokens_count": 9124199
94
+ }
95
+ }
96
+ },
97
+ "num": 4934
98
+ }
eval/20260527_000425/20260527_000425/reports/dir-ppo-llama3.1-8b/trivia_qa.json ADDED
@@ -0,0 +1,93 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "name": "dir-ppo-llama3.1-8b@trivia_qa",
3
+ "dataset_name": "trivia_qa",
4
+ "dataset_pretty_name": "TriviaQA",
5
+ "dataset_description": "\n## Overview\n\nTriviaQA is a large-scale reading comprehension dataset containing over 650K question-answer-evidence triples. Questions are collected from trivia enthusiast websites and paired with Wikipedia articles as evidence documents.\n\n## Task Description\n\n- **Task Type**: Reading Comprehension / Question Answering\n- **Input**: Question with Wikipedia context passage\n- **Output**: Answer extracted or generated from context\n- **Domain**: General knowledge trivia questions\n\n## Key Features\n\n- 650K+ question-answer-evidence triples\n- Questions written by trivia enthusiasts (naturally challenging)\n- Multiple valid answer aliases for flexible evaluation\n- Wikipedia articles provide evidence passages\n- Tests both reading comprehension and knowledge retrieval\n\n## Evaluation Notes\n\n- Default configuration uses **0-shot** evaluation\n- Uses the Wikipedia reading comprehension subset (rc.wikipedia)\n- Answers should follow the format: \"ANSWER: [ANSWER]\"\n- Supports inclusion-based matching for answer comparison\n- Evaluates on validation split\n",
6
+ "model_name": "dir-ppo-llama3.1-8b",
7
+ "score": 0.0407,
8
+ "metrics": [
9
+ {
10
+ "name": "mean_acc",
11
+ "num": 3689,
12
+ "score": 0.0407,
13
+ "macro_score": 0.0407,
14
+ "categories": [
15
+ {
16
+ "name": [
17
+ "default"
18
+ ],
19
+ "num": 3689,
20
+ "score": 0.0407,
21
+ "macro_score": 0.0407,
22
+ "subsets": [
23
+ {
24
+ "name": "rc.wikipedia",
25
+ "score": 0.0407,
26
+ "num": 3689
27
+ }
28
+ ]
29
+ }
30
+ ]
31
+ }
32
+ ],
33
+ "analysis": "N/A",
34
+ "perf_metrics": {
35
+ "summary": {
36
+ "n_samples": 3689,
37
+ "latency": {
38
+ "mean": 1.375701,
39
+ "std": 4.526989,
40
+ "min": 0.115312,
41
+ "25%": 0.658521,
42
+ "50%": 0.914985,
43
+ "75%": 1.307014,
44
+ "90%": 1.959365,
45
+ "99%": 5.197951,
46
+ "max": 113.092732
47
+ },
48
+ "throughput": {
49
+ "avg_output_tps": 42.44,
50
+ "avg_req_ps": 0.7269
51
+ },
52
+ "usage": {
53
+ "input_tokens": {
54
+ "mean": 3750.501762,
55
+ "std": 2268.520724,
56
+ "min": 116.0,
57
+ "25%": 1732.0,
58
+ "50%": 3504.0,
59
+ "75%": 5585.0,
60
+ "90%": 7181.6,
61
+ "99%": 8087.0,
62
+ "max": 8190.0
63
+ },
64
+ "output_tokens": {
65
+ "mean": 58.387097,
66
+ "std": 291.002579,
67
+ "min": 2.0,
68
+ "25%": 21.0,
69
+ "50%": 30.0,
70
+ "75%": 45.0,
71
+ "90%": 83.0,
72
+ "99%": 286.36,
73
+ "max": 7337.0
74
+ },
75
+ "total_tokens": {
76
+ "mean": 3808.888859,
77
+ "std": 2283.354375,
78
+ "min": 134.0,
79
+ "25%": 1788.0,
80
+ "50%": 3565.0,
81
+ "75%": 5643.0,
82
+ "90%": 7250.4,
83
+ "99%": 8175.48,
84
+ "max": 8192.0
85
+ },
86
+ "total_input_tokens": 13835601,
87
+ "total_output_tokens": 215390,
88
+ "total_tokens_count": 14050991
89
+ }
90
+ }
91
+ },
92
+ "num": 3689
93
+ }
eval/20260527_000425/20260527_000425/reports/report.html ADDED
The diff for this file is too large to render. See raw diff
 
eval/20260527_000425/20260527_000425/reviews/dir-ppo-llama3.1-8b/humaneval_openai_humaneval.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
eval/20260527_000425/20260527_000425/reviews/dir-ppo-llama3.1-8b/ifeval_default.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
eval/20260527_000425/20260527_000425/reviews/dir-ppo-llama3.1-8b/race_high.jsonl ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9bca1ce5a3af7d111c3e98c1d4268a70b41e6a5b06d9d43dd98169389cf01a46
3
+ size 38411680
eval/20260527_000425/20260527_000425/reviews/dir-ppo-llama3.1-8b/race_middle.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
eval/20260527_000425/20260527_000425/reviews/dir-ppo-llama3.1-8b/trivia_qa_rc.wikipedia.jsonl ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b7a4814f193d0768bd1440bb673f997cc476dd2ff24d0e31317483a2f69ae12e
3
+ size 124882928
eval/20260527_000425/configs/task_config.yaml ADDED
@@ -0,0 +1,381 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ agent_config: null
2
+ analysis_report: false
3
+ api_url: http://127.0.0.1:8801/v1
4
+ chat_template: null
5
+ collect_perf: true
6
+ dataset_args:
7
+ humaneval:
8
+ aggregation: mean_and_pass_at_k
9
+ data_statistics: null
10
+ dataset_id: opencompass/humaneval
11
+ default_subset: default
12
+ description: '
13
+
14
+ ## Overview
15
+
16
+
17
+ HumanEval is a benchmark for evaluating the code generation capabilities of
18
+ language models. It consists of 164 hand-written Python programming problems
19
+ with function signatures, docstrings, and comprehensive test cases.
20
+
21
+
22
+ ## Task Description
23
+
24
+
25
+ - **Task Type**: Code Generation (Python)
26
+
27
+ - **Input**: Function signature with docstring describing the expected behavior
28
+
29
+ - **Output**: Complete Python function implementation
30
+
31
+ - **Languages**: Python only
32
+
33
+
34
+ ## Key Features
35
+
36
+
37
+ - 164 hand-crafted programming problems
38
+
39
+ - Each problem includes a function signature, docstring, and test cases
40
+
41
+ - Problems range from simple string manipulation to complex algorithms
42
+
43
+ - Canonical solutions provided for reference
44
+
45
+ - Automatic correctness verification through test execution
46
+
47
+
48
+ ## Evaluation Notes
49
+
50
+
51
+ - **Security Warning**: By default, code is executed in the local environment.
52
+ We strongly recommend using sandbox execution for safety. See the [sandbox documentation](https://evalscope.readthedocs.io/en/latest/user_guides/sandbox.html)
53
+ for details.
54
+
55
+ - Supports `pass@k` metric calculation for measuring generation quality
56
+
57
+ - Default timeout is 4 seconds per problem
58
+
59
+ - Code is extracted from markdown code blocks if present
60
+
61
+ '
62
+ eval_split: test
63
+ extra_params: {}
64
+ few_shot_num: 0
65
+ few_shot_prompt_template: null
66
+ few_shot_random: false
67
+ filters: null
68
+ force_redownload: false
69
+ metric_list:
70
+ - acc
71
+ name: humaneval
72
+ output_types:
73
+ - generation
74
+ paper_url: null
75
+ pretty_name: HumanEval
76
+ prompt_template: 'Read the following function signature and docstring, and fully
77
+ implement the function described. Your response should only contain the code
78
+ for this function.
79
+
80
+ {question}'
81
+ query_template: null
82
+ review_timeout: 4
83
+ sample_example: null
84
+ sandbox_config:
85
+ image: python:3.11-slim
86
+ tools_config:
87
+ python_executor: {}
88
+ shell_executor: {}
89
+ shuffle: false
90
+ shuffle_choices: false
91
+ subset_list:
92
+ - openai_humaneval
93
+ system_prompt: null
94
+ tags:
95
+ - Coding
96
+ train_split: null
97
+ ifeval:
98
+ aggregation: mean
99
+ data_statistics: null
100
+ dataset_id: opencompass/ifeval
101
+ default_subset: default
102
+ description: "\n## Overview\n\nIFEval (Instruction-Following Eval) is a benchmark\
103
+ \ for evaluating how well language models follow explicit, verifiable instructions.\
104
+ \ It contains prompts with specific formatting, content, or structural requirements\
105
+ \ that can be objectively verified.\n\n## Task Description\n\n- **Task Type**:\
106
+ \ Instruction Following Evaluation\n- **Input**: Prompts with explicit, verifiable\
107
+ \ constraints\n- **Output**: Response that follows all specified instructions\n\
108
+ - **Constraint Types**: Format, length, keywords, structure, etc.\n\n## Key\
109
+ \ Features\n\n- ~500 prompts with 25 types of verifiable instructions\n- Instructions\
110
+ \ are objectively checkable (not subjective)\n- Examples: \"write exactly 3\
111
+ \ paragraphs\", \"include the word X\", \"use bullet points\"\n- Tests instruction\
112
+ \ comprehension and compliance\n- No ambiguity in evaluation criteria\n\n##\
113
+ \ Evaluation Notes\n\n- Default configuration uses **0-shot** evaluation\n-\
114
+ \ Four metrics available:\n - `prompt_level_strict`: All instructions in prompt\
115
+ \ must be followed\n - `prompt_level_loose`: Some tolerance for minor deviations\n\
116
+ \ - `inst_level_strict`: Per-instruction accuracy (strict)\n - `inst_level_loose`:\
117
+ \ Per-instruction accuracy (loose)\n- `prompt_level_strict` is the primary metric\n\
118
+ - Automatic verification of instruction compliance\n"
119
+ eval_split: train
120
+ extra_params: {}
121
+ few_shot_num: 0
122
+ few_shot_prompt_template: null
123
+ few_shot_random: false
124
+ filters: null
125
+ force_redownload: false
126
+ metric_list:
127
+ - prompt_level_strict
128
+ - inst_level_strict
129
+ - prompt_level_loose
130
+ - inst_level_loose
131
+ name: ifeval
132
+ output_types:
133
+ - generation
134
+ paper_url: null
135
+ pretty_name: IFEval
136
+ prompt_template: ''
137
+ query_template: null
138
+ review_timeout: null
139
+ sample_example: null
140
+ sandbox_config: {}
141
+ shuffle: false
142
+ shuffle_choices: false
143
+ subset_list:
144
+ - default
145
+ system_prompt: null
146
+ tags:
147
+ - InstructionFollowing
148
+ train_split: null
149
+ race:
150
+ aggregation: mean
151
+ data_statistics: null
152
+ dataset_id: evalscope/race
153
+ default_subset: default
154
+ description: '
155
+
156
+ ## Overview
157
+
158
+
159
+ RACE (ReAding Comprehension from Examinations) is a large-scale reading comprehension
160
+ benchmark collected from Chinese middle school and high school English examinations.
161
+ It tests comprehensive reading comprehension abilities.
162
+
163
+
164
+ ## Task Description
165
+
166
+
167
+ - **Task Type**: Reading Comprehension (Multiple-Choice)
168
+
169
+ - **Input**: Article passage with question and 4 answer choices
170
+
171
+ - **Output**: Correct answer letter (A, B, C, or D)
172
+
173
+ - **Difficulty Levels**: Middle school and High school
174
+
175
+
176
+ ## Key Features
177
+
178
+
179
+ - 28,000+ passages with 100,000 questions
180
+
181
+ - Real examination questions for authentic difficulty
182
+
183
+ - Two subsets: middle (easier) and high (harder)
184
+
185
+ - Tests various comprehension skills (inference, vocabulary, main idea, etc.)
186
+
187
+ - Diverse article topics and question types
188
+
189
+
190
+ ## Evaluation Notes
191
+
192
+
193
+ - Default configuration uses **3-shot** examples
194
+
195
+ - Maximum few-shot number is 3 (context length consideration)
196
+
197
+ - Uses Chain-of-Thought (CoT) prompting
198
+
199
+ - Two subsets available: `high` and `middle`
200
+
201
+ - Evaluates on test split
202
+
203
+ '
204
+ eval_split: test
205
+ extra_params: {}
206
+ few_shot_num: 3
207
+ few_shot_prompt_template: null
208
+ few_shot_random: false
209
+ filters: null
210
+ force_redownload: false
211
+ metric_list:
212
+ - acc
213
+ name: race
214
+ output_types:
215
+ - generation
216
+ paper_url: null
217
+ pretty_name: RACE
218
+ prompt_template: 'Answer the following multiple choice question. The last line
219
+ of your response should be of the following format: ''ANSWER: [LETTER]'' (without
220
+ quotes) where [LETTER] is one of {letters}. Think step by step before answering.
221
+
222
+
223
+ {question}
224
+
225
+
226
+ {choices}'
227
+ query_template: null
228
+ review_timeout: null
229
+ sample_example: null
230
+ sandbox_config: {}
231
+ shuffle: false
232
+ shuffle_choices: false
233
+ subset_list:
234
+ - high
235
+ - middle
236
+ system_prompt: null
237
+ tags:
238
+ - Reasoning
239
+ - MCQ
240
+ train_split: train
241
+ trivia_qa:
242
+ aggregation: mean
243
+ data_statistics: null
244
+ dataset_id: evalscope/trivia_qa
245
+ default_subset: default
246
+ description: '
247
+
248
+ ## Overview
249
+
250
+
251
+ TriviaQA is a large-scale reading comprehension dataset containing over 650K
252
+ question-answer-evidence triples. Questions are collected from trivia enthusiast
253
+ websites and paired with Wikipedia articles as evidence documents.
254
+
255
+
256
+ ## Task Description
257
+
258
+
259
+ - **Task Type**: Reading Comprehension / Question Answering
260
+
261
+ - **Input**: Question with Wikipedia context passage
262
+
263
+ - **Output**: Answer extracted or generated from context
264
+
265
+ - **Domain**: General knowledge trivia questions
266
+
267
+
268
+ ## Key Features
269
+
270
+
271
+ - 650K+ question-answer-evidence triples
272
+
273
+ - Questions written by trivia enthusiasts (naturally challenging)
274
+
275
+ - Multiple valid answer aliases for flexible evaluation
276
+
277
+ - Wikipedia articles provide evidence passages
278
+
279
+ - Tests both reading comprehension and knowledge retrieval
280
+
281
+
282
+ ## Evaluation Notes
283
+
284
+
285
+ - Default configuration uses **0-shot** evaluation
286
+
287
+ - Uses the Wikipedia reading comprehension subset (rc.wikipedia)
288
+
289
+ - Answers should follow the format: "ANSWER: [ANSWER]"
290
+
291
+ - Supports inclusion-based matching for answer comparison
292
+
293
+ - Evaluates on validation split
294
+
295
+ '
296
+ eval_split: validation
297
+ extra_params: {}
298
+ few_shot_num: 0
299
+ few_shot_prompt_template: null
300
+ few_shot_random: false
301
+ filters: null
302
+ force_redownload: false
303
+ metric_list:
304
+ - acc:
305
+ allow_inclusion: true
306
+ name: trivia_qa
307
+ output_types:
308
+ - generation
309
+ paper_url: null
310
+ pretty_name: TriviaQA
311
+ prompt_template: 'Read the content and answer the following question.
312
+
313
+
314
+ Content: {content}
315
+
316
+
317
+ Question: {question}
318
+
319
+
320
+ Keep your The last line of your response should be of the form "ANSWER: [ANSWER]"
321
+ (without quotes) where [ANSWER] is the answer to the problem.
322
+
323
+ '
324
+ query_template: null
325
+ review_timeout: null
326
+ sample_example: null
327
+ sandbox_config: {}
328
+ shuffle: false
329
+ shuffle_choices: false
330
+ subset_list:
331
+ - rc.wikipedia
332
+ system_prompt: null
333
+ tags:
334
+ - QA
335
+ - ReadingComprehension
336
+ train_split: null
337
+ dataset_dir: /root/.cache/modelscope/hub/datasets
338
+ dataset_hub: modelscope
339
+ datasets:
340
+ - ifeval
341
+ - race
342
+ - trivia_qa
343
+ - humaneval
344
+ debug: false
345
+ enable_progress_tracker: false
346
+ eval_backend: Native
347
+ eval_batch_size: 16
348
+ eval_config: null
349
+ eval_type: server
350
+ evalscope_version: 1.7.1
351
+ generation_config:
352
+ batch_size: 16
353
+ do_sample: true
354
+ max_new_tokens: 2048
355
+ temperature: 0.7
356
+ ignore_errors: true
357
+ judge_model_args: {}
358
+ judge_strategy: auto
359
+ judge_worker_num: 1
360
+ limit: null
361
+ model: dir-ppo-llama3.1-8b
362
+ model_args: {}
363
+ model_id: dir-ppo-llama3.1-8b
364
+ model_task: text_generation
365
+ no_timestamp: false
366
+ repeats: 1
367
+ rerun_review: false
368
+ sandbox:
369
+ default_config: {}
370
+ enabled: false
371
+ engine: docker
372
+ manager_config: {}
373
+ pool_size: null
374
+ sandbox_manager_config: {}
375
+ sandbox_type: docker
376
+ seed: 42
377
+ stream: null
378
+ timeout: null
379
+ use_cache: null
380
+ use_sandbox: false
381
+ work_dir: ./outputs/20260527_000425
eval/20260527_000425/logs/eval_log.log ADDED
The diff for this file is too large to render. See raw diff
 
eval/20260527_000425/predictions/dir-ppo-llama3.1-8b/ifeval_default.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
eval/20260527_000425/predictions/dir-ppo-llama3.1-8b/race_high.jsonl ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d595bed6f072c3ef9ed0e19010e34b124818146eb724a04531238384b16fe08e
3
+ size 40497772
eval/20260527_000425/predictions/dir-ppo-llama3.1-8b/race_middle.jsonl ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b0ea7aff3ae009236736a2c2ca8c02a74eedf3b6f87b4091de2918a67bca719c
3
+ size 10885732
eval/20260527_000425/predictions/dir-ppo-llama3.1-8b/trivia_qa_rc.wikipedia.jsonl ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:64d916009fd65f0108bed676b6b0d28785cae1a41b955cf7e466df9454a28f69
3
+ size 46729079
eval/20260527_000425/reports/dir-ppo-llama3.1-8b/ifeval.json ADDED
@@ -0,0 +1,162 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "name": "dir-ppo-llama3.1-8b@ifeval",
3
+ "dataset_name": "ifeval",
4
+ "dataset_pretty_name": "IFEval",
5
+ "dataset_description": "\n## Overview\n\nIFEval (Instruction-Following Eval) is a benchmark for evaluating how well language models follow explicit, verifiable instructions. It contains prompts with specific formatting, content, or structural requirements that can be objectively verified.\n\n## Task Description\n\n- **Task Type**: Instruction Following Evaluation\n- **Input**: Prompts with explicit, verifiable constraints\n- **Output**: Response that follows all specified instructions\n- **Constraint Types**: Format, length, keywords, structure, etc.\n\n## Key Features\n\n- ~500 prompts with 25 types of verifiable instructions\n- Instructions are objectively checkable (not subjective)\n- Examples: \"write exactly 3 paragraphs\", \"include the word X\", \"use bullet points\"\n- Tests instruction comprehension and compliance\n- No ambiguity in evaluation criteria\n\n## Evaluation Notes\n\n- Default configuration uses **0-shot** evaluation\n- Four metrics available:\n - `prompt_level_strict`: All instructions in prompt must be followed\n - `prompt_level_loose`: Some tolerance for minor deviations\n - `inst_level_strict`: Per-instruction accuracy (strict)\n - `inst_level_loose`: Per-instruction accuracy (loose)\n- `prompt_level_strict` is the primary metric\n- Automatic verification of instruction compliance\n",
6
+ "model_name": "dir-ppo-llama3.1-8b",
7
+ "score": 0.6981,
8
+ "metrics": [
9
+ {
10
+ "name": "mean_prompt_level_strict",
11
+ "num": 540,
12
+ "score": 0.6981,
13
+ "macro_score": 0.6981,
14
+ "categories": [
15
+ {
16
+ "name": [
17
+ "default"
18
+ ],
19
+ "num": 540,
20
+ "score": 0.6981,
21
+ "macro_score": 0.6981,
22
+ "subsets": [
23
+ {
24
+ "name": "default",
25
+ "score": 0.6981,
26
+ "num": 540
27
+ }
28
+ ]
29
+ }
30
+ ]
31
+ },
32
+ {
33
+ "name": "mean_inst_level_strict",
34
+ "num": 540,
35
+ "score": 0.7901,
36
+ "macro_score": 0.7901,
37
+ "categories": [
38
+ {
39
+ "name": [
40
+ "default"
41
+ ],
42
+ "num": 540,
43
+ "score": 0.7901,
44
+ "macro_score": 0.7901,
45
+ "subsets": [
46
+ {
47
+ "name": "default",
48
+ "score": 0.7901,
49
+ "num": 540
50
+ }
51
+ ]
52
+ }
53
+ ]
54
+ },
55
+ {
56
+ "name": "mean_prompt_level_loose",
57
+ "num": 540,
58
+ "score": 0.787,
59
+ "macro_score": 0.787,
60
+ "categories": [
61
+ {
62
+ "name": [
63
+ "default"
64
+ ],
65
+ "num": 540,
66
+ "score": 0.787,
67
+ "macro_score": 0.787,
68
+ "subsets": [
69
+ {
70
+ "name": "default",
71
+ "score": 0.787,
72
+ "num": 540
73
+ }
74
+ ]
75
+ }
76
+ ]
77
+ },
78
+ {
79
+ "name": "mean_inst_level_loose",
80
+ "num": 540,
81
+ "score": 0.8623,
82
+ "macro_score": 0.8623,
83
+ "categories": [
84
+ {
85
+ "name": [
86
+ "default"
87
+ ],
88
+ "num": 540,
89
+ "score": 0.8623,
90
+ "macro_score": 0.8623,
91
+ "subsets": [
92
+ {
93
+ "name": "default",
94
+ "score": 0.8623,
95
+ "num": 540
96
+ }
97
+ ]
98
+ }
99
+ ]
100
+ }
101
+ ],
102
+ "analysis": "N/A",
103
+ "perf_metrics": {
104
+ "summary": {
105
+ "n_samples": 541,
106
+ "latency": {
107
+ "mean": 7.710747,
108
+ "std": 16.563807,
109
+ "min": 0.066271,
110
+ "25%": 2.457928,
111
+ "50%": 5.100689,
112
+ "75%": 8.237214,
113
+ "90%": 12.235768,
114
+ "99%": 107.019331,
115
+ "max": 160.99844
116
+ },
117
+ "throughput": {
118
+ "avg_output_tps": 52.84,
119
+ "avg_req_ps": 0.1297
120
+ },
121
+ "usage": {
122
+ "input_tokens": {
123
+ "mean": 80.924214,
124
+ "std": 23.92573,
125
+ "min": 48.0,
126
+ "25%": 67.0,
127
+ "50%": 76.0,
128
+ "75%": 90.0,
129
+ "90%": 107.0,
130
+ "99%": 135.6,
131
+ "max": 395.0
132
+ },
133
+ "output_tokens": {
134
+ "mean": 407.428835,
135
+ "std": 862.651879,
136
+ "min": 3.0,
137
+ "25%": 127.0,
138
+ "50%": 265.0,
139
+ "75%": 436.0,
140
+ "90%": 642.0,
141
+ "99%": 5846.8,
142
+ "max": 8138.0
143
+ },
144
+ "total_tokens": {
145
+ "mean": 488.35305,
146
+ "std": 861.312936,
147
+ "min": 69.0,
148
+ "25%": 212.0,
149
+ "50%": 346.0,
150
+ "75%": 521.0,
151
+ "90%": 725.0,
152
+ "99%": 5930.8,
153
+ "max": 8192.0
154
+ },
155
+ "total_input_tokens": 43780,
156
+ "total_output_tokens": 220419,
157
+ "total_tokens_count": 264199
158
+ }
159
+ }
160
+ },
161
+ "num": 540
162
+ }
eval/20260527_000425/reports/dir-ppo-llama3.1-8b/race.json ADDED
@@ -0,0 +1,98 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "name": "dir-ppo-llama3.1-8b@race",
3
+ "dataset_name": "race",
4
+ "dataset_pretty_name": "RACE",
5
+ "dataset_description": "\n## Overview\n\nRACE (ReAding Comprehension from Examinations) is a large-scale reading comprehension benchmark collected from Chinese middle school and high school English examinations. It tests comprehensive reading comprehension abilities.\n\n## Task Description\n\n- **Task Type**: Reading Comprehension (Multiple-Choice)\n- **Input**: Article passage with question and 4 answer choices\n- **Output**: Correct answer letter (A, B, C, or D)\n- **Difficulty Levels**: Middle school and High school\n\n## Key Features\n\n- 28,000+ passages with 100,000 questions\n- Real examination questions for authentic difficulty\n- Two subsets: middle (easier) and high (harder)\n- Tests various comprehension skills (inference, vocabulary, main idea, etc.)\n- Diverse article topics and question types\n\n## Evaluation Notes\n\n- Default configuration uses **3-shot** examples\n- Maximum few-shot number is 3 (context length consideration)\n- Uses Chain-of-Thought (CoT) prompting\n- Two subsets available: `high` and `middle`\n- Evaluates on test split\n",
6
+ "model_name": "dir-ppo-llama3.1-8b",
7
+ "score": 0.833,
8
+ "metrics": [
9
+ {
10
+ "name": "mean_acc",
11
+ "num": 4934,
12
+ "score": 0.833,
13
+ "macro_score": 0.833,
14
+ "categories": [
15
+ {
16
+ "name": [
17
+ "default"
18
+ ],
19
+ "num": 4934,
20
+ "score": 0.833,
21
+ "macro_score": 0.8433,
22
+ "subsets": [
23
+ {
24
+ "name": "high",
25
+ "score": 0.8188,
26
+ "num": 3498
27
+ },
28
+ {
29
+ "name": "middle",
30
+ "score": 0.8677,
31
+ "num": 1436
32
+ }
33
+ ]
34
+ }
35
+ ]
36
+ }
37
+ ],
38
+ "analysis": "N/A",
39
+ "perf_metrics": {
40
+ "summary": {
41
+ "n_samples": 4934,
42
+ "latency": {
43
+ "mean": 3.518324,
44
+ "std": 4.350995,
45
+ "min": 0.375683,
46
+ "25%": 2.235433,
47
+ "50%": 3.250558,
48
+ "75%": 4.360585,
49
+ "90%": 5.39817,
50
+ "99%": 7.618212,
51
+ "max": 139.063051
52
+ },
53
+ "throughput": {
54
+ "avg_output_tps": 51.33,
55
+ "avg_req_ps": 0.2842
56
+ },
57
+ "usage": {
58
+ "input_tokens": {
59
+ "mean": 1668.64471,
60
+ "std": 334.066106,
61
+ "min": 958.0,
62
+ "25%": 1274.0,
63
+ "50%": 1824.0,
64
+ "75%": 1885.0,
65
+ "90%": 1953.0,
66
+ "99%": 2269.67,
67
+ "max": 2584.0
68
+ },
69
+ "output_tokens": {
70
+ "mean": 180.605188,
71
+ "std": 216.808972,
72
+ "min": 20.0,
73
+ "25%": 114.0,
74
+ "50%": 168.0,
75
+ "75%": 228.0,
76
+ "90%": 274.0,
77
+ "99%": 387.01,
78
+ "max": 6932.0
79
+ },
80
+ "total_tokens": {
81
+ "mean": 1849.249899,
82
+ "std": 410.659869,
83
+ "min": 1028.0,
84
+ "25%": 1479.0,
85
+ "50%": 1975.0,
86
+ "75%": 2084.0,
87
+ "90%": 2180.0,
88
+ "99%": 2527.39,
89
+ "max": 8192.0
90
+ },
91
+ "total_input_tokens": 8233093,
92
+ "total_output_tokens": 891106,
93
+ "total_tokens_count": 9124199
94
+ }
95
+ }
96
+ },
97
+ "num": 4934
98
+ }
eval/20260527_000425/reviews/dir-ppo-llama3.1-8b/ifeval_default.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
eval/20260527_000425/reviews/dir-ppo-llama3.1-8b/race_high.jsonl ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9bca1ce5a3af7d111c3e98c1d4268a70b41e6a5b06d9d43dd98169389cf01a46
3
+ size 38411680
eval/20260527_000425/reviews/dir-ppo-llama3.1-8b/race_middle.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
eval/20260527_000425/reviews/dir-ppo-llama3.1-8b/trivia_qa_rc.wikipedia.jsonl ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:fb259e7f4c082c3794ef1c4b4671d1bf7b24a4bee17839a9e079287f132d033d
3
+ size 45028591