Commit
·
aa17c3f
1
Parent(s):
1c82b01
Remove eval folder (Moved to bigscience/evaluation-results)
Browse filesThis view is limited to 50 files because it contains too many changes.
See raw diff
- evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/ar/Answer_Given_options/results.json +0 -9
- evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/ar/Choose_Story_Ending/results.json +0 -9
- evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/ar/Generate_Ending/results.json +0 -9
- evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/ar/Novel_Correct_Ending/results.json +0 -9
- evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/ar/Story_Continuation_and_Options/results.json +0 -9
- evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/es/Answer_Given_options/results.json +0 -9
- evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/es/Choose_Story_Ending/results.json +0 -9
- evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/es/Generate_Ending/results.json +0 -9
- evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/es/Novel_Correct_Ending/results.json +0 -9
- evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/es/Story_Continuation_and_Options/results.json +0 -9
- evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/eu/Answer_Given_options/results.json +0 -9
- evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/eu/Choose_Story_Ending/results.json +0 -9
- evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/eu/Generate_Ending/results.json +0 -9
- evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/eu/Novel_Correct_Ending/results.json +0 -9
- evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/eu/Story_Continuation_and_Options/results.json +0 -9
- evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/hi/Answer_Given_options/results.json +0 -9
- evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/hi/Choose_Story_Ending/results.json +0 -9
- evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/hi/Generate_Ending/results.json +0 -9
- evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/hi/Novel_Correct_Ending/results.json +0 -9
- evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/hi/Story_Continuation_and_Options/results.json +0 -9
- evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/id/Answer_Given_options/results.json +0 -9
- evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/id/Choose_Story_Ending/results.json +0 -9
- evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/id/Generate_Ending/results.json +0 -9
- evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/id/Novel_Correct_Ending/results.json +0 -9
- evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/id/Story_Continuation_and_Options/results.json +0 -9
- evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/zh/Answer_Given_options/results.json +0 -9
- evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/zh/Choose_Story_Ending/results.json +0 -9
- evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/zh/Generate_Ending/results.json +0 -9
- evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/zh/Novel_Correct_Ending/results.json +0 -9
- evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/zh/Story_Continuation_and_Options/results.json +0 -9
- evaluation_bloomz-3b/evaluation_l1/Muennighoff_xwinograd/en/Replace/results.json +0 -9
- evaluation_bloomz-3b/evaluation_l1/Muennighoff_xwinograd/en/True_or_False/results.json +0 -9
- evaluation_bloomz-3b/evaluation_l1/Muennighoff_xwinograd/en/does_underscore_refer_to/results.json +0 -9
- evaluation_bloomz-3b/evaluation_l1/Muennighoff_xwinograd/en/stand_for/results.json +0 -9
- evaluation_bloomz-3b/evaluation_l1/Muennighoff_xwinograd/en/underscore_refer_to/results.json +0 -9
- evaluation_bloomz-3b/evaluation_l1/Muennighoff_xwinograd/fr/Replace/results.json +0 -9
- evaluation_bloomz-3b/evaluation_l1/Muennighoff_xwinograd/fr/True_or_False/results.json +0 -9
- evaluation_bloomz-3b/evaluation_l1/Muennighoff_xwinograd/fr/does_underscore_refer_to/results.json +0 -9
- evaluation_bloomz-3b/evaluation_l1/Muennighoff_xwinograd/fr/stand_for/results.json +0 -9
- evaluation_bloomz-3b/evaluation_l1/Muennighoff_xwinograd/fr/underscore_refer_to/results.json +0 -9
- evaluation_bloomz-3b/evaluation_l1/Muennighoff_xwinograd/pt/Replace/results.json +0 -9
- evaluation_bloomz-3b/evaluation_l1/Muennighoff_xwinograd/pt/True_or_False/results.json +0 -9
- evaluation_bloomz-3b/evaluation_l1/Muennighoff_xwinograd/pt/does_underscore_refer_to/results.json +0 -9
- evaluation_bloomz-3b/evaluation_l1/Muennighoff_xwinograd/pt/stand_for/results.json +0 -9
- evaluation_bloomz-3b/evaluation_l1/Muennighoff_xwinograd/pt/underscore_refer_to/results.json +0 -9
- evaluation_bloomz-3b/evaluation_l1/Muennighoff_xwinograd/zh/Replace/results.json +0 -9
- evaluation_bloomz-3b/evaluation_l1/Muennighoff_xwinograd/zh/True_or_False/results.json +0 -9
- evaluation_bloomz-3b/evaluation_l1/Muennighoff_xwinograd/zh/does_underscore_refer_to/results.json +0 -9
- evaluation_bloomz-3b/evaluation_l1/Muennighoff_xwinograd/zh/stand_for/results.json +0 -9
- evaluation_bloomz-3b/evaluation_l1/Muennighoff_xwinograd/zh/underscore_refer_to/results.json +0 -9
evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/ar/Answer_Given_options/results.json
DELETED
|
@@ -1,9 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
-
"dataset_config_name": "ar",
|
| 4 |
-
"template_name": "Answer Given options",
|
| 5 |
-
"evaluation": {
|
| 6 |
-
"accuracy": 0.7511581733951026
|
| 7 |
-
},
|
| 8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='ar', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Answer Given options', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/ar/Choose_Story_Ending/results.json
DELETED
|
@@ -1,9 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
-
"dataset_config_name": "ar",
|
| 4 |
-
"template_name": "Choose Story Ending",
|
| 5 |
-
"evaluation": {
|
| 6 |
-
"accuracy": 0.8060886829913965
|
| 7 |
-
},
|
| 8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='ar', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Choose Story Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/ar/Generate_Ending/results.json
DELETED
|
@@ -1,9 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
-
"dataset_config_name": "ar",
|
| 4 |
-
"template_name": "Generate Ending",
|
| 5 |
-
"evaluation": {
|
| 6 |
-
"accuracy": 0.5684976836532097
|
| 7 |
-
},
|
| 8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='ar', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Generate Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/ar/Novel_Correct_Ending/results.json
DELETED
|
@@ -1,9 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
-
"dataset_config_name": "ar",
|
| 4 |
-
"template_name": "Novel Correct Ending",
|
| 5 |
-
"evaluation": {
|
| 6 |
-
"accuracy": 0.7822634017207147
|
| 7 |
-
},
|
| 8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='ar', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Novel Correct Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/ar/Story_Continuation_and_Options/results.json
DELETED
|
@@ -1,9 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
-
"dataset_config_name": "ar",
|
| 4 |
-
"template_name": "Story Continuation and Options",
|
| 5 |
-
"evaluation": {
|
| 6 |
-
"accuracy": 0.7908669755129054
|
| 7 |
-
},
|
| 8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='ar', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Story Continuation and Options', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/es/Answer_Given_options/results.json
DELETED
|
@@ -1,9 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
-
"dataset_config_name": "es",
|
| 4 |
-
"template_name": "Answer Given options",
|
| 5 |
-
"evaluation": {
|
| 6 |
-
"accuracy": 0.7498345466578424
|
| 7 |
-
},
|
| 8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='es', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Answer Given options', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/es/Choose_Story_Ending/results.json
DELETED
|
@@ -1,9 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
-
"dataset_config_name": "es",
|
| 4 |
-
"template_name": "Choose Story Ending",
|
| 5 |
-
"evaluation": {
|
| 6 |
-
"accuracy": 0.8590337524818001
|
| 7 |
-
},
|
| 8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='es', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Choose Story Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/es/Generate_Ending/results.json
DELETED
|
@@ -1,9 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
-
"dataset_config_name": "es",
|
| 4 |
-
"template_name": "Generate Ending",
|
| 5 |
-
"evaluation": {
|
| 6 |
-
"accuracy": 0.6260754467240238
|
| 7 |
-
},
|
| 8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='es', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Generate Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/es/Novel_Correct_Ending/results.json
DELETED
|
@@ -1,9 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
-
"dataset_config_name": "es",
|
| 4 |
-
"template_name": "Novel Correct Ending",
|
| 5 |
-
"evaluation": {
|
| 6 |
-
"accuracy": 0.8166776968894772
|
| 7 |
-
},
|
| 8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='es', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Novel Correct Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/es/Story_Continuation_and_Options/results.json
DELETED
|
@@ -1,9 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
-
"dataset_config_name": "es",
|
| 4 |
-
"template_name": "Story Continuation and Options",
|
| 5 |
-
"evaluation": {
|
| 6 |
-
"accuracy": 0.8352084712111185
|
| 7 |
-
},
|
| 8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='es', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Story Continuation and Options', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/eu/Answer_Given_options/results.json
DELETED
|
@@ -1,9 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
-
"dataset_config_name": "eu",
|
| 4 |
-
"template_name": "Answer Given options",
|
| 5 |
-
"evaluation": {
|
| 6 |
-
"accuracy": 0.6029119788219722
|
| 7 |
-
},
|
| 8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='eu', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Answer Given options', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/eu/Choose_Story_Ending/results.json
DELETED
|
@@ -1,9 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
-
"dataset_config_name": "eu",
|
| 4 |
-
"template_name": "Choose Story Ending",
|
| 5 |
-
"evaluation": {
|
| 6 |
-
"accuracy": 0.7094639311714097
|
| 7 |
-
},
|
| 8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='eu', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Choose Story Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/eu/Generate_Ending/results.json
DELETED
|
@@ -1,9 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
-
"dataset_config_name": "eu",
|
| 4 |
-
"template_name": "Generate Ending",
|
| 5 |
-
"evaluation": {
|
| 6 |
-
"accuracy": 0.5314361350099271
|
| 7 |
-
},
|
| 8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='eu', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Generate Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/eu/Novel_Correct_Ending/results.json
DELETED
|
@@ -1,9 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
-
"dataset_config_name": "eu",
|
| 4 |
-
"template_name": "Novel Correct Ending",
|
| 5 |
-
"evaluation": {
|
| 6 |
-
"accuracy": 0.6743878226340172
|
| 7 |
-
},
|
| 8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='eu', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Novel Correct Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/eu/Story_Continuation_and_Options/results.json
DELETED
|
@@ -1,9 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
-
"dataset_config_name": "eu",
|
| 4 |
-
"template_name": "Story Continuation and Options",
|
| 5 |
-
"evaluation": {
|
| 6 |
-
"accuracy": 0.7074784910655195
|
| 7 |
-
},
|
| 8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='eu', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Story Continuation and Options', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/hi/Answer_Given_options/results.json
DELETED
|
@@ -1,9 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
-
"dataset_config_name": "hi",
|
| 4 |
-
"template_name": "Answer Given options",
|
| 5 |
-
"evaluation": {
|
| 6 |
-
"accuracy": 0.6836532097948379
|
| 7 |
-
},
|
| 8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='hi', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Answer Given options', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/hi/Choose_Story_Ending/results.json
DELETED
|
@@ -1,9 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
-
"dataset_config_name": "hi",
|
| 4 |
-
"template_name": "Choose Story Ending",
|
| 5 |
-
"evaluation": {
|
| 6 |
-
"accuracy": 0.7888815354070152
|
| 7 |
-
},
|
| 8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='hi', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Choose Story Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/hi/Generate_Ending/results.json
DELETED
|
@@ -1,9 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
-
"dataset_config_name": "hi",
|
| 4 |
-
"template_name": "Generate Ending",
|
| 5 |
-
"evaluation": {
|
| 6 |
-
"accuracy": 0.5810721376571807
|
| 7 |
-
},
|
| 8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='hi', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Generate Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/hi/Novel_Correct_Ending/results.json
DELETED
|
@@ -1,9 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
-
"dataset_config_name": "hi",
|
| 4 |
-
"template_name": "Novel Correct Ending",
|
| 5 |
-
"evaluation": {
|
| 6 |
-
"accuracy": 0.7597617471872932
|
| 7 |
-
},
|
| 8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='hi', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Novel Correct Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/hi/Story_Continuation_and_Options/results.json
DELETED
|
@@ -1,9 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
-
"dataset_config_name": "hi",
|
| 4 |
-
"template_name": "Story Continuation and Options",
|
| 5 |
-
"evaluation": {
|
| 6 |
-
"accuracy": 0.7723362011912641
|
| 7 |
-
},
|
| 8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='hi', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Story Continuation and Options', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/id/Answer_Given_options/results.json
DELETED
|
@@ -1,9 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
-
"dataset_config_name": "id",
|
| 4 |
-
"template_name": "Answer Given options",
|
| 5 |
-
"evaluation": {
|
| 6 |
-
"accuracy": 0.7127729980145598
|
| 7 |
-
},
|
| 8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='id', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Answer Given options', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/id/Choose_Story_Ending/results.json
DELETED
|
@@ -1,9 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
-
"dataset_config_name": "id",
|
| 4 |
-
"template_name": "Choose Story Ending",
|
| 5 |
-
"evaluation": {
|
| 6 |
-
"accuracy": 0.8299139642620781
|
| 7 |
-
},
|
| 8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='id', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Choose Story Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/id/Generate_Ending/results.json
DELETED
|
@@ -1,9 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
-
"dataset_config_name": "id",
|
| 4 |
-
"template_name": "Generate Ending",
|
| 5 |
-
"evaluation": {
|
| 6 |
-
"accuracy": 0.6035737921906023
|
| 7 |
-
},
|
| 8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='id', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Generate Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/id/Novel_Correct_Ending/results.json
DELETED
|
@@ -1,9 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
-
"dataset_config_name": "id",
|
| 4 |
-
"template_name": "Novel Correct Ending",
|
| 5 |
-
"evaluation": {
|
| 6 |
-
"accuracy": 0.7180675049636003
|
| 7 |
-
},
|
| 8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='id', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Novel Correct Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/id/Story_Continuation_and_Options/results.json
DELETED
|
@@ -1,9 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
-
"dataset_config_name": "id",
|
| 4 |
-
"template_name": "Story Continuation and Options",
|
| 5 |
-
"evaluation": {
|
| 6 |
-
"accuracy": 0.8021178027796162
|
| 7 |
-
},
|
| 8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='id', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Story Continuation and Options', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/zh/Answer_Given_options/results.json
DELETED
|
@@ -1,9 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
-
"dataset_config_name": "zh",
|
| 4 |
-
"template_name": "Answer Given options",
|
| 5 |
-
"evaluation": {
|
| 6 |
-
"accuracy": 0.786234281932495
|
| 7 |
-
},
|
| 8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Answer Given options', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/zh/Choose_Story_Ending/results.json
DELETED
|
@@ -1,9 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
-
"dataset_config_name": "zh",
|
| 4 |
-
"template_name": "Choose Story Ending",
|
| 5 |
-
"evaluation": {
|
| 6 |
-
"accuracy": 0.8431502316346791
|
| 7 |
-
},
|
| 8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Choose Story Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/zh/Generate_Ending/results.json
DELETED
|
@@ -1,9 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
-
"dataset_config_name": "zh",
|
| 4 |
-
"template_name": "Generate Ending",
|
| 5 |
-
"evaluation": {
|
| 6 |
-
"accuracy": 0.6022501654533422
|
| 7 |
-
},
|
| 8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Generate Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/zh/Novel_Correct_Ending/results.json
DELETED
|
@@ -1,9 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
-
"dataset_config_name": "zh",
|
| 4 |
-
"template_name": "Novel Correct Ending",
|
| 5 |
-
"evaluation": {
|
| 6 |
-
"accuracy": 0.8365320979483786
|
| 7 |
-
},
|
| 8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Novel Correct Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz-3b/evaluation_l1/Muennighoff_xstory_cloze/zh/Story_Continuation_and_Options/results.json
DELETED
|
@@ -1,9 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
-
"dataset_config_name": "zh",
|
| 4 |
-
"template_name": "Story Continuation and Options",
|
| 5 |
-
"evaluation": {
|
| 6 |
-
"accuracy": 0.8259430840502978
|
| 7 |
-
},
|
| 8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Story Continuation and Options', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz-3b/evaluation_l1/Muennighoff_xwinograd/en/Replace/results.json
DELETED
|
@@ -1,9 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"dataset_name": "Muennighoff/xwinograd",
|
| 3 |
-
"dataset_config_name": "en",
|
| 4 |
-
"template_name": "Replace",
|
| 5 |
-
"evaluation": {
|
| 6 |
-
"accuracy": 0.5922580645161291
|
| 7 |
-
},
|
| 8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='en', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='Replace', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz-3b/evaluation_l1/Muennighoff_xwinograd/en/True_or_False/results.json
DELETED
|
@@ -1,9 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"dataset_name": "Muennighoff/xwinograd",
|
| 3 |
-
"dataset_config_name": "en",
|
| 4 |
-
"template_name": "True or False",
|
| 5 |
-
"evaluation": {
|
| 6 |
-
"accuracy": 0.5333333333333333
|
| 7 |
-
},
|
| 8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='en', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='True or False', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz-3b/evaluation_l1/Muennighoff_xwinograd/en/does_underscore_refer_to/results.json
DELETED
|
@@ -1,9 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"dataset_name": "Muennighoff/xwinograd",
|
| 3 |
-
"dataset_config_name": "en",
|
| 4 |
-
"template_name": "does underscore refer to",
|
| 5 |
-
"evaluation": {
|
| 6 |
-
"accuracy": 0.5483870967741935
|
| 7 |
-
},
|
| 8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='en', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='does underscore refer to', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz-3b/evaluation_l1/Muennighoff_xwinograd/en/stand_for/results.json
DELETED
|
@@ -1,9 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"dataset_name": "Muennighoff/xwinograd",
|
| 3 |
-
"dataset_config_name": "en",
|
| 4 |
-
"template_name": "stand for",
|
| 5 |
-
"evaluation": {
|
| 6 |
-
"accuracy": 0.5006451612903225
|
| 7 |
-
},
|
| 8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='en', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='stand for', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz-3b/evaluation_l1/Muennighoff_xwinograd/en/underscore_refer_to/results.json
DELETED
|
@@ -1,9 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"dataset_name": "Muennighoff/xwinograd",
|
| 3 |
-
"dataset_config_name": "en",
|
| 4 |
-
"template_name": "underscore refer to",
|
| 5 |
-
"evaluation": {
|
| 6 |
-
"accuracy": 0.5574193548387096
|
| 7 |
-
},
|
| 8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='en', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='underscore refer to', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz-3b/evaluation_l1/Muennighoff_xwinograd/fr/Replace/results.json
DELETED
|
@@ -1,9 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"dataset_name": "Muennighoff/xwinograd",
|
| 3 |
-
"dataset_config_name": "fr",
|
| 4 |
-
"template_name": "Replace",
|
| 5 |
-
"evaluation": {
|
| 6 |
-
"accuracy": 0.4939759036144578
|
| 7 |
-
},
|
| 8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='fr', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='Replace', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz-3b/evaluation_l1/Muennighoff_xwinograd/fr/True_or_False/results.json
DELETED
|
@@ -1,9 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"dataset_name": "Muennighoff/xwinograd",
|
| 3 |
-
"dataset_config_name": "fr",
|
| 4 |
-
"template_name": "True or False",
|
| 5 |
-
"evaluation": {
|
| 6 |
-
"accuracy": 0.5060240963855421
|
| 7 |
-
},
|
| 8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='fr', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='True or False', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz-3b/evaluation_l1/Muennighoff_xwinograd/fr/does_underscore_refer_to/results.json
DELETED
|
@@ -1,9 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"dataset_name": "Muennighoff/xwinograd",
|
| 3 |
-
"dataset_config_name": "fr",
|
| 4 |
-
"template_name": "does underscore refer to",
|
| 5 |
-
"evaluation": {
|
| 6 |
-
"accuracy": 0.4819277108433735
|
| 7 |
-
},
|
| 8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='fr', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='does underscore refer to', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz-3b/evaluation_l1/Muennighoff_xwinograd/fr/stand_for/results.json
DELETED
|
@@ -1,9 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"dataset_name": "Muennighoff/xwinograd",
|
| 3 |
-
"dataset_config_name": "fr",
|
| 4 |
-
"template_name": "stand for",
|
| 5 |
-
"evaluation": {
|
| 6 |
-
"accuracy": 0.4939759036144578
|
| 7 |
-
},
|
| 8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='fr', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='stand for', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz-3b/evaluation_l1/Muennighoff_xwinograd/fr/underscore_refer_to/results.json
DELETED
|
@@ -1,9 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"dataset_name": "Muennighoff/xwinograd",
|
| 3 |
-
"dataset_config_name": "fr",
|
| 4 |
-
"template_name": "underscore refer to",
|
| 5 |
-
"evaluation": {
|
| 6 |
-
"accuracy": 0.5301204819277109
|
| 7 |
-
},
|
| 8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='fr', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='underscore refer to', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz-3b/evaluation_l1/Muennighoff_xwinograd/pt/Replace/results.json
DELETED
|
@@ -1,9 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"dataset_name": "Muennighoff/xwinograd",
|
| 3 |
-
"dataset_config_name": "pt",
|
| 4 |
-
"template_name": "Replace",
|
| 5 |
-
"evaluation": {
|
| 6 |
-
"accuracy": 0.5285171102661597
|
| 7 |
-
},
|
| 8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='pt', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='Replace', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz-3b/evaluation_l1/Muennighoff_xwinograd/pt/True_or_False/results.json
DELETED
|
@@ -1,9 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"dataset_name": "Muennighoff/xwinograd",
|
| 3 |
-
"dataset_config_name": "pt",
|
| 4 |
-
"template_name": "True or False",
|
| 5 |
-
"evaluation": {
|
| 6 |
-
"accuracy": 0.5361216730038023
|
| 7 |
-
},
|
| 8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='pt', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='True or False', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz-3b/evaluation_l1/Muennighoff_xwinograd/pt/does_underscore_refer_to/results.json
DELETED
|
@@ -1,9 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"dataset_name": "Muennighoff/xwinograd",
|
| 3 |
-
"dataset_config_name": "pt",
|
| 4 |
-
"template_name": "does underscore refer to",
|
| 5 |
-
"evaluation": {
|
| 6 |
-
"accuracy": 0.5247148288973384
|
| 7 |
-
},
|
| 8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='pt', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='does underscore refer to', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz-3b/evaluation_l1/Muennighoff_xwinograd/pt/stand_for/results.json
DELETED
|
@@ -1,9 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"dataset_name": "Muennighoff/xwinograd",
|
| 3 |
-
"dataset_config_name": "pt",
|
| 4 |
-
"template_name": "stand for",
|
| 5 |
-
"evaluation": {
|
| 6 |
-
"accuracy": 0.4828897338403042
|
| 7 |
-
},
|
| 8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='pt', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='stand for', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz-3b/evaluation_l1/Muennighoff_xwinograd/pt/underscore_refer_to/results.json
DELETED
|
@@ -1,9 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"dataset_name": "Muennighoff/xwinograd",
|
| 3 |
-
"dataset_config_name": "pt",
|
| 4 |
-
"template_name": "underscore refer to",
|
| 5 |
-
"evaluation": {
|
| 6 |
-
"accuracy": 0.5019011406844106
|
| 7 |
-
},
|
| 8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='pt', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='underscore refer to', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz-3b/evaluation_l1/Muennighoff_xwinograd/zh/Replace/results.json
DELETED
|
@@ -1,9 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"dataset_name": "Muennighoff/xwinograd",
|
| 3 |
-
"dataset_config_name": "zh",
|
| 4 |
-
"template_name": "Replace",
|
| 5 |
-
"evaluation": {
|
| 6 |
-
"accuracy": 0.6091269841269841
|
| 7 |
-
},
|
| 8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='Replace', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz-3b/evaluation_l1/Muennighoff_xwinograd/zh/True_or_False/results.json
DELETED
|
@@ -1,9 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"dataset_name": "Muennighoff/xwinograd",
|
| 3 |
-
"dataset_config_name": "zh",
|
| 4 |
-
"template_name": "True or False",
|
| 5 |
-
"evaluation": {
|
| 6 |
-
"accuracy": 0.5515873015873016
|
| 7 |
-
},
|
| 8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='True or False', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz-3b/evaluation_l1/Muennighoff_xwinograd/zh/does_underscore_refer_to/results.json
DELETED
|
@@ -1,9 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"dataset_name": "Muennighoff/xwinograd",
|
| 3 |
-
"dataset_config_name": "zh",
|
| 4 |
-
"template_name": "does underscore refer to",
|
| 5 |
-
"evaluation": {
|
| 6 |
-
"accuracy": 0.5357142857142857
|
| 7 |
-
},
|
| 8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='does underscore refer to', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz-3b/evaluation_l1/Muennighoff_xwinograd/zh/stand_for/results.json
DELETED
|
@@ -1,9 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"dataset_name": "Muennighoff/xwinograd",
|
| 3 |
-
"dataset_config_name": "zh",
|
| 4 |
-
"template_name": "stand for",
|
| 5 |
-
"evaluation": {
|
| 6 |
-
"accuracy": 0.5238095238095238
|
| 7 |
-
},
|
| 8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='stand for', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz-3b/evaluation_l1/Muennighoff_xwinograd/zh/underscore_refer_to/results.json
DELETED
|
@@ -1,9 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"dataset_name": "Muennighoff/xwinograd",
|
| 3 |
-
"dataset_config_name": "zh",
|
| 4 |
-
"template_name": "underscore refer to",
|
| 5 |
-
"evaluation": {
|
| 6 |
-
"accuracy": 0.5674603174603174
|
| 7 |
-
},
|
| 8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='underscore refer to', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|