Add new example
#6
by
ybelkada
- opened
This view is limited to 50 files because it contains too many changes.
See the raw diff here.
- README.md +10 -18
- config.json +2 -2
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/ar/Answer_Given_options/results.json +9 -0
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/ar/Choose_Story_Ending/results.json +9 -0
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/ar/Generate_Ending/results.json +9 -0
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/ar/Novel_Correct_Ending/results.json +9 -0
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/ar/Story_Continuation_and_Options/results.json +9 -0
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/es/Answer_Given_options/results.json +9 -0
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/es/Choose_Story_Ending/results.json +9 -0
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/es/Generate_Ending/results.json +9 -0
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/es/Novel_Correct_Ending/results.json +9 -0
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/es/Story_Continuation_and_Options/results.json +9 -0
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/eu/Answer_Given_options/results.json +9 -0
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/eu/Choose_Story_Ending/results.json +9 -0
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/eu/Generate_Ending/results.json +9 -0
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/eu/Novel_Correct_Ending/results.json +9 -0
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/eu/Story_Continuation_and_Options/results.json +9 -0
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/hi/Answer_Given_options/results.json +9 -0
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/hi/Choose_Story_Ending/results.json +9 -0
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/hi/Generate_Ending/results.json +9 -0
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/hi/Novel_Correct_Ending/results.json +9 -0
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/hi/Story_Continuation_and_Options/results.json +9 -0
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/id/Answer_Given_options/results.json +9 -0
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/id/Choose_Story_Ending/results.json +9 -0
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/id/Generate_Ending/results.json +9 -0
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/id/Novel_Correct_Ending/results.json +9 -0
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/id/Story_Continuation_and_Options/results.json +9 -0
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/zh/Answer_Given_options/results.json +9 -0
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/zh/Choose_Story_Ending/results.json +9 -0
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/zh/Generate_Ending/results.json +9 -0
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/zh/Novel_Correct_Ending/results.json +9 -0
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/zh/Story_Continuation_and_Options/results.json +9 -0
- evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/en/Replace/results.json +9 -0
- evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/en/True_or_False/results.json +9 -0
- evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/en/does_underscore_refer_to/results.json +9 -0
- evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/en/stand_for/results.json +9 -0
- evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/en/underscore_refer_to/results.json +9 -0
- evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/fr/Replace/results.json +9 -0
- evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/fr/True_or_False/results.json +9 -0
- evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/fr/does_underscore_refer_to/results.json +9 -0
- evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/fr/stand_for/results.json +9 -0
- evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/fr/underscore_refer_to/results.json +9 -0
- evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/pt/Replace/results.json +9 -0
- evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/pt/True_or_False/results.json +9 -0
- evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/pt/does_underscore_refer_to/results.json +9 -0
- evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/pt/stand_for/results.json +9 -0
- evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/pt/underscore_refer_to/results.json +9 -0
- evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/zh/Replace/results.json +9 -0
- evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/zh/True_or_False/results.json +9 -0
- evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/zh/does_underscore_refer_to/results.json +9 -0
README.md
CHANGED
|
@@ -64,7 +64,6 @@ programming_language:
|
|
| 64 |
- Scala
|
| 65 |
- TypeScript
|
| 66 |
pipeline_tag: text-generation
|
| 67 |
-
inference: false
|
| 68 |
widget:
|
| 69 |
- text: "一个传奇的开端,一个不灭的神话,这不仅仅是一部电影,而是作为一个走进新时代的标签,永远彪炳史册。Would you rate the previous review as positive, neutral or negative?"
|
| 70 |
example_title: "zh-en sentiment"
|
|
@@ -86,13 +85,6 @@ widget:
|
|
| 86 |
example_title: "es-en fable"
|
| 87 |
- text: "Write a fable about wood elves living in a forest that is suddenly invaded by ogres. The fable is a masterpiece that has achieved praise worldwide and its moral is \"Violence is the last refuge of the incompetent\". Fable (in Hindi):"
|
| 88 |
example_title: "hi-en fable"
|
| 89 |
-
- text: "How many sides does a rectangle and heptagon have, when
|
| 90 |
-
combined? Answer this question with some math.
|
| 91 |
-
Ein Rechteck hat 4 Seiten. Ein Siebeneck hat 7 Seiten.
|
| 92 |
-
In Kombination haben sie 4 + 7 = 11 Seiten.
|
| 93 |
-
كم عدد الأضلاع التي يجمعها المربع والمثلث؟
|
| 94 |
-
Répondez à cette question en chinois."
|
| 95 |
-
example_title: "en-de-ar-fr-zh math"
|
| 96 |
model-index:
|
| 97 |
- name: bloomz
|
| 98 |
results:
|
|
@@ -680,11 +672,10 @@ model-index:
|
|
| 680 |
|
| 681 |
- **Repository:** [bigscience-workshop/xmtf](https://github.com/bigscience-workshop/xmtf)
|
| 682 |
- **Paper:** [Crosslingual Generalization through Multitask Finetuning](https://arxiv.org/abs/2211.01786)
|
| 683 |
-
- **Point of Contact:** [Niklas Muennighoff](mailto:
|
| 684 |
- **Languages:** Refer to [bloom](https://huggingface.co/bigscience/bloom) for pretraining & [xP3](https://huggingface.co/datasets/bigscience/xP3) for finetuning language proportions. It understands both pretraining & finetuning languages.
|
| 685 |
- **BLOOMZ & mT0 Model Family:**
|
| 686 |
|
| 687 |
-
<div class="max-w-full overflow-auto">
|
| 688 |
<table>
|
| 689 |
<tr>
|
| 690 |
<th colspan="12">Multitask finetuned on <a style="font-weight:bold" href=https://huggingface.co/datasets/bigscience/xP3>xP3</a>. Recommended for prompting in English.
|
|
@@ -705,8 +696,8 @@ model-index:
|
|
| 705 |
</tr>
|
| 706 |
<tr>
|
| 707 |
<td>Finetuned Model</td>
|
| 708 |
-
<td><a href=https://huggingface.co/bigscience/mt0-small>mt0-small</a></td>
|
| 709 |
<td><a href=https://huggingface.co/bigscience/mt0-base>mt0-base</a></td>
|
|
|
|
| 710 |
<td><a href=https://huggingface.co/bigscience/mt0-large>mt0-large</a></td>
|
| 711 |
<td><a href=https://huggingface.co/bigscience/mt0-xl>mt0-xl</a></td>
|
| 712 |
<td><a href=https://huggingface.co/bigscience/mt0-xxl>mt0-xxl</a></td>
|
|
@@ -754,8 +745,8 @@ model-index:
|
|
| 754 |
<th colspan="12">Original pretrained checkpoints. Not recommended.</th>
|
| 755 |
<tr>
|
| 756 |
<td>Pretrained Model</td>
|
| 757 |
-
<td><a href=https://huggingface.co/google/mt5-small>mt5-small</a></td>
|
| 758 |
<td><a href=https://huggingface.co/google/mt5-base>mt5-base</a></td>
|
|
|
|
| 759 |
<td><a href=https://huggingface.co/google/mt5-large>mt5-large</a></td>
|
| 760 |
<td><a href=https://huggingface.co/google/mt5-xl>mt5-xl</a></td>
|
| 761 |
<td><a href=https://huggingface.co/google/mt5-xxl>mt5-xxl</a></td>
|
|
@@ -767,7 +758,6 @@ model-index:
|
|
| 767 |
<td><a href=https://huggingface.co/bigscience/bloom>bloom</a></td>
|
| 768 |
</tr>
|
| 769 |
</table>
|
| 770 |
-
</div>
|
| 771 |
|
| 772 |
|
| 773 |
# Use
|
|
@@ -883,10 +873,12 @@ We refer to Table 7 from our [paper](https://arxiv.org/abs/2211.01786) & [bigsci
|
|
| 883 |
|
| 884 |
# Citation
|
| 885 |
```bibtex
|
| 886 |
-
@
|
| 887 |
-
|
| 888 |
-
|
| 889 |
-
|
| 890 |
-
|
|
|
|
|
|
|
| 891 |
}
|
| 892 |
```
|
|
|
|
| 64 |
- Scala
|
| 65 |
- TypeScript
|
| 66 |
pipeline_tag: text-generation
|
|
|
|
| 67 |
widget:
|
| 68 |
- text: "一个传奇的开端,一个不灭的神话,这不仅仅是一部电影,而是作为一个走进新时代的标签,永远彪炳史册。Would you rate the previous review as positive, neutral or negative?"
|
| 69 |
example_title: "zh-en sentiment"
|
|
|
|
| 85 |
example_title: "es-en fable"
|
| 86 |
- text: "Write a fable about wood elves living in a forest that is suddenly invaded by ogres. The fable is a masterpiece that has achieved praise worldwide and its moral is \"Violence is the last refuge of the incompetent\". Fable (in Hindi):"
|
| 87 |
example_title: "hi-en fable"
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 88 |
model-index:
|
| 89 |
- name: bloomz
|
| 90 |
results:
|
|
|
|
| 672 |
|
| 673 |
- **Repository:** [bigscience-workshop/xmtf](https://github.com/bigscience-workshop/xmtf)
|
| 674 |
- **Paper:** [Crosslingual Generalization through Multitask Finetuning](https://arxiv.org/abs/2211.01786)
|
| 675 |
+
- **Point of Contact:** [Niklas Muennighoff](mailto:niklas@hf.co)
|
| 676 |
- **Languages:** Refer to [bloom](https://huggingface.co/bigscience/bloom) for pretraining & [xP3](https://huggingface.co/datasets/bigscience/xP3) for finetuning language proportions. It understands both pretraining & finetuning languages.
|
| 677 |
- **BLOOMZ & mT0 Model Family:**
|
| 678 |
|
|
|
|
| 679 |
<table>
|
| 680 |
<tr>
|
| 681 |
<th colspan="12">Multitask finetuned on <a style="font-weight:bold" href=https://huggingface.co/datasets/bigscience/xP3>xP3</a>. Recommended for prompting in English.
|
|
|
|
| 696 |
</tr>
|
| 697 |
<tr>
|
| 698 |
<td>Finetuned Model</td>
|
|
|
|
| 699 |
<td><a href=https://huggingface.co/bigscience/mt0-base>mt0-base</a></td>
|
| 700 |
+
<td><a href=https://huggingface.co/bigscience/mt0-small>mt0-small</a></td>
|
| 701 |
<td><a href=https://huggingface.co/bigscience/mt0-large>mt0-large</a></td>
|
| 702 |
<td><a href=https://huggingface.co/bigscience/mt0-xl>mt0-xl</a></td>
|
| 703 |
<td><a href=https://huggingface.co/bigscience/mt0-xxl>mt0-xxl</a></td>
|
|
|
|
| 745 |
<th colspan="12">Original pretrained checkpoints. Not recommended.</th>
|
| 746 |
<tr>
|
| 747 |
<td>Pretrained Model</td>
|
|
|
|
| 748 |
<td><a href=https://huggingface.co/google/mt5-base>mt5-base</a></td>
|
| 749 |
+
<td><a href=https://huggingface.co/google/mt5-small>mt5-small</a></td>
|
| 750 |
<td><a href=https://huggingface.co/google/mt5-large>mt5-large</a></td>
|
| 751 |
<td><a href=https://huggingface.co/google/mt5-xl>mt5-xl</a></td>
|
| 752 |
<td><a href=https://huggingface.co/google/mt5-xxl>mt5-xxl</a></td>
|
|
|
|
| 758 |
<td><a href=https://huggingface.co/bigscience/bloom>bloom</a></td>
|
| 759 |
</tr>
|
| 760 |
</table>
|
|
|
|
| 761 |
|
| 762 |
|
| 763 |
# Use
|
|
|
|
| 873 |
|
| 874 |
# Citation
|
| 875 |
```bibtex
|
| 876 |
+
@misc{muennighoff2022crosslingual,
|
| 877 |
+
title={Crosslingual Generalization through Multitask Finetuning},
|
| 878 |
+
author={Niklas Muennighoff and Thomas Wang and Lintang Sutawika and Adam Roberts and Stella Biderman and Teven Le Scao and M Saiful Bari and Sheng Shen and Zheng-Xin Yong and Hailey Schoelkopf and Xiangru Tang and Dragomir Radev and Alham Fikri Aji and Khalid Almubarak and Samuel Albanie and Zaid Alyafeai and Albert Webson and Edward Raff and Colin Raffel},
|
| 879 |
+
year={2022},
|
| 880 |
+
eprint={2211.01786},
|
| 881 |
+
archivePrefix={arXiv},
|
| 882 |
+
primaryClass={cs.CL}
|
| 883 |
}
|
| 884 |
```
|
config.json
CHANGED
|
@@ -2,7 +2,7 @@
|
|
| 2 |
"apply_residual_connection_post_layernorm": false,
|
| 3 |
"attention_dropout": 0.0,
|
| 4 |
"architectures": [
|
| 5 |
-
"
|
| 6 |
],
|
| 7 |
"attention_softmax_in_fp32": true,
|
| 8 |
"seq_length": 2048,
|
|
@@ -22,4 +22,4 @@
|
|
| 22 |
"transformers_version": "4.21.0",
|
| 23 |
"use_cache": true,
|
| 24 |
"vocab_size": 250880
|
| 25 |
-
}
|
|
|
|
| 2 |
"apply_residual_connection_post_layernorm": false,
|
| 3 |
"attention_dropout": 0.0,
|
| 4 |
"architectures": [
|
| 5 |
+
"BloomModel"
|
| 6 |
],
|
| 7 |
"attention_softmax_in_fp32": true,
|
| 8 |
"seq_length": 2048,
|
|
|
|
| 22 |
"transformers_version": "4.21.0",
|
| 23 |
"use_cache": true,
|
| 24 |
"vocab_size": 250880
|
| 25 |
+
}
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/ar/Answer_Given_options/results.json
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
+
"dataset_config_name": "ar",
|
| 4 |
+
"template_name": "Answer Given options",
|
| 5 |
+
"evaluation": {
|
| 6 |
+
"accuracy": 0.7968232958305758
|
| 7 |
+
},
|
| 8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='ar', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Answer Given options', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
+
}
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/ar/Choose_Story_Ending/results.json
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
+
"dataset_config_name": "ar",
|
| 4 |
+
"template_name": "Choose Story Ending",
|
| 5 |
+
"evaluation": {
|
| 6 |
+
"accuracy": 0.9232296492389146
|
| 7 |
+
},
|
| 8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='ar', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Choose Story Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
+
}
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/ar/Generate_Ending/results.json
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
+
"dataset_config_name": "ar",
|
| 4 |
+
"template_name": "Generate Ending",
|
| 5 |
+
"evaluation": {
|
| 6 |
+
"accuracy": 0.6677696889477167
|
| 7 |
+
},
|
| 8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='ar', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Generate Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
+
}
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/ar/Novel_Correct_Ending/results.json
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
+
"dataset_config_name": "ar",
|
| 4 |
+
"template_name": "Novel Correct Ending",
|
| 5 |
+
"evaluation": {
|
| 6 |
+
"accuracy": 0.9265387160820648
|
| 7 |
+
},
|
| 8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='ar', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Novel Correct Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
+
}
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/ar/Story_Continuation_and_Options/results.json
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
+
"dataset_config_name": "ar",
|
| 4 |
+
"template_name": "Story Continuation and Options",
|
| 5 |
+
"evaluation": {
|
| 6 |
+
"accuracy": 0.9126406353408338
|
| 7 |
+
},
|
| 8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='ar', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Story Continuation and Options', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
+
}
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/es/Answer_Given_options/results.json
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
+
"dataset_config_name": "es",
|
| 4 |
+
"template_name": "Answer Given options",
|
| 5 |
+
"evaluation": {
|
| 6 |
+
"accuracy": 0.8729318332230311
|
| 7 |
+
},
|
| 8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='es', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Answer Given options', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
+
}
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/es/Choose_Story_Ending/results.json
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
+
"dataset_config_name": "es",
|
| 4 |
+
"template_name": "Choose Story Ending",
|
| 5 |
+
"evaluation": {
|
| 6 |
+
"accuracy": 0.9417604235605559
|
| 7 |
+
},
|
| 8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='es', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Choose Story Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
+
}
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/es/Generate_Ending/results.json
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
+
"dataset_config_name": "es",
|
| 4 |
+
"template_name": "Generate Ending",
|
| 5 |
+
"evaluation": {
|
| 6 |
+
"accuracy": 0.7359364659166115
|
| 7 |
+
},
|
| 8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='es', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Generate Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
+
}
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/es/Novel_Correct_Ending/results.json
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
+
"dataset_config_name": "es",
|
| 4 |
+
"template_name": "Novel Correct Ending",
|
| 5 |
+
"evaluation": {
|
| 6 |
+
"accuracy": 0.9430840502978161
|
| 7 |
+
},
|
| 8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='es', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Novel Correct Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
+
}
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/es/Story_Continuation_and_Options/results.json
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
+
"dataset_config_name": "es",
|
| 4 |
+
"template_name": "Story Continuation and Options",
|
| 5 |
+
"evaluation": {
|
| 6 |
+
"accuracy": 0.9318332230311053
|
| 7 |
+
},
|
| 8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='es', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Story Continuation and Options', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
+
}
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/eu/Answer_Given_options/results.json
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
+
"dataset_config_name": "eu",
|
| 4 |
+
"template_name": "Answer Given options",
|
| 5 |
+
"evaluation": {
|
| 6 |
+
"accuracy": 0.7054930509596293
|
| 7 |
+
},
|
| 8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='eu', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Answer Given options', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
+
}
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/eu/Choose_Story_Ending/results.json
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
+
"dataset_config_name": "eu",
|
| 4 |
+
"template_name": "Choose Story Ending",
|
| 5 |
+
"evaluation": {
|
| 6 |
+
"accuracy": 0.8663136995367307
|
| 7 |
+
},
|
| 8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='eu', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Choose Story Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
+
}
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/eu/Generate_Ending/results.json
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
+
"dataset_config_name": "eu",
|
| 4 |
+
"template_name": "Generate Ending",
|
| 5 |
+
"evaluation": {
|
| 6 |
+
"accuracy": 0.6320317670416943
|
| 7 |
+
},
|
| 8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='eu', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Generate Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
+
}
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/eu/Novel_Correct_Ending/results.json
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
+
"dataset_config_name": "eu",
|
| 4 |
+
"template_name": "Novel Correct Ending",
|
| 5 |
+
"evaluation": {
|
| 6 |
+
"accuracy": 0.8689609530112509
|
| 7 |
+
},
|
| 8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='eu', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Novel Correct Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
+
}
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/eu/Story_Continuation_and_Options/results.json
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
+
"dataset_config_name": "eu",
|
| 4 |
+
"template_name": "Story Continuation and Options",
|
| 5 |
+
"evaluation": {
|
| 6 |
+
"accuracy": 0.8524156187954997
|
| 7 |
+
},
|
| 8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='eu', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Story Continuation and Options', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
+
}
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/hi/Answer_Given_options/results.json
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
+
"dataset_config_name": "hi",
|
| 4 |
+
"template_name": "Answer Given options",
|
| 5 |
+
"evaluation": {
|
| 6 |
+
"accuracy": 0.798808735936466
|
| 7 |
+
},
|
| 8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='hi', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Answer Given options', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
+
}
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/hi/Choose_Story_Ending/results.json
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
+
"dataset_config_name": "hi",
|
| 4 |
+
"template_name": "Choose Story Ending",
|
| 5 |
+
"evaluation": {
|
| 6 |
+
"accuracy": 0.8702845797485109
|
| 7 |
+
},
|
| 8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='hi', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Choose Story Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
+
}
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/hi/Generate_Ending/results.json
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
+
"dataset_config_name": "hi",
|
| 4 |
+
"template_name": "Generate Ending",
|
| 5 |
+
"evaluation": {
|
| 6 |
+
"accuracy": 0.6604897418927862
|
| 7 |
+
},
|
| 8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='hi', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Generate Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
+
}
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/hi/Novel_Correct_Ending/results.json
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
+
"dataset_config_name": "hi",
|
| 4 |
+
"template_name": "Novel Correct Ending",
|
| 5 |
+
"evaluation": {
|
| 6 |
+
"accuracy": 0.8788881535407015
|
| 7 |
+
},
|
| 8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='hi', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Novel Correct Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
+
}
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/hi/Story_Continuation_and_Options/results.json
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
+
"dataset_config_name": "hi",
|
| 4 |
+
"template_name": "Story Continuation and Options",
|
| 5 |
+
"evaluation": {
|
| 6 |
+
"accuracy": 0.870946393117141
|
| 7 |
+
},
|
| 8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='hi', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Story Continuation and Options', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
+
}
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/id/Answer_Given_options/results.json
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
+
"dataset_config_name": "id",
|
| 4 |
+
"template_name": "Answer Given options",
|
| 5 |
+
"evaluation": {
|
| 6 |
+
"accuracy": 0.8557246856386499
|
| 7 |
+
},
|
| 8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='id', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Answer Given options', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
+
}
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/id/Choose_Story_Ending/results.json
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
+
"dataset_config_name": "id",
|
| 4 |
+
"template_name": "Choose Story Ending",
|
| 5 |
+
"evaluation": {
|
| 6 |
+
"accuracy": 0.9212442091330245
|
| 7 |
+
},
|
| 8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='id', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Choose Story Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
+
}
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/id/Generate_Ending/results.json
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
+
"dataset_config_name": "id",
|
| 4 |
+
"template_name": "Generate Ending",
|
| 5 |
+
"evaluation": {
|
| 6 |
+
"accuracy": 0.7041694242223693
|
| 7 |
+
},
|
| 8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='id', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Generate Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
+
}
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/id/Novel_Correct_Ending/results.json
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
+
"dataset_config_name": "id",
|
| 4 |
+
"template_name": "Novel Correct Ending",
|
| 5 |
+
"evaluation": {
|
| 6 |
+
"accuracy": 0.9205823957643945
|
| 7 |
+
},
|
| 8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='id', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Novel Correct Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
+
}
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/id/Story_Continuation_and_Options/results.json
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
+
"dataset_config_name": "id",
|
| 4 |
+
"template_name": "Story Continuation and Options",
|
| 5 |
+
"evaluation": {
|
| 6 |
+
"accuracy": 0.9066843150231635
|
| 7 |
+
},
|
| 8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='id', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Story Continuation and Options', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
+
}
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/zh/Answer_Given_options/results.json
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
+
"dataset_config_name": "zh",
|
| 4 |
+
"template_name": "Answer Given options",
|
| 5 |
+
"evaluation": {
|
| 6 |
+
"accuracy": 0.900066181336863
|
| 7 |
+
},
|
| 8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Answer Given options', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
+
}
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/zh/Choose_Story_Ending/results.json
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
+
"dataset_config_name": "zh",
|
| 4 |
+
"template_name": "Choose Story Ending",
|
| 5 |
+
"evaluation": {
|
| 6 |
+
"accuracy": 0.9232296492389146
|
| 7 |
+
},
|
| 8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Choose Story Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
+
}
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/zh/Generate_Ending/results.json
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
+
"dataset_config_name": "zh",
|
| 4 |
+
"template_name": "Generate Ending",
|
| 5 |
+
"evaluation": {
|
| 6 |
+
"accuracy": 0.684976836532098
|
| 7 |
+
},
|
| 8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Generate Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
+
}
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/zh/Novel_Correct_Ending/results.json
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
+
"dataset_config_name": "zh",
|
| 4 |
+
"template_name": "Novel Correct Ending",
|
| 5 |
+
"evaluation": {
|
| 6 |
+
"accuracy": 0.9311714096624751
|
| 7 |
+
},
|
| 8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Novel Correct Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
+
}
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/zh/Story_Continuation_and_Options/results.json
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_name": "Muennighoff/xstory_cloze",
|
| 3 |
+
"dataset_config_name": "zh",
|
| 4 |
+
"template_name": "Story Continuation and Options",
|
| 5 |
+
"evaluation": {
|
| 6 |
+
"accuracy": 0.9199205823957644
|
| 7 |
+
},
|
| 8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Story Continuation and Options', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
+
}
|
evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/en/Replace/results.json
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_name": "Muennighoff/xwinograd",
|
| 3 |
+
"dataset_config_name": "en",
|
| 4 |
+
"template_name": "Replace",
|
| 5 |
+
"evaluation": {
|
| 6 |
+
"accuracy": 0.6847311827956989
|
| 7 |
+
},
|
| 8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='en', dataset_name='Muennighoff/xwinograd', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Replace', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
+
}
|
evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/en/True_or_False/results.json
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_name": "Muennighoff/xwinograd",
|
| 3 |
+
"dataset_config_name": "en",
|
| 4 |
+
"template_name": "True or False",
|
| 5 |
+
"evaluation": {
|
| 6 |
+
"accuracy": 0.5135483870967742
|
| 7 |
+
},
|
| 8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='en', dataset_name='Muennighoff/xwinograd', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='True or False', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
+
}
|
evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/en/does_underscore_refer_to/results.json
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_name": "Muennighoff/xwinograd",
|
| 3 |
+
"dataset_config_name": "en",
|
| 4 |
+
"template_name": "does underscore refer to",
|
| 5 |
+
"evaluation": {
|
| 6 |
+
"accuracy": 0.6787096774193548
|
| 7 |
+
},
|
| 8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='en', dataset_name='Muennighoff/xwinograd', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='does underscore refer to', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
+
}
|
evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/en/stand_for/results.json
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_name": "Muennighoff/xwinograd",
|
| 3 |
+
"dataset_config_name": "en",
|
| 4 |
+
"template_name": "stand for",
|
| 5 |
+
"evaluation": {
|
| 6 |
+
"accuracy": 0.5053763440860215
|
| 7 |
+
},
|
| 8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='en', dataset_name='Muennighoff/xwinograd', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='stand for', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
+
}
|
evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/en/underscore_refer_to/results.json
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_name": "Muennighoff/xwinograd",
|
| 3 |
+
"dataset_config_name": "en",
|
| 4 |
+
"template_name": "underscore refer to",
|
| 5 |
+
"evaluation": {
|
| 6 |
+
"accuracy": 0.690752688172043
|
| 7 |
+
},
|
| 8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='en', dataset_name='Muennighoff/xwinograd', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='underscore refer to', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
+
}
|
evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/fr/Replace/results.json
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_name": "Muennighoff/xwinograd",
|
| 3 |
+
"dataset_config_name": "fr",
|
| 4 |
+
"template_name": "Replace",
|
| 5 |
+
"evaluation": {
|
| 6 |
+
"accuracy": 0.6506024096385542
|
| 7 |
+
},
|
| 8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='fr', dataset_name='Muennighoff/xwinograd', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Replace', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
+
}
|
evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/fr/True_or_False/results.json
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_name": "Muennighoff/xwinograd",
|
| 3 |
+
"dataset_config_name": "fr",
|
| 4 |
+
"template_name": "True or False",
|
| 5 |
+
"evaluation": {
|
| 6 |
+
"accuracy": 0.4939759036144578
|
| 7 |
+
},
|
| 8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='fr', dataset_name='Muennighoff/xwinograd', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='True or False', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
+
}
|
evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/fr/does_underscore_refer_to/results.json
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_name": "Muennighoff/xwinograd",
|
| 3 |
+
"dataset_config_name": "fr",
|
| 4 |
+
"template_name": "does underscore refer to",
|
| 5 |
+
"evaluation": {
|
| 6 |
+
"accuracy": 0.6867469879518072
|
| 7 |
+
},
|
| 8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='fr', dataset_name='Muennighoff/xwinograd', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='does underscore refer to', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
+
}
|
evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/fr/stand_for/results.json
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_name": "Muennighoff/xwinograd",
|
| 3 |
+
"dataset_config_name": "fr",
|
| 4 |
+
"template_name": "stand for",
|
| 5 |
+
"evaluation": {
|
| 6 |
+
"accuracy": 0.46987951807228917
|
| 7 |
+
},
|
| 8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='fr', dataset_name='Muennighoff/xwinograd', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='stand for', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
+
}
|
evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/fr/underscore_refer_to/results.json
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_name": "Muennighoff/xwinograd",
|
| 3 |
+
"dataset_config_name": "fr",
|
| 4 |
+
"template_name": "underscore refer to",
|
| 5 |
+
"evaluation": {
|
| 6 |
+
"accuracy": 0.6626506024096386
|
| 7 |
+
},
|
| 8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='fr', dataset_name='Muennighoff/xwinograd', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='underscore refer to', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
+
}
|
evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/pt/Replace/results.json
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_name": "Muennighoff/xwinograd",
|
| 3 |
+
"dataset_config_name": "pt",
|
| 4 |
+
"template_name": "Replace",
|
| 5 |
+
"evaluation": {
|
| 6 |
+
"accuracy": 0.6349809885931559
|
| 7 |
+
},
|
| 8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='pt', dataset_name='Muennighoff/xwinograd', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='Replace', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
+
}
|
evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/pt/True_or_False/results.json
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_name": "Muennighoff/xwinograd",
|
| 3 |
+
"dataset_config_name": "pt",
|
| 4 |
+
"template_name": "True or False",
|
| 5 |
+
"evaluation": {
|
| 6 |
+
"accuracy": 0.4866920152091255
|
| 7 |
+
},
|
| 8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='pt', dataset_name='Muennighoff/xwinograd', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='True or False', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
+
}
|
evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/pt/does_underscore_refer_to/results.json
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_name": "Muennighoff/xwinograd",
|
| 3 |
+
"dataset_config_name": "pt",
|
| 4 |
+
"template_name": "does underscore refer to",
|
| 5 |
+
"evaluation": {
|
| 6 |
+
"accuracy": 0.6387832699619772
|
| 7 |
+
},
|
| 8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='pt', dataset_name='Muennighoff/xwinograd', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='does underscore refer to', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
+
}
|
evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/pt/stand_for/results.json
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_name": "Muennighoff/xwinograd",
|
| 3 |
+
"dataset_config_name": "pt",
|
| 4 |
+
"template_name": "stand for",
|
| 5 |
+
"evaluation": {
|
| 6 |
+
"accuracy": 0.49429657794676807
|
| 7 |
+
},
|
| 8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='pt', dataset_name='Muennighoff/xwinograd', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='stand for', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
+
}
|
evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/pt/underscore_refer_to/results.json
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_name": "Muennighoff/xwinograd",
|
| 3 |
+
"dataset_config_name": "pt",
|
| 4 |
+
"template_name": "underscore refer to",
|
| 5 |
+
"evaluation": {
|
| 6 |
+
"accuracy": 0.6425855513307985
|
| 7 |
+
},
|
| 8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='pt', dataset_name='Muennighoff/xwinograd', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='underscore refer to', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
+
}
|
evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/zh/Replace/results.json
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_name": "Muennighoff/xwinograd",
|
| 3 |
+
"dataset_config_name": "zh",
|
| 4 |
+
"template_name": "Replace",
|
| 5 |
+
"evaluation": {
|
| 6 |
+
"accuracy": 0.6865079365079365
|
| 7 |
+
},
|
| 8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xwinograd', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Replace', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
+
}
|
evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/zh/True_or_False/results.json
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_name": "Muennighoff/xwinograd",
|
| 3 |
+
"dataset_config_name": "zh",
|
| 4 |
+
"template_name": "True or False",
|
| 5 |
+
"evaluation": {
|
| 6 |
+
"accuracy": 0.5277777777777778
|
| 7 |
+
},
|
| 8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xwinograd', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='True or False', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
+
}
|
evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/zh/does_underscore_refer_to/results.json
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_name": "Muennighoff/xwinograd",
|
| 3 |
+
"dataset_config_name": "zh",
|
| 4 |
+
"template_name": "does underscore refer to",
|
| 5 |
+
"evaluation": {
|
| 6 |
+
"accuracy": 0.6884920634920635
|
| 7 |
+
},
|
| 8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xwinograd', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='does underscore refer to', tokenizer_name=None, use_slow_tokenizer=False)"
|
| 9 |
+
}
|