diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/direct_cot/prompt_2/afrimgsm_cot_lin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/direct_cot/prompt_2/afrimgsm_cot_lin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..840a99acfacba2dacfe8c2883fed6445373f72e7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/direct_cot/prompt_2/afrimgsm_cot_lin.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: lin +include: afrimgsm_cot_yaml +task: afrimgsm_cot_lin_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/direct_cot/prompt_4/afrimgsm_cot_eng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/direct_cot/prompt_4/afrimgsm_cot_eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fa82cb876081070e9a300dd1471f18c78a8cc311 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/direct_cot/prompt_4/afrimgsm_cot_eng.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng +doc_to_text: "Answer the given question with the step by step solution appropriate\ + \ numerical value, ensuring that the response is clear and without any supplementary\ + \ information. \n\nQuestion: {{question}} \nStep by step answer: " +include: afrimgsm_cot_yaml +task: afrimgsm_cot_eng_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/direct_cot/prompt_4/afrimgsm_cot_zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/direct_cot/prompt_4/afrimgsm_cot_zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2641b2e58c631b88f85ee8c156a9028c8b319c86 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/direct_cot/prompt_4/afrimgsm_cot_zul.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: zul +doc_to_text: "Answer the given question with the step by step solution appropriate\ + \ numerical value, ensuring that the response is clear and without any supplementary\ + \ information. \n\nQuestion: {{question}} \nStep by step answer: " +include: afrimgsm_cot_yaml +task: afrimgsm_cot_zul_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/direct_cot/prompt_5/afrimgsm_cot_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/direct_cot/prompt_5/afrimgsm_cot_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e2c6a5cc7fb2a54afcfdb98b2d176bda42b23aaf --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/direct_cot/prompt_5/afrimgsm_cot_ibo.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: ibo +doc_to_text: "For mathematical questions provided in Igbo language. Supply the accurate\ + \ step by step answer to the provided question. \n\nQuestion: {{question}} \nStep\ + \ by step answer: " +include: afrimgsm_cot_yaml +task: afrimgsm_cot_ibo_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/direct_cot/prompt_5/afrimgsm_cot_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/direct_cot/prompt_5/afrimgsm_cot_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..36c19a99fdf7dfe8410a2d3183e31f633455932c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/direct_cot/prompt_5/afrimgsm_cot_kin.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: kin +doc_to_text: "For mathematical questions provided in Kinyarwanda language. Supply\ + \ the accurate step by step answer to the provided question. \n\nQuestion: {{question}}\ + \ \nStep by step answer: " +include: afrimgsm_cot_yaml +task: afrimgsm_cot_kin_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/direct_cot/prompt_5/afrimgsm_cot_lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/direct_cot/prompt_5/afrimgsm_cot_lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..918a3e31484a4f42b5c78bd53a5ad2194e23507d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/direct_cot/prompt_5/afrimgsm_cot_lug.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: lug +doc_to_text: "For mathematical questions provided in Luganda language. Supply the\ + \ accurate step by step answer to the provided question. \n\nQuestion: {{question}}\ + \ \nStep by step answer: " +include: afrimgsm_cot_yaml +task: afrimgsm_cot_lug_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/direct_cot/prompt_5/afrimgsm_cot_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/direct_cot/prompt_5/afrimgsm_cot_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dd0436e7e8ad0a80e32554fb9991158fe66df7be --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/direct_cot/prompt_5/afrimgsm_cot_twi.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: twi +doc_to_text: "For mathematical questions provided in Twi language. Supply the accurate\ + \ step by step answer to the provided question. \n\nQuestion: {{question}} \nStep\ + \ by step answer: " +include: afrimgsm_cot_yaml +task: afrimgsm_cot_twi_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/direct_cot/prompt_5/afrimgsm_cot_vai.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/direct_cot/prompt_5/afrimgsm_cot_vai.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0b348be3a2a337a96936ddafe9315fe060dc8516 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/direct_cot/prompt_5/afrimgsm_cot_vai.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: vai +doc_to_text: "For mathematical questions provided in Vai language. Supply the accurate\ + \ step by step answer to the provided question. \n\nQuestion: {{question}} \nStep\ + \ by step answer: " +include: afrimgsm_cot_yaml +task: afrimgsm_cot_vai_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/direct_cot/prompt_5/afrimgsm_cot_zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/direct_cot/prompt_5/afrimgsm_cot_zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0b6ef0037a6bb530304d7fd5031c2f6816d678a3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/direct_cot/prompt_5/afrimgsm_cot_zul.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: zul +doc_to_text: "For mathematical questions provided in Zulu language. Supply the accurate\ + \ step by step answer to the provided question. \n\nQuestion: {{question}} \nStep\ + \ by step answer: " +include: afrimgsm_cot_yaml +task: afrimgsm_cot_zul_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0f067e53f525c95640c43cd94f02dfca0e4702a8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_amh.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: amh +include: afrimgsm_translate_yaml +task: afrimgsm_translate_amh_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_ewe.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_ewe.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1420deed028ca24caea7a72b96bcc53494f7e186 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_ewe.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: ewe +include: afrimgsm_translate_yaml +task: afrimgsm_translate_ewe_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..768bcab970c2b6af4edbdaa0897ea4f24e1d0eb5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_hau.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: hau +include: afrimgsm_translate_yaml +task: afrimgsm_translate_hau_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_sna.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_sna.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2f8826ab4d61ab2a2803c73fd770db538a4db0f0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_sna.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: sna +include: afrimgsm_translate_yaml +task: afrimgsm_translate_sna_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3aede319a2ad47e5e0e0a65c18ca8a85da812971 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_swa.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: swa +include: afrimgsm_translate_yaml +task: afrimgsm_translate_swa_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c8e23103ee7a482a1f84002c3dc40bc10758e659 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_twi.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: twi +include: afrimgsm_translate_yaml +task: afrimgsm_translate_twi_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4b97922fc5d1d051a0a98623a994b695dbf68c86 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_wol.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: wol +include: afrimgsm_translate_yaml +task: afrimgsm_translate_wol_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_yaml new file mode 100644 index 0000000000000000000000000000000000000000..614510895a32f30082ccd6eb5cbdfa87766c4473 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_yaml @@ -0,0 +1,32 @@ +tag: afrimgsm_tt_tasks +dataset_path: masakhane/afrimgsm-translate-test +output_type: generate_until +test_split: test +doc_to_target: '{% if answer is not none %}{{answer[21:]}}{% else %}{{answer_number|string}}{% endif %}' +doc_to_text: '{% if answer is not none %}{{question+"\nAnswer:"}}{% else %}{{"Question: "+question+"\nAnswer:"}}{% endif %}' +target_delimiter: "" +generation_kwargs: + do_sample: false + until: + - 'Question:' + - + - <|im_end|> +filter_list: + - name: remove_whitespace + filter: + - function: remove_whitespace + - function: take_first + - filter: + - function: regex + group_select: -1 + regex_pattern: (-?[$0-9.,]{2,})|(-?[0-9]+) + - function: take_first + name: flexible-extract +metric_list: + - metric: exact_match + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true +metadata: + version: 2.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..49b559be5c1f62ab67498131d29ac0e0091f622d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_amh.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: amh +include: afrimgsm_translate_yaml +task: afrimgsm_translate_amh_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_ewe.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_ewe.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cf82f8624adc0bf9b60194b889ee8ccc7df76c70 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_ewe.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: ewe +include: afrimgsm_translate_yaml +task: afrimgsm_translate_ewe_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_fra.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_fra.yaml new file mode 100644 index 0000000000000000000000000000000000000000..924ac026e258ab6da6bc0c2be9c55e73f48a3457 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_fra.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fra +include: afrimgsm_translate_yaml +task: afrimgsm_translate_fra_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..466ced5c435be500ff453e9242097050cfcc587c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_ibo.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: ibo +include: afrimgsm_translate_yaml +task: afrimgsm_translate_ibo_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_lin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_lin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..72aa73d209de818f83d596148e756752bc44c754 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_lin.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: lin +include: afrimgsm_translate_yaml +task: afrimgsm_translate_lin_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..88ae24a26d652729a5c52696f8c34fcec358dfd8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_lug.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: lug +include: afrimgsm_translate_yaml +task: afrimgsm_translate_lug_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_orm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_orm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7e2ffcc32241ed36d042591921737f9c91deabcc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_orm.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: orm +include: afrimgsm_translate_yaml +task: afrimgsm_translate_orm_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_sot.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_sot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5bd7e53ca6ec65492978eb7e2bc95a0569b496e8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_sot.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: sot +include: afrimgsm_translate_yaml +task: afrimgsm_translate_sot_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5134b3c423c88832393b6eb7b74338b8f7e97807 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_swa.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: swa +include: afrimgsm_translate_yaml +task: afrimgsm_translate_swa_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f6135d99ce5181855cb78f1a0beba3f357c7082c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_twi.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: twi +include: afrimgsm_translate_yaml +task: afrimgsm_translate_twi_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..db00be881c3170721857cfb3fb685b3adfec4dfc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_wol.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: wol +include: afrimgsm_translate_yaml +task: afrimgsm_translate_wol_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3be8dd64c7e7aaacebb296c8f51397d4c50fe3ec --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_xho.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: xho +include: afrimgsm_translate_yaml +task: afrimgsm_translate_xho_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..01a54e15eea260f6a4ae5fa28e0ea4c627dc904c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_yor.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: yor +include: afrimgsm_translate_yaml +task: afrimgsm_translate_yor_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5f7e74df5ba87e25f35e0e13762cd2116797fe65 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_zul.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: zul +include: afrimgsm_translate_yaml +task: afrimgsm_translate_zul_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..04a14a1bf6b69e10873e181d46fcd51e2e901126 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_amh.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: amh +include: afrimgsm_translate_yaml +task: afrimgsm_translate_amh_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_ewe.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_ewe.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3cda09e47c41ecc23f2e15b16c869fd6e3f13d87 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_ewe.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: ewe +include: afrimgsm_translate_yaml +task: afrimgsm_translate_ewe_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_fra.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_fra.yaml new file mode 100644 index 0000000000000000000000000000000000000000..49c95be2cd94d718bd6cb1754eee6a79ae6f5ae8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_fra.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fra +include: afrimgsm_translate_yaml +task: afrimgsm_translate_fra_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9d16ac8faf84785902baabb92a1a48bcf383a35a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_hau.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: hau +include: afrimgsm_translate_yaml +task: afrimgsm_translate_hau_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2bbb66ff41f6c55799d77a19e29bfd8e01f0d61d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_ibo.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: ibo +include: afrimgsm_translate_yaml +task: afrimgsm_translate_ibo_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..488061a306ebcbc4e9f53e78214e25d5391475fa --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_kin.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: kin +include: afrimgsm_translate_yaml +task: afrimgsm_translate_kin_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_lin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_lin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..928ba457061168ae1524db65a4f03f6ec202349d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_lin.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: lin +include: afrimgsm_translate_yaml +task: afrimgsm_translate_lin_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bdc0c80788f1e0e0e4db9bc38d77aab7e2f0f8df --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_lug.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: lug +include: afrimgsm_translate_yaml +task: afrimgsm_translate_lug_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_orm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_orm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..04ec7565c10592c8ff143d0d2a944986e2870c1f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_orm.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: orm +include: afrimgsm_translate_yaml +task: afrimgsm_translate_orm_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_sna.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_sna.yaml new file mode 100644 index 0000000000000000000000000000000000000000..22ab7bde213b00599cee3e97ef8e4995f2ded97a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_sna.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: sna +include: afrimgsm_translate_yaml +task: afrimgsm_translate_sna_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_sot.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_sot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..617340d07bdab98b185c8e441e1a2e08bca3930b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_sot.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: sot +include: afrimgsm_translate_yaml +task: afrimgsm_translate_sot_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..337ad6e470631085a735ee2133211c8e9258600e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_swa.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: swa +include: afrimgsm_translate_yaml +task: afrimgsm_translate_swa_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..eb13aba534c504473735b6cb67c90efab5db9095 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_twi.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: twi +include: afrimgsm_translate_yaml +task: afrimgsm_translate_twi_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f759e6aa6aebd9f6c6f82087dc2756d68a71025e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_wol.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: wol +include: afrimgsm_translate_yaml +task: afrimgsm_translate_wol_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..50ab80df1d0589e592b072f71628395fe118ef8b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_xho.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: xho +include: afrimgsm_translate_yaml +task: afrimgsm_translate_xho_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_yaml new file mode 100644 index 0000000000000000000000000000000000000000..544fa0ccc13fbbeb9eebc4f0eb2cb78b5c68e183 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_yaml @@ -0,0 +1,32 @@ +tag: afrimgsm_tt_tasks +dataset_path: masakhane/afrimgsm-translate-test +output_type: generate_until +test_split: test +doc_to_target: '{% if answer is not none %}{{answer[21:]}}{% else %}{{answer_number|string}}{% endif %}' +doc_to_text: "Solve the following math question \n\nQuestion: {{question}} \nAnswer: " +target_delimiter: "" +generation_kwargs: + do_sample: false + until: + - 'Question:' + - + - <|im_end|> +filter_list: + - name: remove_whitespace + filter: + - function: remove_whitespace + - function: take_first + - filter: + - function: regex + group_select: -1 + regex_pattern: (-?[$0-9.,]{2,})|(-?[0-9]+) + - function: take_first + name: flexible-extract +metric_list: + - metric: exact_match + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true +metadata: + version: 2.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ee1c7917f6356e01b54ad4c545d3f08b2dfcf8a6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_yor.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: yor +include: afrimgsm_translate_yaml +task: afrimgsm_translate_yor_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f3e21704f20ceba6e34273f8a2d5b1d87b648dda --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_zul.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: zul +include: afrimgsm_translate_yaml +task: afrimgsm_translate_zul_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..60387afce4fa5801492c968fc97a1519fdbeb5a8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_amh.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: amh +doc_to_text: "Answer the given question with the appropriate numerical value, ensuring\ + \ that the response is clear and without any supplementary information. \n\nQuestion:\ + \ {{question}} \nAnswer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_amh_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_ewe.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_ewe.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7633bc3efd95541fd6f30aaa8d469fa993a0375e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_ewe.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: ewe +doc_to_text: "Answer the given question with the appropriate numerical value, ensuring\ + \ that the response is clear and without any supplementary information. \n\nQuestion:\ + \ {{question}} \nAnswer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_ewe_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_fra.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_fra.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a8e16ea929f8d5639f388c339671f1253554fd1d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_fra.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: fra +doc_to_text: "Answer the given question with the appropriate numerical value, ensuring\ + \ that the response is clear and without any supplementary information. \n\nQuestion:\ + \ {{question}} \nAnswer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_fra_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9828205094c1b447ce422da748ba347d448f3b0b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_hau.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: hau +doc_to_text: "Answer the given question with the appropriate numerical value, ensuring\ + \ that the response is clear and without any supplementary information. \n\nQuestion:\ + \ {{question}} \nAnswer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_hau_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b8acf8d0fb40630e1c19e54ed5b87687f2bf2897 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_ibo.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: ibo +doc_to_text: "Answer the given question with the appropriate numerical value, ensuring\ + \ that the response is clear and without any supplementary information. \n\nQuestion:\ + \ {{question}} \nAnswer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_ibo_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..74ac117344b5625f1a8671cb938e7d911e849eea --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_kin.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: kin +doc_to_text: "Answer the given question with the appropriate numerical value, ensuring\ + \ that the response is clear and without any supplementary information. \n\nQuestion:\ + \ {{question}} \nAnswer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_kin_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_lin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_lin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6cf113619e461a91f0a4517dae08109055ed743f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_lin.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: lin +doc_to_text: "Answer the given question with the appropriate numerical value, ensuring\ + \ that the response is clear and without any supplementary information. \n\nQuestion:\ + \ {{question}} \nAnswer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_lin_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5dffbdb899140911f0dab979581fc7b0d52674bf --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_lug.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: lug +doc_to_text: "Answer the given question with the appropriate numerical value, ensuring\ + \ that the response is clear and without any supplementary information. \n\nQuestion:\ + \ {{question}} \nAnswer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_lug_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_orm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_orm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..30f776f4b6c6a05a98b79234dbf45f269170dc0b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_orm.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: orm +doc_to_text: "Answer the given question with the appropriate numerical value, ensuring\ + \ that the response is clear and without any supplementary information. \n\nQuestion:\ + \ {{question}} \nAnswer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_orm_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_sna.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_sna.yaml new file mode 100644 index 0000000000000000000000000000000000000000..63efa2505e80032a7ef347f1f986f0ec0952c07c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_sna.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: sna +doc_to_text: "Answer the given question with the appropriate numerical value, ensuring\ + \ that the response is clear and without any supplementary information. \n\nQuestion:\ + \ {{question}} \nAnswer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_sna_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_sot.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_sot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..19b86220a5eeb0319c7732add1ed7d969ab72c67 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_sot.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: sot +doc_to_text: "Answer the given question with the appropriate numerical value, ensuring\ + \ that the response is clear and without any supplementary information. \n\nQuestion:\ + \ {{question}} \nAnswer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_sot_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5fe7e7475ae49642169275f3aff639ba4d3fbeda --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_twi.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: twi +doc_to_text: "Answer the given question with the appropriate numerical value, ensuring\ + \ that the response is clear and without any supplementary information. \n\nQuestion:\ + \ {{question}} \nAnswer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_twi_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a8fb5640def2bb261c461a80b66ba50324231317 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_wol.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: wol +doc_to_text: "Answer the given question with the appropriate numerical value, ensuring\ + \ that the response is clear and without any supplementary information. \n\nQuestion:\ + \ {{question}} \nAnswer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_wol_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fdb63749c47b38edfb97d20998f4fb8d4479075d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_xho.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: xho +doc_to_text: "Answer the given question with the appropriate numerical value, ensuring\ + \ that the response is clear and without any supplementary information. \n\nQuestion:\ + \ {{question}} \nAnswer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_xho_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_yaml new file mode 100644 index 0000000000000000000000000000000000000000..2d3903948dd4a8ae2c285e33eb2551f554d2310c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_yaml @@ -0,0 +1,31 @@ +tag: afrimgsm_tt_tasks +dataset_path: masakhane/afrimgsm-translate-test +output_type: generate_until +test_split: test +doc_to_target: '{% if answer is not none %}{{answer[21:]}}{% else %}{{answer_number|string}}{% endif %}' +target_delimiter: "" +generation_kwargs: + do_sample: false + until: + - 'Question:' + - + - <|im_end|> +filter_list: + - name: remove_whitespace + filter: + - function: remove_whitespace + - function: take_first + - filter: + - function: regex + group_select: -1 + regex_pattern: (-?[$0-9.,]{2,})|(-?[0-9]+) + - function: take_first + name: flexible-extract +metric_list: + - metric: exact_match + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true +metadata: + version: 2.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f5cb74d41b85d71f851f6c3f160edbc85d006151 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_yor.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: yor +doc_to_text: "Answer the given question with the appropriate numerical value, ensuring\ + \ that the response is clear and without any supplementary information. \n\nQuestion:\ + \ {{question}} \nAnswer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_yor_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6f0a068e9f662d567bb01a208e46a6e0b2d014d2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_zul.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: zul +doc_to_text: "Answer the given question with the appropriate numerical value, ensuring\ + \ that the response is clear and without any supplementary information. \n\nQuestion:\ + \ {{question}} \nAnswer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_zul_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..48ca09aaafafe7360ca9b0c2872d99e63ff16619 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_amh.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: amh +doc_to_text: "For mathematical questions provided in Amharic language. Supply the\ + \ accurate numeric answer to the provided question. \n\nQuestion: {{question}} \n\ + Answer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_amh_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_ewe.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_ewe.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a4a254f0a2d766380528dde5d2ad9ac7eb67bb2d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_ewe.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: ewe +doc_to_text: "For mathematical questions provided in Ewe language. Supply the accurate\ + \ numeric answer to the provided question. \n\nQuestion: {{question}} \nAnswer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_ewe_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_fra.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_fra.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ac62304550bb0f5b8705c7a9ba3f5934278be55d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_fra.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: fra +doc_to_text: "For mathematical questions provided in French language. Supply the accurate\ + \ numeric answer to the provided question. \n\nQuestion: {{question}} \nAnswer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_fra_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..695f1f373464fb93811f28644c0b849fe72de9ed --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_hau.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: hau +doc_to_text: "For mathematical questions provided in Hausa language. Supply the accurate\ + \ numeric answer to the provided question. \n\nQuestion: {{question}} \nAnswer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_hau_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7fd530e7409124357c091d42cbaf5608473976c0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_ibo.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: ibo +doc_to_text: "For mathematical questions provided in Igbo language. Supply the accurate\ + \ numeric answer to the provided question. \n\nQuestion: {{question}} \nAnswer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_ibo_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..52ea0a78a2053f7958583167d613ca209b43e22a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_kin.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: kin +doc_to_text: "For mathematical questions provided in Kinyarwanda language. Supply\ + \ the accurate numeric answer to the provided question. \n\nQuestion: {{question}}\ + \ \nAnswer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_kin_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_lin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_lin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..07cf6a6b0e871bbf004d809d4fcff2f7f063a2a9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_lin.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: lin +doc_to_text: "For mathematical questions provided in Lingala language. Supply the\ + \ accurate numeric answer to the provided question. \n\nQuestion: {{question}} \n\ + Answer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_lin_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fa3461beb8fc3c9f05dd1a58c5ca7e7de4ac6cb3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_lug.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: lug +doc_to_text: "For mathematical questions provided in Luganda language. Supply the\ + \ accurate numeric answer to the provided question. \n\nQuestion: {{question}} \n\ + Answer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_lug_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_orm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_orm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c1a00385f578feeb5fb5071c7f2835574a68a30e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_orm.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: orm +doc_to_text: "For mathematical questions provided in Oromo language. Supply the accurate\ + \ numeric answer to the provided question. \n\nQuestion: {{question}} \nAnswer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_orm_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_sna.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_sna.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c7f08a786ecac0379aa660a26e5619055e97063a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_sna.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: sna +doc_to_text: "For mathematical questions provided in chiShona language. Supply the\ + \ accurate numeric answer to the provided question. \n\nQuestion: {{question}} \n\ + Answer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_sna_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_sot.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_sot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b258204f4eaed7d4889c597189ab910956e9dbb5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_sot.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: sot +doc_to_text: "For mathematical questions provided in Sesotho language. Supply the\ + \ accurate numeric answer to the provided question. \n\nQuestion: {{question}} \n\ + Answer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_sot_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a950c84d3c617353fd9492e6a1d2a028fd836881 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_swa.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: swa +doc_to_text: "For mathematical questions provided in Swahili language. Supply the\ + \ accurate numeric answer to the provided question. \n\nQuestion: {{question}} \n\ + Answer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_swa_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0a0488295249513408b1d33de3a246b663cc523a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_twi.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: twi +doc_to_text: "For mathematical questions provided in Twi language. Supply the accurate\ + \ numeric answer to the provided question. \n\nQuestion: {{question}} \nAnswer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_twi_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c308cc7f57b90ddfe959132e75aad3cc5a0b6f01 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_xho.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: xho +doc_to_text: "For mathematical questions provided in isiXhosa language. Supply the\ + \ accurate numeric answer to the provided question. \n\nQuestion: {{question}} \n\ + Answer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_xho_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_yaml new file mode 100644 index 0000000000000000000000000000000000000000..2d3903948dd4a8ae2c285e33eb2551f554d2310c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_yaml @@ -0,0 +1,31 @@ +tag: afrimgsm_tt_tasks +dataset_path: masakhane/afrimgsm-translate-test +output_type: generate_until +test_split: test +doc_to_target: '{% if answer is not none %}{{answer[21:]}}{% else %}{{answer_number|string}}{% endif %}' +target_delimiter: "" +generation_kwargs: + do_sample: false + until: + - 'Question:' + - + - <|im_end|> +filter_list: + - name: remove_whitespace + filter: + - function: remove_whitespace + - function: take_first + - filter: + - function: regex + group_select: -1 + regex_pattern: (-?[$0-9.,]{2,})|(-?[0-9]+) + - function: take_first + name: flexible-extract +metric_list: + - metric: exact_match + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true +metadata: + version: 2.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f2a0a0fd28fc895e25d309a8f5aaf64769e7de6c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_yor.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: yor +doc_to_text: "For mathematical questions provided in Yoruba language. Supply the accurate\ + \ numeric answer to the provided question. \n\nQuestion: {{question}} \nAnswer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_yor_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/afrimgsm_tt_cot.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/afrimgsm_tt_cot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d43ddd233b785cbfba006785de6db94bb4eb5d97 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/afrimgsm_tt_cot.yaml @@ -0,0 +1,9 @@ +group: afrimgsm_tt_cot-irokobench +task: + - afrimgsm_tt_cot_tasks +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true +metadata: + version: 2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..da7764e81c0665c53c129f42d61629460a74ea1e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_amh.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: amh +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_amh_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_fra.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_fra.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a16b91ffb250880c9217e63d6e8c1e46c7d4021c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_fra.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fra +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_fra_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3bee8575de4f774c0ae7510e3a074023919dbbe3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_hau.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: hau +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_hau_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e6f495eaeb6738e0b8b9524341c1bc2453b4c00f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_ibo.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: ibo +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_ibo_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..83c9565d54a2786fea141b6775681172cf49b592 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_lug.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: lug +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_lug_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_sna.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_sna.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e99d3aa7b59d92691fc89687fcf951a993487f89 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_sna.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: sna +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_sna_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a0b4f9716cb900a71167b85a68ac402470aec3f0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_twi.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: twi +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_twi_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_vai.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_vai.yaml new file mode 100644 index 0000000000000000000000000000000000000000..76c18a3f9172645111dbcb26e0183b6d48f3fc69 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_vai.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: vai +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_vai_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cf093766dc687b5a092d984ab8cda6514ffd56a9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_yor.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: yor +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_yor_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1bb302a4d0f2268543c228ba13d0744509c4911d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_zul.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: zul +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_zul_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e9f97735373ee9e47f6234b27002f6f12edb1a13 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_amh.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: amh +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_amh_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_fra.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_fra.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b496c775817645ce477d84749de8e71f0badbc22 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_fra.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fra +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_fra_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4d8a986acd1dd95e9560555b44f2d3f5aed5395d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_kin.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: kin +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_kin_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_orm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_orm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b325b2ce931cbcf05cee50293845043081097387 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_orm.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: orm +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_orm_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_sna.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_sna.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3e85255881b279d8d4f578bfcbfd96355e8af3cd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_sna.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: sna +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_sna_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_vai.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_vai.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9186f1e00c540dc64bb89ca619439e8b927162d5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_vai.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: vai +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_vai_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..52a0e1ca5144bda20184bcc081768098d945a239 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_xho.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: xho +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_xho_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2452b0fae4a9f939ed115736056091302b9cfa78 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_yor.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: yor +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_yor_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2ce8151b79849bbc85be11e8cd535bf7a1e12ede --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_zul.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: zul +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_zul_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_ewe.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_ewe.yaml new file mode 100644 index 0000000000000000000000000000000000000000..52dc345f0cc5f2ad617db14a22327d4b2fc298bd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_ewe.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: ewe +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_ewe_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_fra.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_fra.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d2b7582c34e26f4e6e9cd87aa1c675788a23ccae --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_fra.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fra +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_fra_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d57be8c8c4c9dbaebaef523ff4bd9310df1ebc40 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_hau.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: hau +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_hau_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2b555b3e314cc6b50970f06649e7d1f662370073 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_kin.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: kin +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_kin_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_sot.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_sot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5990be74d1cceb58eff6a4f9648d1f3de0e11d8a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_sot.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: sot +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_sot_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..78ef85fc25c0da4894843624afad970c9ff572b0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_twi.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: twi +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_twi_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_vai.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_vai.yaml new file mode 100644 index 0000000000000000000000000000000000000000..25ec4e8fd71e2423370182180efc7b1bc7843e38 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_vai.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: vai +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_vai_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7815a0a5f4b77403334b41baebb2e529eda19d1a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_wol.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: wol +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_wol_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e45afd3a0ecc2ce8656e70ebfe29cad1f6ff06ba --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_xho.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: xho +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_xho_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..39e18cb48d1f214a13f5beb5a1e2c4d4f34855af --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_yor.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: yor +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_yor_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4f73f15f9ef2ba8f0d75e0a0a74cbf00e84cf8e5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_amh.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: amh +doc_to_text: "Answer the given question with the step by step solution appropriate\ + \ numerical value, ensuring that the response is clear and without any supplementary\ + \ information. \n\nQuestion: {{question}} \nStep by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_amh_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_fra.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_fra.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6a2f70ca6327e79044f7ed1685282ab8803fcc7d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_fra.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: fra +doc_to_text: "Answer the given question with the step by step solution appropriate\ + \ numerical value, ensuring that the response is clear and without any supplementary\ + \ information. \n\nQuestion: {{question}} \nStep by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_fra_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c5e7903c88c05de0ac942fd0894b523209dad320 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_hau.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: hau +doc_to_text: "Answer the given question with the step by step solution appropriate\ + \ numerical value, ensuring that the response is clear and without any supplementary\ + \ information. \n\nQuestion: {{question}} \nStep by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_hau_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c1b57c1395efb57ed2d8e31df9f2878cfbad59ad --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_kin.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: kin +doc_to_text: "Answer the given question with the step by step solution appropriate\ + \ numerical value, ensuring that the response is clear and without any supplementary\ + \ information. \n\nQuestion: {{question}} \nStep by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_kin_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_lin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_lin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..81cdb1ff41ede1fd2ceca5c5ebaf099df5df5d45 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_lin.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: lin +doc_to_text: "Answer the given question with the step by step solution appropriate\ + \ numerical value, ensuring that the response is clear and without any supplementary\ + \ information. \n\nQuestion: {{question}} \nStep by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_lin_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a949b0289211c9443a6a73880598c762a8c5a8f9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_lug.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: lug +doc_to_text: "Answer the given question with the step by step solution appropriate\ + \ numerical value, ensuring that the response is clear and without any supplementary\ + \ information. \n\nQuestion: {{question}} \nStep by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_lug_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_sot.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_sot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bf83c8f0209f627b01da77a7ab033ab93a276891 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_sot.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: sot +doc_to_text: "Answer the given question with the step by step solution appropriate\ + \ numerical value, ensuring that the response is clear and without any supplementary\ + \ information. \n\nQuestion: {{question}} \nStep by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_sot_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..87b581b8a5fdb5e9d2d0eede15f5c98b04f88693 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_swa.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: swa +doc_to_text: "Answer the given question with the step by step solution appropriate\ + \ numerical value, ensuring that the response is clear and without any supplementary\ + \ information. \n\nQuestion: {{question}} \nStep by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_swa_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..223901eb208af30d0a8dc549f7d021d343e17076 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_twi.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: twi +doc_to_text: "Answer the given question with the step by step solution appropriate\ + \ numerical value, ensuring that the response is clear and without any supplementary\ + \ information. \n\nQuestion: {{question}} \nStep by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_twi_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_vai.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_vai.yaml new file mode 100644 index 0000000000000000000000000000000000000000..92ce3892451070777235d493be10bbe9811ad05d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_vai.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: vai +doc_to_text: "Answer the given question with the step by step solution appropriate\ + \ numerical value, ensuring that the response is clear and without any supplementary\ + \ information. \n\nQuestion: {{question}} \nStep by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_vai_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c626fde4e599303ff73e490d3cbc38b6335d6755 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_wol.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: wol +doc_to_text: "Answer the given question with the step by step solution appropriate\ + \ numerical value, ensuring that the response is clear and without any supplementary\ + \ information. \n\nQuestion: {{question}} \nStep by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_wol_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..285b679cc0a7da31dc3d719918ead217d5287b9e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_xho.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: xho +doc_to_text: "Answer the given question with the step by step solution appropriate\ + \ numerical value, ensuring that the response is clear and without any supplementary\ + \ information. \n\nQuestion: {{question}} \nStep by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_xho_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_yaml new file mode 100644 index 0000000000000000000000000000000000000000..241787c7aa25a0ac46e2556efdb8db633e7a0719 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_yaml @@ -0,0 +1,32 @@ +tag: afrimgsm_tt_cot_tasks +dataset_path: masakhane/afrimgsm-translate-test +dataset_name: null # Overridden by language-specific config. +output_type: generate_until +test_split: test +doc_to_target: '{% if answer is not none %}{{answer[21:]}}{% else %}{{answer_number|string}}{% endif %}' +generation_kwargs: + do_sample: false + until: + - 'Question:' + - + - <|im_end|> +metric_list: + - metric: exact_match + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true +filter_list: + - name: "strict-match" + filter: + - function: "regex" + regex_pattern: "The answer is (\\-?[0-9\\.\\,]+)" + - function: "take_first" + - filter: + - function: regex + group_select: -1 + regex_pattern: (-?[$0-9.,]{2,})|(-?[0-9]+) + - function: take_first + name: flexible-extract +metadata: + version: 2.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f76f4cd109026c0c94a578674c4d8140201e5bfa --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_yor.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: yor +doc_to_text: "Answer the given question with the step by step solution appropriate\ + \ numerical value, ensuring that the response is clear and without any supplementary\ + \ information. \n\nQuestion: {{question}} \nStep by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_yor_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7023a5540612c3f7a568313f7d0d26e541839c49 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_zul.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: zul +doc_to_text: "Answer the given question with the step by step solution appropriate\ + \ numerical value, ensuring that the response is clear and without any supplementary\ + \ information. \n\nQuestion: {{question}} \nStep by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_zul_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d64f088f46c537580055f91f3eaa347187531da4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_amh.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: amh +doc_to_text: "For mathematical questions provided in Amharic language. Supply the\ + \ accurate step by step answer to the provided question. \n\nQuestion: {{question}}\ + \ \nStep by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_amh_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_ewe.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_ewe.yaml new file mode 100644 index 0000000000000000000000000000000000000000..de4aa48d48be2c3ea31b03cb497c4b881ab09ead --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_ewe.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: ewe +doc_to_text: "For mathematical questions provided in Ewe language. Supply the accurate\ + \ step by step answer to the provided question. \n\nQuestion: {{question}} \nStep\ + \ by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_ewe_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_fra.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_fra.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5cf15ea1ad1e7937cb46cf7f5c09014e00e7fea7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_fra.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: fra +doc_to_text: "For mathematical questions provided in French language. Supply the accurate\ + \ step by step answer to the provided question. \n\nQuestion: {{question}} \nStep\ + \ by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_fra_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0dfa643c4cf14b0bbd871f6293f5b142a6a0337a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_hau.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: hau +doc_to_text: "For mathematical questions provided in Hausa language. Supply the accurate\ + \ step by step answer to the provided question. \n\nQuestion: {{question}} \nStep\ + \ by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_hau_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..959f389070977606a0d330573c4c4015754b80d7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_ibo.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: ibo +doc_to_text: "For mathematical questions provided in Igbo language. Supply the accurate\ + \ step by step answer to the provided question. \n\nQuestion: {{question}} \nStep\ + \ by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_ibo_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..85ff4196d50b187779c8c39b62cfcb5448ddb258 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_kin.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: kin +doc_to_text: "For mathematical questions provided in Kinyarwanda language. Supply\ + \ the accurate step by step answer to the provided question. \n\nQuestion: {{question}}\ + \ \nStep by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_kin_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ac0fde85c1607b570abb2c0fa602d0f680ef3fe1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_lug.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: lug +doc_to_text: "For mathematical questions provided in Luganda language. Supply the\ + \ accurate step by step answer to the provided question. \n\nQuestion: {{question}}\ + \ \nStep by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_lug_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_orm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_orm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bcf34106d924c07a92e63eaacdd3f112e860c25d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_orm.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: orm +doc_to_text: "For mathematical questions provided in Oromo language. Supply the accurate\ + \ step by step answer to the provided question. \n\nQuestion: {{question}} \nStep\ + \ by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_orm_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_sot.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_sot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9fc015cdf75b34a6147183cfd3ad2f8bbe4d4660 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_sot.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: sot +doc_to_text: "For mathematical questions provided in Sesotho language. Supply the\ + \ accurate step by step answer to the provided question. \n\nQuestion: {{question}}\ + \ \nStep by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_sot_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..179af86738feaf0656c65e260cf465284c9bc3e5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_swa.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: swa +doc_to_text: "For mathematical questions provided in Swahili language. Supply the\ + \ accurate step by step answer to the provided question. \n\nQuestion: {{question}}\ + \ \nStep by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_swa_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ebb680a6f5e54727adabbdd517819e90ca9c2b96 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_twi.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: twi +doc_to_text: "For mathematical questions provided in Twi language. Supply the accurate\ + \ step by step answer to the provided question. \n\nQuestion: {{question}} \nStep\ + \ by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_twi_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_vai.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_vai.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3d2848648e5a752973ca5a21a38b0a2fb82b4127 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_vai.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: vai +doc_to_text: "For mathematical questions provided in Vai language. Supply the accurate\ + \ step by step answer to the provided question. \n\nQuestion: {{question}} \nStep\ + \ by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_vai_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..799cc29fbbca2a4a328cec6bf679cc68ad51139f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_wol.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: wol +doc_to_text: "For mathematical questions provided in Wolof language. Supply the accurate\ + \ step by step answer to the provided question. \n\nQuestion: {{question}} \nStep\ + \ by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_wol_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7969fdbabd911f8fe4ffdfb9f7e47364c3a80857 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_xho.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: xho +doc_to_text: "For mathematical questions provided in isiXhosa language. Supply the\ + \ accurate step by step answer to the provided question. \n\nQuestion: {{question}}\ + \ \nStep by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_xho_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_yaml new file mode 100644 index 0000000000000000000000000000000000000000..241787c7aa25a0ac46e2556efdb8db633e7a0719 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_yaml @@ -0,0 +1,32 @@ +tag: afrimgsm_tt_cot_tasks +dataset_path: masakhane/afrimgsm-translate-test +dataset_name: null # Overridden by language-specific config. +output_type: generate_until +test_split: test +doc_to_target: '{% if answer is not none %}{{answer[21:]}}{% else %}{{answer_number|string}}{% endif %}' +generation_kwargs: + do_sample: false + until: + - 'Question:' + - + - <|im_end|> +metric_list: + - metric: exact_match + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true +filter_list: + - name: "strict-match" + filter: + - function: "regex" + regex_pattern: "The answer is (\\-?[0-9\\.\\,]+)" + - function: "take_first" + - filter: + - function: regex + group_select: -1 + regex_pattern: (-?[$0-9.,]{2,})|(-?[0-9]+) + - function: take_first + name: flexible-extract +metadata: + version: 2.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0d05de223110d6e434bcf75bb2f9cf71957b76d3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_yor.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: yor +doc_to_text: "For mathematical questions provided in Yoruba language. Supply the accurate\ + \ step by step answer to the provided question. \n\nQuestion: {{question}} \nStep\ + \ by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_yor_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..68329068941f34bf7e53739334e8101c46a0ecfe --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_zul.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: zul +doc_to_text: "For mathematical questions provided in Zulu language. Supply the accurate\ + \ step by step answer to the provided question. \n\nQuestion: {{question}} \nStep\ + \ by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_zul_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/README.md b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/README.md new file mode 100644 index 0000000000000000000000000000000000000000..f7f7ed4d82f04224440a0d164d2cc24c0e758990 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/README.md @@ -0,0 +1,50 @@ +# MathQA + +### Paper + +IrokoBench: A New Benchmark for African Languages in the Age of Large Language Models +https://arxiv.org/pdf/2406.03368 + +IrokoBench is a human-translated benchmark dataset for 16 typologically diverse +low-resource African languages covering three tasks: natural language inference (AfriXNLI), +mathematical reasoning (AfriMGSM), and multi-choice knowledge-based QA (AfriMMLU). + + +### Citation + +``` +@misc{adelani2024irokobenchnewbenchmarkafrican, + title={IrokoBench: A New Benchmark for African Languages in the Age of Large Language Models}, + author={David Ifeoluwa Adelani and Jessica Ojo and Israel Abebe Azime and Jian Yun Zhuang and Jesujoba O. Alabi and Xuanli He and Millicent Ochieng and Sara Hooker and Andiswa Bukula and En-Shiun Annie Lee and Chiamaka Chukwuneke and Happy Buzaaba and Blessing Sibanda and Godson Kalipe and Jonathan Mukiibi and Salomon Kabongo and Foutse Yuehgoh and Mmasibidi Setaka and Lolwethu Ndolela and Nkiruka Odu and Rooweither Mabuya and Shamsuddeen Hassan Muhammad and Salomey Osei and Sokhar Samb and Tadesse Kebede Guge and Pontus Stenetorp}, + year={2024}, + eprint={2406.03368}, + archivePrefix={arXiv}, + primaryClass={cs.CL}, + url={https://arxiv.org/abs/2406.03368}, +} +``` + +### Groups and Tasks + +#### Groups + +* `afrimmlu`: All afrimmlu tasks +* `afrimmlu_direct`: afrimmlu_direct evaluates models performance on the curated dataset +* `afrimmlu_translate`: afrimmlu_translate evaluates models in translate-test setting + +#### Tasks +* `afrimmlu_direct_{language_code}`: each task evaluates for one language +* `afrimmlu_translate_{language_code}`: each task evaluates for one language + +### Checklist + +For adding novel benchmarks/datasets to the library: +* [x] Is the task an existing benchmark in the literature? + * [x] Have you referenced the original paper that introduced the task? + * [ ] If yes, does the original paper provide a reference implementation? If so, have you checked against the reference implementation and documented how to run such a test? + +If other tasks on this dataset are already supported: +* [x] Is the "Main" variant of this task clearly denoted? +* [x] Have you provided a short sentence in a README on what each new variant adds / evaluates? +* [x] Have you noted which, if any, published evaluation setups are matched by this variant? + * [x] Checked for equivalence with v0.3.0 LM Evaluation Harness diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct new file mode 100644 index 0000000000000000000000000000000000000000..a3e17f711f6eac83c52fad1d3f0314a01f08d169 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct @@ -0,0 +1,37 @@ +tag: + - afrimmlu_tasks + - afrimmlu_tasks_prompt_1 + - afrobench_mmlu_tasks +dataset_path: masakhane/afrimmlu +dataset_name: null +output_type: multiple_choice +validation_split: validation +test_split: test +fewshot_split: validation +doc_to_text: !function utils.doc_to_text +doc_to_target: "{{['A', 'B', 'C', 'D'].index(answer)}}" +doc_to_choice: !function utils.doc_to_choice +should_decontaminate: true +doc_to_decontamination_query: "Question: {{question}}\nAnswer:" +metric_list: + - metric: f1 + aggregation: !function utils.weighted_f1_score + # aggregation: mean + average: weighted + hf_evaluate: true + higher_is_better: True + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" + - metric: acc + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_eng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..18a34c7b719cdef3254e2472399b7fdd3121d543 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_eng.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: eng +include: afrimmlu_direct +task: afrimmlu_direct_eng_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_ewe.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_ewe.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e85bd7dc7dcc4fa2f5ea90f4f540a0a75b160dbd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_ewe.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: ewe +include: afrimmlu_direct +task: afrimmlu_direct_ewe_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b438ea3198a18caf74d44a97f2d4752337edc082 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_hau.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: hau +include: afrimmlu_direct +task: afrimmlu_direct_hau_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_lin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_lin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7059c941d2dffd27c8eda15dc1fc087a626455a2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_lin.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: lin +include: afrimmlu_direct +task: afrimmlu_direct_lin_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_orm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_orm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5047ae98abd0d3ea8ebeb98233ae4fb1ebb42dab --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_orm.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: orm +include: afrimmlu_direct +task: afrimmlu_direct_orm_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_sna.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_sna.yaml new file mode 100644 index 0000000000000000000000000000000000000000..17222f95253270fdcff74177fbd0474cb75660b3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_sna.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: sna +include: afrimmlu_direct +task: afrimmlu_direct_sna_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_sot.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_sot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c62ce9bf4957545dc39f96d7bd6dc60ce60e868a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_sot.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: sot +include: afrimmlu_direct +task: afrimmlu_direct_sot_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fb270c949a7a250e4f4810a5086a80ff22716f1f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_twi.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: twi +include: afrimmlu_direct +task: afrimmlu_direct_twi_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e3de56f8d3c9fe1f283712ea359ead734a413d93 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_yor.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: yor +include: afrimmlu_direct +task: afrimmlu_direct_yor_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..86c56fec097dc4c636070f6c0ab0750a23bb2435 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_zul.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: zul +include: afrimmlu_direct +task: afrimmlu_direct_zul_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct new file mode 100644 index 0000000000000000000000000000000000000000..fefabf7e0b52e644d1e9d922c8f899607eab6075 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct @@ -0,0 +1,37 @@ +tag: + - afrimmlu_tasks + - afrimmlu_tasks_prompt_2 + - afrobench_mmlu_tasks +dataset_path: masakhane/afrimmlu +dataset_name: null +output_type: multiple_choice +validation_split: validation +test_split: test +fewshot_split: validation +doc_to_text: !function utils.doc_to_text +doc_to_target: "{{['A', 'B', 'C', 'D'].index(answer)}}" +doc_to_choice: !function utils.doc_to_choice +should_decontaminate: true +doc_to_decontamination_query: "Question: {{question}}\nAnswer:" +metric_list: + - metric: f1 + aggregation: !function utils.weighted_f1_score + # aggregation: mean + average: weighted + hf_evaluate: true + higher_is_better: True + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" + - metric: acc + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_eng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c46eca5e68356372fc43c1b1908e45667ff05d12 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_eng.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: eng +include: afrimmlu_direct +task: afrimmlu_direct_eng_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_ewe.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_ewe.yaml new file mode 100644 index 0000000000000000000000000000000000000000..26acfcfa93b3019a746a5e9a78e4cdb48871c978 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_ewe.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: ewe +include: afrimmlu_direct +task: afrimmlu_direct_ewe_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0cf7db0e4c0585c2cde8f0645e64438546ba5818 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_ibo.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: ibo +include: afrimmlu_direct +task: afrimmlu_direct_ibo_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..963f7cd2cdee47fda381af3cbe1a63c56601b709 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_wol.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: wol +include: afrimmlu_direct +task: afrimmlu_direct_wol_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9da0589a8bfe5791ff3764fd26e1f375a836b701 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_xho.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: xho +include: afrimmlu_direct +task: afrimmlu_direct_xho_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..39b365418eda1be6e4c344d4e32717045d2eafda --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_yor.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: yor +include: afrimmlu_direct +task: afrimmlu_direct_yor_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/fewshot.sh b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/fewshot.sh new file mode 100644 index 0000000000000000000000000000000000000000..c69c48d7dff4e2495485023187dc162742c7ca6a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/fewshot.sh @@ -0,0 +1,8 @@ +lm_eval --model hf \ + --model_args pretrained=masakhane/African-ultrachat-alpaca \ + --tasks afrimmlu_direct_amh,afrimmlu_direct_eng,afrimmlu_direct_ewe,afrimmlu_direct_fra,afrimmlu_direct_hau,afrimmlu_direct_ibo,afrimmlu_direct_kin,afrimmlu_direct_lin,afrimmlu_direct_lug,afrimmlu_direct_orm,afrimmlu_direct_sna,afrimmlu_direct_sot,afrimmlu_direct_twi,afrimmlu_direct_wol,afrimmlu_direct_xho,afrimmlu_direct_yor,afrimmlu_direct_zul \ + --device cuda:0 \ + --batch_size 1 \ + --num_fewshot 0 \ + --verbosity DEBUG \ + --wandb_args project=afrimmlu diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..9d02b342b2e3c9f3d3bd66d3f62330aa53c9159c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/utils.py @@ -0,0 +1,32 @@ +from lm_eval.utils import weighted_f1_score + + +def doc_to_choice(doc): + choices = eval(doc["choices"]) + return choices + + +def doc_to_text(doc): + output = """You are a highly knowledgeable and intelligent artificial intelligence + model answers multiple-choice questions about '{subject}' + + Question: '''{question}''' + + Choices: + A: ''{choice1}''' + B: ''{choice2}''' + C: ''{choice3}''' + D: ''{choice4}''' + + Answer: """ + + choices = eval(doc["choices"]) + text = output.format( + subject=doc["subject"], + question=doc["question"], + choice1=choices[0], + choice2=choices[1], + choice3=choices[2], + choice4=choices[3], + ) + return text diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/direct/prompt_5/afrixnli_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/direct/prompt_5/afrixnli_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1891bfd0592a8d3c6e2f0e98619bbeee234a852f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/direct/prompt_5/afrixnli_xho.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: xho +doc_to_text: "Based on the given statement, is the following claim 'true', 'false',\ + \ or 'inconclusive'. \nStatement: {{premise}} \nClaim: {{hypothesis}}" +include: afrixnli_yaml +task: afrixnli_xho_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/lai prompt/translate/afrixnli_manual_translate_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/lai prompt/translate/afrixnli_manual_translate_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..aa79494a59c3d529cefe3afc6793c113136ba4a9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/lai prompt/translate/afrixnli_manual_translate_amh.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: amh +include: afrixnli_manual_translate_yaml +task: afrixnli_manual_translate_amh diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/lai prompt/translate/afrixnli_manual_translate_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/lai prompt/translate/afrixnli_manual_translate_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5874ae5f6cd2cac4296b3abaa5568a4dd7d2188a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/lai prompt/translate/afrixnli_manual_translate_kin.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: kin +include: afrixnli_manual_translate_yaml +task: afrixnli_manual_translate_kin diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/lai prompt/translate/afrixnli_manual_translate_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/lai prompt/translate/afrixnli_manual_translate_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1825ec27a5eb16b4229385555d699d33338d7c86 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/lai prompt/translate/afrixnli_manual_translate_yor.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: yor +include: afrixnli_manual_translate_yaml +task: afrixnli_manual_translate_yor diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/lai prompt/translate/afrixnli_manual_translate_zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/lai prompt/translate/afrixnli_manual_translate_zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4bf52549d730596f90caf263fe6299bbc705095b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/lai prompt/translate/afrixnli_manual_translate_zul.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: zul +include: afrixnli_manual_translate_yaml +task: afrixnli_manual_translate_zul diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_1/afrixnli_translate_fra.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_1/afrixnli_translate_fra.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3dc72af6f3dbcc41ab83a2adcf384b952e97a4d5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_1/afrixnli_translate_fra.yaml @@ -0,0 +1,15 @@ +# Generated by utils.py +dataset_name: fra +doc_to_text: 'Please identify whether the premise entails or contradicts the hypothesis + in the following premise and hypothesis. The answer should be exact entailment, + contradiction, or neutral. + + + Premise: {premise} + + Hypothesis: {hypothesis} + + + Is it entailment, contradiction, or neutral?' +include: afrixnli_translate_yaml +task: afrixnli_translate_fra_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_1/afrixnli_translate_lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_1/afrixnli_translate_lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bf45d957a3d4f169630b3c3405b348a0ceefe1b1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_1/afrixnli_translate_lug.yaml @@ -0,0 +1,15 @@ +# Generated by utils.py +dataset_name: lug +doc_to_text: 'Please identify whether the premise entails or contradicts the hypothesis + in the following premise and hypothesis. The answer should be exact entailment, + contradiction, or neutral. + + + Premise: {premise} + + Hypothesis: {hypothesis} + + + Is it entailment, contradiction, or neutral?' +include: afrixnli_translate_yaml +task: afrixnli_translate_lug_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_1/afrixnli_translate_sna.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_1/afrixnli_translate_sna.yaml new file mode 100644 index 0000000000000000000000000000000000000000..13103dd7a2f339f02b280dd3c67d8ec27807c86a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_1/afrixnli_translate_sna.yaml @@ -0,0 +1,15 @@ +# Generated by utils.py +dataset_name: sna +doc_to_text: 'Please identify whether the premise entails or contradicts the hypothesis + in the following premise and hypothesis. The answer should be exact entailment, + contradiction, or neutral. + + + Premise: {premise} + + Hypothesis: {hypothesis} + + + Is it entailment, contradiction, or neutral?' +include: afrixnli_translate_yaml +task: afrixnli_translate_sna_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_1/afrixnli_translate_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_1/afrixnli_translate_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..824bb17aa23f3a1a546d369a6d3118272ee05c54 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_1/afrixnli_translate_swa.yaml @@ -0,0 +1,15 @@ +# Generated by utils.py +dataset_name: swa +doc_to_text: 'Please identify whether the premise entails or contradicts the hypothesis + in the following premise and hypothesis. The answer should be exact entailment, + contradiction, or neutral. + + + Premise: {premise} + + Hypothesis: {hypothesis} + + + Is it entailment, contradiction, or neutral?' +include: afrixnli_translate_yaml +task: afrixnli_translate_swa_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_1/afrixnli_translate_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_1/afrixnli_translate_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9d971c3e5fe6ca65e7568da81b647d2ff8f20696 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_1/afrixnli_translate_twi.yaml @@ -0,0 +1,15 @@ +# Generated by utils.py +dataset_name: twi +doc_to_text: 'Please identify whether the premise entails or contradicts the hypothesis + in the following premise and hypothesis. The answer should be exact entailment, + contradiction, or neutral. + + + Premise: {premise} + + Hypothesis: {hypothesis} + + + Is it entailment, contradiction, or neutral?' +include: afrixnli_translate_yaml +task: afrixnli_translate_twi_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_1/afrixnli_translate_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_1/afrixnli_translate_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c907a2bf453d99c048822ad93feb12781004d2d0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_1/afrixnli_translate_xho.yaml @@ -0,0 +1,15 @@ +# Generated by utils.py +dataset_name: xho +doc_to_text: 'Please identify whether the premise entails or contradicts the hypothesis + in the following premise and hypothesis. The answer should be exact entailment, + contradiction, or neutral. + + + Premise: {premise} + + Hypothesis: {hypothesis} + + + Is it entailment, contradiction, or neutral?' +include: afrixnli_translate_yaml +task: afrixnli_translate_xho_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_1/afrixnli_translate_yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_1/afrixnli_translate_yaml new file mode 100644 index 0000000000000000000000000000000000000000..832b51493a0f17996ebca680f7151e38b59168d3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_1/afrixnli_translate_yaml @@ -0,0 +1,27 @@ +tag: afrixnli_tt_tasks +dataset_path: masakhane/afrixnli-translate-test +dataset_name: null +output_type: multiple_choice +test_split: test +fewshot_split: test +doc_to_target: !function utils.doc_to_target +doc_to_choice: + - "entailment" + - "neutral" + - "contradiction" +should_decontaminate: true +doc_to_decontamination_query: premise +metric_list: + - metric: f1 + aggregation: !function utils.weighted_f1_score + average: weighted + higher_is_better: True + ignore_case: true + ignore_punctuation: true + - metric: acc + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_2/afrixnli_translate_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_2/afrixnli_translate_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0810f6b37b9c83815fab4de50d3bc42b2c01624e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_2/afrixnli_translate_amh.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: amh +include: afrixnli_translate_yaml +task: afrixnli_translate_amh_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_2/afrixnli_translate_ewe.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_2/afrixnli_translate_ewe.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d7aec16a61ab04415327addd832893ab6e4e531c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_2/afrixnli_translate_ewe.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: ewe +include: afrixnli_translate_yaml +task: afrixnli_translate_ewe_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_2/afrixnli_translate_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_2/afrixnli_translate_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..688778c3195c54868f0f3d1d9f56e17c167205f9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_2/afrixnli_translate_hau.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: hau +include: afrixnli_translate_yaml +task: afrixnli_translate_hau_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_2/afrixnli_translate_lin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_2/afrixnli_translate_lin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..27e88a5b7664a515e7d6cc901962f201913670d6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_2/afrixnli_translate_lin.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: lin +include: afrixnli_translate_yaml +task: afrixnli_translate_lin_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_2/afrixnli_translate_orm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_2/afrixnli_translate_orm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..db1a3ea1cbeff8ddd8c5b2ba49d3d553410c1b06 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_2/afrixnli_translate_orm.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: orm +include: afrixnli_translate_yaml +task: afrixnli_translate_orm_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_2/afrixnli_translate_sna.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_2/afrixnli_translate_sna.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fa110774a6fa86e29253307163e22c82f35a9ac0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_2/afrixnli_translate_sna.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: sna +include: afrixnli_translate_yaml +task: afrixnli_translate_sna_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_2/afrixnli_translate_sot.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_2/afrixnli_translate_sot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3133308b2810a3a61e1037ebe58c02ba0da0a2ef --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_2/afrixnli_translate_sot.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: sot +include: afrixnli_translate_yaml +task: afrixnli_translate_sot_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_2/afrixnli_translate_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_2/afrixnli_translate_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c229de3dd2b40b902e7ef93de0cc06a6da04d8af --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_2/afrixnli_translate_twi.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: twi +include: afrixnli_translate_yaml +task: afrixnli_translate_twi_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_2/afrixnli_translate_wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_2/afrixnli_translate_wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..87844c49639fa350e52877d967f64d5ea95cf28e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_2/afrixnli_translate_wol.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: wol +include: afrixnli_translate_yaml +task: afrixnli_translate_wol_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_2/afrixnli_translate_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_2/afrixnli_translate_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..63fa3ffc1c69d89637dca73da5050c840593aa4a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_2/afrixnli_translate_xho.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: xho +include: afrixnli_translate_yaml +task: afrixnli_translate_xho_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_2/afrixnli_translate_yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_2/afrixnli_translate_yaml new file mode 100644 index 0000000000000000000000000000000000000000..8ad87afcd99a5c875ccf1f7d2761afdea2a118aa --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_2/afrixnli_translate_yaml @@ -0,0 +1,31 @@ +tag: afrixnli_tt_tasks +dataset_path: masakhane/afrixnli-translate-test +dataset_name: null +output_type: multiple_choice +test_split: test +fewshot_split: test +doc_to_text: "{{premise}}\nQuestion: {{hypothesis}} True, False, or Neither?\nAnswer:" +# True = entailment +# False = contradiction +# Neither = neutral +doc_to_target: !function utils.doc_to_target +doc_to_choice: + - "True" + - "Neither" + - "False" +should_decontaminate: true +doc_to_decontamination_query: premise +metric_list: + - metric: f1 + aggregation: !function utils.weighted_f1_score + average: weighted + higher_is_better: True + ignore_case: true + ignore_punctuation: true + - metric: acc + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_2/afrixnli_translate_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_2/afrixnli_translate_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7dfc9bd6d46fb8982402dd85531d1d312a8a07b8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_2/afrixnli_translate_yor.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: yor +include: afrixnli_translate_yaml +task: afrixnli_translate_yor_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_2/afrixnli_translate_zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_2/afrixnli_translate_zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0878c4e03e985eced95c234fa079c3d92999a982 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_2/afrixnli_translate_zul.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: zul +include: afrixnli_translate_yaml +task: afrixnli_translate_zul_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_2/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_2/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..5d1ac19e19b2e855c957e75f1c778366dfbc7e55 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_2/utils.py @@ -0,0 +1,6 @@ +from lm_eval.utils import weighted_f1_score + + +def doc_to_target(doc): + replacements = {0: "True", 1: "Neither", 2: "False"} + return replacements[doc["label"]] diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6fb06d0f1f7dc3c62e7c51a6395b1a79f1759878 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_amh.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: amh +doc_to_text: "Given the following premise and hypothesis in Amharic, identify if the\ + \ premise entails, contradicts, or is neutral towards the hypothesis. Please respond\ + \ with exact 'entailment', 'contradiction', or 'neutral'. \n\nPremise: {{premise}}\ + \ \nHypothesis: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_amh_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_ewe.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_ewe.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0d550f9daff83651b4099d3ad4228de0afab6ac4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_ewe.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: ewe +doc_to_text: "Given the following premise and hypothesis in Ewe, identify if the premise\ + \ entails, contradicts, or is neutral towards the hypothesis. Please respond with\ + \ exact 'entailment', 'contradiction', or 'neutral'. \n\nPremise: {{premise}} \n\ + Hypothesis: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_ewe_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_fra.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_fra.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3156466c388362d06414b613e963f0f9fcb1465f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_fra.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: fra +doc_to_text: "Given the following premise and hypothesis in French, identify if the\ + \ premise entails, contradicts, or is neutral towards the hypothesis. Please respond\ + \ with exact 'entailment', 'contradiction', or 'neutral'. \n\nPremise: {{premise}}\ + \ \nHypothesis: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_fra_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6981da83291b927f5fd5908f62e039227ded3a9e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_kin.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: kin +doc_to_text: "Given the following premise and hypothesis in Kinyarwanda, identify\ + \ if the premise entails, contradicts, or is neutral towards the hypothesis. Please\ + \ respond with exact 'entailment', 'contradiction', or 'neutral'. \n\nPremise: {{premise}}\ + \ \nHypothesis: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_kin_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_lin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_lin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1984416f0a828871d5e29ca46bde91c300440feb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_lin.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: lin +doc_to_text: "Given the following premise and hypothesis in Lingala, identify if the\ + \ premise entails, contradicts, or is neutral towards the hypothesis. Please respond\ + \ with exact 'entailment', 'contradiction', or 'neutral'. \n\nPremise: {{premise}}\ + \ \nHypothesis: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_lin_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..32a7ad2a161818c6d84b9002d6690b95cc86af3c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_lug.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: lug +doc_to_text: "Given the following premise and hypothesis in Luganda, identify if the\ + \ premise entails, contradicts, or is neutral towards the hypothesis. Please respond\ + \ with exact 'entailment', 'contradiction', or 'neutral'. \n\nPremise: {{premise}}\ + \ \nHypothesis: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_lug_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_orm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_orm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d3923a80cdc78c59dabb63bb3fe9dda6e8e572a9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_orm.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: orm +doc_to_text: "Given the following premise and hypothesis in Oromo, identify if the\ + \ premise entails, contradicts, or is neutral towards the hypothesis. Please respond\ + \ with exact 'entailment', 'contradiction', or 'neutral'. \n\nPremise: {{premise}}\ + \ \nHypothesis: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_orm_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_sna.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_sna.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d7dbf17e8b0520fc15bc6d6b337c959f828268ef --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_sna.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: sna +doc_to_text: "Given the following premise and hypothesis in chiShona, identify if\ + \ the premise entails, contradicts, or is neutral towards the hypothesis. Please\ + \ respond with exact 'entailment', 'contradiction', or 'neutral'. \n\nPremise: {{premise}}\ + \ \nHypothesis: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_sna_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_sot.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_sot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c4e89ec90608180f281765d321c6c1b0220171e0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_sot.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: sot +doc_to_text: "Given the following premise and hypothesis in Sesotho, identify if the\ + \ premise entails, contradicts, or is neutral towards the hypothesis. Please respond\ + \ with exact 'entailment', 'contradiction', or 'neutral'. \n\nPremise: {{premise}}\ + \ \nHypothesis: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_sot_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a3c5243b35264d5b26065a63c9c3babb8c714ec7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_swa.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: swa +doc_to_text: "Given the following premise and hypothesis in Swahili, identify if the\ + \ premise entails, contradicts, or is neutral towards the hypothesis. Please respond\ + \ with exact 'entailment', 'contradiction', or 'neutral'. \n\nPremise: {{premise}}\ + \ \nHypothesis: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_swa_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e7e8568701be5958b2da080d5d6c0885e83bb370 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_twi.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: twi +doc_to_text: "Given the following premise and hypothesis in Twi, identify if the premise\ + \ entails, contradicts, or is neutral towards the hypothesis. Please respond with\ + \ exact 'entailment', 'contradiction', or 'neutral'. \n\nPremise: {{premise}} \n\ + Hypothesis: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_twi_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2cf0b08eff851ad0832adad093f6338222f3bc74 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_wol.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: wol +doc_to_text: "Given the following premise and hypothesis in Wolof, identify if the\ + \ premise entails, contradicts, or is neutral towards the hypothesis. Please respond\ + \ with exact 'entailment', 'contradiction', or 'neutral'. \n\nPremise: {{premise}}\ + \ \nHypothesis: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_wol_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a4dafa34f6cd5b69e115de88e04cd24b3473c5fd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_xho.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: xho +doc_to_text: "Given the following premise and hypothesis in isiXhosa, identify if\ + \ the premise entails, contradicts, or is neutral towards the hypothesis. Please\ + \ respond with exact 'entailment', 'contradiction', or 'neutral'. \n\nPremise: {{premise}}\ + \ \nHypothesis: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_xho_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_yaml new file mode 100644 index 0000000000000000000000000000000000000000..832b51493a0f17996ebca680f7151e38b59168d3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_yaml @@ -0,0 +1,27 @@ +tag: afrixnli_tt_tasks +dataset_path: masakhane/afrixnli-translate-test +dataset_name: null +output_type: multiple_choice +test_split: test +fewshot_split: test +doc_to_target: !function utils.doc_to_target +doc_to_choice: + - "entailment" + - "neutral" + - "contradiction" +should_decontaminate: true +doc_to_decontamination_query: premise +metric_list: + - metric: f1 + aggregation: !function utils.weighted_f1_score + average: weighted + higher_is_better: True + ignore_case: true + ignore_punctuation: true + - metric: acc + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f5c01ca54eaba74fd968ce7847f43b2fe4b373fd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_yor.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: yor +doc_to_text: "Given the following premise and hypothesis in Yoruba, identify if the\ + \ premise entails, contradicts, or is neutral towards the hypothesis. Please respond\ + \ with exact 'entailment', 'contradiction', or 'neutral'. \n\nPremise: {{premise}}\ + \ \nHypothesis: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_yor_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..c455a3045a9be8b7318b96e23d9f061add6a342e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/utils.py @@ -0,0 +1,21 @@ +from lm_eval.utils import weighted_f1_score + + +def doc_to_text(doc): + output = """You are an NLP assistant whose purpose is to solve Natural Language Inference (NLI) problems + + Please identify whether the premise entails or contradicts the hypothesis in the following premise + and hypothesis. The answer should be exact entailment, contradiction, or neutral. + + Premise: {premise} + Hypothesis: {hypothesis} + + Is it entailment, contradiction, or neutral?""" + + text = output.format(premise=doc["premise"], hypothesis=doc["hypothesis"]) + return text + + +def doc_to_target(doc): + replacements = {0: "entailment", 1: "neutral", 2: "contradiction"} + return replacements[doc["label"]] diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b5f972e7070ac1ed50b7ed177daa706d520e3a4f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_amh.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: amh +doc_to_text: "You are an expert in Natural Language Inference (NLI) specializing in\ + \ the Amharic language.\nAnalyze the premise and hypothesis given in Amharic, and\ + \ determine the relationship between them.\n Respond with one of the following options:\ + \ 'entailment', 'contradiction', or 'neutral'. \n\nPremise: {{premise}} \nHypothesis:\ + \ {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_amh_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_fra.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_fra.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2ad718c708ddb9c82c3e0ac35c7b805c0c364a64 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_fra.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: fra +doc_to_text: "You are an expert in Natural Language Inference (NLI) specializing in\ + \ the French language.\nAnalyze the premise and hypothesis given in French, and\ + \ determine the relationship between them.\n Respond with one of the following options:\ + \ 'entailment', 'contradiction', or 'neutral'. \n\nPremise: {{premise}} \nHypothesis:\ + \ {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_fra_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dd65f366bc2de430404434c7918e3a0bb69aba03 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_hau.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: hau +doc_to_text: "You are an expert in Natural Language Inference (NLI) specializing in\ + \ the Hausa language.\nAnalyze the premise and hypothesis given in Hausa, and determine\ + \ the relationship between them.\n Respond with one of the following options: 'entailment',\ + \ 'contradiction', or 'neutral'. \n\nPremise: {{premise}} \nHypothesis: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_hau_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..13df12642e743ceb4867a4d515ed2d20edce7486 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_ibo.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: ibo +doc_to_text: "You are an expert in Natural Language Inference (NLI) specializing in\ + \ the Igbo language.\nAnalyze the premise and hypothesis given in Igbo, and determine\ + \ the relationship between them.\n Respond with one of the following options: 'entailment',\ + \ 'contradiction', or 'neutral'. \n\nPremise: {{premise}} \nHypothesis: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_ibo_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..198d88750287ee0fc28507ddcdab0146b1c2f734 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_kin.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: kin +doc_to_text: "You are an expert in Natural Language Inference (NLI) specializing in\ + \ the Kinyarwanda language.\nAnalyze the premise and hypothesis given in Kinyarwanda,\ + \ and determine the relationship between them.\n Respond with one of the following\ + \ options: 'entailment', 'contradiction', or 'neutral'. \n\nPremise: {{premise}}\ + \ \nHypothesis: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_kin_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_lin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_lin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b25856cfdcbb7b0c0df2aba81322d650585e9d9a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_lin.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: lin +doc_to_text: "You are an expert in Natural Language Inference (NLI) specializing in\ + \ the Lingala language.\nAnalyze the premise and hypothesis given in Lingala, and\ + \ determine the relationship between them.\n Respond with one of the following options:\ + \ 'entailment', 'contradiction', or 'neutral'. \n\nPremise: {{premise}} \nHypothesis:\ + \ {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_lin_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..633c173c0b75903c81934391e1e9f07a8de9b7f7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_lug.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: lug +doc_to_text: "You are an expert in Natural Language Inference (NLI) specializing in\ + \ the Luganda language.\nAnalyze the premise and hypothesis given in Luganda, and\ + \ determine the relationship between them.\n Respond with one of the following options:\ + \ 'entailment', 'contradiction', or 'neutral'. \n\nPremise: {{premise}} \nHypothesis:\ + \ {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_lug_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_orm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_orm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e63f93eb89351af076593eff1623123301bbda9c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_orm.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: orm +doc_to_text: "You are an expert in Natural Language Inference (NLI) specializing in\ + \ the Oromo language.\nAnalyze the premise and hypothesis given in Oromo, and determine\ + \ the relationship between them.\n Respond with one of the following options: 'entailment',\ + \ 'contradiction', or 'neutral'. \n\nPremise: {{premise}} \nHypothesis: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_orm_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_sna.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_sna.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6fcb4e063159ab48e9e73423b895658a74cafa9e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_sna.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: sna +doc_to_text: "You are an expert in Natural Language Inference (NLI) specializing in\ + \ the chiShona language.\nAnalyze the premise and hypothesis given in chiShona,\ + \ and determine the relationship between them.\n Respond with one of the following\ + \ options: 'entailment', 'contradiction', or 'neutral'. \n\nPremise: {{premise}}\ + \ \nHypothesis: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_sna_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_sot.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_sot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..358e4b353eb73065bf91b5471127faf5b2f1675f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_sot.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: sot +doc_to_text: "You are an expert in Natural Language Inference (NLI) specializing in\ + \ the Sesotho language.\nAnalyze the premise and hypothesis given in Sesotho, and\ + \ determine the relationship between them.\n Respond with one of the following options:\ + \ 'entailment', 'contradiction', or 'neutral'. \n\nPremise: {{premise}} \nHypothesis:\ + \ {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_sot_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8ce271ed77c52dbce949a6a83ccd1313d26c9b25 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_swa.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: swa +doc_to_text: "You are an expert in Natural Language Inference (NLI) specializing in\ + \ the Swahili language.\nAnalyze the premise and hypothesis given in Swahili, and\ + \ determine the relationship between them.\n Respond with one of the following options:\ + \ 'entailment', 'contradiction', or 'neutral'. \n\nPremise: {{premise}} \nHypothesis:\ + \ {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_swa_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8171e0daa98aa93614800f3c77a42d2b699ee2a7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_twi.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: twi +doc_to_text: "You are an expert in Natural Language Inference (NLI) specializing in\ + \ the Twi language.\nAnalyze the premise and hypothesis given in Twi, and determine\ + \ the relationship between them.\n Respond with one of the following options: 'entailment',\ + \ 'contradiction', or 'neutral'. \n\nPremise: {{premise}} \nHypothesis: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_twi_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b2662662dff2263023dd2bce1afeb86bb09ce262 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_wol.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: wol +doc_to_text: "You are an expert in Natural Language Inference (NLI) specializing in\ + \ the Wolof language.\nAnalyze the premise and hypothesis given in Wolof, and determine\ + \ the relationship between them.\n Respond with one of the following options: 'entailment',\ + \ 'contradiction', or 'neutral'. \n\nPremise: {{premise}} \nHypothesis: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_wol_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5aa3a9d171a4dabe831d5a6126ca61384d218d81 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_xho.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: xho +doc_to_text: "You are an expert in Natural Language Inference (NLI) specializing in\ + \ the isiXhosa language.\nAnalyze the premise and hypothesis given in isiXhosa,\ + \ and determine the relationship between them.\n Respond with one of the following\ + \ options: 'entailment', 'contradiction', or 'neutral'. \n\nPremise: {{premise}}\ + \ \nHypothesis: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_xho_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_yaml new file mode 100644 index 0000000000000000000000000000000000000000..832b51493a0f17996ebca680f7151e38b59168d3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_yaml @@ -0,0 +1,27 @@ +tag: afrixnli_tt_tasks +dataset_path: masakhane/afrixnli-translate-test +dataset_name: null +output_type: multiple_choice +test_split: test +fewshot_split: test +doc_to_target: !function utils.doc_to_target +doc_to_choice: + - "entailment" + - "neutral" + - "contradiction" +should_decontaminate: true +doc_to_decontamination_query: premise +metric_list: + - metric: f1 + aggregation: !function utils.weighted_f1_score + average: weighted + higher_is_better: True + ignore_case: true + ignore_punctuation: true + - metric: acc + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..478e5043431c73fe2448474b593ae02002d6a722 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_yor.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: yor +doc_to_text: "You are an expert in Natural Language Inference (NLI) specializing in\ + \ the Yoruba language.\nAnalyze the premise and hypothesis given in Yoruba, and\ + \ determine the relationship between them.\n Respond with one of the following options:\ + \ 'entailment', 'contradiction', or 'neutral'. \n\nPremise: {{premise}} \nHypothesis:\ + \ {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_yor_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c0dc06e6bce0aae18769cd2259aee6699be3b0bd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_zul.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: zul +doc_to_text: "You are an expert in Natural Language Inference (NLI) specializing in\ + \ the Zulu language.\nAnalyze the premise and hypothesis given in Zulu, and determine\ + \ the relationship between them.\n Respond with one of the following options: 'entailment',\ + \ 'contradiction', or 'neutral'. \n\nPremise: {{premise}} \nHypothesis: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_zul_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..d97a0a288508e817ab695e637fb157a08c813808 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/utils.py @@ -0,0 +1,19 @@ +from lm_eval.utils import weighted_f1_score + + +def doc_to_text(doc): + output = """Please identify whether the premise entails or contradicts the hypothesis in the following premise + and hypothesis. The answer should be exact entailment, contradiction, or neutral. + + Premise: {premise} + Hypothesis: {hypothesis} + + Is it entailment, contradiction, or neutral?""" + + text = output.format(premise=doc["premise"], hypothesis=doc["hypothesis"]) + return text + + +def doc_to_target(doc): + replacements = {0: "entailment", 1: "neutral", 2: "contradiction"} + return replacements[doc["label"]] diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3079712ce60b0b9b7e5846d3e1d9b16383c8cf97 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_amh.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: amh +doc_to_text: "Based on the given statement, is the following claim 'true', 'false',\ + \ or 'inconclusive'. \nStatement: {{premise}} \nClaim: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_amh_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_ewe.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_ewe.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6eb452db53361c5c048e75a807de20b3528414ec --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_ewe.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: ewe +doc_to_text: "Based on the given statement, is the following claim 'true', 'false',\ + \ or 'inconclusive'. \nStatement: {{premise}} \nClaim: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_ewe_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_fra.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_fra.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d6ddf49332ba4ec1671a7a20e036e7d4906c2097 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_fra.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: fra +doc_to_text: "Based on the given statement, is the following claim 'true', 'false',\ + \ or 'inconclusive'. \nStatement: {{premise}} \nClaim: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_fra_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..09d182f7f12e9debba53ed9bd5b1249c38b63a53 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_hau.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: hau +doc_to_text: "Based on the given statement, is the following claim 'true', 'false',\ + \ or 'inconclusive'. \nStatement: {{premise}} \nClaim: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_hau_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b5bf1555a454879018ee16c3ed60dd1f71cbdbbe --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_ibo.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: ibo +doc_to_text: "Based on the given statement, is the following claim 'true', 'false',\ + \ or 'inconclusive'. \nStatement: {{premise}} \nClaim: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_ibo_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f0cbe9c2c78612111af3ce29e4a8b846879f3060 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_kin.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: kin +doc_to_text: "Based on the given statement, is the following claim 'true', 'false',\ + \ or 'inconclusive'. \nStatement: {{premise}} \nClaim: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_kin_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_lin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_lin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..159116be57762adddff7368d1731a867a8daa152 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_lin.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: lin +doc_to_text: "Based on the given statement, is the following claim 'true', 'false',\ + \ or 'inconclusive'. \nStatement: {{premise}} \nClaim: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_lin_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9448fa28cb2561631dfa78fbeccb4bc054c867f6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_lug.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: lug +doc_to_text: "Based on the given statement, is the following claim 'true', 'false',\ + \ or 'inconclusive'. \nStatement: {{premise}} \nClaim: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_lug_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_orm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_orm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..64621cb4b784c2e4c16bfe220acb69d7ea17cb8f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_orm.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: orm +doc_to_text: "Based on the given statement, is the following claim 'true', 'false',\ + \ or 'inconclusive'. \nStatement: {{premise}} \nClaim: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_orm_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_sna.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_sna.yaml new file mode 100644 index 0000000000000000000000000000000000000000..788bae3068b2b40ed1ae4419a8ddbfc9094fe71c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_sna.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: sna +doc_to_text: "Based on the given statement, is the following claim 'true', 'false',\ + \ or 'inconclusive'. \nStatement: {{premise}} \nClaim: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_sna_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_sot.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_sot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..617dd9f88db6d283eef224126b36a0fcf2e35158 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_sot.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: sot +doc_to_text: "Based on the given statement, is the following claim 'true', 'false',\ + \ or 'inconclusive'. \nStatement: {{premise}} \nClaim: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_sot_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..81a159252ca45302bf5c88c448489a76bd342270 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_swa.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: swa +doc_to_text: "Based on the given statement, is the following claim 'true', 'false',\ + \ or 'inconclusive'. \nStatement: {{premise}} \nClaim: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_swa_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cb9f115fb3c01ea8a509a7347f6a369ae1f9c819 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_twi.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: twi +doc_to_text: "Based on the given statement, is the following claim 'true', 'false',\ + \ or 'inconclusive'. \nStatement: {{premise}} \nClaim: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_twi_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a5f4eb0c2eeaf10d04b2978cf3ffb821a7123889 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_wol.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: wol +doc_to_text: "Based on the given statement, is the following claim 'true', 'false',\ + \ or 'inconclusive'. \nStatement: {{premise}} \nClaim: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_wol_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d085919b777d18fff97397bdd32d1c0cf2c1c316 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_xho.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: xho +doc_to_text: "Based on the given statement, is the following claim 'true', 'false',\ + \ or 'inconclusive'. \nStatement: {{premise}} \nClaim: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_xho_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_yaml new file mode 100644 index 0000000000000000000000000000000000000000..3047238439e371f99664aed214a6589b33528e66 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_yaml @@ -0,0 +1,27 @@ +tag: afrixnli_tt_tasks +dataset_path: masakhane/afrixnli-translate-test +dataset_name: null +output_type: multiple_choice +test_split: test +fewshot_split: test +doc_to_target: !function utils.doc_to_target +doc_to_choice: + - "true" + - "inconclusive" + - "false" +should_decontaminate: true +doc_to_decontamination_query: premise +metric_list: + - metric: f1 + aggregation: !function utils.weighted_f1_score + average: weighted + higher_is_better: True + ignore_case: true + ignore_punctuation: true + - metric: acc + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d963646034069b77a78fe5284b106b6f74718a6a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_zul.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: zul +doc_to_text: "Based on the given statement, is the following claim 'true', 'false',\ + \ or 'inconclusive'. \nStatement: {{premise}} \nClaim: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_zul_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..6b9cb312b25a4c21bdd3d6a5e0a4e8e160451e4a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/utils.py @@ -0,0 +1,6 @@ +from lm_eval.utils import weighted_f1_score + + +def doc_to_target(doc): + replacements = {0: "true", 1: "false", 2: "inconclusive"} + return replacements[doc["label"]] diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/README.md b/lm-evaluation-harness/lm_eval/tasks/afrobench/README.md new file mode 100644 index 0000000000000000000000000000000000000000..a6ab3ceef1b37e94f1c191d1931648b7b669a49e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/README.md @@ -0,0 +1,72 @@ +# AfroBench + +### Paper + +Title: `AfroBench: How Good are Large Language Models on African Languages?` + +Paper Link: https://arxiv.org/abs/2311.07978 + +## Abstract +> Large-scale multilingual evaluations, such as MEGA, often include only a handful of African languages due to the scarcity of high-quality evaluation data and the limited discoverability of existing African datasets. This lack of representation hinders comprehensive LLM evaluation across a diverse range of languages and tasks. To address these challenges, we introduce AfroBench -- a multi-task benchmark for evaluating the performance of LLMs across 64 African languages, 15 tasks and 22 datasets. AfroBench consists of nine natural language understanding datasets, six text generation datasets, six knowledge and question answering tasks, and one mathematical reasoning task. We present results comparing the performance of prompting LLMs to fine-tuned baselines based on BERT and T5-style models. Our results suggest large gaps in performance between high-resource languages, such as English, and African languages across most tasks; but performance also varies based on the availability of monolingual data resources. Our findings confirm that performance on African languages continues to remain a hurdle for current LLMs, underscoring the need for additional efforts to close this gap. + +HomePage: https://mcgill-nlp.github.io/AfroBench/ + +### Groups, and Tasks +#### Groups +* `afrobench` : Runs all that tasks, datasets and prompts in this benchmark +* `afrobench_lite`: Runs the lite version of the benchmark which includes; afrimgsm, afrimmlu, afrixnli, sib, intent, adr and flores + +Dataset specific grouping that listing all prompts, allowing users to review or edit them. +* `adr` `afrihate` `afrisenti` `belebele` `african_flores` `injongointent` `mafand` `masakhaner` `masakhapos` `naijarc` `nollysenti` `african_ntrex` `openai_mmlu` `salt` `sib` `uhura` `xlsum` + + +#### Task Tags +* `adr_tasks`: all datasets in this benchmark relating to Automatic Diacritics Restoration task +* `afrihate_tasks`: all datasets in this benchmark relating to Hate Speech detection task +* `afrimgsm_tasks`: all datasets in this benchmark relating to Mathematical reasoning task +* `afrixnli_tasks`: all datasets in this benchmark relating to Natural Language Inference task +* `afrobench_xqa_tasks`: all datasets in this benchmark relating to Crosslingual QA (XQA) task +* `afrobench_sentiment_tasks`: all datasets in this benchmark relating to Sentiment Classification task +* `afrobench_MT_tasks`: all datasets in this benchmark relating to Machine Translation task +* `afrobench_TC_tasks`: all datasets in this benchmark relating to Topic Classification task +* `afrobench_mmlu_tasks`: all datasets in this benchmark relating to MMLU task +* `injongointent_tasks`: all datasets in this benchmark relating to Intent Detection task +* `masakhaner_tasks`: all datasets in this benchmark relating to Named Entity Recognition (NER) task +* `masakhapos_tasks`: all datasets in this benchmark relating to Part of Speech Tagging (POS) task +* `RC_tasks`: all datasets in this benchmark relating to Reading Comprehension task +* `uhura_arc_easy_tasks`: all datasets in this benchmark relating to Arc-Easy (XQA) task +* `xlsum_tasks`: all datasets in this benchmark relating to Summarization task + + +We've included sample run scripts for easier integration with the benchmark: [sample run scripts](./sample_run_scripts) + +For better understanding of the run interface see [interface.md](../../../docs/interface.md) + +All dataset used in this benchmark are available at [huggingface](https://huggingface.co/collections/masakhane/afrobench-67dbf553ebf5701c2207f883) + +### Citation + +``` +@misc{ojo2025afrobenchgoodlargelanguage, + title={AfroBench: How Good are Large Language Models on African Languages?}, + author={Jessica Ojo and Odunayo Ogundepo and Akintunde Oladipo and Kelechi Ogueji and Jimmy Lin and Pontus Stenetorp and David Ifeoluwa Adelani}, + year={2025}, + eprint={2311.07978}, + archivePrefix={arXiv}, + primaryClass={cs.CL}, + url={https://arxiv.org/abs/2311.07978}, +} +``` +Please cite datasets used. Citations for individual datasets are included in their respective repository readme files within this benchmark. +### Checklist + +For adding novel benchmarks/datasets to the library: +* [ ] Is the task an existing benchmark in the literature? + * [ ] Have you referenced the original paper that introduced the task? + * [ ] If yes, does the original paper provide a reference implementation? If so, have you checked against the reference implementation and documented how to run such a test? The original paper doesn't have an associated implementation, but there is an official entry in [BigBench](https://github.com/google/BIG-bench/tree/main/bigbench/benchmark_tasks/social_iqa). I use the same prompting format as BigBench. + + +If other tasks on this dataset are already supported: +* [ ] Is the "Main" variant of this task clearly denoted? +* [ ] Have you provided a short sentence in a README on what each new variant adds / evaluates? +* [ ] Have you noted which, if any, published evaluation setups are matched by this variant? diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/README.md b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/README.md new file mode 100644 index 0000000000000000000000000000000000000000..cb09567dcd2f461e3adba531bddd570c21ebfdf5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/README.md @@ -0,0 +1,7 @@ +# Automatic Diacritics Restoration (ADR) + +Automatic Diacritics Restoration (ADR) is the task of restoring diacritical marks in text where they have been omitted or removed. +This process is essential for languages where diacritics alter pronunciation, meaning, or grammatical structure. +ADR requires the model to have a deep understanding of linguistic context, syntax, and semantics to accurately predict and reinsert the appropriate diacritics. + +As part of this benchmark project, we utilise the mafand dataset to curate a dataset specifically for ADR. We focus on five languages: Gbomola, Fon, Igbo, Wolof, and Yoruba. diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/afridiacritics.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/afridiacritics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..34d60eef66acd9829ebb6e60ca6b85e8616a32d8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/afridiacritics.yaml @@ -0,0 +1,13 @@ +group: adr +task: + - adr_prompt_1 + - adr_prompt_2 + - adr_prompt_3 + - adr_prompt_4 + - adr_prompt_5 +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/gen_utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/gen_utils.py new file mode 100644 index 0000000000000000000000000000000000000000..ff6e63e3456abf809a1068f4abeea8ac93b49e94 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/gen_utils.py @@ -0,0 +1,105 @@ +import argparse +import os + +import yaml + + +class FunctionTag: + def __init__(self, value): + self.value = value + + +def prompt_func(mode, lang): + prompt_map = { + "prompt_1": "Please restore the missing diacritics in the following sentence: {{text}}. Return output sentence only", + "prompt_2": "Given a sentence without diacritics, add the appropriate diacritics to make it grammatically " + "and semantically correct. \nSentence: {{text}}. Return output sentence only", + "prompt_3": f"This text is in {lang}. Restore all diacritical marks to their proper places in the " + "following sentence: {{text}}. Return output sentence only", + "prompt_4": f"You are a linguist specializing in diacritical marks for {lang}. " + f"Add the appropriate diacritics to this {lang} sentence: " + "{{text}}. Return output sentence only", + "prompt_5": f"You are a linguist specializing in diacritical marks for {lang}. Diacritics are essential for " + f"proper pronunciation and meaning in {lang}. You are tasked with converting {lang} sentences " + "without diacritics into their correctly accented forms. Here's the input: {{text}}. " + "Return output sentence only", + } + return prompt_map[mode] + + +def gen_lang_yamls(output_dir: str, overwrite: bool, mode: str) -> None: + """ + Generate a yaml file for each language. + + :param output_dir: The directory to output the files to. + :param overwrite: Whether to overwrite files if they already exist. + """ + err = [] + languages = { + "fon": "Fon", + "bbj": "Gbomala", + "ibo": "Igbo", + "wol": "Wolof", + "yor": "Yoruba", + } + + for lang in languages.keys(): + try: + file_name = f"afridiacritics_{lang}.yaml" + task_name = f"afridiacritics_{lang}_{mode}" + yaml_template = "afridiacritics_yaml" + yaml_details = { + "include": yaml_template, + "task": task_name, + "dataset_name": lang, + "doc_to_text": prompt_func(mode, languages[lang]), + } + os.makedirs(f"{output_dir}/{mode}", exist_ok=True) + with open( + f"{output_dir}/{mode}/{file_name}", + "w" if overwrite else "x", + encoding="utf8", + ) as f: + f.write("# Generated by utils.py\n") + yaml.dump( + yaml_details, + f, + allow_unicode=True, + ) + except FileExistsError: + err.append(file_name) + + if len(err) > 0: + raise FileExistsError( + "Files were not created because they already exist (use --overwrite flag):" + f" {', '.join(err)}" + ) + + +def main() -> None: + """Parse CLI args and generate language-specific yaml files.""" + parser = argparse.ArgumentParser() + parser.add_argument( + "--overwrite", + default=True, + action="store_true", + help="Overwrite files if they already exist", + ) + parser.add_argument( + "--output-dir", + default="./", + help="Directory to write yaml files to", + ) + parser.add_argument( + "--mode", + default="prompt_1", + choices=["prompt_1", "prompt_2", "prompt_3", "prompt_4", "prompt_5"], + help="Prompt number", + ) + args = parser.parse_args() + + gen_lang_yamls(output_dir=args.output_dir, overwrite=args.overwrite, mode=args.mode) + + +if __name__ == "__main__": + main() diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_1/afridiacritics_bbj.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_1/afridiacritics_bbj.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f3eb26ebae6f723c03591aa73eb29f2256fb0e4a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_1/afridiacritics_bbj.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: bbj +doc_to_text: 'Please restore the missing diacritics in the following sentence: {{text}}. + Return output sentence only' +include: afridiacritics_yaml +task: afridiacritics_bbj_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_1/afridiacritics_fon.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_1/afridiacritics_fon.yaml new file mode 100644 index 0000000000000000000000000000000000000000..874832d5d00799deb9235d2f04960684f2b91770 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_1/afridiacritics_fon.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: fon +doc_to_text: 'Please restore the missing diacritics in the following sentence: {{text}}. + Return output sentence only' +include: afridiacritics_yaml +task: afridiacritics_fon_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_1/afridiacritics_yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_1/afridiacritics_yaml new file mode 100644 index 0000000000000000000000000000000000000000..53cebaee05c9e7a65779ad12faaa0a9ee40c7c8b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_1/afridiacritics_yaml @@ -0,0 +1,25 @@ +tag: +- adr_tasks +- adr_prompt_1 +dataset_path: masakhane/diacritics-restoration +dataset_kwargs: {trust_remote_code: True} +doc_to_target: target +output_type: generate_until +fewshot_split: dev +test_split: test +training_split: train +metric_list: + - metric: bleu + aggregation: bleu + higher_is_better: true + - metric: chrf + aggregation: chrf + higher_is_better: true +generation_kwargs: + do_sample: false + until: + - '' + - + - <|im_end|> +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_1/afridiacritics_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_1/afridiacritics_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8e98af10abec2009d32b112694923f45c17473af --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_1/afridiacritics_yor.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: yor +doc_to_text: 'Please restore the missing diacritics in the following sentence: {{text}}. + Return output sentence only' +include: afridiacritics_yaml +task: afridiacritics_yor_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_2/afridiacritics_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_2/afridiacritics_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c1ebac101f9a69722b21bbfe65ccd224f811e8d3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_2/afridiacritics_ibo.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: ibo +doc_to_text: "Given a sentence without diacritics, add the appropriate diacritics\ + \ to make it grammatically and semantically correct. \nSentence: {{text}}. Return\ + \ output sentence only" +include: afridiacritics_yaml +task: afridiacritics_ibo_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_2/afridiacritics_wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_2/afridiacritics_wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b8448d6ffce9f67252621cf6085fc575dace588e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_2/afridiacritics_wol.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: wol +doc_to_text: "Given a sentence without diacritics, add the appropriate diacritics\ + \ to make it grammatically and semantically correct. \nSentence: {{text}}. Return\ + \ output sentence only" +include: afridiacritics_yaml +task: afridiacritics_wol_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_2/afridiacritics_yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_2/afridiacritics_yaml new file mode 100644 index 0000000000000000000000000000000000000000..a0cc722d890f6a64939417f39f860532c4cd342b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_2/afridiacritics_yaml @@ -0,0 +1,25 @@ +tag: +- adr_tasks +- adr_prompt_2 +dataset_path: masakhane/diacritics-restoration +dataset_kwargs: {trust_remote_code: True} +doc_to_target: target +output_type: generate_until +fewshot_split: dev +test_split: test +training_split: train +metric_list: + - metric: bleu + aggregation: bleu + higher_is_better: true + - metric: chrf + aggregation: chrf + higher_is_better: true +generation_kwargs: + do_sample: false + until: + - '' + - + - <|im_end|> +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_2/afridiacritics_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_2/afridiacritics_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..eb95f5e234add5c178efc181bddb1fc87f9ce19d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_2/afridiacritics_yor.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: yor +doc_to_text: "Given a sentence without diacritics, add the appropriate diacritics\ + \ to make it grammatically and semantically correct. \nSentence: {{text}}. Return\ + \ output sentence only" +include: afridiacritics_yaml +task: afridiacritics_yor_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_3/afridiacritics_bbj.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_3/afridiacritics_bbj.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a50b40c535d778cb4bd564455fbfdcf43415a53d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_3/afridiacritics_bbj.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: bbj +doc_to_text: 'This text is in Gbomala. Restore all diacritical marks to their proper + places in the following sentence: {{text}}. Return output sentence only' +include: afridiacritics_yaml +task: afridiacritics_bbj_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_3/afridiacritics_fon.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_3/afridiacritics_fon.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5b0909ce9dd69f46cdca75ebdc325f452b25a462 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_3/afridiacritics_fon.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: fon +doc_to_text: 'This text is in Fon. Restore all diacritical marks to their proper places + in the following sentence: {{text}}. Return output sentence only' +include: afridiacritics_yaml +task: afridiacritics_fon_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_4/afridiacritics_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_4/afridiacritics_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..21e3a53fefcb4ae41eb00a406a3319f14ed60aba --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_4/afridiacritics_yor.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: yor +doc_to_text: 'You are a linguist specializing in diacritical marks for Yoruba. Add + the appropriate diacritics to this Yoruba sentence: {{text}}. Return output sentence + only' +include: afridiacritics_yaml +task: afridiacritics_yor_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_5/afridiacritics_bbj.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_5/afridiacritics_bbj.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b1bcc833c73d0a789700c1b50b8636163620ed27 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_5/afridiacritics_bbj.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: bbj +doc_to_text: 'You are a linguist specializing in diacritical marks for Gbomala. Diacritics + are essential for proper pronunciation and meaning in Gbomala. You are tasked with + converting Gbomala sentences without diacritics into their correctly accented forms. + Here''s the input: {{text}}. Return output sentence only' +include: afridiacritics_yaml +task: afridiacritics_bbj_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_5/afridiacritics_fon.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_5/afridiacritics_fon.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3a1c55f813b4c4b7d08daff74cf32040b85e2b35 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_5/afridiacritics_fon.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: fon +doc_to_text: 'You are a linguist specializing in diacritical marks for Fon. Diacritics + are essential for proper pronunciation and meaning in Fon. You are tasked with converting + Fon sentences without diacritics into their correctly accented forms. Here''s the + input: {{text}}. Return output sentence only' +include: afridiacritics_yaml +task: afridiacritics_fon_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_5/afridiacritics_yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_5/afridiacritics_yaml new file mode 100644 index 0000000000000000000000000000000000000000..aaad3306e7270e78cdd2f83dd8ffeb790520134d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_5/afridiacritics_yaml @@ -0,0 +1,25 @@ +tag: +- adr_tasks +- adr_prompt_5 +dataset_path: masakhane/diacritics-restoration +dataset_kwargs: {trust_remote_code: True} +doc_to_target: target +output_type: generate_until +fewshot_split: dev +test_split: test +training_split: train +metric_list: + - metric: bleu + aggregation: bleu + higher_is_better: true + - metric: chrf + aggregation: chrf + higher_is_better: true +generation_kwargs: + do_sample: false + until: + - '' + - + - <|im_end|> +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_1/afriqa_fon.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_1/afriqa_fon.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c51196157f95e96315d0321e4b53859ef8e5ae35 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_1/afriqa_fon.yaml @@ -0,0 +1,12 @@ +# Generated by utils.py +dataset_name: fon +doc_to_text: 'Your task is to answer a qestion given a context.Make sure you respond + with the shortest span containing the answer in the context. + + Question: {{question_lang}} + + Context: {{context}} + + Answer:' +include: afriqa +task: afriqa_fon_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_1/afriqa_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_1/afriqa_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dbdebe14dccc3eda73ee706e3327a13a43a3aa81 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_1/afriqa_swa.yaml @@ -0,0 +1,15 @@ +# Generated by utils.py +dataset_name: swa +doc_to_text: 'Your task is to answer a qestion given a context.Make sure you respond + with the shortest span containing the answer in the context. + + Question: {{question_lang}} + + Context: {{context}} + + Answer:' +include: afriqa +fewshot_split: test +fewshot_config: + sampler: first_n +task: afriqa_swa_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_5/afriqa b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_5/afriqa new file mode 100644 index 0000000000000000000000000000000000000000..fab00068beb951dbab88d4baa870fabfced4f820 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_5/afriqa @@ -0,0 +1,42 @@ +tag: + - afrobench_xqa_tasks + - afriqa_prompt_5 +dataset_kwargs: {trust_remote_code: True} +dataset_path: masakhane/afriqa-gold-passages +dataset_name: null +output_type: generate_until +test_split: test +fewshot_split: train +doc_to_target: answer_pivot +should_decontaminate: true +doc_to_decontamination_query: question_lang +generation_kwargs: + until: + - "\n" + do_sample: false + temperature: 0.0 +filter_list: + - name: remove_whitespace + filter: + - function: remove_whitespace + - function: take_first +target_delimiter: " " +metric_list: + - metric: exact_match + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true + - "." + - "," + - "\\$" + - metric: f1 + aggregation: !function utils.f1 + higher_is_better: true + ignore_case: true + ignore_punctuation: true + - "." + - "," + - "\\$" +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_1/afrisenti b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_1/afrisenti new file mode 100644 index 0000000000000000000000000000000000000000..69ef6b2bc08bbc198e2c6610c7c40041db4d20a4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_1/afrisenti @@ -0,0 +1,41 @@ +tag: + - afrobench_sentiment_tasks + - afrisenti_prompt_1 +task: null +dataset_path: masakhane/afrisenti +dataset_name: null +dataset_kwargs: {trust_remote_code: True} +output_type: multiple_choice +validation_split: validation +test_split: test +fewshot_split: train +doc_to_text: 'Does this statement; "{{tweet}}" have a Neutral, Positive or Negative sentiment? Labels only' +doc_to_target: label +doc_to_choice: + - "negative" + - "positive" + - "neutral" +should_decontaminate: true +doc_to_decontamination_query: tweet +metric_list: + - metric: f1 + aggregation: !function utils.weighted_f1_score + # aggregation: mean + average: weighted + hf_evaluate: true + higher_is_better: True + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" + - metric: acc + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_1/afrisenti_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_1/afrisenti_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f0ab9071abbc0211b8048743db43347bb5df1583 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_1/afrisenti_hau.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: hau +include: afrisenti +task: afrisenti_hau_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_1/afrisenti_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_1/afrisenti_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b0176d08764dae9a5fd8af57dc903b6b55ab0124 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_1/afrisenti_ibo.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: ibo +include: afrisenti +task: afrisenti_ibo_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_1/afrisenti_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_1/afrisenti_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..75bb717a6e22d931404d2c7cfc919cfa8f99453d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_1/afrisenti_kin.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: kin +include: afrisenti +task: afrisenti_kin_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_1/afrisenti_tso.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_1/afrisenti_tso.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d500693270946b6581020bc68a38612bdfd4f033 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_1/afrisenti_tso.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: tso +include: afrisenti +task: afrisenti_tso_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_1/afrisenti_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_1/afrisenti_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fda98c2c82c6e323eae8c96843e657e07a4d9665 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_1/afrisenti_yor.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: yor +include: afrisenti +task: afrisenti_yor_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_1/run.sh b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_1/run.sh new file mode 100644 index 0000000000000000000000000000000000000000..50d1a1338f87330219dd4c6f79fa85ef918bb21c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_1/run.sh @@ -0,0 +1,34 @@ +#!/bin/bash + +models=( + + "google/gemma-1.1-7b-it" + "CohereForAI/aya-101" + "meta-llama/Llama-2-7b-chat-hf" + "meta-llama/Meta-Llama-3-8B-Instruct" + "google/gemma-2-9b-it" + "bigscience/mt0-xxl" + "google/gemma-2-27b-it" + "meta-llama/Meta-Llama-3-70B-Instruct" +) +task=afrisenti_amh_prompt_1,afrisenti_arq_prompt_1,afrisenti_ary_prompt_1,afrisenti_hau_prompt_1,afrisenti_ibo_prompt_1,afrisenti_kin_prompt_1,afrisenti_pcm_prompt_1,afrisenti_por_prompt_1,afrisenti_swa_prompt_1,afrisenti_tir_prompt_1,afrisenti_tso_prompt_1,afrisenti_twi_prompt_1,afrisenti_yor_prompt_1 + +for model in "${models[@]}" +do + echo "Evaluating model: $model" + for fewshot in 0 5 + do + export OUTPUT_DIR=results/$fewshot + + mkdir -p "$OUTPUT_DIR" + + lm_eval --model hf \ + --model_args "pretrained=${model}" \ + --tasks $task\ + --device cuda:0 \ + --batch_size 16 \ + --output_path "$OUTPUT_DIR" \ + --num_fewshot $fewshot \ + --verbosity DEBUG + done +done diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_2/afrisenti_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_2/afrisenti_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d97b2c25787d9338546dba3707afcda31ad31269 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_2/afrisenti_amh.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: amh +doc_to_text: Does this Amharic statement; '{{tweet}}' have a Neutral, Positive or + Negative sentiment? Labels only +include: afrisenti +task: afrisenti_amh_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_2/afrisenti_ary.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_2/afrisenti_ary.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4e76d385b3dc5ee5bbbe817336ec9d78feef7eb7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_2/afrisenti_ary.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: ary +doc_to_text: Does this Moroccan Arabic statement; '{{tweet}}' have a Neutral, Positive + or Negative sentiment? Labels only +include: afrisenti +task: afrisenti_ary_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_2/afrisenti_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_2/afrisenti_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f7b0ccb2811b30acc98fe594af33b0a38fb2a88b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_2/afrisenti_hau.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: hau +doc_to_text: Does this Hausa statement; '{{tweet}}' have a Neutral, Positive or Negative + sentiment? Labels only +include: afrisenti +task: afrisenti_hau_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_2/afrisenti_orm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_2/afrisenti_orm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e8abbbfbd73687a4249238a3f2ad988b85634531 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_2/afrisenti_orm.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: orm +doc_to_text: Does this Oromo statement; '{{tweet}}' have a Neutral, Positive or Negative + sentiment? Labels only +include: afrisenti +task: afrisenti_orm_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_2/afrisenti_pcm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_2/afrisenti_pcm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4dd98925299e9b872d49bdb9bea0ebd2b48d1ec7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_2/afrisenti_pcm.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: pcm +doc_to_text: Does this Nigerian Pidgin statement; '{{tweet}}' have a Neutral, Positive + or Negative sentiment? Labels only +include: afrisenti +task: afrisenti_pcm_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_2/afrisenti_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_2/afrisenti_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..496da1a1d1e2b4fcfa004e918af85e7321a1ed29 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_2/afrisenti_swa.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: swa +doc_to_text: Does this Swahili statement; '{{tweet}}' have a Neutral, Positive or + Negative sentiment? Labels only +include: afrisenti +task: afrisenti_swa_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_2/afrisenti_tir.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_2/afrisenti_tir.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3899c992ed3180d76f2ff677c148026da6d5e9a1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_2/afrisenti_tir.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: tir +doc_to_text: Does this Tigrinya statement; '{{tweet}}' have a Neutral, Positive or + Negative sentiment? Labels only +include: afrisenti +task: afrisenti_tir_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_2/run.sh b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_2/run.sh new file mode 100644 index 0000000000000000000000000000000000000000..48797912512124c9c5287dcdd654e5fa04a029b0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_2/run.sh @@ -0,0 +1,33 @@ +#!/bin/bash + +models=( + + "google/gemma-1.1-7b-it" + "CohereForAI/aya-101" + "meta-llama/Llama-2-7b-chat-hf" + "meta-llama/Meta-Llama-3-8B-Instruct" + "google/gemma-2-9b-it" + "bigscience/mt0-xxl" + "google/gemma-2-27b-it" + "meta-llama/Meta-Llama-3-70B-Instruct" +) + +for model in "${models[@]}" +do + echo "Evaluating model: $model" + for fewshot in 0 5 + do + export OUTPUT_DIR=./results/$fewshot + + mkdir -p "$OUTPUT_DIR" + + lm_eval --model hf \ + --model_args "pretrained=${model},parallelize: true" \ + --tasks afribench\ + --batch_size 256 \ + --output_path "$OUTPUT_DIR" \ + --num_fewshot $fewshot \ + --verbosity DEBUG \ + --limit 2 + done +done diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_2/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_2/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..3e735e2deb1f9c53152c072615aebe8ba3acb90b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_2/utils.py @@ -0,0 +1 @@ +from lm_eval.utils import weighted_f1_score diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_3/afrisenti_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_3/afrisenti_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2645b72befa6b0048986f68aa14b4d5ed60027dc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_3/afrisenti_amh.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: amh +doc_to_text: "You are an assistant able to detect sentiments in tweets. \n\nGiven\ + \ the sentiment labels Neutral, Positive or Negative; what is the sentiment of the\ + \ Amharic statement below? Return only the labels. \n\ntext: {{tweet}} \nlabel:" +include: afrisenti +task: afrisenti_amh_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_3/afrisenti_arq.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_3/afrisenti_arq.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0b90f690e93f249e8c4668bb74a667ff19a39247 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_3/afrisenti_arq.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: arq +doc_to_text: "You are an assistant able to detect sentiments in tweets. \n\nGiven\ + \ the sentiment labels Neutral, Positive or Negative; what is the sentiment of the\ + \ Algerian Arabic statement below? Return only the labels. \n\ntext: {{tweet}} \n\ + label:" +include: afrisenti +task: afrisenti_arq_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_3/afrisenti_ary.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_3/afrisenti_ary.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ba11ee3e5146db8bfef7680becfb95a6f0a9b6aa --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_3/afrisenti_ary.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: ary +doc_to_text: "You are an assistant able to detect sentiments in tweets. \n\nGiven\ + \ the sentiment labels Neutral, Positive or Negative; what is the sentiment of the\ + \ Moroccan Arabic statement below? Return only the labels. \n\ntext: {{tweet}} \n\ + label:" +include: afrisenti +task: afrisenti_ary_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_3/afrisenti_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_3/afrisenti_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4f4e6b3fb3252929fcd2d0f30240ca8d4553a009 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_3/afrisenti_hau.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: hau +doc_to_text: "You are an assistant able to detect sentiments in tweets. \n\nGiven\ + \ the sentiment labels Neutral, Positive or Negative; what is the sentiment of the\ + \ Hausa statement below? Return only the labels. \n\ntext: {{tweet}} \nlabel:" +include: afrisenti +task: afrisenti_hau_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_3/afrisenti_pcm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_3/afrisenti_pcm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..eb0ac8ff3bc1b3e0c8efa9fe9afdc40ea0c0690a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_3/afrisenti_pcm.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: pcm +doc_to_text: "You are an assistant able to detect sentiments in tweets. \n\nGiven\ + \ the sentiment labels Neutral, Positive or Negative; what is the sentiment of the\ + \ Nigerian Pidgin statement below? Return only the labels. \n\ntext: {{tweet}} \n\ + label:" +include: afrisenti +task: afrisenti_pcm_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_3/afrisenti_por.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_3/afrisenti_por.yaml new file mode 100644 index 0000000000000000000000000000000000000000..821a4355b044844d3608c01950b949ce5f292ba2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_3/afrisenti_por.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: por +doc_to_text: "You are an assistant able to detect sentiments in tweets. \n\nGiven\ + \ the sentiment labels Neutral, Positive or Negative; what is the sentiment of the\ + \ Mozambique Portuguese statement below? Return only the labels. \n\ntext: {{tweet}}\ + \ \nlabel:" +include: afrisenti +task: afrisenti_por_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_4/afrisenti_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_4/afrisenti_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8a30a72a45d38f6f4221a1e6ceaef93dd5472bbd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_4/afrisenti_amh.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: amh +doc_to_text: "Label the following text as Neutral, Positive, or Negative. Provide\ + \ only the label as your response. \n\ntext: {{tweet}} \nlabel: " +include: afrisenti +task: afrisenti_amh_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_4/afrisenti_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_4/afrisenti_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5e7e9a443e5926dfe4de5074e1416417d38b4447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_4/afrisenti_hau.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: hau +doc_to_text: "Label the following text as Neutral, Positive, or Negative. Provide\ + \ only the label as your response. \n\ntext: {{tweet}} \nlabel: " +include: afrisenti +task: afrisenti_hau_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_4/afrisenti_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_4/afrisenti_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7061ab63523b3b5cb34ec1bb0d35db11fdefd5d0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_4/afrisenti_kin.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: kin +doc_to_text: "Label the following text as Neutral, Positive, or Negative. Provide\ + \ only the label as your response. \n\ntext: {{tweet}} \nlabel: " +include: afrisenti +task: afrisenti_kin_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_4/afrisenti_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_4/afrisenti_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b4366d1f27902651502a96adfc9c09fb8d82dfde --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_4/afrisenti_twi.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: twi +doc_to_text: "Label the following text as Neutral, Positive, or Negative. Provide\ + \ only the label as your response. \n\ntext: {{tweet}} \nlabel: " +include: afrisenti +task: afrisenti_twi_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_4/afrisenti_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_4/afrisenti_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8394706cdfbfb1124d4c25c987901d5f338cd5f7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_4/afrisenti_yor.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: yor +doc_to_text: "Label the following text as Neutral, Positive, or Negative. Provide\ + \ only the label as your response. \n\ntext: {{tweet}} \nlabel: " +include: afrisenti +task: afrisenti_yor_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_4/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_4/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..3e735e2deb1f9c53152c072615aebe8ba3acb90b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_4/utils.py @@ -0,0 +1 @@ +from lm_eval.utils import weighted_f1_score diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_5/afrisenti b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_5/afrisenti new file mode 100644 index 0000000000000000000000000000000000000000..5107bb80d5333a462afda9a8efb62a6fd039a733 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_5/afrisenti @@ -0,0 +1,39 @@ +tag: + - afrobench_sentiment_tasks + - afrisenti_prompt_5 +dataset_path: masakhane/afrisenti +dataset_name: null +dataset_kwargs: {trust_remote_code: True} +output_type: multiple_choice +validation_split: validation +test_split: test +fewshot_split: train +doc_to_target: label +doc_to_choice: + - "negative" + - "positive" + - "neutral" +should_decontaminate: true +doc_to_decontamination_query: 'Text: {{tweet}} \nlabel:' +metric_list: + - metric: f1 + aggregation: !function utils.weighted_f1_score + # aggregation: mean + average: weighted + hf_evaluate: true + higher_is_better: True + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" + - metric: acc + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_5/afrisenti_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_5/afrisenti_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dba7b17964cdc320243578f9e9249a00caea4a44 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_5/afrisenti_ibo.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: ibo +doc_to_text: "You are tasked with performing sentiment classification on the following\ + \ Igbo text. For each input, classify the sentiment as positive, negative, or neutral.\ + \ Use the following guidelines: \n\n Positive: The text expresses happiness, satisfaction,\ + \ or optimism. \nNegative: The text conveys disappointment, dissatisfaction, or\ + \ pessimism. \nNeutral: The text is factual, objective, or without strong emotional\ + \ undertones. \n\nIf the text contains both positive and negative sentiments, choose\ + \ the dominant sentiment. For ambiguous or unclear sentiments, select the label\ + \ that best reflects the overall tone. Please provide a single classification for\ + \ each input.\n\ntext: {{tweet}} \nlabel: " +include: afrisenti +task: afrisenti_ibo_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_5/afrisenti_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_5/afrisenti_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..16ea6f0c5734b4795acb06b5d854d5cb33b97c05 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_5/afrisenti_kin.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: kin +doc_to_text: "You are tasked with performing sentiment classification on the following\ + \ Kinyarwanda text. For each input, classify the sentiment as positive, negative,\ + \ or neutral. Use the following guidelines: \n\n Positive: The text expresses happiness,\ + \ satisfaction, or optimism. \nNegative: The text conveys disappointment, dissatisfaction,\ + \ or pessimism. \nNeutral: The text is factual, objective, or without strong emotional\ + \ undertones. \n\nIf the text contains both positive and negative sentiments, choose\ + \ the dominant sentiment. For ambiguous or unclear sentiments, select the label\ + \ that best reflects the overall tone. Please provide a single classification for\ + \ each input.\n\ntext: {{tweet}} \nlabel: " +include: afrisenti +task: afrisenti_kin_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_5/afrisenti_tir.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_5/afrisenti_tir.yaml new file mode 100644 index 0000000000000000000000000000000000000000..47a67e1c9358682be41792e77ef9e85fb1735baf --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_5/afrisenti_tir.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: tir +doc_to_text: "You are tasked with performing sentiment classification on the following\ + \ Tigrinya text. For each input, classify the sentiment as positive, negative, or\ + \ neutral. Use the following guidelines: \n\n Positive: The text expresses happiness,\ + \ satisfaction, or optimism. \nNegative: The text conveys disappointment, dissatisfaction,\ + \ or pessimism. \nNeutral: The text is factual, objective, or without strong emotional\ + \ undertones. \n\nIf the text contains both positive and negative sentiments, choose\ + \ the dominant sentiment. For ambiguous or unclear sentiments, select the label\ + \ that best reflects the overall tone. Please provide a single classification for\ + \ each input.\n\ntext: {{tweet}} \nlabel: " +include: afrisenti +task: afrisenti_tir_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_5/afrisenti_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_5/afrisenti_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c0b4fe03da64d5f1940dc9aeebe29de2cea09227 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_5/afrisenti_twi.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: twi +doc_to_text: "You are tasked with performing sentiment classification on the following\ + \ Twi text. For each input, classify the sentiment as positive, negative, or neutral.\ + \ Use the following guidelines: \n\n Positive: The text expresses happiness, satisfaction,\ + \ or optimism. \nNegative: The text conveys disappointment, dissatisfaction, or\ + \ pessimism. \nNeutral: The text is factual, objective, or without strong emotional\ + \ undertones. \n\nIf the text contains both positive and negative sentiments, choose\ + \ the dominant sentiment. For ambiguous or unclear sentiments, select the label\ + \ that best reflects the overall tone. Please provide a single classification for\ + \ each input.\n\ntext: {{tweet}} \nlabel: " +include: afrisenti +task: afrisenti_twi_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_5/xx.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_5/xx.py new file mode 100644 index 0000000000000000000000000000000000000000..375facffa5030cdde562a9ab4474193a9d45f597 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_5/xx.py @@ -0,0 +1,8 @@ +# data = load_dataset('HausaNLP/AfriSenti-Twitter', 'yor', trust_remote_code=True) +# print(data) + +import torch + + +print(torch.cuda.is_available()) # Should return True +print(torch.cuda.device_count()) diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrobench-lite.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrobench-lite.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a23c050a2d09646492778e80dbc3a30dc281f580 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrobench-lite.yaml @@ -0,0 +1,15 @@ +group: afrobench_lite +task: + - afrimgsm_cot_tasks + - afrimmlu_tasks + - afrixnli_tasks + - belebele_tasks + - sib_tasks + - african_flores_tasks + - injongointent_tasks +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7ee55e8eb291cd47abc402f7c4323d2f95caee1e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_amh.yaml @@ -0,0 +1,17 @@ +# Generated by utils.py +dataset_name: amh_Ethi +doc_to_text: 'P: {{flores_passage}} + + Q: {{question.strip()}} + + A: {{mc_answer1}} + + B: {{mc_answer2}} + + C: {{mc_answer3}} + + D: {{mc_answer4}} + + Please choose the correct answer from the options above:' +include: belebele +task: belebele_amh_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_lin_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_lin_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..41494a7263855a44fd217ac6d7cee38e714a8597 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_lin_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: lin_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Lingala: {{sentence_lin_Latn}} \nEnglish: " +include: flores +task: flores_lin_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_luo_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_luo_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7a22ec7db9d0cf3d3497ab0367c32a2aef602513 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_luo_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: luo_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Luo: {{sentence_luo_Latn}} \nEnglish: " +include: flores +task: flores_luo_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_nso_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_nso_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2409753c8a86d873fc33b68cbaba493b154fd947 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_nso_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: nso_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Northern Sotho: {{sentence_nso_Latn}} \nEnglish: " +include: flores +task: flores_nso_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_run_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_run_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e00eb85718ffff0eb4fd84a1ce50fc4ff92c9988 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_run_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: run_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Rundi: {{sentence_run_Latn}} \nEnglish: " +include: flores +task: flores_run_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_sag_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_sag_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e7f43c6b6cf91d52311d4f0981086afcd833d8b3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_sag_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: sag_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Sango: {{sentence_sag_Latn}} \nEnglish: " +include: flores +task: flores_sag_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_sna_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_sna_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d63b4c6baf598d665509cadcaf1c1613f7f2c77d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_sna_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: sna_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Shona: {{sentence_sna_Latn}} \nEnglish: " +include: flores +task: flores_sna_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_som_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_som_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f625c559f3c1dd1b0490d42c29515dfeaef28d68 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_som_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: som_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Somali: {{sentence_som_Latn}} \nEnglish: " +include: flores +task: flores_som_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_tir_Ethi-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_tir_Ethi-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cc13ae0413578e722f0f7c7e1c724e565b7a10c6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_tir_Ethi-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: tir_Ethi-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Tigrinya: {{sentence_tir_Ethi}} \nEnglish: " +include: flores +task: flores_tir_Ethi-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_tsn_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_tsn_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2a6c4e1c820c15ff4e1b3f2d8db43c102a4b206e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_tsn_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: tsn_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Setswana: {{sentence_tsn_Latn}} \nEnglish: " +include: flores +task: flores_tsn_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_tso_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_tso_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0d473ab03b198ff0691265f278a6aec6e688e967 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_tso_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: tso_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Tsonga: {{sentence_tso_Latn}} \nEnglish: " +include: flores +task: flores_tso_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_tum_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_tum_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c491f25b514977bd8554a0d53212525b028a6e41 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_tum_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: tum_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Tumbuka: {{sentence_tum_Latn}} \nEnglish: " +include: flores +task: flores_tum_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_twi_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_twi_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2d8ad29e375918e52cbd3337ac6965a461c7f7fc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_twi_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: twi_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Twi: {{sentence_twi_Latn}} \nEnglish: " +include: flores +task: flores_twi_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_tzm_Tfng-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_tzm_Tfng-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ba4624651a68c219eefd7c9157f3bce58d751c48 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_tzm_Tfng-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: tzm_Tfng-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Central Atlas Tamazight: {{sentence_tzm_Tfng}} \nEnglish: " +include: flores +task: flores_tzm_Tfng-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_umb_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_umb_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0758003ad3bb766fad0fa545d39ba976d4443f26 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_umb_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: umb_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Umbundu: {{sentence_umb_Latn}} \nEnglish: " +include: flores +task: flores_umb_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_wol_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_wol_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..914e6c128220583cb7f2e064e0eed56948bbdfd9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_wol_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: wol_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Wolof: {{sentence_wol_Latn}} \nEnglish: " +include: flores +task: flores_wol_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_xho_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_xho_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bc130fb0447aa7e617debe3cf1086f55ac96aec6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_xho_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: xho_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Xhosa: {{sentence_xho_Latn}} \nEnglish: " +include: flores +task: flores_xho_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_yor_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_yor_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0ea0fbc4e65b9320f1d4bd701a98819c17b69ab0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_yor_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: yor_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Yoruba: {{sentence_yor_Latn}} \nEnglish: " +include: flores +task: flores_yor_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_zul_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_zul_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ea070b30e7822d9e55135f3842766e68f94c1f2c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_zul_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: zul_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Zulu: {{sentence_zul_Latn}} \nEnglish: " +include: flores +task: flores_zul_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-ace_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-ace_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d2ed60660bd3f565484d16440ce9fb2d82f6a555 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-ace_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-ace_Latn +doc_to_target: sentence_ace_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nAcehnese (Latin script): " +include: flores +task: flores_eng_Latn-ace_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-arz_Arab.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-arz_Arab.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a42fa079b4501f6402eebd3241c26f14b1e5af6e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-arz_Arab.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-arz_Arab +doc_to_target: sentence_arz_Arab +doc_to_text: "English: {{sentence_eng_Latn}} \nEgyptian Arabic: " +include: flores +task: flores_eng_Latn-arz_Arab_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-ban_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-ban_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f43a4b71131da9cf555964b79a6258ce7f36c2ef --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-ban_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-ban_Latn +doc_to_target: sentence_ban_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nBalinese: " +include: flores +task: flores_eng_Latn-ban_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-bem_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-bem_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..252117ef888300c0dfaa64f2cacc55bf66292136 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-bem_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-bem_Latn +doc_to_target: sentence_bem_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nBemba: " +include: flores +task: flores_eng_Latn-bem_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-dik_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-dik_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..36dea9d3a9dc3371a576315540f439af9e38b4e5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-dik_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-dik_Latn +doc_to_target: sentence_dik_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nSouthwestern Dinka: " +include: flores +task: flores_eng_Latn-dik_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-fuv_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-fuv_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8855378737fa985bf35a840c78f81d34f8542305 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-fuv_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-fuv_Latn +doc_to_target: sentence_fuv_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nNigerian Fulfulde: " +include: flores +task: flores_eng_Latn-fuv_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-kab_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-kab_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fd22cb7de77e624bea297d7011aa18aab3408b10 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-kab_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-kab_Latn +doc_to_target: sentence_kab_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nKabyle: " +include: flores +task: flores_eng_Latn-kab_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-kam_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-kam_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..802ae7dca4e9b4165f2028cfce4be8c08c1b0edd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-kam_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-kam_Latn +doc_to_target: sentence_kam_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nKamba: " +include: flores +task: flores_eng_Latn-kam_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-kbp_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-kbp_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9cc1afd5e9af72a8a7f73a6789cb2dc0af1e9c39 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-kbp_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-kbp_Latn +doc_to_target: sentence_kbp_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nKabiyè: " +include: flores +task: flores_eng_Latn-kbp_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-kea_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-kea_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..55d3e7767c8533eb9f0f94c37a33fdb628c2b27a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-kea_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-kea_Latn +doc_to_target: sentence_kea_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nKabuverdianu: " +include: flores +task: flores_eng_Latn-kea_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-kin_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-kin_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..28530541f153ad52724b5d0aa13eca176fa73c29 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-kin_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-kin_Latn +doc_to_target: sentence_kin_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nKinyarwanda: " +include: flores +task: flores_eng_Latn-kin_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-kmb_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-kmb_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5619209f346582b64e51b904288181fc18bc34d6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-kmb_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-kmb_Latn +doc_to_target: sentence_kmb_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nKimbundu: " +include: flores +task: flores_eng_Latn-kmb_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-knc_Arab.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-knc_Arab.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fba5e257d7c5154ef3d89ace61d5f7fe2397ccc8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-knc_Arab.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-knc_Arab +doc_to_target: sentence_knc_Arab +doc_to_text: "English: {{sentence_eng_Latn}} \nCentral Kanuri (Arabic script): " +include: flores +task: flores_eng_Latn-knc_Arab_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-knc_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-knc_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c1f84d5753ff5e03d9d4d6463beeb9a390d5c2eb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-knc_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-knc_Latn +doc_to_target: sentence_knc_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nCentral Kanuri (Latin script): " +include: flores +task: flores_eng_Latn-knc_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-kon_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-kon_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d6d8ef32d897edfa3086e7b93c39efa41a907e75 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-kon_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-kon_Latn +doc_to_target: sentence_kon_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nKikongo: " +include: flores +task: flores_eng_Latn-kon_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-lin_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-lin_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f998f3590b5ebdc35388a9a875a6358366684260 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-lin_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-lin_Latn +doc_to_target: sentence_lin_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nLingala: " +include: flores +task: flores_eng_Latn-lin_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-lug_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-lug_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3416989fdba2777eb13664a7db4409f091bb0b75 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-lug_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-lug_Latn +doc_to_target: sentence_lug_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nLuganda: " +include: flores +task: flores_eng_Latn-lug_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-luo_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-luo_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8a56e1482037f8728945929f991a9217a9d91f05 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-luo_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-luo_Latn +doc_to_target: sentence_luo_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nLuo: " +include: flores +task: flores_eng_Latn-luo_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-mos_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-mos_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..393862689847cd6d3c6701696f57bea6f190c564 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-mos_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-mos_Latn +doc_to_target: sentence_mos_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nMossi: " +include: flores +task: flores_eng_Latn-mos_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-nso_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-nso_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..86bd9c6bcdc72f750dd3ae245c992e754ec6b55b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-nso_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-nso_Latn +doc_to_target: sentence_nso_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nNorthern Sotho: " +include: flores +task: flores_eng_Latn-nso_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-nus_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-nus_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5ac9148958f11e460266f4ff55aab4b44263074c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-nus_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-nus_Latn +doc_to_target: sentence_nus_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nNuer: " +include: flores +task: flores_eng_Latn-nus_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-nya_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-nya_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f4e35d78e708d25ad57abf36bb3ef4230f1acd66 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-nya_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-nya_Latn +doc_to_target: sentence_nya_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nNyanja: " +include: flores +task: flores_eng_Latn-nya_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-plt_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-plt_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e07ffcd257e91028807b37bec7c259f04fd3adb9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-plt_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-plt_Latn +doc_to_target: sentence_plt_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nPlateau Malagasy: " +include: flores +task: flores_eng_Latn-plt_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-run_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-run_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cad3666bdb35072af569f2405ea106c487121d57 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-run_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-run_Latn +doc_to_target: sentence_run_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nRundi: " +include: flores +task: flores_eng_Latn-run_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-sag_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-sag_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9eaa3c8995add71ccf7052a621d92e76cd06861c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-sag_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-sag_Latn +doc_to_target: sentence_sag_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nSango: " +include: flores +task: flores_eng_Latn-sag_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-sna_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-sna_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..16f70ba79f218b67b8e3efb730cde9d903ea38b3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-sna_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-sna_Latn +doc_to_target: sentence_sna_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nShona: " +include: flores +task: flores_eng_Latn-sna_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-som_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-som_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b628b7a4eddf4a073c2b6a77c6ed295c0f9cca17 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-som_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-som_Latn +doc_to_target: sentence_som_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nSomali: " +include: flores +task: flores_eng_Latn-som_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-sot_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-sot_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..62655dff56701879705027502b23986c3bd96f78 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-sot_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-sot_Latn +doc_to_target: sentence_sot_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nSouthern Sotho: " +include: flores +task: flores_eng_Latn-sot_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-ssw_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-ssw_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c247e565f839de423ac4aeecc79198189471d126 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-ssw_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-ssw_Latn +doc_to_target: sentence_ssw_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nSwati: " +include: flores +task: flores_eng_Latn-ssw_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-sun_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-sun_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ee3c4a5712ea83c5ab676c949c77192ba4f84735 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-sun_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-sun_Latn +doc_to_target: sentence_sun_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nSundanese: " +include: flores +task: flores_eng_Latn-sun_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-swh_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-swh_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2b464c16601d0e7e385a6df32b83fcde41d24c91 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-swh_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-swh_Latn +doc_to_target: sentence_swh_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nSwahili: " +include: flores +task: flores_eng_Latn-swh_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-taq_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-taq_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cc50d54faa83f621f08241f59baf6a14e4b6c674 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-taq_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-taq_Latn +doc_to_target: sentence_taq_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nTamasheq: " +include: flores +task: flores_eng_Latn-taq_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-taq_Tfng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-taq_Tfng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c8c0045338ad054dd48750ae88767f958f0b9e4f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-taq_Tfng.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-taq_Tfng +doc_to_target: sentence_taq_Tfng +doc_to_text: "English: {{sentence_eng_Latn}} \nTamasheq (Tifinagh script): " +include: flores +task: flores_eng_Latn-taq_Tfng_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-tir_Ethi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-tir_Ethi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2d3110696c8c1088d2ad2683d8d9d45b3415038d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-tir_Ethi.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-tir_Ethi +doc_to_target: sentence_tir_Ethi +doc_to_text: "English: {{sentence_eng_Latn}} \nTigrinya: " +include: flores +task: flores_eng_Latn-tir_Ethi_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-tso_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-tso_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..85bca5e9ffdf083cca344501fb818d7e0e60b732 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-tso_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-tso_Latn +doc_to_target: sentence_tso_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nTsonga: " +include: flores +task: flores_eng_Latn-tso_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-tzm_Tfng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-tzm_Tfng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..28728f412fdbe74d85d062a316eeb385b04b94a1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-tzm_Tfng.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-tzm_Tfng +doc_to_target: sentence_tzm_Tfng +doc_to_text: "English: {{sentence_eng_Latn}} \nCentral Atlas Tamazight: " +include: flores +task: flores_eng_Latn-tzm_Tfng_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-umb_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-umb_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bd95ac316ab006df8c4a52867ae3fdafafa36da2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-umb_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-umb_Latn +doc_to_target: sentence_umb_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nUmbundu: " +include: flores +task: flores_eng_Latn-umb_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-zul_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-zul_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..62de546051295ff6b413870f9eee5e806151f4cd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-zul_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-zul_Latn +doc_to_target: sentence_zul_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nZulu: " +include: flores +task: flores_eng_Latn-zul_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_ace_Arab-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_ace_Arab-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dd54b6c84dc428721616791f698ce55f5064aef4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_ace_Arab-eng_Latn.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: ace_Arab-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Acehnese (Arabic\ + \ script) sentences to English \nAcehnese (Arabic script): {{sentence_ace_Arab}}\n\ + English: " +include: flores +task: flores_ace_Arab-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_ace_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_ace_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b0814b27f80b3ac85e70179a124708ed5f9c3ac4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_ace_Latn-eng_Latn.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: ace_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Acehnese (Latin\ + \ script) sentences to English \nAcehnese (Latin script): {{sentence_ace_Latn}}\n\ + English: " +include: flores +task: flores_ace_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_acq_Arab-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_acq_Arab-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d1464b4965d566d628dcfa12583003a972af29aa --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_acq_Arab-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: acq_Arab-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Ta’izzi-Adeni\ + \ Arabic sentences to English \nTa’izzi-Adeni Arabic: {{sentence_acq_Arab}}\nEnglish: " +include: flores +task: flores_acq_Arab-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_aeb_Arab-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_aeb_Arab-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2bbded5ff0d48bf2bbd582dbfa97cfa4d222b5c9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_aeb_Arab-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: aeb_Arab-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Tunisian Arabic\ + \ sentences to English \nTunisian Arabic: {{sentence_aeb_Arab}}\nEnglish: " +include: flores +task: flores_aeb_Arab-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_afr_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_afr_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7b5847d8a997310bd191a3e0009d24abdedb6e4f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_afr_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: afr_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Afrikaans sentences\ + \ to English \nAfrikaans: {{sentence_afr_Latn}}\nEnglish: " +include: flores +task: flores_afr_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_aka_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_aka_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5f9493c5ea18493d4b0a2f2070135b5fb01c0692 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_aka_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: aka_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Akan sentences\ + \ to English \nAkan: {{sentence_aka_Latn}}\nEnglish: " +include: flores +task: flores_aka_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_ary_Arab-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_ary_Arab-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..feecf4510ac45bf5d364cf1782942e388c2119eb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_ary_Arab-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: ary_Arab-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Moroccan Arabic\ + \ sentences to English \nMoroccan Arabic: {{sentence_ary_Arab}}\nEnglish: " +include: flores +task: flores_ary_Arab-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_arz_Arab-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_arz_Arab-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..13f3e18b5d6ecd8d911741e4fe1d3ee7720f81df --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_arz_Arab-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: arz_Arab-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Egyptian Arabic\ + \ sentences to English \nEgyptian Arabic: {{sentence_arz_Arab}}\nEnglish: " +include: flores +task: flores_arz_Arab-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_bam_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_bam_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a258d264b26a9c02be0ca900f058500b19f0c256 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_bam_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: bam_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Bambara sentences\ + \ to English \nBambara: {{sentence_bam_Latn}}\nEnglish: " +include: flores +task: flores_bam_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_ban_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_ban_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c19cf00874080b980b18e811f61d31a332ca2a5b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_ban_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: ban_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Balinese sentences\ + \ to English \nBalinese: {{sentence_ban_Latn}}\nEnglish: " +include: flores +task: flores_ban_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_bem_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_bem_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9500a3b37c033a026795f47cc74c6a7df94325c1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_bem_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: bem_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Bemba sentences\ + \ to English \nBemba: {{sentence_bem_Latn}}\nEnglish: " +include: flores +task: flores_bem_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_cjk_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_cjk_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..58185199f7dcfcff5a9c5c4ba501ea7edaa4b26f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_cjk_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: cjk_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Chokwe sentences\ + \ to English \nChokwe: {{sentence_cjk_Latn}}\nEnglish: " +include: flores +task: flores_cjk_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_dik_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_dik_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4c9090a56c686036f07c63b3b92bafad068d811e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_dik_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: dik_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Southwestern Dinka\ + \ sentences to English \nSouthwestern Dinka: {{sentence_dik_Latn}}\nEnglish: " +include: flores +task: flores_dik_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_dyu_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_dyu_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..47187fb0817ddf7c04041f6b2b9ef358138dfdf8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_dyu_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: dyu_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Dyula sentences\ + \ to English \nDyula: {{sentence_dyu_Latn}}\nEnglish: " +include: flores +task: flores_dyu_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_ewe_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_ewe_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8838bc3a03742e9e4da5a387f995dcc94018b5d9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_ewe_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: ewe_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Ewe sentences\ + \ to English \nEwe: {{sentence_ewe_Latn}}\nEnglish: " +include: flores +task: flores_ewe_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_fon_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_fon_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7874a87cecb89c402a0e0c1ffea473f4283cc58f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_fon_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: fon_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Fon sentences\ + \ to English \nFon: {{sentence_fon_Latn}}\nEnglish: " +include: flores +task: flores_fon_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_fra_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_fra_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bb84246ef4b7942f1ecac940e1f80d1664faef3e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_fra_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: fra_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following French sentences\ + \ to English \nFrench: {{sentence_fra_Latn}}\nEnglish: " +include: flores +task: flores_fra_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_fuv_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_fuv_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0686706d533616d7c1809313c1f5bd302b3c1a45 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_fuv_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: fuv_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Nigerian Fulfulde\ + \ sentences to English \nNigerian Fulfulde: {{sentence_fuv_Latn}}\nEnglish: " +include: flores +task: flores_fuv_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_gaz_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_gaz_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f0ba07a6112f9dc2b740a868a5df7673dfba7650 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_gaz_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: gaz_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Oromo sentences\ + \ to English \nOromo: {{sentence_gaz_Latn}}\nEnglish: " +include: flores +task: flores_gaz_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_hau_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_hau_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..85647455d58aabfb824508bae282e15f63ccaa18 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_hau_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: hau_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Hausa sentences\ + \ to English \nHausa: {{sentence_hau_Latn}}\nEnglish: " +include: flores +task: flores_hau_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_ibo_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_ibo_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c401f1e75be4f83b89a714d40eaaa97ffe677e36 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_ibo_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: ibo_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Igbo sentences\ + \ to English \nIgbo: {{sentence_ibo_Latn}}\nEnglish: " +include: flores +task: flores_ibo_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_kab_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_kab_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c82946b9143ce0706f9768bb078d6bdc6541ebbc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_kab_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: kab_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Kabyle sentences\ + \ to English \nKabyle: {{sentence_kab_Latn}}\nEnglish: " +include: flores +task: flores_kab_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_kam_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_kam_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8661bf6e5694498af0e29ca23e9e037ddd77adc9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_kam_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: kam_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Kamba sentences\ + \ to English \nKamba: {{sentence_kam_Latn}}\nEnglish: " +include: flores +task: flores_kam_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_kea_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_kea_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d078c293ab75a286b9ac717f9322d8fd92d0b585 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_kea_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: kea_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Kabuverdianu sentences\ + \ to English \nKabuverdianu: {{sentence_kea_Latn}}\nEnglish: " +include: flores +task: flores_kea_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_kik_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_kik_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..346dcb98be91749b59747194a47cefe40ca3eef4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_kik_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: kik_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Kikuyu sentences\ + \ to English \nKikuyu: {{sentence_kik_Latn}}\nEnglish: " +include: flores +task: flores_kik_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_kin_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_kin_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7210e7e6b21c0458983917f659518ba666d45d0a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_kin_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: kin_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Kinyarwanda sentences\ + \ to English \nKinyarwanda: {{sentence_kin_Latn}}\nEnglish: " +include: flores +task: flores_kin_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_kmb_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_kmb_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b3dc8d5ac6d52ccfb84f6b8fc416ab61eaa7c007 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_kmb_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: kmb_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Kimbundu sentences\ + \ to English \nKimbundu: {{sentence_kmb_Latn}}\nEnglish: " +include: flores +task: flores_kmb_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_knc_Arab-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_knc_Arab-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..37d5d624ff0f55b15649c5468f215b069efd4bcb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_knc_Arab-eng_Latn.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: knc_Arab-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Central Kanuri\ + \ (Arabic script) sentences to English \nCentral Kanuri (Arabic script): {{sentence_knc_Arab}}\n\ + English: " +include: flores +task: flores_knc_Arab-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_kon_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_kon_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..63b3539cf9f361ba4b0786596bf43d934c85478c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_kon_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: kon_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Kikongo sentences\ + \ to English \nKikongo: {{sentence_kon_Latn}}\nEnglish: " +include: flores +task: flores_kon_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_lin_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_lin_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..82543c70d5b26ed5b79288c0775a6d21216bfbe8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_lin_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: lin_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Lingala sentences\ + \ to English \nLingala: {{sentence_lin_Latn}}\nEnglish: " +include: flores +task: flores_lin_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_lua_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_lua_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..af0796cc47138b67137e58ba44835ffa9ebf8596 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_lua_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: lua_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Luba-Kasai sentences\ + \ to English \nLuba-Kasai: {{sentence_lua_Latn}}\nEnglish: " +include: flores +task: flores_lua_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_lug_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_lug_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..eb9f47bcaf95caaac8de7730dba8f662dac0230c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_lug_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: lug_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Luganda sentences\ + \ to English \nLuganda: {{sentence_lug_Latn}}\nEnglish: " +include: flores +task: flores_lug_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_luo_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_luo_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6000ab87662f6753b7dd98d97dc1c057c6c23b58 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_luo_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: luo_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Luo sentences\ + \ to English \nLuo: {{sentence_luo_Latn}}\nEnglish: " +include: flores +task: flores_luo_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_sna_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_sna_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f0c98f038617264c525edf3d0df5325e05da55ce --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_sna_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: sna_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Shona sentences\ + \ to English \nShona: {{sentence_sna_Latn}}\nEnglish: " +include: flores +task: flores_sna_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_ssw_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_ssw_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5ae236e5cbc21cba7724cb345d78ed20097b351b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_ssw_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: ssw_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Swati sentences\ + \ to English \nSwati: {{sentence_ssw_Latn}}\nEnglish: " +include: flores +task: flores_ssw_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_sun_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_sun_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5a697a2194eaf477a40ca3f56787caaf563d2179 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_sun_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: sun_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Sundanese sentences\ + \ to English \nSundanese: {{sentence_sun_Latn}}\nEnglish: " +include: flores +task: flores_sun_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-ace_Arab.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-ace_Arab.yaml new file mode 100644 index 0000000000000000000000000000000000000000..30890d9150172e44d453679cea878790d1153f95 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-ace_Arab.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: eng_Latn-ace_Arab +doc_to_target: sentence_ace_Arab +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Acehnese (Arabic script) \nEnglish: {{sentence_eng_Latn}} \nAcehnese (Arabic\ + \ script): " +include: flores +task: flores_eng_Latn-ace_Arab_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-acq_Arab.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-acq_Arab.yaml new file mode 100644 index 0000000000000000000000000000000000000000..630c824e342e8032926204ca41cfdbc6472c35eb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-acq_Arab.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-acq_Arab +doc_to_target: sentence_acq_Arab +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Ta’izzi-Adeni Arabic \nEnglish: {{sentence_eng_Latn}} \nTa’izzi-Adeni Arabic: " +include: flores +task: flores_eng_Latn-acq_Arab_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-aeb_Arab.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-aeb_Arab.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0df4f642f499163c22733c8d1c7397f9949054c6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-aeb_Arab.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-aeb_Arab +doc_to_target: sentence_aeb_Arab +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Tunisian Arabic \nEnglish: {{sentence_eng_Latn}} \nTunisian Arabic: " +include: flores +task: flores_eng_Latn-aeb_Arab_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-aka_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-aka_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..624149c7fb2c031e4483050382e7fe12a09ad32f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-aka_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-aka_Latn +doc_to_target: sentence_aka_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Akan \nEnglish: {{sentence_eng_Latn}} \nAkan: " +include: flores +task: flores_eng_Latn-aka_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-ary_Arab.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-ary_Arab.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bb814d766327c71ebfd38d4ca046f2484aa3d3d1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-ary_Arab.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-ary_Arab +doc_to_target: sentence_ary_Arab +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Moroccan Arabic \nEnglish: {{sentence_eng_Latn}} \nMoroccan Arabic: " +include: flores +task: flores_eng_Latn-ary_Arab_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-cjk_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-cjk_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..38c4ea6ff6d0358ddf49b934b4f21549fb7b14d2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-cjk_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-cjk_Latn +doc_to_target: sentence_cjk_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Chokwe \nEnglish: {{sentence_eng_Latn}} \nChokwe: " +include: flores +task: flores_eng_Latn-cjk_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-hau_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-hau_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..aad0a48b3277c8150cb5679b4b8b77636d04b5c6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-hau_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-hau_Latn +doc_to_target: sentence_hau_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Hausa \nEnglish: {{sentence_eng_Latn}} \nHausa: " +include: flores +task: flores_eng_Latn-hau_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-kbp_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-kbp_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b04cbdf144d5a9718f5a1f9ae38158952e6975e7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-kbp_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-kbp_Latn +doc_to_target: sentence_kbp_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Kabiyè \nEnglish: {{sentence_eng_Latn}} \nKabiyè: " +include: flores +task: flores_eng_Latn-kbp_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_twi_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_twi_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4cc0d674c8ced1f005eef5a94c87a886a6f176aa --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_twi_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: twi_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Twi and English linguist, translate the following Twi sentences\ + \ to English \nTwi: {{sentence_twi_Latn}}\nEnglish: " +include: flores +task: flores_twi_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_tzm_Tfng-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_tzm_Tfng-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d3575ccb2a766a722cbc88880f34656af8cdb3a4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_tzm_Tfng-eng_Latn.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: tzm_Tfng-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Central Atlas Tamazight and English linguist, translate the following\ + \ Central Atlas Tamazight sentences to English \nCentral Atlas Tamazight: {{sentence_tzm_Tfng}}\n\ + English: " +include: flores +task: flores_tzm_Tfng-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_yor_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_yor_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5bbd8eb967ea1f4c6afe09bef65e379c4fed9c25 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_yor_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: yor_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Yoruba and English linguist, translate the following Yoruba sentences\ + \ to English \nYoruba: {{sentence_yor_Latn}}\nEnglish: " +include: flores +task: flores_yor_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-knc_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-knc_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..262d0c1f3b8efc51c35e7154f27a9a4e6ed1405f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-knc_Latn.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: eng_Latn-knc_Latn +doc_to_target: sentence_knc_Latn +doc_to_text: "As a Central Kanuri (Latin script) and English linguist, translate the\ + \ following English sentences to Central Kanuri (Latin script) \nEnglish: {{sentence_eng_Latn}}\ + \ \nCentral Kanuri (Latin script): " +include: flores +task: flores_eng_Latn-knc_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-lug_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-lug_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4dfc626b9fdbcde3de0383b5d512365137da00b5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-lug_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-lug_Latn +doc_to_target: sentence_lug_Latn +doc_to_text: "As a Luganda and English linguist, translate the following English sentences\ + \ to Luganda \nEnglish: {{sentence_eng_Latn}} \nLuganda: " +include: flores +task: flores_eng_Latn-lug_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-nso_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-nso_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..44839d82cac5af78f74b2f382d36cbd74f93baa6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-nso_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-nso_Latn +doc_to_target: sentence_nso_Latn +doc_to_text: "As a Northern Sotho and English linguist, translate the following English\ + \ sentences to Northern Sotho \nEnglish: {{sentence_eng_Latn}} \nNorthern Sotho: " +include: flores +task: flores_eng_Latn-nso_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-nus_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-nus_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..387e4341f0761727d3e07a8748291aac574c727f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-nus_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-nus_Latn +doc_to_target: sentence_nus_Latn +doc_to_text: "As a Nuer and English linguist, translate the following English sentences\ + \ to Nuer \nEnglish: {{sentence_eng_Latn}} \nNuer: " +include: flores +task: flores_eng_Latn-nus_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-plt_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-plt_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..afc81158cbed0e268746ee51c1d4e1071f4315e0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-plt_Latn.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: eng_Latn-plt_Latn +doc_to_target: sentence_plt_Latn +doc_to_text: "As a Plateau Malagasy and English linguist, translate the following\ + \ English sentences to Plateau Malagasy \nEnglish: {{sentence_eng_Latn}} \nPlateau\ + \ Malagasy: " +include: flores +task: flores_eng_Latn-plt_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-run_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-run_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..519700cd32de76f039a6f1d3ce16a4c539278334 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-run_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-run_Latn +doc_to_target: sentence_run_Latn +doc_to_text: "As a Rundi and English linguist, translate the following English sentences\ + \ to Rundi \nEnglish: {{sentence_eng_Latn}} \nRundi: " +include: flores +task: flores_eng_Latn-run_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-sag_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-sag_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fa99b16137861e1e9fb4f19669dbb71977fd3cc1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-sag_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-sag_Latn +doc_to_target: sentence_sag_Latn +doc_to_text: "As a Sango and English linguist, translate the following English sentences\ + \ to Sango \nEnglish: {{sentence_eng_Latn}} \nSango: " +include: flores +task: flores_eng_Latn-sag_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-sna_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-sna_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fd7ac49ac5854133223a599569981f4c27d19a21 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-sna_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-sna_Latn +doc_to_target: sentence_sna_Latn +doc_to_text: "As a Shona and English linguist, translate the following English sentences\ + \ to Shona \nEnglish: {{sentence_eng_Latn}} \nShona: " +include: flores +task: flores_eng_Latn-sna_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-som_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-som_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..17870addf00ed2c4f0c4e126fc094a06aabc8027 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-som_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-som_Latn +doc_to_target: sentence_som_Latn +doc_to_text: "As a Somali and English linguist, translate the following English sentences\ + \ to Somali \nEnglish: {{sentence_eng_Latn}} \nSomali: " +include: flores +task: flores_eng_Latn-som_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-ssw_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-ssw_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0dbd162772b8938aef4f05da00ac3da0ce3be530 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-ssw_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-ssw_Latn +doc_to_target: sentence_ssw_Latn +doc_to_text: "As a Swati and English linguist, translate the following English sentences\ + \ to Swati \nEnglish: {{sentence_eng_Latn}} \nSwati: " +include: flores +task: flores_eng_Latn-ssw_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-sun_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-sun_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0f8f6339450e8af7318e570290370a21037ba98d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-sun_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-sun_Latn +doc_to_target: sentence_sun_Latn +doc_to_text: "As a Sundanese and English linguist, translate the following English\ + \ sentences to Sundanese \nEnglish: {{sentence_eng_Latn}} \nSundanese: " +include: flores +task: flores_eng_Latn-sun_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-swh_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-swh_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..20971c5383cbcce97b3743262adc35c9d2dfadcf --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-swh_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-swh_Latn +doc_to_target: sentence_swh_Latn +doc_to_text: "As a Swahili and English linguist, translate the following English sentences\ + \ to Swahili \nEnglish: {{sentence_eng_Latn}} \nSwahili: " +include: flores +task: flores_eng_Latn-swh_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-taq_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-taq_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bdb06f77b78b9ca296eb960fc62a63a94e990fd8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-taq_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-taq_Latn +doc_to_target: sentence_taq_Latn +doc_to_text: "As a Tamasheq and English linguist, translate the following English\ + \ sentences to Tamasheq \nEnglish: {{sentence_eng_Latn}} \nTamasheq: " +include: flores +task: flores_eng_Latn-taq_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-taq_Tfng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-taq_Tfng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d690651ddb21eaf3022475e98dc8f5ddf72f073b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-taq_Tfng.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: eng_Latn-taq_Tfng +doc_to_target: sentence_taq_Tfng +doc_to_text: "As a Tamasheq (Tifinagh script) and English linguist, translate the\ + \ following English sentences to Tamasheq (Tifinagh script) \nEnglish: {{sentence_eng_Latn}}\ + \ \nTamasheq (Tifinagh script): " +include: flores +task: flores_eng_Latn-taq_Tfng_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-tir_Ethi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-tir_Ethi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c6b3ba347ab935d9c87a48262a9cafb259985ea0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-tir_Ethi.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-tir_Ethi +doc_to_target: sentence_tir_Ethi +doc_to_text: "As a Tigrinya and English linguist, translate the following English\ + \ sentences to Tigrinya \nEnglish: {{sentence_eng_Latn}} \nTigrinya: " +include: flores +task: flores_eng_Latn-tir_Ethi_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-tsn_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-tsn_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..845626f5f347436a0fcd2c04fe3446cb806da44e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-tsn_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-tsn_Latn +doc_to_target: sentence_tsn_Latn +doc_to_text: "As a Setswana and English linguist, translate the following English\ + \ sentences to Setswana \nEnglish: {{sentence_eng_Latn}} \nSetswana: " +include: flores +task: flores_eng_Latn-tsn_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-tso_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-tso_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..958411f89ea39c77cc329ce5f14795761ec1f1a9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-tso_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-tso_Latn +doc_to_target: sentence_tso_Latn +doc_to_text: "As a Tsonga and English linguist, translate the following English sentences\ + \ to Tsonga \nEnglish: {{sentence_eng_Latn}} \nTsonga: " +include: flores +task: flores_eng_Latn-tso_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-tum_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-tum_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..95e6efa7dbdd8ea987bd680954e69e97742c452d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-tum_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-tum_Latn +doc_to_target: sentence_tum_Latn +doc_to_text: "As a Tumbuka and English linguist, translate the following English sentences\ + \ to Tumbuka \nEnglish: {{sentence_eng_Latn}} \nTumbuka: " +include: flores +task: flores_eng_Latn-tum_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-twi_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-twi_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0dcb20543ccff51461bc2af582fcdfaa5855d25f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-twi_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-twi_Latn +doc_to_target: sentence_twi_Latn +doc_to_text: "As a Twi and English linguist, translate the following English sentences\ + \ to Twi \nEnglish: {{sentence_eng_Latn}} \nTwi: " +include: flores +task: flores_eng_Latn-twi_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-tzm_Tfng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-tzm_Tfng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..887344c67c0fe6cd1658c8172b55a252bfc9f12b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-tzm_Tfng.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: eng_Latn-tzm_Tfng +doc_to_target: sentence_tzm_Tfng +doc_to_text: "As a Central Atlas Tamazight and English linguist, translate the following\ + \ English sentences to Central Atlas Tamazight \nEnglish: {{sentence_eng_Latn}}\ + \ \nCentral Atlas Tamazight: " +include: flores +task: flores_eng_Latn-tzm_Tfng_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-umb_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-umb_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b8c4adc0139f6f2d9ce4fcd910e3f558b96df78d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-umb_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-umb_Latn +doc_to_target: sentence_umb_Latn +doc_to_text: "As a Umbundu and English linguist, translate the following English sentences\ + \ to Umbundu \nEnglish: {{sentence_eng_Latn}} \nUmbundu: " +include: flores +task: flores_eng_Latn-umb_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-wol_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-wol_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..66ad25794e67d54123f2a63ffad5e98d12c6ce59 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-wol_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-wol_Latn +doc_to_target: sentence_wol_Latn +doc_to_text: "As a Wolof and English linguist, translate the following English sentences\ + \ to Wolof \nEnglish: {{sentence_eng_Latn}} \nWolof: " +include: flores +task: flores_eng_Latn-wol_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-xho_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-xho_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8cd2fe08ec7bcd000182a7a9f080e739b6d82289 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-xho_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-xho_Latn +doc_to_target: sentence_xho_Latn +doc_to_text: "As a Xhosa and English linguist, translate the following English sentences\ + \ to Xhosa \nEnglish: {{sentence_eng_Latn}} \nXhosa: " +include: flores +task: flores_eng_Latn-xho_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-yor_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-yor_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..09562458138acb1d146e5319816b01e2205351ce --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-yor_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-yor_Latn +doc_to_target: sentence_yor_Latn +doc_to_text: "As a Yoruba and English linguist, translate the following English sentences\ + \ to Yoruba \nEnglish: {{sentence_eng_Latn}} \nYoruba: " +include: flores +task: flores_eng_Latn-yor_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-zul_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-zul_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..15b41952e52b3acfee85294c372ecb954f8b37cc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-zul_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-zul_Latn +doc_to_target: sentence_zul_Latn +doc_to_text: "As a Zulu and English linguist, translate the following English sentences\ + \ to Zulu \nEnglish: {{sentence_eng_Latn}} \nZulu: " +include: flores +task: flores_eng_Latn-zul_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent new file mode 100644 index 0000000000000000000000000000000000000000..a77bc5c95941392779b960df6ad26ebebe5ba96d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent @@ -0,0 +1,75 @@ +tag: +- injongointent_tasks +- injongointent_prompt_1 +dataset_path: masakhane/InjongoIntent +dataset_name: null +output_type: multiple_choice +validation_split: validation +test_split: test +fewshot_split: validation +doc_to_target: intent +doc_to_choice: + - alarm + - balance + - bill_balance + - book_flight + - book_hotel + - calendar_update + - cancel_reservation + - car_rental + - confirm_reservation + - cook_time + - exchange_rate + - food_last + - freeze_account + - ingredients_list + - interest_rate + - international_visa + - make_call + - meal_suggestion + - min_payment + - pay_bill + - pin_change + - play_music + - plug_type + - recipe + - restaurant_reservation + - restaurant_reviews + - restaurant_suggestion + - share_location + - shopping_list_update + - spending_history + - text + - time + - timezone + - transactions + - transfer + - translate + - travel_notification + - travel_suggestion + - update_playlist + - weather +should_decontaminate: true +doc_to_decontamination_query: text +metric_list: + - metric: f1 + aggregation: !function utils.weighted_f1_score + # aggregation: mean + average: weighted + hf_evaluate: true + higher_is_better: True + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" + - metric: acc + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_lin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_lin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2cc08df484bf58f9eaf2d498074eb1ac5dc72338 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_lin.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: lin +doc_to_text: 'Given the text: ''{{text}}'', determine the correct intent from the + following list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Only + output one intent from the list.' +include: injongointent +task: injongointent_lin_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b1a421577bbe8ff9ce34b00da45fdb3efcb22e9d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_lug.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: lug +doc_to_text: 'Given the text: ''{{text}}'', determine the correct intent from the + following list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Only + output one intent from the list.' +include: injongointent +task: injongointent_lug_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fc3d797c05c018baf599dcee28bccc0dd5c5ab72 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_swa.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: swa +doc_to_text: 'Given the text: ''{{text}}'', determine the correct intent from the + following list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Only + output one intent from the list.' +include: injongointent +task: injongointent_swa_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7d359d2f8e4cb91028705cc8c76842c9b5e78c3c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_wol.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: wol +doc_to_text: 'Given the text: ''{{text}}'', determine the correct intent from the + following list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Only + output one intent from the list.' +include: injongointent +task: injongointent_wol_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4d9c173aac2832665724358697d59d3bf8f38e56 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_xho.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: xho +doc_to_text: 'Given the text: ''{{text}}'', determine the correct intent from the + following list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Only + output one intent from the list.' +include: injongointent +task: injongointent_xho_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1d38a78141e1912ff995f6d87c0753f317ec6ad0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_zul.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: zul +doc_to_text: 'Given the text: ''{{text}}'', determine the correct intent from the + following list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Only + output one intent from the list.' +include: injongointent +task: injongointent_zul_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..3e735e2deb1f9c53152c072615aebe8ba3acb90b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/utils.py @@ -0,0 +1 @@ +from lm_eval.utils import weighted_f1_score diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent new file mode 100644 index 0000000000000000000000000000000000000000..dfcb82678a61524c08bfd2d7e2d2ec0a50330f27 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent @@ -0,0 +1,75 @@ +tag: +- injongointent_tasks +- injongointent_prompt_2 +dataset_path: masakhane/InjongoIntent +dataset_name: null +output_type: multiple_choice +validation_split: validation +test_split: test +fewshot_split: validation +doc_to_target: intent +doc_to_choice: + - alarm + - balance + - bill_balance + - book_flight + - book_hotel + - calendar_update + - cancel_reservation + - car_rental + - confirm_reservation + - cook_time + - exchange_rate + - food_last + - freeze_account + - ingredients_list + - interest_rate + - international_visa + - make_call + - meal_suggestion + - min_payment + - pay_bill + - pin_change + - play_music + - plug_type + - recipe + - restaurant_reservation + - restaurant_reviews + - restaurant_suggestion + - share_location + - shopping_list_update + - spending_history + - text + - time + - timezone + - transactions + - transfer + - translate + - travel_notification + - travel_suggestion + - update_playlist + - weather +should_decontaminate: true +doc_to_decontamination_query: text +metric_list: + - metric: f1 + aggregation: !function utils.weighted_f1_score + # aggregation: mean + average: weighted + hf_evaluate: true + higher_is_better: True + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" + - metric: acc + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7c1b2189ac768bef8c6263ccc41a003cd00ee6d8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_amh.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: amh +doc_to_text: 'Analyze the text: ''{{text}}''. Choose the most appropriate intent from + these options: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Respond + with only the selected intent.' +include: injongointent +task: injongointent_amh_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_ewe.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_ewe.yaml new file mode 100644 index 0000000000000000000000000000000000000000..58eb914472609d29836edb92f5c676deac50166a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_ewe.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: ewe +doc_to_text: 'Analyze the text: ''{{text}}''. Choose the most appropriate intent from + these options: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Respond + with only the selected intent.' +include: injongointent +task: injongointent_ewe_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e7745369ac6af719ea71e9f4e3032bcc7b66a8ac --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_hau.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: hau +doc_to_text: 'Analyze the text: ''{{text}}''. Choose the most appropriate intent from + these options: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Respond + with only the selected intent.' +include: injongointent +task: injongointent_hau_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b47052d71829c170849a07957a0412b8f21bddb2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_ibo.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: ibo +doc_to_text: 'Analyze the text: ''{{text}}''. Choose the most appropriate intent from + these options: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Respond + with only the selected intent.' +include: injongointent +task: injongointent_ibo_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..457935674bd84bb39e96d8ba000497430963f983 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_kin.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: kin +doc_to_text: 'Analyze the text: ''{{text}}''. Choose the most appropriate intent from + these options: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Respond + with only the selected intent.' +include: injongointent +task: injongointent_kin_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_lin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_lin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..54e7fcb71bb91ce4272dbd4723b7e2059da71c12 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_lin.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: lin +doc_to_text: 'Analyze the text: ''{{text}}''. Choose the most appropriate intent from + these options: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Respond + with only the selected intent.' +include: injongointent +task: injongointent_lin_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..96aa42fc9e2bed859ca386091267ca48b9566399 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_lug.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: lug +doc_to_text: 'Analyze the text: ''{{text}}''. Choose the most appropriate intent from + these options: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Respond + with only the selected intent.' +include: injongointent +task: injongointent_lug_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_orm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_orm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..872f96c542cd697bef3b0f92294e9247cedb458a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_orm.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: orm +doc_to_text: 'Analyze the text: ''{{text}}''. Choose the most appropriate intent from + these options: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Respond + with only the selected intent.' +include: injongointent +task: injongointent_orm_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_sna.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_sna.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a62dfe34a4611319a56bc06f7bb01c447ac7ad6f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_sna.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: sna +doc_to_text: 'Analyze the text: ''{{text}}''. Choose the most appropriate intent from + these options: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Respond + with only the selected intent.' +include: injongointent +task: injongointent_sna_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_sot.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_sot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e9ca6a5675529c7bb3701d332c20eb1c2af19d53 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_sot.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: sot +doc_to_text: 'Analyze the text: ''{{text}}''. Choose the most appropriate intent from + these options: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Respond + with only the selected intent.' +include: injongointent +task: injongointent_sot_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..339f66ac8d569245f6e8cab2f3456fcea24713d4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_swa.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: swa +doc_to_text: 'Analyze the text: ''{{text}}''. Choose the most appropriate intent from + these options: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Respond + with only the selected intent.' +include: injongointent +task: injongointent_swa_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b758bce0493c722c0d995cdbf8cc5e4b409e4af9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_twi.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: twi +doc_to_text: 'Analyze the text: ''{{text}}''. Choose the most appropriate intent from + these options: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Respond + with only the selected intent.' +include: injongointent +task: injongointent_twi_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9b573a444b22b2e34ed10ee9f75a914f16177bfc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_wol.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: wol +doc_to_text: 'Analyze the text: ''{{text}}''. Choose the most appropriate intent from + these options: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Respond + with only the selected intent.' +include: injongointent +task: injongointent_wol_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f2c02205fdb7b9701096089c26d6813fff802dd7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_xho.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: xho +doc_to_text: 'Analyze the text: ''{{text}}''. Choose the most appropriate intent from + these options: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Respond + with only the selected intent.' +include: injongointent +task: injongointent_xho_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4c821736809b4a37d94dedc82e6943594585ce35 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_yor.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: yor +doc_to_text: 'Analyze the text: ''{{text}}''. Choose the most appropriate intent from + these options: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Respond + with only the selected intent.' +include: injongointent +task: injongointent_yor_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c8a541b66bfb14f692c1230871f2482452eb7347 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_zul.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: zul +doc_to_text: 'Analyze the text: ''{{text}}''. Choose the most appropriate intent from + these options: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Respond + with only the selected intent.' +include: injongointent +task: injongointent_zul_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..3e735e2deb1f9c53152c072615aebe8ba3acb90b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/utils.py @@ -0,0 +1 @@ +from lm_eval.utils import weighted_f1_score diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent new file mode 100644 index 0000000000000000000000000000000000000000..afdf43cfc10b75238debbd5dbab36ac493872025 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent @@ -0,0 +1,75 @@ +tag: +- injongointent_tasks +- injongointent_prompt_3 +dataset_path: masakhane/InjongoIntent +dataset_name: null +output_type: multiple_choice +validation_split: validation +test_split: test +fewshot_split: validation +doc_to_target: intent +doc_to_choice: + - alarm + - balance + - bill_balance + - book_flight + - book_hotel + - calendar_update + - cancel_reservation + - car_rental + - confirm_reservation + - cook_time + - exchange_rate + - food_last + - freeze_account + - ingredients_list + - interest_rate + - international_visa + - make_call + - meal_suggestion + - min_payment + - pay_bill + - pin_change + - play_music + - plug_type + - recipe + - restaurant_reservation + - restaurant_reviews + - restaurant_suggestion + - share_location + - shopping_list_update + - spending_history + - text + - time + - timezone + - transactions + - transfer + - translate + - travel_notification + - travel_suggestion + - update_playlist + - weather +should_decontaminate: true +doc_to_decontamination_query: text +metric_list: + - metric: f1 + aggregation: !function utils.weighted_f1_score + # aggregation: mean + average: weighted + hf_evaluate: true + higher_is_better: True + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" + - metric: acc + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7bd62c5b00eedcfb1cd254ae679426955cacb8a0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_amh.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: amh +doc_to_text: 'You are a linguistic analyst trained to understand user intent. Based + on the text: ''{{text}}'', choose the intent that best matches from this list: [alarm, + balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather]. Return only the intent.' +include: injongointent +task: injongointent_amh_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_eng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..258f0cfab4a4dc9ee8f7f8baceb2c2bac35c0c02 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_eng.yaml @@ -0,0 +1,16 @@ +# Generated by utils.py +dataset_name: eng +validation_split: train +test_split: test +fewshot_split: train +doc_to_text: 'You are a linguistic analyst trained to understand user intent. Based + on the text: ''{{text}}'', choose the intent that best matches from this list: [alarm, + balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather]. Return only the intent.' +include: injongointent +task: injongointent_eng_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_ewe.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_ewe.yaml new file mode 100644 index 0000000000000000000000000000000000000000..12688cc9d97dcacc50419a91707ac2f5fa47363b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_ewe.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: ewe +doc_to_text: 'You are a linguistic analyst trained to understand user intent. Based + on the text: ''{{text}}'', choose the intent that best matches from this list: [alarm, + balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather]. Return only the intent.' +include: injongointent +task: injongointent_ewe_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8414a09bae932411fd51cfbcf9d5efbbb2d93a45 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_hau.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: hau +doc_to_text: 'You are a linguistic analyst trained to understand user intent. Based + on the text: ''{{text}}'', choose the intent that best matches from this list: [alarm, + balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather]. Return only the intent.' +include: injongointent +task: injongointent_hau_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8f254438388e95b551fd9344758851b0a9fc8768 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_ibo.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: ibo +doc_to_text: 'You are a linguistic analyst trained to understand user intent. Based + on the text: ''{{text}}'', choose the intent that best matches from this list: [alarm, + balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather]. Return only the intent.' +include: injongointent +task: injongointent_ibo_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b946cf004f775642ad8f0b9d5995820b97187f0a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_kin.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: kin +doc_to_text: 'You are a linguistic analyst trained to understand user intent. Based + on the text: ''{{text}}'', choose the intent that best matches from this list: [alarm, + balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather]. Return only the intent.' +include: injongointent +task: injongointent_kin_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_lin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_lin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4a485d5ce807f0f9200ccdf535d5427697561bbd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_lin.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: lin +doc_to_text: 'You are a linguistic analyst trained to understand user intent. Based + on the text: ''{{text}}'', choose the intent that best matches from this list: [alarm, + balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather]. Return only the intent.' +include: injongointent +task: injongointent_lin_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..71376ec3f8a3ae742cc694ea0d19c57ad9187b0b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_lug.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: lug +doc_to_text: 'You are a linguistic analyst trained to understand user intent. Based + on the text: ''{{text}}'', choose the intent that best matches from this list: [alarm, + balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather]. Return only the intent.' +include: injongointent +task: injongointent_lug_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_orm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_orm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..706f3a908a221dd3bc876adae4ddc40b6b4cb6ea --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_orm.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: orm +doc_to_text: 'You are a linguistic analyst trained to understand user intent. Based + on the text: ''{{text}}'', choose the intent that best matches from this list: [alarm, + balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather]. Return only the intent.' +include: injongointent +task: injongointent_orm_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_sna.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_sna.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f4aca73782fbcc8516983e588271bb336f674f2b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_sna.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: sna +doc_to_text: 'You are a linguistic analyst trained to understand user intent. Based + on the text: ''{{text}}'', choose the intent that best matches from this list: [alarm, + balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather]. Return only the intent.' +include: injongointent +task: injongointent_sna_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_sot.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_sot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..57e27afab1edb9509d4f9901bea7fc114f249a09 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_sot.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: sot +doc_to_text: 'You are a linguistic analyst trained to understand user intent. Based + on the text: ''{{text}}'', choose the intent that best matches from this list: [alarm, + balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather]. Return only the intent.' +include: injongointent +task: injongointent_sot_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6cb4886d1f1d9ba9c87cbc78a436dad83d72487e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_swa.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: swa +doc_to_text: 'You are a linguistic analyst trained to understand user intent. Based + on the text: ''{{text}}'', choose the intent that best matches from this list: [alarm, + balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather]. Return only the intent.' +include: injongointent +task: injongointent_swa_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e8623bf33bcc89d2b25390b487617159263327f6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_twi.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: twi +doc_to_text: 'You are a linguistic analyst trained to understand user intent. Based + on the text: ''{{text}}'', choose the intent that best matches from this list: [alarm, + balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather]. Return only the intent.' +include: injongointent +task: injongointent_twi_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..afc3cf4a907143eb7a13dca4c4783a0c715e1524 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_wol.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: wol +doc_to_text: 'You are a linguistic analyst trained to understand user intent. Based + on the text: ''{{text}}'', choose the intent that best matches from this list: [alarm, + balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather]. Return only the intent.' +include: injongointent +task: injongointent_wol_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9f41aa561fbd6a6da39e9f90d93c4fa909ffbf72 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_xho.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: xho +doc_to_text: 'You are a linguistic analyst trained to understand user intent. Based + on the text: ''{{text}}'', choose the intent that best matches from this list: [alarm, + balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather]. Return only the intent.' +include: injongointent +task: injongointent_xho_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3a5d5686de20ff32a1a4207ea2711d2176464df5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_yor.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: yor +doc_to_text: 'You are a linguistic analyst trained to understand user intent. Based + on the text: ''{{text}}'', choose the intent that best matches from this list: [alarm, + balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather]. Return only the intent.' +include: injongointent +task: injongointent_yor_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f857ff065b13b0eb050107b884a7b57f795dc779 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_zul.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: zul +doc_to_text: 'You are a linguistic analyst trained to understand user intent. Based + on the text: ''{{text}}'', choose the intent that best matches from this list: [alarm, + balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather]. Return only the intent.' +include: injongointent +task: injongointent_zul_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..3e735e2deb1f9c53152c072615aebe8ba3acb90b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/utils.py @@ -0,0 +1 @@ +from lm_eval.utils import weighted_f1_score diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent new file mode 100644 index 0000000000000000000000000000000000000000..5d5c05ae113bdcef17764decefc59f335ddb3ba3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent @@ -0,0 +1,75 @@ +tag: +- injongointent_tasks +- injongointent_prompt_4 +dataset_path: masakhane/InjongoIntent +dataset_name: null +output_type: multiple_choice +validation_split: validation +test_split: test +fewshot_split: validation +doc_to_target: intent +doc_to_choice: + - alarm + - balance + - bill_balance + - book_flight + - book_hotel + - calendar_update + - cancel_reservation + - car_rental + - confirm_reservation + - cook_time + - exchange_rate + - food_last + - freeze_account + - ingredients_list + - interest_rate + - international_visa + - make_call + - meal_suggestion + - min_payment + - pay_bill + - pin_change + - play_music + - plug_type + - recipe + - restaurant_reservation + - restaurant_reviews + - restaurant_suggestion + - share_location + - shopping_list_update + - spending_history + - text + - time + - timezone + - transactions + - transfer + - translate + - travel_notification + - travel_suggestion + - update_playlist + - weather +should_decontaminate: true +doc_to_decontamination_query: text +metric_list: + - metric: f1 + aggregation: !function utils.weighted_f1_score + # aggregation: mean + average: weighted + hf_evaluate: true + higher_is_better: True + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" + - metric: acc + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..aa14ee5b178f7577c036039b089678bcfa697a04 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_amh.yaml @@ -0,0 +1,14 @@ +# Generated by utils.py +dataset_name: amh +doc_to_text: 'You are a Amharic linguistic analyst trained to understand Amharic user + intent. Based on the Amharictext: ''{{text}}'', choose the intent that best matches + from this list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Return + only the intent.' +include: injongointent +task: injongointent_amh_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_eng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..853e64965251e37e59efe72554557b2a378e358f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_eng.yaml @@ -0,0 +1,17 @@ +# Generated by utils.py +dataset_name: eng +validation_split: train +test_split: test +fewshot_split: train +doc_to_text: 'You are a English linguistic analyst trained to understand English user + intent. Based on the English text: ''{{text}}'', choose the intent that best matches + from this list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Return + only the intent.' +include: injongointent +task: injongointent_eng_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_ewe.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_ewe.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f61a3db57d2bcf2af8cf30dc65ac08b896013fbe --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_ewe.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: ewe +doc_to_text: 'You are a Ewe linguistic analyst trained to understand Ewe user intent. + Based on the Ewetext: ''{{text}}'', choose the intent that best matches from this + list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather]. Return only the intent.' +include: injongointent +task: injongointent_ewe_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fdef34cb847fbc8008ed496826bc2cb79361a546 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_hau.yaml @@ -0,0 +1,14 @@ +# Generated by utils.py +dataset_name: hau +doc_to_text: 'You are a Hausa linguistic analyst trained to understand Hausa user + intent. Based on the Hausatext: ''{{text}}'', choose the intent that best matches + from this list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Return + only the intent.' +include: injongointent +task: injongointent_hau_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..23b59831ed97b56ed983d26871d155b1f72a176b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_ibo.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: ibo +doc_to_text: 'You are a Igbo linguistic analyst trained to understand Igbo user intent. + Based on the Igbotext: ''{{text}}'', choose the intent that best matches from this + list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather]. Return only the intent.' +include: injongointent +task: injongointent_ibo_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..28f05aeb00423bb80f2f070763f3f22345ae4776 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_kin.yaml @@ -0,0 +1,14 @@ +# Generated by utils.py +dataset_name: kin +doc_to_text: 'You are a Kinyarwanda linguistic analyst trained to understand Kinyarwanda + user intent. Based on the Kinyarwandatext: ''{{text}}'', choose the intent that + best matches from this list: [alarm, balance, bill_balance, book_flight, book_hotel, + calendar_update, cancel_reservation, car_rental, confirm_reservation, cook_time, + exchange_rate, food_last, freeze_account, ingredients_list, interest_rate, international_visa, + make_call, meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, + recipe, restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Return + only the intent.' +include: injongointent +task: injongointent_kin_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_lin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_lin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..df991d89146112468bae83c7c5fe87eef307dc49 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_lin.yaml @@ -0,0 +1,14 @@ +# Generated by utils.py +dataset_name: lin +doc_to_text: 'You are a Lingala linguistic analyst trained to understand Lingala user + intent. Based on the Lingalatext: ''{{text}}'', choose the intent that best matches + from this list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Return + only the intent.' +include: injongointent +task: injongointent_lin_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c1abb66edb31a6a76e987e665f91f630fe8d3416 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_lug.yaml @@ -0,0 +1,14 @@ +# Generated by utils.py +dataset_name: lug +doc_to_text: 'You are a Luganda linguistic analyst trained to understand Luganda user + intent. Based on the Lugandatext: ''{{text}}'', choose the intent that best matches + from this list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Return + only the intent.' +include: injongointent +task: injongointent_lug_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_orm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_orm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..195ff4a232e782d38bae93b33648535934ff07e4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_orm.yaml @@ -0,0 +1,14 @@ +# Generated by utils.py +dataset_name: orm +doc_to_text: 'You are a Oromo linguistic analyst trained to understand Oromo user + intent. Based on the Oromotext: ''{{text}}'', choose the intent that best matches + from this list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Return + only the intent.' +include: injongointent +task: injongointent_orm_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_sna.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_sna.yaml new file mode 100644 index 0000000000000000000000000000000000000000..23d066c3d8c8b41e7184f867ba492d58a8736d82 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_sna.yaml @@ -0,0 +1,14 @@ +# Generated by utils.py +dataset_name: sna +doc_to_text: 'You are a Shona linguistic analyst trained to understand Shona user + intent. Based on the Shonatext: ''{{text}}'', choose the intent that best matches + from this list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Return + only the intent.' +include: injongointent +task: injongointent_sna_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_sot.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_sot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..82102a21e2e7bb0a3d57a6fff4dd678b587d1d95 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_sot.yaml @@ -0,0 +1,14 @@ +# Generated by utils.py +dataset_name: sot +doc_to_text: 'You are a Sotho linguistic analyst trained to understand Sotho user + intent. Based on the Sothotext: ''{{text}}'', choose the intent that best matches + from this list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Return + only the intent.' +include: injongointent +task: injongointent_sot_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..031ffbb40ceba3233e5f396e38a523f2bca81ad9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_swa.yaml @@ -0,0 +1,14 @@ +# Generated by utils.py +dataset_name: swa +doc_to_text: 'You are a Swahili linguistic analyst trained to understand Swahili user + intent. Based on the Swahilitext: ''{{text}}'', choose the intent that best matches + from this list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Return + only the intent.' +include: injongointent +task: injongointent_swa_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a569b3cec8f6858d041e9d0c0c876720558b808b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_twi.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: twi +doc_to_text: 'You are a Twi linguistic analyst trained to understand Twi user intent. + Based on the Twitext: ''{{text}}'', choose the intent that best matches from this + list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather]. Return only the intent.' +include: injongointent +task: injongointent_twi_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a55398ab4161ec54e971863cd1b4bfb41330b093 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_wol.yaml @@ -0,0 +1,14 @@ +# Generated by utils.py +dataset_name: wol +doc_to_text: 'You are a Wolof linguistic analyst trained to understand Wolof user + intent. Based on the Woloftext: ''{{text}}'', choose the intent that best matches + from this list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Return + only the intent.' +include: injongointent +task: injongointent_wol_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d773a1756e17eb5cd21b3bc550b5847acaae671c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_xho.yaml @@ -0,0 +1,14 @@ +# Generated by utils.py +dataset_name: xho +doc_to_text: 'You are a Xhosa linguistic analyst trained to understand Xhosa user + intent. Based on the Xhosatext: ''{{text}}'', choose the intent that best matches + from this list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Return + only the intent.' +include: injongointent +task: injongointent_xho_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..af01d9f3e8efc1d57003194dd0201dd76bd76fbf --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_yor.yaml @@ -0,0 +1,14 @@ +# Generated by utils.py +dataset_name: yor +doc_to_text: 'You are a Yoruba linguistic analyst trained to understand Yoruba user + intent. Based on the Yorubatext: ''{{text}}'', choose the intent that best matches + from this list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Return + only the intent.' +include: injongointent +task: injongointent_yor_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3b6e5aace303095eccb19b8a011559555cee7eb2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_zul.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: zul +doc_to_text: 'You are a Zulu linguistic analyst trained to understand Zulu user intent. + Based on the Zulutext: ''{{text}}'', choose the intent that best matches from this + list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather]. Return only the intent.' +include: injongointent +task: injongointent_zul_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..3e735e2deb1f9c53152c072615aebe8ba3acb90b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/utils.py @@ -0,0 +1 @@ +from lm_eval.utils import weighted_f1_score diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent new file mode 100644 index 0000000000000000000000000000000000000000..0012857bdaa787ad8bf9ba345330844c8b266a8e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent @@ -0,0 +1,75 @@ +tag: +- injongointent_tasks +- injongointent_prompt_5 +dataset_path: masakhane/InjongoIntent +dataset_name: null +output_type: multiple_choice +validation_split: validation +test_split: test +fewshot_split: validation +doc_to_target: intent +doc_to_choice: + - alarm + - balance + - bill_balance + - book_flight + - book_hotel + - calendar_update + - cancel_reservation + - car_rental + - confirm_reservation + - cook_time + - exchange_rate + - food_last + - freeze_account + - ingredients_list + - interest_rate + - international_visa + - make_call + - meal_suggestion + - min_payment + - pay_bill + - pin_change + - play_music + - plug_type + - recipe + - restaurant_reservation + - restaurant_reviews + - restaurant_suggestion + - share_location + - shopping_list_update + - spending_history + - text + - time + - timezone + - transactions + - transfer + - translate + - travel_notification + - travel_suggestion + - update_playlist + - weather +should_decontaminate: true +doc_to_decontamination_query: text +metric_list: + - metric: f1 + aggregation: !function utils.weighted_f1_score + # aggregation: mean + average: weighted + hf_evaluate: true + higher_is_better: True + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" + - metric: acc + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2a6623a9387a98d2d21791f51d0a8690b756c9a4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_amh.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: amh +doc_to_text: 'The following text is in Amharic: ''{{text}}''. Given the list of intents: + [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather], identify the intent expressed in the + text. Return only the identified intent.' +include: injongointent +task: injongointent_amh_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_eng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4dcbebbbd00cf8a3abb7f5e848d3a6b8520d46a3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_eng.yaml @@ -0,0 +1,16 @@ +# Generated by utils.py +dataset_name: eng +validation_split: train +test_split: test +fewshot_split: train +doc_to_text: 'The following text is in English: ''{{text}}''. Given the list of intents: + [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather], identify the intent expressed in the + text. Return only the identified intent.' +include: injongointent +task: injongointent_eng_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_ewe.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_ewe.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0cab84252dcbfdc4d1eefcd2f4aa36b031edd7ba --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_ewe.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: ewe +doc_to_text: 'The following text is in Ewe: ''{{text}}''. Given the list of intents: + [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather], identify the intent expressed in the + text. Return only the identified intent.' +include: injongointent +task: injongointent_ewe_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b6275db8383eddada828f7cb1963d554b4f8d658 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_hau.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: hau +doc_to_text: 'The following text is in Hausa: ''{{text}}''. Given the list of intents: + [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather], identify the intent expressed in the + text. Return only the identified intent.' +include: injongointent +task: injongointent_hau_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..518ec898411de5d141ac16137578e710fc7bb60e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_ibo.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: ibo +doc_to_text: 'The following text is in Igbo: ''{{text}}''. Given the list of intents: + [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather], identify the intent expressed in the + text. Return only the identified intent.' +include: injongointent +task: injongointent_ibo_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..348535c679af3c08da162c812fbfd700557e8326 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_kin.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: kin +doc_to_text: 'The following text is in Kinyarwanda: ''{{text}}''. Given the list of + intents: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather], identify + the intent expressed in the text. Return only the identified intent.' +include: injongointent +task: injongointent_kin_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_lin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_lin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..75bbf4ec5935505c554d5c7b58934188d1591532 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_lin.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: lin +doc_to_text: 'The following text is in Lingala: ''{{text}}''. Given the list of intents: + [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather], identify the intent expressed in the + text. Return only the identified intent.' +include: injongointent +task: injongointent_lin_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..49b7f6faddd5ad59c079de1a1986e33ef4a29311 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_lug.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: lug +doc_to_text: 'The following text is in Luganda: ''{{text}}''. Given the list of intents: + [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather], identify the intent expressed in the + text. Return only the identified intent.' +include: injongointent +task: injongointent_lug_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_orm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_orm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..72a7686934892b2f21f3098b302dc601050b394b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_orm.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: orm +doc_to_text: 'The following text is in Oromo: ''{{text}}''. Given the list of intents: + [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather], identify the intent expressed in the + text. Return only the identified intent.' +include: injongointent +task: injongointent_orm_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_sna.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_sna.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8931b65ce15999ea1215cf420a671baed32a51c2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_sna.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: sna +doc_to_text: 'The following text is in Shona: ''{{text}}''. Given the list of intents: + [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather], identify the intent expressed in the + text. Return only the identified intent.' +include: injongointent +task: injongointent_sna_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_sot.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_sot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5a8d0328e75acaead01f4e08fb4c6ed26eed6314 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_sot.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: sot +doc_to_text: 'The following text is in Sotho: ''{{text}}''. Given the list of intents: + [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather], identify the intent expressed in the + text. Return only the identified intent.' +include: injongointent +task: injongointent_sot_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1da6be32a6c64ec6583c70a2f0cbaa3d6aa3d435 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_swa.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: swa +doc_to_text: 'The following text is in Swahili: ''{{text}}''. Given the list of intents: + [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather], identify the intent expressed in the + text. Return only the identified intent.' +include: injongointent +task: injongointent_swa_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cc78ae4f60d64d9d30a5bf52064fda54e0bce9ef --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_twi.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: twi +doc_to_text: 'The following text is in Twi: ''{{text}}''. Given the list of intents: + [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather], identify the intent expressed in the + text. Return only the identified intent.' +include: injongointent +task: injongointent_twi_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9c71483ec36660b62bc945b4ebf89fdb77034e85 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_wol.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: wol +doc_to_text: 'The following text is in Wolof: ''{{text}}''. Given the list of intents: + [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather], identify the intent expressed in the + text. Return only the identified intent.' +include: injongointent +task: injongointent_wol_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8d8b543fc5cac462b31a9ae806aa3ce0f60b6ba9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_xho.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: xho +doc_to_text: 'The following text is in Xhosa: ''{{text}}''. Given the list of intents: + [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather], identify the intent expressed in the + text. Return only the identified intent.' +include: injongointent +task: injongointent_xho_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cbe285688f78a2ec210740e0294e4a3c6fde8dd7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_yor.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: yor +doc_to_text: 'The following text is in Yoruba: ''{{text}}''. Given the list of intents: + [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather], identify the intent expressed in the + text. Return only the identified intent.' +include: injongointent +task: injongointent_yor_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7ba384db5a8b2f7f17601ad7fff0bed8327ecc1c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_zul.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: zul +doc_to_text: 'The following text is in Zulu: ''{{text}}''. Given the list of intents: + [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather], identify the intent expressed in the + text. Return only the identified intent.' +include: injongointent +task: injongointent_zul_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..3e735e2deb1f9c53152c072615aebe8ba3acb90b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/utils.py @@ -0,0 +1 @@ +from lm_eval.utils import weighted_f1_score diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/README.md b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/README.md new file mode 100644 index 0000000000000000000000000000000000000000..9e7eea17598d29defff07bb37c4f47efaa446547 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/README.md @@ -0,0 +1,73 @@ +# + +## Paper +Title: `A Few Thousand Translations Go a Long Way! Leveraging Pre-trained Models for African News Translation` + +Paper Link: https://aclanthology.org/2022.naacl-main.223/ + +## Abstract +>Recent advances in the pre-training of language models leverage large-scale datasets to create multilingual models. However, low-resource languages are mostly left out in these datasets. This is primarily because many widely spoken languages are not well represented on the web and therefore excluded from the large-scale crawls used to create datasets. Furthermore, downstream users of these models are restricted to the selection of languages originally chosen for pre-training. This work investigates how to optimally leverage existing pre-trained models to create low-resource translation systems for 16 African languages. We focus on two questions: 1) How can pre-trained models be used for languages not included in the initial pre-training? and 2) How can the resulting translation models effectively transfer to new domains? To answer these questions, we create a new African news corpus covering 16 languages, of which eight languages are not part of any existing evaluation dataset. We demonstrate that the most effective strategy for transferring both to additional languages and to additional domains is to fine-tune large pre-trained models on small quantities of high-quality translation data. + +HomePage: https://github.com/masakhane-io/lafand-mt + +### Citation + +``` +@inproceedings{adelani-etal-2022-thousand, + title = "A Few Thousand Translations Go a Long Way! Leveraging Pre-trained Models for {A}frican News Translation", + author = "Adelani, David and + Alabi, Jesujoba and + Fan, Angela and + Kreutzer, Julia and + Shen, Xiaoyu and + Reid, Machel and + Ruiter, Dana and + Klakow, Dietrich and + Nabende, Peter and + Chang, Ernie and + Gwadabe, Tajuddeen and + Sackey, Freshia and + Dossou, Bonaventure F. P. and + Emezue, Chris and + Leong, Colin and + Beukman, Michael and + Muhammad, Shamsuddeen and + Jarso, Guyo and + Yousuf, Oreen and + Niyongabo Rubungo, Andre and + Hacheme, Gilles and + Wairagala, Eric Peter and + Nasir, Muhammad Umair and + Ajibade, Benjamin and + Ajayi, Tunde and + Gitau, Yvonne and + Abbott, Jade and + Ahmed, Mohamed and + Ochieng, Millicent and + Aremu, Anuoluwapo and + Ogayo, Perez and + Mukiibi, Jonathan and + Ouoba Kabore, Fatoumata and + Kalipe, Godson and + Mbaye, Derguene and + Tapo, Allahsera Auguste and + Memdjokam Koagne, Victoire and + Munkoh-Buabeng, Edwin and + Wagner, Valencia and + Abdulmumin, Idris and + Awokoya, Ayodele and + Buzaaba, Happy and + Sibanda, Blessing and + Bukula, Andiswa and + Manthalu, Sam", + booktitle = "Proceedings of the 2022 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies", + month = jul, + year = "2022", + address = "Seattle, United States", + publisher = "Association for Computational Linguistics", + url = "https://aclanthology.org/2022.naacl-main.223", + doi = "10.18653/v1/2022.naacl-main.223", + pages = "3053--3070", + abstract = "Recent advances in the pre-training for language models leverage large-scale datasets to create multilingual models. However, low-resource languages are mostly left out in these datasets. This is primarily because many widely spoken languages that are not well represented on the web and therefore excluded from the large-scale crawls for datasets. Furthermore, downstream users of these models are restricted to the selection of languages originally chosen for pre-training. This work investigates how to optimally leverage existing pre-trained models to create low-resource translation systems for 16 African languages. We focus on two questions: 1) How can pre-trained models be used for languages not included in the initial pretraining? and 2) How can the resulting translation models effectively transfer to new domains? To answer these questions, we create a novel African news corpus covering 16 languages, of which eight languages are not part of any existing evaluation dataset. We demonstrate that the most effective strategy for transferring both additional languages and additional domains is to leverage small quantities of high-quality translation data to fine-tune large pre-trained models.", +} +``` diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/gen_utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/gen_utils.py new file mode 100644 index 0000000000000000000000000000000000000000..c260a321a419b3013545b738850af5f796a1bc32 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/gen_utils.py @@ -0,0 +1,147 @@ +import argparse +import os + +import yaml + + +class FunctionTag: + def __init__(self, value): + self.value = value + + +def prompt_func(mode, lang, lang_dict): + language_column_name = f"{lang}_text" + prompt_map = { + "prompt_1": "You are an advanced Translator, a specialized assistant designed to translate documents from " + f"{lang_dict[lang]} into English. \nYour main goal is to ensure translations are grammatically " + f"correct and human-oriented. \n{lang_dict[lang]}: {{{{{language_column_name}}}}} \nEnglish: ", + "prompt_1_reverse": "You are an advanced Translator, a specialized assistant designed to translate documents " + f"from English into {lang_dict[lang]}. \nYour main goal is to ensure translations are " + f"grammatically correct and human-oriented. " + f"\nEnglish: {{eng_text}} \n{lang_dict[lang]}: ", + "prompt_2": f"{lang_dict[lang]} sentence: {{{{{language_column_name}}}}} \nEnglish sentence: ", + "prompt_2_reverse": "English sentence: {{eng_text}} " + f"\n{lang_dict[lang]} sentence: ", + "prompt_3": f"You are a translation expert. Translate the following {lang_dict[lang]} sentences to English \n" + f"{lang_dict[lang]} sentence: {{{{{language_column_name}}}}}\nEnglish sentence: ", + "prompt_3_reverse": f"You are a translation expert. Translate the following English sentences to " + f"{lang_dict[lang]} " + "\nEnglish sentence: {{eng_text}} " + f"\n{lang_dict[lang]} sentence: ", + } + return prompt_map[mode] + + +def gen_lang_yamls(output_dir: str, overwrite: bool, mode: str, reverse: bool) -> None: + """ + Generate a yaml file for each language. + + :param output_dir: The directory to output the files to. + :param overwrite: Whether to overwrite files if they already exist. + """ + err = [] + languages = { + "amh": "Amharic", + "bam": "Bambara", + "bbj": "Gbomala", + "ewe": "Ewe", + "fon": "Fon", + "hau": "Hausa", + "ibo": "Igbo", + "kin": "Kinyarwanda", + "lug": "Luganda", + "luo": "Luo", + "mos": "Mossi", + "nya": "Chichewa", + "pcm": "Nigerian Pidgin", + "sna": "Shona", + "swa": "Swahili", + "tsn": "Setswana", + "twi": "Twi", + "wol": "Wolof", + "xho": "Xhosa", + "yor": "Yoruba", + "zul": "Zulu", + } + + french_langs = ["bam", "bbj", "ewe", "fon", "wol", "mos"] + + for lang in languages.keys(): + try: + norm_lang = f"{lang}-en" if lang not in french_langs else f"{lang}-fr" + reverse_lang = f"en-{lang}" if lang not in french_langs else f"fr-{lang}" + dataset_name = norm_lang if reverse else reverse_lang + file_name = f"mafand_{dataset_name}.yaml" + task_name = f"mafand_{dataset_name}_{mode}" + yaml_template = "mafand" + yaml_details = { + "include": yaml_template, + "task": task_name, + "dataset_name": reverse_lang, + } + file_dir = ( + f"{output_dir}/{mode}/african-english" + if reverse + else f"{output_dir}/{mode}/english-african" + ) + os.makedirs(file_dir, exist_ok=True) + with open( + f"{file_dir}/{file_name}", + "w" if overwrite else "x", + encoding="utf8", + ) as f: + f.write("# Generated by utils.py\n") + yaml.dump( + yaml_details, + f, + allow_unicode=True, + ) + + except FileExistsError: + err.append(file_name) + + if len(err) > 0: + raise FileExistsError( + "Files were not created because they already exist (use --overwrite flag):" + f" {', '.join(err)}" + ) + + +def main() -> None: + """Parse CLI args and generate language-specific yaml files.""" + parser = argparse.ArgumentParser() + parser.add_argument( + "--overwrite", + default=True, + action="store_true", + help="Overwrite files if they already exist", + ) + parser.add_argument( + "--output-dir", + default="./", + help="Directory to write yaml files to", + ) + parser.add_argument( + "--mode", + default="prompt_3", + choices=["prompt_1", "prompt_2", "prompt_3"], + help="Prompt number", + ) + parser.add_argument( + "--reverse", + default=True, + choices=[True, False], + help="Reverse the translation direction", + ) + args = parser.parse_args() + + gen_lang_yamls( + output_dir=args.output_dir, + overwrite=args.overwrite, + mode=args.mode, + reverse=args.reverse, + ) + + +if __name__ == "__main__": + main() diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/mafand.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/mafand.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ef8619addf5df6be4d59e31540c6ed6663e6b189 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/mafand.yaml @@ -0,0 +1,14 @@ +group: mafand +task: + - mafand_eng-afr_prompt_1 + - mafand_eng-afr_prompt_2 + - mafand_eng-afr_prompt_3 + - mafand_afr-eng_prompt_1 + - mafand_afr-eng_prompt_2 + - mafand_afr-eng_prompt_3 +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand new file mode 100644 index 0000000000000000000000000000000000000000..4f2047be0877cfda1e41acc76e491478c1f8f8f0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand @@ -0,0 +1,28 @@ +tag: +- mafand_tasks +- mafand_afr-eng +- mafand_afr-eng_prompt_1 +- afrobench_MT_tasks +dataset_path: masakhane/mafand +dataset_kwargs: {trust_remote_code: True} +output_type: generate_until +validation_split: validation +fewshot_split: validation +test_split: test +doc_to_target: !function utils.get_target +doc_to_text: !function utils.create_text_prompt_1 +metric_list: + - metric: bleu + aggregation: bleu + higher_is_better: true + - metric: chrf + aggregation: chrf + higher_is_better: true +generation_kwargs: + until: + - "\n" + do_sample: false + temperature: 0.0 +repeats: 1 +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_amh-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_amh-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..95e87fd8aeb3df35fd529338e719683805a78f18 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_amh-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-amh +include: mafand +task: mafand_amh-en_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_bam-fr.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_bam-fr.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dbc612ac327d46f46c4df459d558c8429d2089dd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_bam-fr.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fr-bam +include: mafand +task: mafand_bam-fr_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_bbj-fr.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_bbj-fr.yaml new file mode 100644 index 0000000000000000000000000000000000000000..abe64f9b2a1dbf14bcf60f9f4f80df24f65821ed --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_bbj-fr.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fr-bbj +include: mafand +task: mafand_bbj-fr_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_ewe-fr.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_ewe-fr.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ecd9b38bcdfda7354569bc002e3ed10ff573449f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_ewe-fr.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fr-ewe +include: mafand +task: mafand_ewe-fr_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_fon-fr.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_fon-fr.yaml new file mode 100644 index 0000000000000000000000000000000000000000..705cfbb855261a2cc14d841bde89661fa5d6be75 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_fon-fr.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fr-fon +include: mafand +task: mafand_fon-fr_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_hau-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_hau-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3b84d9cecabfd92350a9ab63585f38fcfff328d7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_hau-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-hau +include: mafand +task: mafand_hau-en_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_ibo-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_ibo-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d78c91bb1fc61797c9554b9cf8acb7c099e53919 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_ibo-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-ibo +include: mafand +task: mafand_ibo-en_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_kin-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_kin-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..954c036e80de22ad705888d66705e58b2a15f689 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_kin-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-kin +include: mafand +task: mafand_kin-en_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_lug-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_lug-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..671c072adc79858fc9eb1f2c32b99420b26fcb95 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_lug-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-lug +include: mafand +task: mafand_lug-en_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_luo-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_luo-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a1d5965f08d62f2a8147a39878f844d714800ade --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_luo-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-luo +include: mafand +task: mafand_luo-en_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_mos-fr.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_mos-fr.yaml new file mode 100644 index 0000000000000000000000000000000000000000..da085707bc8ec37730535867ad2faa80e23bfa20 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_mos-fr.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fr-mos +include: mafand +task: mafand_mos-fr_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_nya-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_nya-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3bc2426687b5784602b1bc1bcb64c453d79d39ab --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_nya-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-nya +include: mafand +task: mafand_nya-en_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_pcm-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_pcm-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f6fb5dee6d393fbe6a6172e347e382cee2aecfb5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_pcm-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-pcm +include: mafand +task: mafand_pcm-en_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_sna-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_sna-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..283517d6ac4f9f4c8e462a63002d11a681c438b3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_sna-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-sna +include: mafand +task: mafand_sna-en_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_swa-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_swa-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..476bba42d82461b241b2b8a396baefe20c154206 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_swa-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-swa +include: mafand +task: mafand_swa-en_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_tsn-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_tsn-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a94c5b6ea0f870db19b4fc3c692be95a3fc6d455 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_tsn-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-tsn +include: mafand +task: mafand_tsn-en_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_twi-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_twi-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7f5883b12d6875133699a8158dd2151c8c784981 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_twi-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-twi +include: mafand +task: mafand_twi-en_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_wol-fr.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_wol-fr.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bb887188000da8e83aeed287619ed1cd2375e18a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_wol-fr.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fr-wol +include: mafand +task: mafand_wol-fr_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_xho-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_xho-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a0561b4157c15f0fa5f8f3876ab86d07087a5a0a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_xho-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-xho +include: mafand +task: mafand_xho-en_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_yor-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_yor-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ec97ae7d3c4abc88a4774dc69cb32e2d7dced32d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_yor-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-yor +include: mafand +task: mafand_yor-en_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_zul-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_zul-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9649d772d259a5519e87ae522f80b3d591f1b2be --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_zul-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-zul +include: mafand +task: mafand_zul-en_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..0df3a329824d44fa94eb830ae943fa30dd32bab7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/utils.py @@ -0,0 +1,121 @@ +languages = { + "amh": "Amharic", + "bam": "Bambara", + "bbj": "Gbomala", + "ewe": "Ewe", + "fon": "Fon", + "hau": "Hausa", + "ibo": "Igbo", + "kin": "Kinyarwanda", + "lug": "Luganda", + "luo": "Luo", + "mos": "Mossi", + "nya": "Chichewa", + "pcm": "Nigerian Pidgin", + "sna": "Shona", + "swa": "Swahili", + "tsn": "Setswana", + "twi": "Twi", + "wol": "Wolof", + "xho": "Xhosa", + "yor": "Yoruba", + "zul": "Zulu", +} + + +def get_target(doc): + target = ( + doc["translation"]["en"] + if "en" in doc["translation"].keys() + else doc["translation"]["fr"] + ) + return target + + +def get_target_reverse(doc): + target_key = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + target = doc["translation"][target_key] + return target + + +def create_text_prompt_1(doc): + source_key = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + source_sentence = doc["translation"][source_key] + source_lang = "English" if "en" in doc["translation"].keys() else "French" + prompt = ( + "You are an advanced Translator, a specialized assistant designed to translate documents from " + f"{languages[source_key]} into {source_lang}. \nYour main goal is to ensure translations are grammatically " + f"correct and human-oriented. \n{languages[source_key]}: {source_sentence} \n{source_lang}: " + ) + return prompt + + +def create_reverse_prompt_1(doc): + target_lang = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + source_key = "en" if "en" in doc["translation"].keys() else "fr" + source_lang = "English" if source_key == "en" else "French" + source_sentence = doc["translation"][source_key] + prompt = ( + "You are an advanced Translator, a specialized assistant designed to translate documents from " + f"{source_lang} into {languages[target_lang]}. \nYour main goal is to ensure translations are " + f"grammatically correct and human-oriented. \n{source_lang}: {source_sentence} \n{languages[target_lang]}: " + ) + return prompt + + +def create_text_prompt_2(doc): + source_key = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + source_sentence = doc["translation"][source_key] + source_lang = "English" if "en" in doc["translation"].keys() else "French" + prompt = ( + f"{languages[source_key]} sentence: {source_sentence} \n{source_lang} sentence: ", + ) + return prompt + + +def create_reverse_prompt_2(doc): + target_lang = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + source_key = "en" if "en" in doc["translation"].keys() else "fr" + source_lang = "English" if source_key == "en" else "French" + source_sentence = doc["translation"][source_key] + prompt = ( + f"{source_lang} sentence: {source_sentence} \n{languages[target_lang]} sentence: \n", + ) + return prompt + + +def create_text_prompt_3(doc): + source_key = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + source_sentence = doc["translation"][source_key] + source_lang = "English" if "en" in doc["translation"].keys() else "French" + prompt = ( + f"You are a translation expert. Translate the following {languages[source_key]} sentences " + f"to {source_lang}. \n{languages[source_key]} sentence: {source_sentence}\n{source_lang} sentence: " + ) + return prompt + + +def create_reverse_prompt_3(doc): + target_lang = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + source_key = "en" if "en" in doc["translation"].keys() else "fr" + source_lang = "English" if source_key == "en" else "French" + source_sentence = doc["translation"][source_key] + prompt = ( + f"You are a translation expert. Translate the following {source_lang} sentence into {languages[target_lang]}\n" + f"{source_lang} sentence: {source_sentence}\n{languages[target_lang]} sentence: " + ) + return prompt diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand new file mode 100644 index 0000000000000000000000000000000000000000..1d004556267924e0120e95fb516fee72a5d3eb1d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand @@ -0,0 +1,28 @@ +tag: +- mafand_tasks +- mafand_eng-afr +- mafand_eng-afr_prompt_1 +- afrobench_MT_tasks +dataset_path: masakhane/mafand +dataset_kwargs: {trust_remote_code: True} +output_type: generate_until +validation_split: validation +fewshot_split: validation +test_split: test +doc_to_target: !function utils.get_target_reverse +doc_to_text: !function utils.create_reverse_prompt_1 +metric_list: + - metric: bleu + aggregation: bleu + higher_is_better: true + - metric: chrf + aggregation: chrf + higher_is_better: true +generation_kwargs: + until: + - "\n" + do_sample: false + temperature: 0.0 +repeats: 1 +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_en-amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_en-amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8ef9ab288615c42f1c4c04538f0b51e3d2245ac1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_en-amh.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-amh +include: mafand +task: mafand_en-amh_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_en-hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_en-hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0ea577781deda41a7217dbd757158789a80dafc6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_en-hau.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-hau +include: mafand +task: mafand_en-hau_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_en-ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_en-ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..88af221fb31e583d9238c95c05dffb44306a617e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_en-ibo.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-ibo +include: mafand +task: mafand_en-ibo_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_en-kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_en-kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0c415f511ad246eb99509f4bedb979810dfcf20d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_en-kin.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-kin +include: mafand +task: mafand_en-kin_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_en-lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_en-lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..94070e8986b0f44eecb1097c673c7daf4aec8067 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_en-lug.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-lug +include: mafand +task: mafand_en-lug_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_en-luo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_en-luo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fc6b15c67f2919dca92ef84d058a93da6e0271a1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_en-luo.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-luo +include: mafand +task: mafand_en-luo_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_en-nya.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_en-nya.yaml new file mode 100644 index 0000000000000000000000000000000000000000..225a46474ff2c694da98558bfaf665bb8e102248 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_en-nya.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-nya +include: mafand +task: mafand_en-nya_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_en-pcm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_en-pcm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..69380c7182bdfc8f8e25c2ee16b3ced4a5c05d49 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_en-pcm.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-pcm +include: mafand +task: mafand_en-pcm_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_en-sna.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_en-sna.yaml new file mode 100644 index 0000000000000000000000000000000000000000..634d988fd46f0a98b6e5b76e4a42f897edb4871d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_en-sna.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-sna +include: mafand +task: mafand_en-sna_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_en-swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_en-swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bfbf259cbbe8d6d740dc7522f6d8bd6542c790a1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_en-swa.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-swa +include: mafand +task: mafand_en-swa_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_en-tsn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_en-tsn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..faa99ddf5186515253279585aa1439569d7de9b3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_en-tsn.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-tsn +include: mafand +task: mafand_en-tsn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_en-twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_en-twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a9294975da5385d196db9abc9f22d087a2c9cce0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_en-twi.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-twi +include: mafand +task: mafand_en-twi_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_en-xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_en-xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..244f5cabd9901020a136322842ec47fb181faed6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_en-xho.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-xho +include: mafand +task: mafand_en-xho_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_en-yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_en-yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..aa3189779577ecbc58f80ff4973a63d81ea1c6a6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_en-yor.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-yor +include: mafand +task: mafand_en-yor_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_en-zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_en-zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6afdc0c5648d8fec0be6d2d2fca0b9416dccffab --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_en-zul.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-zul +include: mafand +task: mafand_en-zul_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_fr-bam.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_fr-bam.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7c21d96f275a14b81225b3979044a23909bfb023 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_fr-bam.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fr-bam +include: mafand +task: mafand_fr-bam_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_fr-bbj.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_fr-bbj.yaml new file mode 100644 index 0000000000000000000000000000000000000000..76cf07507625677c85772e988ac66741508efbc6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_fr-bbj.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fr-bbj +include: mafand +task: mafand_fr-bbj_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_fr-ewe.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_fr-ewe.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0c7bd6671b7ba6da2a869cf12862046b632396e6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_fr-ewe.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fr-ewe +include: mafand +task: mafand_fr-ewe_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_fr-fon.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_fr-fon.yaml new file mode 100644 index 0000000000000000000000000000000000000000..737d68eba79405276a96151a33786387c33cc148 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_fr-fon.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fr-fon +include: mafand +task: mafand_fr-fon_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_fr-mos.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_fr-mos.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9186a5b9f6670f5187284981d547d91631cc94f6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_fr-mos.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fr-mos +include: mafand +task: mafand_fr-mos_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_fr-wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_fr-wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6e29f5fb98a138f3a131c62b685abc25b60e3f69 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/mafand_fr-wol.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fr-wol +include: mafand +task: mafand_fr-wol_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..0df3a329824d44fa94eb830ae943fa30dd32bab7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/english-african/utils.py @@ -0,0 +1,121 @@ +languages = { + "amh": "Amharic", + "bam": "Bambara", + "bbj": "Gbomala", + "ewe": "Ewe", + "fon": "Fon", + "hau": "Hausa", + "ibo": "Igbo", + "kin": "Kinyarwanda", + "lug": "Luganda", + "luo": "Luo", + "mos": "Mossi", + "nya": "Chichewa", + "pcm": "Nigerian Pidgin", + "sna": "Shona", + "swa": "Swahili", + "tsn": "Setswana", + "twi": "Twi", + "wol": "Wolof", + "xho": "Xhosa", + "yor": "Yoruba", + "zul": "Zulu", +} + + +def get_target(doc): + target = ( + doc["translation"]["en"] + if "en" in doc["translation"].keys() + else doc["translation"]["fr"] + ) + return target + + +def get_target_reverse(doc): + target_key = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + target = doc["translation"][target_key] + return target + + +def create_text_prompt_1(doc): + source_key = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + source_sentence = doc["translation"][source_key] + source_lang = "English" if "en" in doc["translation"].keys() else "French" + prompt = ( + "You are an advanced Translator, a specialized assistant designed to translate documents from " + f"{languages[source_key]} into {source_lang}. \nYour main goal is to ensure translations are grammatically " + f"correct and human-oriented. \n{languages[source_key]}: {source_sentence} \n{source_lang}: " + ) + return prompt + + +def create_reverse_prompt_1(doc): + target_lang = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + source_key = "en" if "en" in doc["translation"].keys() else "fr" + source_lang = "English" if source_key == "en" else "French" + source_sentence = doc["translation"][source_key] + prompt = ( + "You are an advanced Translator, a specialized assistant designed to translate documents from " + f"{source_lang} into {languages[target_lang]}. \nYour main goal is to ensure translations are " + f"grammatically correct and human-oriented. \n{source_lang}: {source_sentence} \n{languages[target_lang]}: " + ) + return prompt + + +def create_text_prompt_2(doc): + source_key = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + source_sentence = doc["translation"][source_key] + source_lang = "English" if "en" in doc["translation"].keys() else "French" + prompt = ( + f"{languages[source_key]} sentence: {source_sentence} \n{source_lang} sentence: ", + ) + return prompt + + +def create_reverse_prompt_2(doc): + target_lang = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + source_key = "en" if "en" in doc["translation"].keys() else "fr" + source_lang = "English" if source_key == "en" else "French" + source_sentence = doc["translation"][source_key] + prompt = ( + f"{source_lang} sentence: {source_sentence} \n{languages[target_lang]} sentence: \n", + ) + return prompt + + +def create_text_prompt_3(doc): + source_key = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + source_sentence = doc["translation"][source_key] + source_lang = "English" if "en" in doc["translation"].keys() else "French" + prompt = ( + f"You are a translation expert. Translate the following {languages[source_key]} sentences " + f"to {source_lang}. \n{languages[source_key]} sentence: {source_sentence}\n{source_lang} sentence: " + ) + return prompt + + +def create_reverse_prompt_3(doc): + target_lang = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + source_key = "en" if "en" in doc["translation"].keys() else "fr" + source_lang = "English" if source_key == "en" else "French" + source_sentence = doc["translation"][source_key] + prompt = ( + f"You are a translation expert. Translate the following {source_lang} sentence into {languages[target_lang]}\n" + f"{source_lang} sentence: {source_sentence}\n{languages[target_lang]} sentence: " + ) + return prompt diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand new file mode 100644 index 0000000000000000000000000000000000000000..eb7ad9883115b42626fd30d223fa90ff6f133384 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand @@ -0,0 +1,28 @@ +tag: +- mafand_tasks +- mafand_afr-eng +- mafand_afr-eng_prompt_3 +- afrobench_MT_tasks +dataset_path: masakhane/mafand +dataset_kwargs: {trust_remote_code: True} +output_type: generate_until +validation_split: validation +fewshot_split: validation +test_split: test +doc_to_target: !function utils.get_target +doc_to_text: !function utils.create_text_prompt_2 +metric_list: + - metric: bleu + aggregation: bleu + higher_is_better: true + - metric: chrf + aggregation: chrf + higher_is_better: true +generation_kwargs: + until: + - "\n" + do_sample: false + temperature: 0.0 +repeats: 1 +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_amh-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_amh-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6db544cb5e17e87e6ebcf606ad59ad9a7435f338 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_amh-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-amh +include: mafand +task: mafand_amh-en_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_bam-fr.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_bam-fr.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0a9f3b3ac09958dc4253d982dfe1c872aefafd7e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_bam-fr.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fr-bam +include: mafand +task: mafand_bam-fr_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_bbj-fr.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_bbj-fr.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b0b42b23787e88bd5ff0d000c1806655e26f65cf --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_bbj-fr.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fr-bbj +include: mafand +task: mafand_bbj-fr_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_ewe-fr.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_ewe-fr.yaml new file mode 100644 index 0000000000000000000000000000000000000000..457c0d1945bfa65f5d5f0e1ccffc30c11aadd452 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_ewe-fr.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fr-ewe +include: mafand +task: mafand_ewe-fr_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_fon-fr.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_fon-fr.yaml new file mode 100644 index 0000000000000000000000000000000000000000..84263d5a9ca41877f02d680660706eaeba28fea2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_fon-fr.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fr-fon +include: mafand +task: mafand_fon-fr_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_hau-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_hau-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..05c31a4670a9300c1175dc2f7e109c024b146301 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_hau-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-hau +include: mafand +task: mafand_hau-en_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_ibo-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_ibo-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3cb4a5b897c2d143ef31d667c9d87fc35f40caa0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_ibo-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-ibo +include: mafand +task: mafand_ibo-en_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_kin-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_kin-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e3e1acf9a92a63ab80dbc0314adebfa59e449490 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_kin-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-kin +include: mafand +task: mafand_kin-en_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_lug-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_lug-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..eb68279d783b71ff94742507d4ed1571a03b6b51 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_lug-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-lug +include: mafand +task: mafand_lug-en_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_luo-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_luo-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..12f199473ee6e4ab1f6a2d3ae6e69eb04ba6a399 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_luo-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-luo +include: mafand +task: mafand_luo-en_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_mos-fr.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_mos-fr.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a723701d3250aa78f7631a3dcfc450301201bf73 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_mos-fr.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fr-mos +include: mafand +task: mafand_mos-fr_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_nya-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_nya-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..24569f00825bb4a2e0419e290ae4ff1bc7e0d312 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_nya-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-nya +include: mafand +task: mafand_nya-en_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_sna-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_sna-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3bf99b9955378aa6a5930d9783ed33dfea96ac95 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_sna-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-sna +include: mafand +task: mafand_sna-en_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_swa-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_swa-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..eb2ada0bba76e9c83a99d8060ac8f4146a1462c7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_swa-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-swa +include: mafand +task: mafand_swa-en_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_tsn-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_tsn-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d16e7e94c5b251eff2adece451f89c7af71fbc30 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_tsn-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-tsn +include: mafand +task: mafand_tsn-en_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_twi-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_twi-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..267337c177fc06e777d62c3a2286084d79266a4c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_twi-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-twi +include: mafand +task: mafand_twi-en_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_wol-fr.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_wol-fr.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f6c67bd8d6df8d81d4d4088b6bbf8e04498afef6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_wol-fr.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fr-wol +include: mafand +task: mafand_wol-fr_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_xho-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_xho-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fd1960d0efbd42ba25132c938944692fbf63b92f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_xho-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-xho +include: mafand +task: mafand_xho-en_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_yor-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_yor-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cb7241ad2cc2fc2f889bab56c4ad3e233a2d2165 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_yor-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-yor +include: mafand +task: mafand_yor-en_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_zul-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_zul-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d44db7a2eeefcc7dc0218f08327ceaee4a6e351a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_zul-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-zul +include: mafand +task: mafand_zul-en_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..0df3a329824d44fa94eb830ae943fa30dd32bab7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/utils.py @@ -0,0 +1,121 @@ +languages = { + "amh": "Amharic", + "bam": "Bambara", + "bbj": "Gbomala", + "ewe": "Ewe", + "fon": "Fon", + "hau": "Hausa", + "ibo": "Igbo", + "kin": "Kinyarwanda", + "lug": "Luganda", + "luo": "Luo", + "mos": "Mossi", + "nya": "Chichewa", + "pcm": "Nigerian Pidgin", + "sna": "Shona", + "swa": "Swahili", + "tsn": "Setswana", + "twi": "Twi", + "wol": "Wolof", + "xho": "Xhosa", + "yor": "Yoruba", + "zul": "Zulu", +} + + +def get_target(doc): + target = ( + doc["translation"]["en"] + if "en" in doc["translation"].keys() + else doc["translation"]["fr"] + ) + return target + + +def get_target_reverse(doc): + target_key = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + target = doc["translation"][target_key] + return target + + +def create_text_prompt_1(doc): + source_key = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + source_sentence = doc["translation"][source_key] + source_lang = "English" if "en" in doc["translation"].keys() else "French" + prompt = ( + "You are an advanced Translator, a specialized assistant designed to translate documents from " + f"{languages[source_key]} into {source_lang}. \nYour main goal is to ensure translations are grammatically " + f"correct and human-oriented. \n{languages[source_key]}: {source_sentence} \n{source_lang}: " + ) + return prompt + + +def create_reverse_prompt_1(doc): + target_lang = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + source_key = "en" if "en" in doc["translation"].keys() else "fr" + source_lang = "English" if source_key == "en" else "French" + source_sentence = doc["translation"][source_key] + prompt = ( + "You are an advanced Translator, a specialized assistant designed to translate documents from " + f"{source_lang} into {languages[target_lang]}. \nYour main goal is to ensure translations are " + f"grammatically correct and human-oriented. \n{source_lang}: {source_sentence} \n{languages[target_lang]}: " + ) + return prompt + + +def create_text_prompt_2(doc): + source_key = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + source_sentence = doc["translation"][source_key] + source_lang = "English" if "en" in doc["translation"].keys() else "French" + prompt = ( + f"{languages[source_key]} sentence: {source_sentence} \n{source_lang} sentence: ", + ) + return prompt + + +def create_reverse_prompt_2(doc): + target_lang = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + source_key = "en" if "en" in doc["translation"].keys() else "fr" + source_lang = "English" if source_key == "en" else "French" + source_sentence = doc["translation"][source_key] + prompt = ( + f"{source_lang} sentence: {source_sentence} \n{languages[target_lang]} sentence: \n", + ) + return prompt + + +def create_text_prompt_3(doc): + source_key = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + source_sentence = doc["translation"][source_key] + source_lang = "English" if "en" in doc["translation"].keys() else "French" + prompt = ( + f"You are a translation expert. Translate the following {languages[source_key]} sentences " + f"to {source_lang}. \n{languages[source_key]} sentence: {source_sentence}\n{source_lang} sentence: " + ) + return prompt + + +def create_reverse_prompt_3(doc): + target_lang = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + source_key = "en" if "en" in doc["translation"].keys() else "fr" + source_lang = "English" if source_key == "en" else "French" + source_sentence = doc["translation"][source_key] + prompt = ( + f"You are a translation expert. Translate the following {source_lang} sentence into {languages[target_lang]}\n" + f"{source_lang} sentence: {source_sentence}\n{languages[target_lang]} sentence: " + ) + return prompt diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..09c21d215e4cab74dda2b30ee110f55dcf2cbcbb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-amh.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-amh +include: mafand +task: mafand_en-amh_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e9a91c76499bf37863e6b55c71ff88205b1e2599 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-hau.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-hau +include: mafand +task: mafand_en-hau_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..09425f64afdb68817d0f462545e67f5a1e2d5f07 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-kin.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-kin +include: mafand +task: mafand_en-kin_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..13c91d36a6e5e0037a301c88fcf814309e29590f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-lug.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-lug +include: mafand +task: mafand_en-lug_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-luo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-luo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..41bb09363d1c100ffea32d38eb2085400bec018b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-luo.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-luo +include: mafand +task: mafand_en-luo_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-sna.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-sna.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ac37187170a032226e1b33dd87fb09c7b9952cf1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-sna.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-sna +include: mafand +task: mafand_en-sna_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5502ffa4e5547433d7cba66191456178c5d3377e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-twi.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-twi +include: mafand +task: mafand_en-twi_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c8c1ffee3415a1373a1b583819e874f968b5cfee --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-xho.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-xho +include: mafand +task: mafand_en-xho_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e54725404ba481e668156618a97f0b11d1a1fb31 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-zul.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-zul +include: mafand +task: mafand_en-zul_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_fr-fon.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_fr-fon.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9710db5b341b91663a782945da57a3a21ae3c1ae --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_fr-fon.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fr-fon +include: mafand +task: mafand_fr-fon_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_fr-mos.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_fr-mos.yaml new file mode 100644 index 0000000000000000000000000000000000000000..682fb19c479f7f4a49a810c1f665c20439e4ee3e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_fr-mos.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fr-mos +include: mafand +task: mafand_fr-mos_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand new file mode 100644 index 0000000000000000000000000000000000000000..eb7ad9883115b42626fd30d223fa90ff6f133384 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand @@ -0,0 +1,28 @@ +tag: +- mafand_tasks +- mafand_afr-eng +- mafand_afr-eng_prompt_3 +- afrobench_MT_tasks +dataset_path: masakhane/mafand +dataset_kwargs: {trust_remote_code: True} +output_type: generate_until +validation_split: validation +fewshot_split: validation +test_split: test +doc_to_target: !function utils.get_target +doc_to_text: !function utils.create_text_prompt_2 +metric_list: + - metric: bleu + aggregation: bleu + higher_is_better: true + - metric: chrf + aggregation: chrf + higher_is_better: true +generation_kwargs: + until: + - "\n" + do_sample: false + temperature: 0.0 +repeats: 1 +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_bbj-fr.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_bbj-fr.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1170c266f4b14550499e7ddf930c50714560ec66 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_bbj-fr.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fr-bbj +include: mafand +task: mafand_bbj-fr_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_kin-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_kin-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..254b22be3883110224aee429313cf2636993e675 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_kin-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-kin +include: mafand +task: mafand_kin-en_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_nya-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_nya-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..de9ec930a1b738673b91ed89916c17109b159e66 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_nya-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-nya +include: mafand +task: mafand_nya-en_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_tsn-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_tsn-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0ee8f4152a80d5af1bfd4b792167021d81e37284 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_tsn-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-tsn +include: mafand +task: mafand_tsn-en_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f64e68162a76b1a0cc33f44d62a0f66b5dc099e0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-hau.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-hau +include: mafand +task: mafand_en-hau_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_fr-bbj.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_fr-bbj.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9170a6b500239357dacf736b06044dd17aff5b25 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_fr-bbj.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fr-bbj +include: mafand +task: mafand_fr-bbj_prompt_3