diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_sot.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_sot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b258204f4eaed7d4889c597189ab910956e9dbb5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_sot.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: sot +doc_to_text: "For mathematical questions provided in Sesotho language. Supply the\ + \ accurate numeric answer to the provided question. \n\nQuestion: {{question}} \n\ + Answer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_sot_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b52cfb72dfe0f89092718c46e6fd4360a5dd3646 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_zul.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: zul +doc_to_text: "For mathematical questions provided in Zulu language. Supply the accurate\ + \ numeric answer to the provided question. \n\nQuestion: {{question}} \nAnswer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_zul_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_ewe.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_ewe.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e65e9298656895f4dab45111420da97559d62023 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_ewe.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: ewe +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_ewe_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e6f495eaeb6738e0b8b9524341c1bc2453b4c00f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_ibo.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: ibo +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_ibo_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_lin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_lin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..22599e98c83a4eae8b0c6b103396195af55fbbee --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_lin.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: lin +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_lin_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_sna.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_sna.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e99d3aa7b59d92691fc89687fcf951a993487f89 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_sna.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: sna +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_sna_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a0b4f9716cb900a71167b85a68ac402470aec3f0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_twi.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: twi +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_twi_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_vai.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_vai.yaml new file mode 100644 index 0000000000000000000000000000000000000000..76c18a3f9172645111dbcb26e0183b6d48f3fc69 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_vai.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: vai +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_vai_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ee0d6fc9babb8f968cdcbbd460bdb83f14e14c06 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_wol.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: wol +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_wol_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_yaml new file mode 100644 index 0000000000000000000000000000000000000000..8ad7f0069cd8b63090b625ef103a37154356782c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_yaml @@ -0,0 +1,33 @@ +tag: afrimgsm_tt_cot_tasks +dataset_path: masakhane/afrimgsm-translate-test +dataset_name: null # Overridden by language-specific config. +output_type: generate_until +test_split: test +doc_to_target: '{% if answer is not none %}{{answer[21:]}}{% else %}{{answer_number|string}}{% endif %}' +doc_to_text: '{% if answer is not none %}{{question+"\nStep-by-Step Answer:"}}{% else %}{{"Question: "+question+"\nStep-by-Step Answer:"}}{% endif %}' +generation_kwargs: + do_sample: false + until: + - 'Question:' + - + - <|im_end|> +metric_list: + - metric: exact_match + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true +filter_list: + - name: "strict-match" + filter: + - function: "regex" + regex_pattern: "The answer is (\\-?[0-9\\.\\,]+)" + - function: "take_first" + - filter: + - function: regex + group_select: -1 + regex_pattern: (-?[$0-9.,]{2,})|(-?[0-9]+) + - function: take_first + name: flexible-extract +metadata: + version: 2.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1bb302a4d0f2268543c228ba13d0744509c4911d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_zul.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: zul +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_zul_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4d8a986acd1dd95e9560555b44f2d3f5aed5395d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_kin.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: kin +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_kin_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_lin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_lin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..70d4032301398b2124ff128ebb9ed1ba4eb0f0ea --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_lin.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: lin +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_lin_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_orm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_orm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b325b2ce931cbcf05cee50293845043081097387 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_orm.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: orm +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_orm_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_sna.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_sna.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3e85255881b279d8d4f578bfcbfd96355e8af3cd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_sna.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: sna +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_sna_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_sot.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_sot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a298b504439fa6c7d8eab548ecf2a0b997eddc9d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_sot.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: sot +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_sot_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e3de9a4c61b8a018007b13e16cd9028d31af92c2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_swa.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: swa +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_swa_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c2e1ab61ec9af7af947ae656d2c7069230ee02c5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_twi.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: twi +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_twi_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_vai.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_vai.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9186f1e00c540dc64bb89ca619439e8b927162d5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_vai.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: vai +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_vai_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..185b406be03d84beef24b6c6fc453a4518d7a66f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_wol.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: wol +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_wol_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..52a0e1ca5144bda20184bcc081768098d945a239 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_xho.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: xho +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_xho_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_yaml new file mode 100644 index 0000000000000000000000000000000000000000..ad059aead35b933cabaf763d549c59592f006fc7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_yaml @@ -0,0 +1,33 @@ +tag: afrimgsm_tt_cot_tasks +dataset_path: masakhane/afrimgsm-translate-test +dataset_name: null # Overridden by language-specific config. +output_type: generate_until +test_split: test +doc_to_target: '{% if answer is not none %}{{answer[21:]}}{% else %}{{answer_number|string}}{% endif %}' +doc_to_text: 'Give direct numerical answers for the question provided. \n\nQuestion: {{question}} \Step-by-Step Answer: ' +generation_kwargs: + do_sample: false + until: + - 'Question:' + - + - <|im_end|> +metric_list: + - metric: exact_match + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true +filter_list: + - name: "strict-match" + filter: + - function: "regex" + regex_pattern: "The answer is (\\-?[0-9\\.\\,]+)" + - function: "take_first" + - filter: + - function: regex + group_select: -1 + regex_pattern: (-?[$0-9.,]{2,})|(-?[0-9]+) + - function: take_first + name: flexible-extract +metadata: + version: 2.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2452b0fae4a9f939ed115736056091302b9cfa78 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_yor.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: yor +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_yor_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2ce8151b79849bbc85be11e8cd535bf7a1e12ede --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_zul.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: zul +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_zul_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b627e57564d6839ec2ffde82c0a125e42a5c5b77 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_amh.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: amh +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_amh_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_ewe.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_ewe.yaml new file mode 100644 index 0000000000000000000000000000000000000000..52dc345f0cc5f2ad617db14a22327d4b2fc298bd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_ewe.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: ewe +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_ewe_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_fra.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_fra.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d2b7582c34e26f4e6e9cd87aa1c675788a23ccae --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_fra.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fra +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_fra_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d57be8c8c4c9dbaebaef523ff4bd9310df1ebc40 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_hau.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: hau +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_hau_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..296ea98fc1dd70f50bb72ddb566d738d11d42f68 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_ibo.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: ibo +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_ibo_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2b555b3e314cc6b50970f06649e7d1f662370073 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_kin.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: kin +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_kin_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_lin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_lin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0ace69b273060675411488fa639f261b2fb39f8a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_lin.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: lin +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_lin_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bd25a1661f0720e68e2851701a1a9ed8f0131950 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_lug.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: lug +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_lug_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_orm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_orm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..698c1474bd76edf931839d75b3111bafe8b0770c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_orm.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: orm +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_orm_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_sna.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_sna.yaml new file mode 100644 index 0000000000000000000000000000000000000000..354df6bfef4ea01a0b40b6361adb5d21470d395e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_sna.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: sna +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_sna_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_sot.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_sot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5990be74d1cceb58eff6a4f9648d1f3de0e11d8a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_sot.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: sot +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_sot_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d86662980bf7078dde5636aba335bab9897619c4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_swa.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: swa +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_swa_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..78ef85fc25c0da4894843624afad970c9ff572b0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_twi.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: twi +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_twi_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_vai.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_vai.yaml new file mode 100644 index 0000000000000000000000000000000000000000..25ec4e8fd71e2423370182180efc7b1bc7843e38 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_vai.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: vai +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_vai_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7815a0a5f4b77403334b41baebb2e529eda19d1a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_wol.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: wol +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_wol_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e45afd3a0ecc2ce8656e70ebfe29cad1f6ff06ba --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_xho.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: xho +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_xho_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_yaml new file mode 100644 index 0000000000000000000000000000000000000000..c0bb7d6661f0b78b5e417269c61f0e7fc028848f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_yaml @@ -0,0 +1,33 @@ +tag: afrimgsm_tt_cot_tasks +dataset_path: masakhane/afrimgsm-translate-test +dataset_name: null # Overridden by language-specific config. +output_type: generate_until +test_split: test +doc_to_target: '{% if answer is not none %}{{answer[21:]}}{% else %}{{answer_number|string}}{% endif %}' +doc_to_text: 'Solve the following math question \n\nQuestion: {{question}} \nStep-by-Step Answer: ' +generation_kwargs: + do_sample: false + until: + - 'Question:' + - + - <|im_end|> +metric_list: + - metric: exact_match + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true +filter_list: + - name: "strict-match" + filter: + - function: "regex" + regex_pattern: "The answer is (\\-?[0-9\\.\\,]+)" + - function: "take_first" + - filter: + - function: regex + group_select: -1 + regex_pattern: (-?[$0-9.,]{2,})|(-?[0-9]+) + - function: take_first + name: flexible-extract +metadata: + version: 2.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..39e18cb48d1f214a13f5beb5a1e2c4d4f34855af --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_yor.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: yor +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_yor_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..08fbc9e15e1d963da67e5723f33b926701ca9503 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_3/afrimgsm_cot_translate_zul.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: zul +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_zul_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4f73f15f9ef2ba8f0d75e0a0a74cbf00e84cf8e5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_amh.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: amh +doc_to_text: "Answer the given question with the step by step solution appropriate\ + \ numerical value, ensuring that the response is clear and without any supplementary\ + \ information. \n\nQuestion: {{question}} \nStep by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_amh_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_ewe.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_ewe.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0d57247b8b512c60f5536cade1bbc7804083b2f5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_ewe.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: ewe +doc_to_text: "Answer the given question with the step by step solution appropriate\ + \ numerical value, ensuring that the response is clear and without any supplementary\ + \ information. \n\nQuestion: {{question}} \nStep by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_ewe_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_fra.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_fra.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6a2f70ca6327e79044f7ed1685282ab8803fcc7d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_fra.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: fra +doc_to_text: "Answer the given question with the step by step solution appropriate\ + \ numerical value, ensuring that the response is clear and without any supplementary\ + \ information. \n\nQuestion: {{question}} \nStep by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_fra_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c5e7903c88c05de0ac942fd0894b523209dad320 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_hau.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: hau +doc_to_text: "Answer the given question with the step by step solution appropriate\ + \ numerical value, ensuring that the response is clear and without any supplementary\ + \ information. \n\nQuestion: {{question}} \nStep by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_hau_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cf15ed077cd3ff2762deaf532d200e117fb6e9dd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_ibo.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: ibo +doc_to_text: "Answer the given question with the step by step solution appropriate\ + \ numerical value, ensuring that the response is clear and without any supplementary\ + \ information. \n\nQuestion: {{question}} \nStep by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_ibo_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c1b57c1395efb57ed2d8e31df9f2878cfbad59ad --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_kin.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: kin +doc_to_text: "Answer the given question with the step by step solution appropriate\ + \ numerical value, ensuring that the response is clear and without any supplementary\ + \ information. \n\nQuestion: {{question}} \nStep by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_kin_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_lin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_lin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..81cdb1ff41ede1fd2ceca5c5ebaf099df5df5d45 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_lin.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: lin +doc_to_text: "Answer the given question with the step by step solution appropriate\ + \ numerical value, ensuring that the response is clear and without any supplementary\ + \ information. \n\nQuestion: {{question}} \nStep by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_lin_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a949b0289211c9443a6a73880598c762a8c5a8f9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_lug.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: lug +doc_to_text: "Answer the given question with the step by step solution appropriate\ + \ numerical value, ensuring that the response is clear and without any supplementary\ + \ information. \n\nQuestion: {{question}} \nStep by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_lug_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_orm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_orm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d4deb09238913761e594ec4967a4dabd9d188b02 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_orm.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: orm +doc_to_text: "Answer the given question with the step by step solution appropriate\ + \ numerical value, ensuring that the response is clear and without any supplementary\ + \ information. \n\nQuestion: {{question}} \nStep by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_orm_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_sna.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_sna.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5ebac1993f84026c0823e52fb41c6574e816b4c7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_sna.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: sna +doc_to_text: "Answer the given question with the step by step solution appropriate\ + \ numerical value, ensuring that the response is clear and without any supplementary\ + \ information. \n\nQuestion: {{question}} \nStep by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_sna_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_sot.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_sot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bf83c8f0209f627b01da77a7ab033ab93a276891 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_sot.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: sot +doc_to_text: "Answer the given question with the step by step solution appropriate\ + \ numerical value, ensuring that the response is clear and without any supplementary\ + \ information. \n\nQuestion: {{question}} \nStep by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_sot_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..87b581b8a5fdb5e9d2d0eede15f5c98b04f88693 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_swa.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: swa +doc_to_text: "Answer the given question with the step by step solution appropriate\ + \ numerical value, ensuring that the response is clear and without any supplementary\ + \ information. \n\nQuestion: {{question}} \nStep by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_swa_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..223901eb208af30d0a8dc549f7d021d343e17076 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_twi.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: twi +doc_to_text: "Answer the given question with the step by step solution appropriate\ + \ numerical value, ensuring that the response is clear and without any supplementary\ + \ information. \n\nQuestion: {{question}} \nStep by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_twi_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_vai.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_vai.yaml new file mode 100644 index 0000000000000000000000000000000000000000..92ce3892451070777235d493be10bbe9811ad05d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_vai.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: vai +doc_to_text: "Answer the given question with the step by step solution appropriate\ + \ numerical value, ensuring that the response is clear and without any supplementary\ + \ information. \n\nQuestion: {{question}} \nStep by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_vai_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c626fde4e599303ff73e490d3cbc38b6335d6755 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_wol.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: wol +doc_to_text: "Answer the given question with the step by step solution appropriate\ + \ numerical value, ensuring that the response is clear and without any supplementary\ + \ information. \n\nQuestion: {{question}} \nStep by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_wol_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..285b679cc0a7da31dc3d719918ead217d5287b9e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_xho.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: xho +doc_to_text: "Answer the given question with the step by step solution appropriate\ + \ numerical value, ensuring that the response is clear and without any supplementary\ + \ information. \n\nQuestion: {{question}} \nStep by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_xho_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_yaml new file mode 100644 index 0000000000000000000000000000000000000000..241787c7aa25a0ac46e2556efdb8db633e7a0719 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_yaml @@ -0,0 +1,32 @@ +tag: afrimgsm_tt_cot_tasks +dataset_path: masakhane/afrimgsm-translate-test +dataset_name: null # Overridden by language-specific config. +output_type: generate_until +test_split: test +doc_to_target: '{% if answer is not none %}{{answer[21:]}}{% else %}{{answer_number|string}}{% endif %}' +generation_kwargs: + do_sample: false + until: + - 'Question:' + - + - <|im_end|> +metric_list: + - metric: exact_match + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true +filter_list: + - name: "strict-match" + filter: + - function: "regex" + regex_pattern: "The answer is (\\-?[0-9\\.\\,]+)" + - function: "take_first" + - filter: + - function: regex + group_select: -1 + regex_pattern: (-?[$0-9.,]{2,})|(-?[0-9]+) + - function: take_first + name: flexible-extract +metadata: + version: 2.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f76f4cd109026c0c94a578674c4d8140201e5bfa --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_yor.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: yor +doc_to_text: "Answer the given question with the step by step solution appropriate\ + \ numerical value, ensuring that the response is clear and without any supplementary\ + \ information. \n\nQuestion: {{question}} \nStep by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_yor_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7023a5540612c3f7a568313f7d0d26e541839c49 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_4/afrimgsm_cot_translate_zul.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: zul +doc_to_text: "Answer the given question with the step by step solution appropriate\ + \ numerical value, ensuring that the response is clear and without any supplementary\ + \ information. \n\nQuestion: {{question}} \nStep by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_zul_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d64f088f46c537580055f91f3eaa347187531da4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_amh.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: amh +doc_to_text: "For mathematical questions provided in Amharic language. Supply the\ + \ accurate step by step answer to the provided question. \n\nQuestion: {{question}}\ + \ \nStep by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_amh_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_ewe.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_ewe.yaml new file mode 100644 index 0000000000000000000000000000000000000000..de4aa48d48be2c3ea31b03cb497c4b881ab09ead --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_ewe.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: ewe +doc_to_text: "For mathematical questions provided in Ewe language. Supply the accurate\ + \ step by step answer to the provided question. \n\nQuestion: {{question}} \nStep\ + \ by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_ewe_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_fra.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_fra.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5cf15ea1ad1e7937cb46cf7f5c09014e00e7fea7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_fra.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: fra +doc_to_text: "For mathematical questions provided in French language. Supply the accurate\ + \ step by step answer to the provided question. \n\nQuestion: {{question}} \nStep\ + \ by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_fra_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0dfa643c4cf14b0bbd871f6293f5b142a6a0337a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_hau.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: hau +doc_to_text: "For mathematical questions provided in Hausa language. Supply the accurate\ + \ step by step answer to the provided question. \n\nQuestion: {{question}} \nStep\ + \ by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_hau_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..959f389070977606a0d330573c4c4015754b80d7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_ibo.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: ibo +doc_to_text: "For mathematical questions provided in Igbo language. Supply the accurate\ + \ step by step answer to the provided question. \n\nQuestion: {{question}} \nStep\ + \ by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_ibo_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..85ff4196d50b187779c8c39b62cfcb5448ddb258 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_kin.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: kin +doc_to_text: "For mathematical questions provided in Kinyarwanda language. Supply\ + \ the accurate step by step answer to the provided question. \n\nQuestion: {{question}}\ + \ \nStep by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_kin_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_lin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_lin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..87db46c9761e033a42739d1aaaf1d14e51989d14 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_lin.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: lin +doc_to_text: "For mathematical questions provided in Lingala language. Supply the\ + \ accurate step by step answer to the provided question. \n\nQuestion: {{question}}\ + \ \nStep by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_lin_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ac0fde85c1607b570abb2c0fa602d0f680ef3fe1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_lug.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: lug +doc_to_text: "For mathematical questions provided in Luganda language. Supply the\ + \ accurate step by step answer to the provided question. \n\nQuestion: {{question}}\ + \ \nStep by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_lug_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_orm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_orm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bcf34106d924c07a92e63eaacdd3f112e860c25d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_orm.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: orm +doc_to_text: "For mathematical questions provided in Oromo language. Supply the accurate\ + \ step by step answer to the provided question. \n\nQuestion: {{question}} \nStep\ + \ by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_orm_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_sna.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_sna.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d5eac98d4e45e5c199cc5a19286c7d0edcd04a9f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_sna.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: sna +doc_to_text: "For mathematical questions provided in chiShona language. Supply the\ + \ accurate step by step answer to the provided question. \n\nQuestion: {{question}}\ + \ \nStep by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_sna_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_sot.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_sot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9fc015cdf75b34a6147183cfd3ad2f8bbe4d4660 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_sot.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: sot +doc_to_text: "For mathematical questions provided in Sesotho language. Supply the\ + \ accurate step by step answer to the provided question. \n\nQuestion: {{question}}\ + \ \nStep by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_sot_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..179af86738feaf0656c65e260cf465284c9bc3e5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_swa.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: swa +doc_to_text: "For mathematical questions provided in Swahili language. Supply the\ + \ accurate step by step answer to the provided question. \n\nQuestion: {{question}}\ + \ \nStep by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_swa_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ebb680a6f5e54727adabbdd517819e90ca9c2b96 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_twi.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: twi +doc_to_text: "For mathematical questions provided in Twi language. Supply the accurate\ + \ step by step answer to the provided question. \n\nQuestion: {{question}} \nStep\ + \ by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_twi_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_vai.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_vai.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3d2848648e5a752973ca5a21a38b0a2fb82b4127 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_vai.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: vai +doc_to_text: "For mathematical questions provided in Vai language. Supply the accurate\ + \ step by step answer to the provided question. \n\nQuestion: {{question}} \nStep\ + \ by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_vai_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..799cc29fbbca2a4a328cec6bf679cc68ad51139f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_wol.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: wol +doc_to_text: "For mathematical questions provided in Wolof language. Supply the accurate\ + \ step by step answer to the provided question. \n\nQuestion: {{question}} \nStep\ + \ by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_wol_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7969fdbabd911f8fe4ffdfb9f7e47364c3a80857 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_xho.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: xho +doc_to_text: "For mathematical questions provided in isiXhosa language. Supply the\ + \ accurate step by step answer to the provided question. \n\nQuestion: {{question}}\ + \ \nStep by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_xho_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_yaml new file mode 100644 index 0000000000000000000000000000000000000000..241787c7aa25a0ac46e2556efdb8db633e7a0719 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_yaml @@ -0,0 +1,32 @@ +tag: afrimgsm_tt_cot_tasks +dataset_path: masakhane/afrimgsm-translate-test +dataset_name: null # Overridden by language-specific config. +output_type: generate_until +test_split: test +doc_to_target: '{% if answer is not none %}{{answer[21:]}}{% else %}{{answer_number|string}}{% endif %}' +generation_kwargs: + do_sample: false + until: + - 'Question:' + - + - <|im_end|> +metric_list: + - metric: exact_match + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true +filter_list: + - name: "strict-match" + filter: + - function: "regex" + regex_pattern: "The answer is (\\-?[0-9\\.\\,]+)" + - function: "take_first" + - filter: + - function: regex + group_select: -1 + regex_pattern: (-?[$0-9.,]{2,})|(-?[0-9]+) + - function: take_first + name: flexible-extract +metadata: + version: 2.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0d05de223110d6e434bcf75bb2f9cf71957b76d3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_yor.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: yor +doc_to_text: "For mathematical questions provided in Yoruba language. Supply the accurate\ + \ step by step answer to the provided question. \n\nQuestion: {{question}} \nStep\ + \ by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_yor_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..68329068941f34bf7e53739334e8101c46a0ecfe --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_5/afrimgsm_cot_translate_zul.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: zul +doc_to_text: "For mathematical questions provided in Zulu language. Supply the accurate\ + \ step by step answer to the provided question. \n\nQuestion: {{question}} \nStep\ + \ by step answer: " +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_zul_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/README.md b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/README.md new file mode 100644 index 0000000000000000000000000000000000000000..f7f7ed4d82f04224440a0d164d2cc24c0e758990 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/README.md @@ -0,0 +1,50 @@ +# MathQA + +### Paper + +IrokoBench: A New Benchmark for African Languages in the Age of Large Language Models +https://arxiv.org/pdf/2406.03368 + +IrokoBench is a human-translated benchmark dataset for 16 typologically diverse +low-resource African languages covering three tasks: natural language inference (AfriXNLI), +mathematical reasoning (AfriMGSM), and multi-choice knowledge-based QA (AfriMMLU). + + +### Citation + +``` +@misc{adelani2024irokobenchnewbenchmarkafrican, + title={IrokoBench: A New Benchmark for African Languages in the Age of Large Language Models}, + author={David Ifeoluwa Adelani and Jessica Ojo and Israel Abebe Azime and Jian Yun Zhuang and Jesujoba O. Alabi and Xuanli He and Millicent Ochieng and Sara Hooker and Andiswa Bukula and En-Shiun Annie Lee and Chiamaka Chukwuneke and Happy Buzaaba and Blessing Sibanda and Godson Kalipe and Jonathan Mukiibi and Salomon Kabongo and Foutse Yuehgoh and Mmasibidi Setaka and Lolwethu Ndolela and Nkiruka Odu and Rooweither Mabuya and Shamsuddeen Hassan Muhammad and Salomey Osei and Sokhar Samb and Tadesse Kebede Guge and Pontus Stenetorp}, + year={2024}, + eprint={2406.03368}, + archivePrefix={arXiv}, + primaryClass={cs.CL}, + url={https://arxiv.org/abs/2406.03368}, +} +``` + +### Groups and Tasks + +#### Groups + +* `afrimmlu`: All afrimmlu tasks +* `afrimmlu_direct`: afrimmlu_direct evaluates models performance on the curated dataset +* `afrimmlu_translate`: afrimmlu_translate evaluates models in translate-test setting + +#### Tasks +* `afrimmlu_direct_{language_code}`: each task evaluates for one language +* `afrimmlu_translate_{language_code}`: each task evaluates for one language + +### Checklist + +For adding novel benchmarks/datasets to the library: +* [x] Is the task an existing benchmark in the literature? + * [x] Have you referenced the original paper that introduced the task? + * [ ] If yes, does the original paper provide a reference implementation? If so, have you checked against the reference implementation and documented how to run such a test? + +If other tasks on this dataset are already supported: +* [x] Is the "Main" variant of this task clearly denoted? +* [x] Have you provided a short sentence in a README on what each new variant adds / evaluates? +* [x] Have you noted which, if any, published evaluation setups are matched by this variant? + * [x] Checked for equivalence with v0.3.0 LM Evaluation Harness diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/afrimmlu.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/afrimmlu.yaml new file mode 100644 index 0000000000000000000000000000000000000000..202c31825bfcdaa8ea974e8f51444bc864ed4306 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/afrimmlu.yaml @@ -0,0 +1,13 @@ +group: afrimmlu-irokobench +task: + - afrimmlu_tasks_prompt_1 + - afrimmlu_tasks_prompt_2 + - afrimmlu_tasks_prompt_3 + - afrimmlu_tasks_prompt_4 + - afrimmlu_tasks_prompt_5 +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true +metadata: + version: 2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct new file mode 100644 index 0000000000000000000000000000000000000000..a3e17f711f6eac83c52fad1d3f0314a01f08d169 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct @@ -0,0 +1,37 @@ +tag: + - afrimmlu_tasks + - afrimmlu_tasks_prompt_1 + - afrobench_mmlu_tasks +dataset_path: masakhane/afrimmlu +dataset_name: null +output_type: multiple_choice +validation_split: validation +test_split: test +fewshot_split: validation +doc_to_text: !function utils.doc_to_text +doc_to_target: "{{['A', 'B', 'C', 'D'].index(answer)}}" +doc_to_choice: !function utils.doc_to_choice +should_decontaminate: true +doc_to_decontamination_query: "Question: {{question}}\nAnswer:" +metric_list: + - metric: f1 + aggregation: !function utils.weighted_f1_score + # aggregation: mean + average: weighted + hf_evaluate: true + higher_is_better: True + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" + - metric: acc + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8a26369b36ee47ed6ac21c448c316acaf90af749 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_amh.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: amh +include: afrimmlu_direct +task: afrimmlu_direct_amh_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_eng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..18a34c7b719cdef3254e2472399b7fdd3121d543 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_eng.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: eng +include: afrimmlu_direct +task: afrimmlu_direct_eng_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_fra.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_fra.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4e8a2875e71f43dbdd148331d24e6440f92ad71f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_fra.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fra +include: afrimmlu_direct +task: afrimmlu_direct_fra_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b438ea3198a18caf74d44a97f2d4752337edc082 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_hau.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: hau +include: afrimmlu_direct +task: afrimmlu_direct_hau_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8b08d48e0a3c42e123c081935d5ffcc71e1c56c7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_ibo.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: ibo +include: afrimmlu_direct +task: afrimmlu_direct_ibo_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..00d82dfa57e6a77d52f476ae78c54edbc677628d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_kin.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: kin +include: afrimmlu_direct +task: afrimmlu_direct_kin_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_lin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_lin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7059c941d2dffd27c8eda15dc1fc087a626455a2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_lin.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: lin +include: afrimmlu_direct +task: afrimmlu_direct_lin_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_orm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_orm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5047ae98abd0d3ea8ebeb98233ae4fb1ebb42dab --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_orm.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: orm +include: afrimmlu_direct +task: afrimmlu_direct_orm_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_sot.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_sot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c62ce9bf4957545dc39f96d7bd6dc60ce60e868a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_sot.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: sot +include: afrimmlu_direct +task: afrimmlu_direct_sot_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..85d85171bfe9d84205a9ab218ed496aed1eecf73 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_amh.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: amh +include: afrimmlu_direct +task: afrimmlu_direct_amh_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_eng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c46eca5e68356372fc43c1b1908e45667ff05d12 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_eng.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: eng +include: afrimmlu_direct +task: afrimmlu_direct_eng_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_fra.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_fra.yaml new file mode 100644 index 0000000000000000000000000000000000000000..47f0bfb14b6cd7eaf34618eb0709f9f3f0c9b666 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_fra.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fra +include: afrimmlu_direct +task: afrimmlu_direct_fra_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_lin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_lin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..51fcea62af8ca4fa64643ef4ea170444ce25beef --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_lin.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: lin +include: afrimmlu_direct +task: afrimmlu_direct_lin_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/fewshot.sh b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/fewshot.sh new file mode 100644 index 0000000000000000000000000000000000000000..c69c48d7dff4e2495485023187dc162742c7ca6a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/fewshot.sh @@ -0,0 +1,8 @@ +lm_eval --model hf \ + --model_args pretrained=masakhane/African-ultrachat-alpaca \ + --tasks afrimmlu_direct_amh,afrimmlu_direct_eng,afrimmlu_direct_ewe,afrimmlu_direct_fra,afrimmlu_direct_hau,afrimmlu_direct_ibo,afrimmlu_direct_kin,afrimmlu_direct_lin,afrimmlu_direct_lug,afrimmlu_direct_orm,afrimmlu_direct_sna,afrimmlu_direct_sot,afrimmlu_direct_twi,afrimmlu_direct_wol,afrimmlu_direct_xho,afrimmlu_direct_yor,afrimmlu_direct_zul \ + --device cuda:0 \ + --batch_size 1 \ + --num_fewshot 0 \ + --verbosity DEBUG \ + --wandb_args project=afrimmlu diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/gen_utils.py b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/gen_utils.py new file mode 100644 index 0000000000000000000000000000000000000000..a195b6b5852d35042c14632597762a3965faae07 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/gen_utils.py @@ -0,0 +1,103 @@ +import argparse +import os + +import yaml + + +class FunctionTag: + def __init__(self, value): + self.value = value + + +def gen_lang_yamls(output_dir: str, overwrite: bool, mode: str) -> None: + """ + Generate a yaml file for each language. + + :param output_dir: The directory to output the files to. + :param overwrite: Whether to overwrite files if they already exist. + """ + err = [] + languages = { + "eng": "English", + "amh": "Amharic", + "ibo": "Igbo", + "fra": "French", + "sna": "chiShona", + "wol": "Wolof", + "ewe": "Ewe", + "lin": "Lingala", + "lug": "Luganda", + "xho": "isiXhosa", + "kin": "Kinyarwanda", + "twi": "Twi", + "zul": "Zulu", + "orm": "Oromo", + "yor": "Yoruba", + "hau": "Hausa", + "sot": "Sesotho", + "swa": "Swahili", + } + + for lang in languages.keys(): + try: + file_name = f"afrimmlu_direct_{lang}.yaml" + task_name = f"afrimmlu_direct_{lang}_{mode}" + yaml_template = "afrimmlu_direct" + if output_dir.split("/")[-1] == "translate": + file_name = f"afrimmlu_translate_{lang}.yaml" + task_name = f"afrimmlu_translate_{lang}_{mode}" + yaml_template = "afrimmlu_translate" + yaml_details = { + "include": yaml_template, + "task": task_name, + "dataset_name": lang, + } + os.makedirs(f"{output_dir}/{mode}", exist_ok=True) + with open( + f"{output_dir}/{mode}/{file_name}", + "w" if overwrite else "x", + encoding="utf8", + ) as f: + f.write("# Generated by utils.py\n") + yaml.dump( + yaml_details, + f, + allow_unicode=True, + ) + except FileExistsError: + err.append(file_name) + + if len(err) > 0: + raise FileExistsError( + "Files were not created because they already exist (use --overwrite flag):" + f" {', '.join(err)}" + ) + + +def main() -> None: + """Parse CLI args and generate language-specific yaml files.""" + parser = argparse.ArgumentParser() + parser.add_argument( + "--overwrite", + default=True, + action="store_true", + help="Overwrite files if they already exist", + ) + parser.add_argument( + "--output-dir", + default="./direct", + help="Directory to write yaml files to", + ) + parser.add_argument( + "--mode", + default="prompt_4", + choices=["prompt_1", "prompt_2", "prompt_3", "prompt_4", "prompt_5"], + help="Prompt number", + ) + args = parser.parse_args() + + gen_lang_yamls(output_dir=args.output_dir, overwrite=args.overwrite, mode=args.mode) + + +if __name__ == "__main__": + main() diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..9d02b342b2e3c9f3d3bd66d3f62330aa53c9159c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/utils.py @@ -0,0 +1,32 @@ +from lm_eval.utils import weighted_f1_score + + +def doc_to_choice(doc): + choices = eval(doc["choices"]) + return choices + + +def doc_to_text(doc): + output = """You are a highly knowledgeable and intelligent artificial intelligence + model answers multiple-choice questions about '{subject}' + + Question: '''{question}''' + + Choices: + A: ''{choice1}''' + B: ''{choice2}''' + C: ''{choice3}''' + D: ''{choice4}''' + + Answer: """ + + choices = eval(doc["choices"]) + text = output.format( + subject=doc["subject"], + question=doc["question"], + choice1=choices[0], + choice2=choices[1], + choice3=choices[2], + choice4=choices[3], + ) + return text diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_ewe.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_ewe.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0d550f9daff83651b4099d3ad4228de0afab6ac4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_ewe.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: ewe +doc_to_text: "Given the following premise and hypothesis in Ewe, identify if the premise\ + \ entails, contradicts, or is neutral towards the hypothesis. Please respond with\ + \ exact 'entailment', 'contradiction', or 'neutral'. \n\nPremise: {{premise}} \n\ + Hypothesis: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_ewe_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_fra.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_fra.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3156466c388362d06414b613e963f0f9fcb1465f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_fra.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: fra +doc_to_text: "Given the following premise and hypothesis in French, identify if the\ + \ premise entails, contradicts, or is neutral towards the hypothesis. Please respond\ + \ with exact 'entailment', 'contradiction', or 'neutral'. \n\nPremise: {{premise}}\ + \ \nHypothesis: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_fra_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_sna.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_sna.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d7dbf17e8b0520fc15bc6d6b337c959f828268ef --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_sna.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: sna +doc_to_text: "Given the following premise and hypothesis in chiShona, identify if\ + \ the premise entails, contradicts, or is neutral towards the hypothesis. Please\ + \ respond with exact 'entailment', 'contradiction', or 'neutral'. \n\nPremise: {{premise}}\ + \ \nHypothesis: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_sna_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e7e8568701be5958b2da080d5d6c0885e83bb370 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_twi.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: twi +doc_to_text: "Given the following premise and hypothesis in Twi, identify if the premise\ + \ entails, contradicts, or is neutral towards the hypothesis. Please respond with\ + \ exact 'entailment', 'contradiction', or 'neutral'. \n\nPremise: {{premise}} \n\ + Hypothesis: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_twi_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a4dafa34f6cd5b69e115de88e04cd24b3473c5fd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_xho.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: xho +doc_to_text: "Given the following premise and hypothesis in isiXhosa, identify if\ + \ the premise entails, contradicts, or is neutral towards the hypothesis. Please\ + \ respond with exact 'entailment', 'contradiction', or 'neutral'. \n\nPremise: {{premise}}\ + \ \nHypothesis: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_xho_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_yaml new file mode 100644 index 0000000000000000000000000000000000000000..832b51493a0f17996ebca680f7151e38b59168d3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_yaml @@ -0,0 +1,27 @@ +tag: afrixnli_tt_tasks +dataset_path: masakhane/afrixnli-translate-test +dataset_name: null +output_type: multiple_choice +test_split: test +fewshot_split: test +doc_to_target: !function utils.doc_to_target +doc_to_choice: + - "entailment" + - "neutral" + - "contradiction" +should_decontaminate: true +doc_to_decontamination_query: premise +metric_list: + - metric: f1 + aggregation: !function utils.weighted_f1_score + average: weighted + higher_is_better: True + ignore_case: true + ignore_punctuation: true + - metric: acc + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f5c01ca54eaba74fd968ce7847f43b2fe4b373fd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_yor.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: yor +doc_to_text: "Given the following premise and hypothesis in Yoruba, identify if the\ + \ premise entails, contradicts, or is neutral towards the hypothesis. Please respond\ + \ with exact 'entailment', 'contradiction', or 'neutral'. \n\nPremise: {{premise}}\ + \ \nHypothesis: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_yor_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fdbbec809a0b21cabc58defd6cbac15d0ea29ff2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_zul.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: zul +doc_to_text: "Given the following premise and hypothesis in Zulu, identify if the\ + \ premise entails, contradicts, or is neutral towards the hypothesis. Please respond\ + \ with exact 'entailment', 'contradiction', or 'neutral'. \n\nPremise: {{premise}}\ + \ \nHypothesis: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_zul_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..c455a3045a9be8b7318b96e23d9f061add6a342e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/utils.py @@ -0,0 +1,21 @@ +from lm_eval.utils import weighted_f1_score + + +def doc_to_text(doc): + output = """You are an NLP assistant whose purpose is to solve Natural Language Inference (NLI) problems + + Please identify whether the premise entails or contradicts the hypothesis in the following premise + and hypothesis. The answer should be exact entailment, contradiction, or neutral. + + Premise: {premise} + Hypothesis: {hypothesis} + + Is it entailment, contradiction, or neutral?""" + + text = output.format(premise=doc["premise"], hypothesis=doc["hypothesis"]) + return text + + +def doc_to_target(doc): + replacements = {0: "entailment", 1: "neutral", 2: "contradiction"} + return replacements[doc["label"]] diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b5f972e7070ac1ed50b7ed177daa706d520e3a4f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_amh.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: amh +doc_to_text: "You are an expert in Natural Language Inference (NLI) specializing in\ + \ the Amharic language.\nAnalyze the premise and hypothesis given in Amharic, and\ + \ determine the relationship between them.\n Respond with one of the following options:\ + \ 'entailment', 'contradiction', or 'neutral'. \n\nPremise: {{premise}} \nHypothesis:\ + \ {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_amh_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_ewe.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_ewe.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ebc775dd705e572f68835e194b6c3c8d745f6e1b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_ewe.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: ewe +doc_to_text: "You are an expert in Natural Language Inference (NLI) specializing in\ + \ the Ewe language.\nAnalyze the premise and hypothesis given in Ewe, and determine\ + \ the relationship between them.\n Respond with one of the following options: 'entailment',\ + \ 'contradiction', or 'neutral'. \n\nPremise: {{premise}} \nHypothesis: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_ewe_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dd65f366bc2de430404434c7918e3a0bb69aba03 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_hau.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: hau +doc_to_text: "You are an expert in Natural Language Inference (NLI) specializing in\ + \ the Hausa language.\nAnalyze the premise and hypothesis given in Hausa, and determine\ + \ the relationship between them.\n Respond with one of the following options: 'entailment',\ + \ 'contradiction', or 'neutral'. \n\nPremise: {{premise}} \nHypothesis: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_hau_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..13df12642e743ceb4867a4d515ed2d20edce7486 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_ibo.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: ibo +doc_to_text: "You are an expert in Natural Language Inference (NLI) specializing in\ + \ the Igbo language.\nAnalyze the premise and hypothesis given in Igbo, and determine\ + \ the relationship between them.\n Respond with one of the following options: 'entailment',\ + \ 'contradiction', or 'neutral'. \n\nPremise: {{premise}} \nHypothesis: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_ibo_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..198d88750287ee0fc28507ddcdab0146b1c2f734 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_kin.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: kin +doc_to_text: "You are an expert in Natural Language Inference (NLI) specializing in\ + \ the Kinyarwanda language.\nAnalyze the premise and hypothesis given in Kinyarwanda,\ + \ and determine the relationship between them.\n Respond with one of the following\ + \ options: 'entailment', 'contradiction', or 'neutral'. \n\nPremise: {{premise}}\ + \ \nHypothesis: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_kin_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_lin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_lin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b25856cfdcbb7b0c0df2aba81322d650585e9d9a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_lin.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: lin +doc_to_text: "You are an expert in Natural Language Inference (NLI) specializing in\ + \ the Lingala language.\nAnalyze the premise and hypothesis given in Lingala, and\ + \ determine the relationship between them.\n Respond with one of the following options:\ + \ 'entailment', 'contradiction', or 'neutral'. \n\nPremise: {{premise}} \nHypothesis:\ + \ {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_lin_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..633c173c0b75903c81934391e1e9f07a8de9b7f7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_lug.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: lug +doc_to_text: "You are an expert in Natural Language Inference (NLI) specializing in\ + \ the Luganda language.\nAnalyze the premise and hypothesis given in Luganda, and\ + \ determine the relationship between them.\n Respond with one of the following options:\ + \ 'entailment', 'contradiction', or 'neutral'. \n\nPremise: {{premise}} \nHypothesis:\ + \ {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_lug_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_sna.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_sna.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6fcb4e063159ab48e9e73423b895658a74cafa9e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_sna.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: sna +doc_to_text: "You are an expert in Natural Language Inference (NLI) specializing in\ + \ the chiShona language.\nAnalyze the premise and hypothesis given in chiShona,\ + \ and determine the relationship between them.\n Respond with one of the following\ + \ options: 'entailment', 'contradiction', or 'neutral'. \n\nPremise: {{premise}}\ + \ \nHypothesis: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_sna_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_sot.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_sot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..358e4b353eb73065bf91b5471127faf5b2f1675f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_sot.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: sot +doc_to_text: "You are an expert in Natural Language Inference (NLI) specializing in\ + \ the Sesotho language.\nAnalyze the premise and hypothesis given in Sesotho, and\ + \ determine the relationship between them.\n Respond with one of the following options:\ + \ 'entailment', 'contradiction', or 'neutral'. \n\nPremise: {{premise}} \nHypothesis:\ + \ {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_sot_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8ce271ed77c52dbce949a6a83ccd1313d26c9b25 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_swa.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: swa +doc_to_text: "You are an expert in Natural Language Inference (NLI) specializing in\ + \ the Swahili language.\nAnalyze the premise and hypothesis given in Swahili, and\ + \ determine the relationship between them.\n Respond with one of the following options:\ + \ 'entailment', 'contradiction', or 'neutral'. \n\nPremise: {{premise}} \nHypothesis:\ + \ {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_swa_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8171e0daa98aa93614800f3c77a42d2b699ee2a7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_twi.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: twi +doc_to_text: "You are an expert in Natural Language Inference (NLI) specializing in\ + \ the Twi language.\nAnalyze the premise and hypothesis given in Twi, and determine\ + \ the relationship between them.\n Respond with one of the following options: 'entailment',\ + \ 'contradiction', or 'neutral'. \n\nPremise: {{premise}} \nHypothesis: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_twi_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b2662662dff2263023dd2bce1afeb86bb09ce262 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_wol.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: wol +doc_to_text: "You are an expert in Natural Language Inference (NLI) specializing in\ + \ the Wolof language.\nAnalyze the premise and hypothesis given in Wolof, and determine\ + \ the relationship between them.\n Respond with one of the following options: 'entailment',\ + \ 'contradiction', or 'neutral'. \n\nPremise: {{premise}} \nHypothesis: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_wol_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5aa3a9d171a4dabe831d5a6126ca61384d218d81 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_xho.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: xho +doc_to_text: "You are an expert in Natural Language Inference (NLI) specializing in\ + \ the isiXhosa language.\nAnalyze the premise and hypothesis given in isiXhosa,\ + \ and determine the relationship between them.\n Respond with one of the following\ + \ options: 'entailment', 'contradiction', or 'neutral'. \n\nPremise: {{premise}}\ + \ \nHypothesis: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_xho_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_yaml new file mode 100644 index 0000000000000000000000000000000000000000..832b51493a0f17996ebca680f7151e38b59168d3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_yaml @@ -0,0 +1,27 @@ +tag: afrixnli_tt_tasks +dataset_path: masakhane/afrixnli-translate-test +dataset_name: null +output_type: multiple_choice +test_split: test +fewshot_split: test +doc_to_target: !function utils.doc_to_target +doc_to_choice: + - "entailment" + - "neutral" + - "contradiction" +should_decontaminate: true +doc_to_decontamination_query: premise +metric_list: + - metric: f1 + aggregation: !function utils.weighted_f1_score + average: weighted + higher_is_better: True + ignore_case: true + ignore_punctuation: true + - metric: acc + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..478e5043431c73fe2448474b593ae02002d6a722 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_yor.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: yor +doc_to_text: "You are an expert in Natural Language Inference (NLI) specializing in\ + \ the Yoruba language.\nAnalyze the premise and hypothesis given in Yoruba, and\ + \ determine the relationship between them.\n Respond with one of the following options:\ + \ 'entailment', 'contradiction', or 'neutral'. \n\nPremise: {{premise}} \nHypothesis:\ + \ {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_yor_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c0dc06e6bce0aae18769cd2259aee6699be3b0bd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/afrixnli_translate_zul.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: zul +doc_to_text: "You are an expert in Natural Language Inference (NLI) specializing in\ + \ the Zulu language.\nAnalyze the premise and hypothesis given in Zulu, and determine\ + \ the relationship between them.\n Respond with one of the following options: 'entailment',\ + \ 'contradiction', or 'neutral'. \n\nPremise: {{premise}} \nHypothesis: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_zul_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..d97a0a288508e817ab695e637fb157a08c813808 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_4/utils.py @@ -0,0 +1,19 @@ +from lm_eval.utils import weighted_f1_score + + +def doc_to_text(doc): + output = """Please identify whether the premise entails or contradicts the hypothesis in the following premise + and hypothesis. The answer should be exact entailment, contradiction, or neutral. + + Premise: {premise} + Hypothesis: {hypothesis} + + Is it entailment, contradiction, or neutral?""" + + text = output.format(premise=doc["premise"], hypothesis=doc["hypothesis"]) + return text + + +def doc_to_target(doc): + replacements = {0: "entailment", 1: "neutral", 2: "contradiction"} + return replacements[doc["label"]] diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3079712ce60b0b9b7e5846d3e1d9b16383c8cf97 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_amh.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: amh +doc_to_text: "Based on the given statement, is the following claim 'true', 'false',\ + \ or 'inconclusive'. \nStatement: {{premise}} \nClaim: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_amh_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_ewe.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_ewe.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6eb452db53361c5c048e75a807de20b3528414ec --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_ewe.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: ewe +doc_to_text: "Based on the given statement, is the following claim 'true', 'false',\ + \ or 'inconclusive'. \nStatement: {{premise}} \nClaim: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_ewe_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_fra.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_fra.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d6ddf49332ba4ec1671a7a20e036e7d4906c2097 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_fra.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: fra +doc_to_text: "Based on the given statement, is the following claim 'true', 'false',\ + \ or 'inconclusive'. \nStatement: {{premise}} \nClaim: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_fra_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..09d182f7f12e9debba53ed9bd5b1249c38b63a53 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_hau.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: hau +doc_to_text: "Based on the given statement, is the following claim 'true', 'false',\ + \ or 'inconclusive'. \nStatement: {{premise}} \nClaim: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_hau_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b5bf1555a454879018ee16c3ed60dd1f71cbdbbe --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_ibo.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: ibo +doc_to_text: "Based on the given statement, is the following claim 'true', 'false',\ + \ or 'inconclusive'. \nStatement: {{premise}} \nClaim: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_ibo_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f0cbe9c2c78612111af3ce29e4a8b846879f3060 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_kin.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: kin +doc_to_text: "Based on the given statement, is the following claim 'true', 'false',\ + \ or 'inconclusive'. \nStatement: {{premise}} \nClaim: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_kin_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_lin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_lin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..159116be57762adddff7368d1731a867a8daa152 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_lin.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: lin +doc_to_text: "Based on the given statement, is the following claim 'true', 'false',\ + \ or 'inconclusive'. \nStatement: {{premise}} \nClaim: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_lin_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9448fa28cb2561631dfa78fbeccb4bc054c867f6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_lug.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: lug +doc_to_text: "Based on the given statement, is the following claim 'true', 'false',\ + \ or 'inconclusive'. \nStatement: {{premise}} \nClaim: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_lug_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_orm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_orm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..64621cb4b784c2e4c16bfe220acb69d7ea17cb8f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_orm.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: orm +doc_to_text: "Based on the given statement, is the following claim 'true', 'false',\ + \ or 'inconclusive'. \nStatement: {{premise}} \nClaim: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_orm_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_sna.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_sna.yaml new file mode 100644 index 0000000000000000000000000000000000000000..788bae3068b2b40ed1ae4419a8ddbfc9094fe71c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_sna.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: sna +doc_to_text: "Based on the given statement, is the following claim 'true', 'false',\ + \ or 'inconclusive'. \nStatement: {{premise}} \nClaim: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_sna_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_sot.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_sot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..617dd9f88db6d283eef224126b36a0fcf2e35158 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_sot.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: sot +doc_to_text: "Based on the given statement, is the following claim 'true', 'false',\ + \ or 'inconclusive'. \nStatement: {{premise}} \nClaim: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_sot_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..81a159252ca45302bf5c88c448489a76bd342270 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_swa.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: swa +doc_to_text: "Based on the given statement, is the following claim 'true', 'false',\ + \ or 'inconclusive'. \nStatement: {{premise}} \nClaim: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_swa_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cb9f115fb3c01ea8a509a7347f6a369ae1f9c819 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_twi.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: twi +doc_to_text: "Based on the given statement, is the following claim 'true', 'false',\ + \ or 'inconclusive'. \nStatement: {{premise}} \nClaim: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_twi_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a5f4eb0c2eeaf10d04b2978cf3ffb821a7123889 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_wol.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: wol +doc_to_text: "Based on the given statement, is the following claim 'true', 'false',\ + \ or 'inconclusive'. \nStatement: {{premise}} \nClaim: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_wol_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d085919b777d18fff97397bdd32d1c0cf2c1c316 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_xho.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: xho +doc_to_text: "Based on the given statement, is the following claim 'true', 'false',\ + \ or 'inconclusive'. \nStatement: {{premise}} \nClaim: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_xho_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_yaml new file mode 100644 index 0000000000000000000000000000000000000000..3047238439e371f99664aed214a6589b33528e66 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_yaml @@ -0,0 +1,27 @@ +tag: afrixnli_tt_tasks +dataset_path: masakhane/afrixnli-translate-test +dataset_name: null +output_type: multiple_choice +test_split: test +fewshot_split: test +doc_to_target: !function utils.doc_to_target +doc_to_choice: + - "true" + - "inconclusive" + - "false" +should_decontaminate: true +doc_to_decontamination_query: premise +metric_list: + - metric: f1 + aggregation: !function utils.weighted_f1_score + average: weighted + higher_is_better: True + ignore_case: true + ignore_punctuation: true + - metric: acc + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d963646034069b77a78fe5284b106b6f74718a6a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_zul.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: zul +doc_to_text: "Based on the given statement, is the following claim 'true', 'false',\ + \ or 'inconclusive'. \nStatement: {{premise}} \nClaim: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_zul_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..6b9cb312b25a4c21bdd3d6a5e0a4e8e160451e4a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/utils.py @@ -0,0 +1,6 @@ +from lm_eval.utils import weighted_f1_score + + +def doc_to_target(doc): + replacements = {0: "true", 1: "false", 2: "inconclusive"} + return replacements[doc["label"]] diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/README.md b/lm-evaluation-harness/lm_eval/tasks/afrobench/README.md new file mode 100644 index 0000000000000000000000000000000000000000..a6ab3ceef1b37e94f1c191d1931648b7b669a49e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/README.md @@ -0,0 +1,72 @@ +# AfroBench + +### Paper + +Title: `AfroBench: How Good are Large Language Models on African Languages?` + +Paper Link: https://arxiv.org/abs/2311.07978 + +## Abstract +> Large-scale multilingual evaluations, such as MEGA, often include only a handful of African languages due to the scarcity of high-quality evaluation data and the limited discoverability of existing African datasets. This lack of representation hinders comprehensive LLM evaluation across a diverse range of languages and tasks. To address these challenges, we introduce AfroBench -- a multi-task benchmark for evaluating the performance of LLMs across 64 African languages, 15 tasks and 22 datasets. AfroBench consists of nine natural language understanding datasets, six text generation datasets, six knowledge and question answering tasks, and one mathematical reasoning task. We present results comparing the performance of prompting LLMs to fine-tuned baselines based on BERT and T5-style models. Our results suggest large gaps in performance between high-resource languages, such as English, and African languages across most tasks; but performance also varies based on the availability of monolingual data resources. Our findings confirm that performance on African languages continues to remain a hurdle for current LLMs, underscoring the need for additional efforts to close this gap. + +HomePage: https://mcgill-nlp.github.io/AfroBench/ + +### Groups, and Tasks +#### Groups +* `afrobench` : Runs all that tasks, datasets and prompts in this benchmark +* `afrobench_lite`: Runs the lite version of the benchmark which includes; afrimgsm, afrimmlu, afrixnli, sib, intent, adr and flores + +Dataset specific grouping that listing all prompts, allowing users to review or edit them. +* `adr` `afrihate` `afrisenti` `belebele` `african_flores` `injongointent` `mafand` `masakhaner` `masakhapos` `naijarc` `nollysenti` `african_ntrex` `openai_mmlu` `salt` `sib` `uhura` `xlsum` + + +#### Task Tags +* `adr_tasks`: all datasets in this benchmark relating to Automatic Diacritics Restoration task +* `afrihate_tasks`: all datasets in this benchmark relating to Hate Speech detection task +* `afrimgsm_tasks`: all datasets in this benchmark relating to Mathematical reasoning task +* `afrixnli_tasks`: all datasets in this benchmark relating to Natural Language Inference task +* `afrobench_xqa_tasks`: all datasets in this benchmark relating to Crosslingual QA (XQA) task +* `afrobench_sentiment_tasks`: all datasets in this benchmark relating to Sentiment Classification task +* `afrobench_MT_tasks`: all datasets in this benchmark relating to Machine Translation task +* `afrobench_TC_tasks`: all datasets in this benchmark relating to Topic Classification task +* `afrobench_mmlu_tasks`: all datasets in this benchmark relating to MMLU task +* `injongointent_tasks`: all datasets in this benchmark relating to Intent Detection task +* `masakhaner_tasks`: all datasets in this benchmark relating to Named Entity Recognition (NER) task +* `masakhapos_tasks`: all datasets in this benchmark relating to Part of Speech Tagging (POS) task +* `RC_tasks`: all datasets in this benchmark relating to Reading Comprehension task +* `uhura_arc_easy_tasks`: all datasets in this benchmark relating to Arc-Easy (XQA) task +* `xlsum_tasks`: all datasets in this benchmark relating to Summarization task + + +We've included sample run scripts for easier integration with the benchmark: [sample run scripts](./sample_run_scripts) + +For better understanding of the run interface see [interface.md](../../../docs/interface.md) + +All dataset used in this benchmark are available at [huggingface](https://huggingface.co/collections/masakhane/afrobench-67dbf553ebf5701c2207f883) + +### Citation + +``` +@misc{ojo2025afrobenchgoodlargelanguage, + title={AfroBench: How Good are Large Language Models on African Languages?}, + author={Jessica Ojo and Odunayo Ogundepo and Akintunde Oladipo and Kelechi Ogueji and Jimmy Lin and Pontus Stenetorp and David Ifeoluwa Adelani}, + year={2025}, + eprint={2311.07978}, + archivePrefix={arXiv}, + primaryClass={cs.CL}, + url={https://arxiv.org/abs/2311.07978}, +} +``` +Please cite datasets used. Citations for individual datasets are included in their respective repository readme files within this benchmark. +### Checklist + +For adding novel benchmarks/datasets to the library: +* [ ] Is the task an existing benchmark in the literature? + * [ ] Have you referenced the original paper that introduced the task? + * [ ] If yes, does the original paper provide a reference implementation? If so, have you checked against the reference implementation and documented how to run such a test? The original paper doesn't have an associated implementation, but there is an official entry in [BigBench](https://github.com/google/BIG-bench/tree/main/bigbench/benchmark_tasks/social_iqa). I use the same prompting format as BigBench. + + +If other tasks on this dataset are already supported: +* [ ] Is the "Main" variant of this task clearly denoted? +* [ ] Have you provided a short sentence in a README on what each new variant adds / evaluates? +* [ ] Have you noted which, if any, published evaluation setups are matched by this variant? diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/README.md b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/README.md new file mode 100644 index 0000000000000000000000000000000000000000..cb09567dcd2f461e3adba531bddd570c21ebfdf5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/README.md @@ -0,0 +1,7 @@ +# Automatic Diacritics Restoration (ADR) + +Automatic Diacritics Restoration (ADR) is the task of restoring diacritical marks in text where they have been omitted or removed. +This process is essential for languages where diacritics alter pronunciation, meaning, or grammatical structure. +ADR requires the model to have a deep understanding of linguistic context, syntax, and semantics to accurately predict and reinsert the appropriate diacritics. + +As part of this benchmark project, we utilise the mafand dataset to curate a dataset specifically for ADR. We focus on five languages: Gbomola, Fon, Igbo, Wolof, and Yoruba. diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/afridiacritics.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/afridiacritics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..34d60eef66acd9829ebb6e60ca6b85e8616a32d8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/afridiacritics.yaml @@ -0,0 +1,13 @@ +group: adr +task: + - adr_prompt_1 + - adr_prompt_2 + - adr_prompt_3 + - adr_prompt_4 + - adr_prompt_5 +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/gen_utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/gen_utils.py new file mode 100644 index 0000000000000000000000000000000000000000..ff6e63e3456abf809a1068f4abeea8ac93b49e94 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/gen_utils.py @@ -0,0 +1,105 @@ +import argparse +import os + +import yaml + + +class FunctionTag: + def __init__(self, value): + self.value = value + + +def prompt_func(mode, lang): + prompt_map = { + "prompt_1": "Please restore the missing diacritics in the following sentence: {{text}}. Return output sentence only", + "prompt_2": "Given a sentence without diacritics, add the appropriate diacritics to make it grammatically " + "and semantically correct. \nSentence: {{text}}. Return output sentence only", + "prompt_3": f"This text is in {lang}. Restore all diacritical marks to their proper places in the " + "following sentence: {{text}}. Return output sentence only", + "prompt_4": f"You are a linguist specializing in diacritical marks for {lang}. " + f"Add the appropriate diacritics to this {lang} sentence: " + "{{text}}. Return output sentence only", + "prompt_5": f"You are a linguist specializing in diacritical marks for {lang}. Diacritics are essential for " + f"proper pronunciation and meaning in {lang}. You are tasked with converting {lang} sentences " + "without diacritics into their correctly accented forms. Here's the input: {{text}}. " + "Return output sentence only", + } + return prompt_map[mode] + + +def gen_lang_yamls(output_dir: str, overwrite: bool, mode: str) -> None: + """ + Generate a yaml file for each language. + + :param output_dir: The directory to output the files to. + :param overwrite: Whether to overwrite files if they already exist. + """ + err = [] + languages = { + "fon": "Fon", + "bbj": "Gbomala", + "ibo": "Igbo", + "wol": "Wolof", + "yor": "Yoruba", + } + + for lang in languages.keys(): + try: + file_name = f"afridiacritics_{lang}.yaml" + task_name = f"afridiacritics_{lang}_{mode}" + yaml_template = "afridiacritics_yaml" + yaml_details = { + "include": yaml_template, + "task": task_name, + "dataset_name": lang, + "doc_to_text": prompt_func(mode, languages[lang]), + } + os.makedirs(f"{output_dir}/{mode}", exist_ok=True) + with open( + f"{output_dir}/{mode}/{file_name}", + "w" if overwrite else "x", + encoding="utf8", + ) as f: + f.write("# Generated by utils.py\n") + yaml.dump( + yaml_details, + f, + allow_unicode=True, + ) + except FileExistsError: + err.append(file_name) + + if len(err) > 0: + raise FileExistsError( + "Files were not created because they already exist (use --overwrite flag):" + f" {', '.join(err)}" + ) + + +def main() -> None: + """Parse CLI args and generate language-specific yaml files.""" + parser = argparse.ArgumentParser() + parser.add_argument( + "--overwrite", + default=True, + action="store_true", + help="Overwrite files if they already exist", + ) + parser.add_argument( + "--output-dir", + default="./", + help="Directory to write yaml files to", + ) + parser.add_argument( + "--mode", + default="prompt_1", + choices=["prompt_1", "prompt_2", "prompt_3", "prompt_4", "prompt_5"], + help="Prompt number", + ) + args = parser.parse_args() + + gen_lang_yamls(output_dir=args.output_dir, overwrite=args.overwrite, mode=args.mode) + + +if __name__ == "__main__": + main() diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_1/afridiacritics_bbj.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_1/afridiacritics_bbj.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f3eb26ebae6f723c03591aa73eb29f2256fb0e4a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_1/afridiacritics_bbj.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: bbj +doc_to_text: 'Please restore the missing diacritics in the following sentence: {{text}}. + Return output sentence only' +include: afridiacritics_yaml +task: afridiacritics_bbj_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_1/afridiacritics_fon.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_1/afridiacritics_fon.yaml new file mode 100644 index 0000000000000000000000000000000000000000..874832d5d00799deb9235d2f04960684f2b91770 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_1/afridiacritics_fon.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: fon +doc_to_text: 'Please restore the missing diacritics in the following sentence: {{text}}. + Return output sentence only' +include: afridiacritics_yaml +task: afridiacritics_fon_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_1/afridiacritics_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_1/afridiacritics_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..983bc3914c421f46fa1adbdd67c15c433996a584 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_1/afridiacritics_ibo.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: ibo +doc_to_text: 'Please restore the missing diacritics in the following sentence: {{text}}. + Return output sentence only' +include: afridiacritics_yaml +task: afridiacritics_ibo_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_1/afridiacritics_wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_1/afridiacritics_wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9067770a3dac5fc4aab23cdad2d15ede76b82de4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_1/afridiacritics_wol.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: wol +doc_to_text: 'Please restore the missing diacritics in the following sentence: {{text}}. + Return output sentence only' +include: afridiacritics_yaml +task: afridiacritics_wol_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_1/afridiacritics_yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_1/afridiacritics_yaml new file mode 100644 index 0000000000000000000000000000000000000000..53cebaee05c9e7a65779ad12faaa0a9ee40c7c8b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_1/afridiacritics_yaml @@ -0,0 +1,25 @@ +tag: +- adr_tasks +- adr_prompt_1 +dataset_path: masakhane/diacritics-restoration +dataset_kwargs: {trust_remote_code: True} +doc_to_target: target +output_type: generate_until +fewshot_split: dev +test_split: test +training_split: train +metric_list: + - metric: bleu + aggregation: bleu + higher_is_better: true + - metric: chrf + aggregation: chrf + higher_is_better: true +generation_kwargs: + do_sample: false + until: + - '' + - + - <|im_end|> +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_1/afridiacritics_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_1/afridiacritics_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8e98af10abec2009d32b112694923f45c17473af --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_1/afridiacritics_yor.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: yor +doc_to_text: 'Please restore the missing diacritics in the following sentence: {{text}}. + Return output sentence only' +include: afridiacritics_yaml +task: afridiacritics_yor_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_2/afridiacritics_bbj.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_2/afridiacritics_bbj.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0f054eea4c29da978b830d3a5eb2571af364f920 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_2/afridiacritics_bbj.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: bbj +doc_to_text: "Given a sentence without diacritics, add the appropriate diacritics\ + \ to make it grammatically and semantically correct. \nSentence: {{text}}. Return\ + \ output sentence only" +include: afridiacritics_yaml +task: afridiacritics_bbj_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_2/afridiacritics_fon.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_2/afridiacritics_fon.yaml new file mode 100644 index 0000000000000000000000000000000000000000..07f7114649ff8f362a0a2072995724290c5224bd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_2/afridiacritics_fon.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: fon +doc_to_text: "Given a sentence without diacritics, add the appropriate diacritics\ + \ to make it grammatically and semantically correct. \nSentence: {{text}}. Return\ + \ output sentence only" +include: afridiacritics_yaml +task: afridiacritics_fon_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_2/afridiacritics_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_2/afridiacritics_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c1ebac101f9a69722b21bbfe65ccd224f811e8d3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_2/afridiacritics_ibo.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: ibo +doc_to_text: "Given a sentence without diacritics, add the appropriate diacritics\ + \ to make it grammatically and semantically correct. \nSentence: {{text}}. Return\ + \ output sentence only" +include: afridiacritics_yaml +task: afridiacritics_ibo_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_2/afridiacritics_wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_2/afridiacritics_wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b8448d6ffce9f67252621cf6085fc575dace588e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_2/afridiacritics_wol.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: wol +doc_to_text: "Given a sentence without diacritics, add the appropriate diacritics\ + \ to make it grammatically and semantically correct. \nSentence: {{text}}. Return\ + \ output sentence only" +include: afridiacritics_yaml +task: afridiacritics_wol_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_2/afridiacritics_yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_2/afridiacritics_yaml new file mode 100644 index 0000000000000000000000000000000000000000..a0cc722d890f6a64939417f39f860532c4cd342b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_2/afridiacritics_yaml @@ -0,0 +1,25 @@ +tag: +- adr_tasks +- adr_prompt_2 +dataset_path: masakhane/diacritics-restoration +dataset_kwargs: {trust_remote_code: True} +doc_to_target: target +output_type: generate_until +fewshot_split: dev +test_split: test +training_split: train +metric_list: + - metric: bleu + aggregation: bleu + higher_is_better: true + - metric: chrf + aggregation: chrf + higher_is_better: true +generation_kwargs: + do_sample: false + until: + - '' + - + - <|im_end|> +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_2/afridiacritics_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_2/afridiacritics_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..eb95f5e234add5c178efc181bddb1fc87f9ce19d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_2/afridiacritics_yor.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: yor +doc_to_text: "Given a sentence without diacritics, add the appropriate diacritics\ + \ to make it grammatically and semantically correct. \nSentence: {{text}}. Return\ + \ output sentence only" +include: afridiacritics_yaml +task: afridiacritics_yor_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_4/afridiacritics_bbj.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_4/afridiacritics_bbj.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7a807b09ee3000f022374e31e61cdb2f2e091f0e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_4/afridiacritics_bbj.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: bbj +doc_to_text: 'You are a linguist specializing in diacritical marks for Gbomala. Add + the appropriate diacritics to this Gbomala sentence: {{text}}. Return output sentence + only' +include: afridiacritics_yaml +task: afridiacritics_bbj_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_4/afridiacritics_fon.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_4/afridiacritics_fon.yaml new file mode 100644 index 0000000000000000000000000000000000000000..11076e685ae5f6a4435a486d3f35db269ded8f51 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_4/afridiacritics_fon.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: fon +doc_to_text: 'You are a linguist specializing in diacritical marks for Fon. Add the + appropriate diacritics to this Fon sentence: {{text}}. Return output sentence only' +include: afridiacritics_yaml +task: afridiacritics_fon_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrobench-lite.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrobench-lite.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a23c050a2d09646492778e80dbc3a30dc281f580 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrobench-lite.yaml @@ -0,0 +1,15 @@ +group: afrobench_lite +task: + - afrimgsm_cot_tasks + - afrimmlu_tasks + - afrixnli_tasks + - belebele_tasks + - sib_tasks + - african_flores_tasks + - injongointent_tasks +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrobench.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrobench.yaml new file mode 100644 index 0000000000000000000000000000000000000000..52234bef5cde6b5695fa6510019bcf37502ddd40 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrobench.yaml @@ -0,0 +1,23 @@ +group: afrobench +task: +# - adr_tasks +## - afrihate_tasks #dataset not publicly available yet +# - afrimgsm_cot_tasks +# - afrixnli_tasks +# - afrobench_xqa_tasks +# - afrobench_sentiment_tasks + - afrobench_MT_tasks +# - afrobench_TC_tasks +# - afrobench_mmlu_tasks +# - injongointent_tasks +# - masakhaner_tasks +# - masakhapos_tasks +# - RC_tasks +# - uhura_arc_easy_tasks +# - xlsum_tasks +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_lin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_lin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..95cad4e2f44fc85e0dd9276c0024de0f1d2be617 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_lin.yaml @@ -0,0 +1,21 @@ +# Generated by utils.py +dataset_name: lin_Latn +doc_to_text: '{{flores_passage}} + + Based on the above passage, answer the following question: + + {{question.strip()}} + + Choices: + + A) {{mc_answer1}} + + B) {{mc_answer2}} + + C) {{mc_answer3}} + + D) {{mc_answer4}} + + Please provide the correct answer from the choices given:' +include: belebele +task: belebele_lin_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_por.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_por.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e8c06382b61eaec15b027533a5553c77d2974180 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_por.yaml @@ -0,0 +1,21 @@ +# Generated by utils.py +dataset_name: por_Latn +doc_to_text: '{{flores_passage}} + + Based on the above passage, answer the following question: + + {{question.strip()}} + + Choices: + + A) {{mc_answer1}} + + B) {{mc_answer2}} + + C) {{mc_answer3}} + + D) {{mc_answer4}} + + Please provide the correct answer from the choices given:' +include: belebele +task: belebele_por_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..afbdcad238a03bf388dd4ddb070f159cf41782ab --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_swa.yaml @@ -0,0 +1,21 @@ +# Generated by utils.py +dataset_name: swh_Latn +doc_to_text: '{{flores_passage}} + + Based on the above passage, answer the following question: + + {{question.strip()}} + + Choices: + + A) {{mc_answer1}} + + B) {{mc_answer2}} + + C) {{mc_answer3}} + + D) {{mc_answer4}} + + Please provide the correct answer from the choices given:' +include: belebele +task: belebele_swa_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_tsn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_tsn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a8f0a28d86b665bc913b1760dc378e8e4bf4146d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_tsn.yaml @@ -0,0 +1,21 @@ +# Generated by utils.py +dataset_name: tsn_Latn +doc_to_text: '{{flores_passage}} + + Based on the above passage, answer the following question: + + {{question.strip()}} + + Choices: + + A) {{mc_answer1}} + + B) {{mc_answer2}} + + C) {{mc_answer3}} + + D) {{mc_answer4}} + + Please provide the correct answer from the choices given:' +include: belebele +task: belebele_tsn_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_tso.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_tso.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4f1a87faa18a1af4729eca19d50b6a86bda83771 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_tso.yaml @@ -0,0 +1,21 @@ +# Generated by utils.py +dataset_name: tso_Latn +doc_to_text: '{{flores_passage}} + + Based on the above passage, answer the following question: + + {{question.strip()}} + + Choices: + + A) {{mc_answer1}} + + B) {{mc_answer2}} + + C) {{mc_answer3}} + + D) {{mc_answer4}} + + Please provide the correct answer from the choices given:' +include: belebele +task: belebele_tso_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0e0f6a629eea7632d7879943813fbd8964de5b8f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_wol.yaml @@ -0,0 +1,21 @@ +# Generated by utils.py +dataset_name: wol_Latn +doc_to_text: '{{flores_passage}} + + Based on the above passage, answer the following question: + + {{question.strip()}} + + Choices: + + A) {{mc_answer1}} + + B) {{mc_answer2}} + + C) {{mc_answer3}} + + D) {{mc_answer4}} + + Please provide the correct answer from the choices given:' +include: belebele +task: belebele_wol_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c3510a4d82e5975dde9272cd83e856780d7a3766 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_xho.yaml @@ -0,0 +1,21 @@ +# Generated by utils.py +dataset_name: xho_Latn +doc_to_text: '{{flores_passage}} + + Based on the above passage, answer the following question: + + {{question.strip()}} + + Choices: + + A) {{mc_answer1}} + + B) {{mc_answer2}} + + C) {{mc_answer3}} + + D) {{mc_answer4}} + + Please provide the correct answer from the choices given:' +include: belebele +task: belebele_xho_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..526e24ef92ecfbada52046a0d72e878ef939e86c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_yor.yaml @@ -0,0 +1,21 @@ +# Generated by utils.py +dataset_name: yor_Latn +doc_to_text: '{{flores_passage}} + + Based on the above passage, answer the following question: + + {{question.strip()}} + + Choices: + + A) {{mc_answer1}} + + B) {{mc_answer2}} + + C) {{mc_answer3}} + + D) {{mc_answer4}} + + Please provide the correct answer from the choices given:' +include: belebele +task: belebele_yor_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7472e5213b9a5a016818e89eaf901611423131b9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_zul.yaml @@ -0,0 +1,21 @@ +# Generated by utils.py +dataset_name: zul_Latn +doc_to_text: '{{flores_passage}} + + Based on the above passage, answer the following question: + + {{question.strip()}} + + Choices: + + A) {{mc_answer1}} + + B) {{mc_answer2}} + + C) {{mc_answer3}} + + D) {{mc_answer4}} + + Please provide the correct answer from the choices given:' +include: belebele +task: belebele_zul_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_afr.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_afr.yaml new file mode 100644 index 0000000000000000000000000000000000000000..01a724719757a2655800615e982a1ff1272dc438 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_afr.yaml @@ -0,0 +1,19 @@ +# Generated by utils.py +dataset_name: afr_Latn +doc_to_text: 'Read the passage: {{flores_passage}} + + Then answer the question: {{question.strip()}} + + Options: + + A. {{mc_answer1}} + + B. {{mc_answer2}} + + C. {{mc_answer3}} + + D. {{mc_answer4}} + + Please choose the correct option from the above list:' +include: belebele +task: belebele_afr_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f707d7c38153cd8304f7f02e331dc00858cb59c3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_amh.yaml @@ -0,0 +1,19 @@ +# Generated by utils.py +dataset_name: amh_Ethi +doc_to_text: 'Read the passage: {{flores_passage}} + + Then answer the question: {{question.strip()}} + + Options: + + A. {{mc_answer1}} + + B. {{mc_answer2}} + + C. {{mc_answer3}} + + D. {{mc_answer4}} + + Please choose the correct option from the above list:' +include: belebele +task: belebele_amh_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_ary.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_ary.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2cf68405132141da73c1c4c0085bffa47c6aab41 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_ary.yaml @@ -0,0 +1,19 @@ +# Generated by utils.py +dataset_name: ary_Arab +doc_to_text: 'Read the passage: {{flores_passage}} + + Then answer the question: {{question.strip()}} + + Options: + + A. {{mc_answer1}} + + B. {{mc_answer2}} + + C. {{mc_answer3}} + + D. {{mc_answer4}} + + Please choose the correct option from the above list:' +include: belebele +task: belebele_ary_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_eng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..62617bf1e3266c46897623097836dbc3337add03 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_eng.yaml @@ -0,0 +1,19 @@ +# Generated by utils.py +dataset_name: eng_Latn +doc_to_text: 'Read the passage: {{flores_passage}} + + Then answer the question: {{question.strip()}} + + Options: + + A. {{mc_answer1}} + + B. {{mc_answer2}} + + C. {{mc_answer3}} + + D. {{mc_answer4}} + + Please choose the correct option from the above list:' +include: belebele +task: belebele_eng_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_fra.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_fra.yaml new file mode 100644 index 0000000000000000000000000000000000000000..05131046414169281a6bdb6046cbfef3f461939f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_fra.yaml @@ -0,0 +1,19 @@ +# Generated by utils.py +dataset_name: fra_Latn +doc_to_text: 'Read the passage: {{flores_passage}} + + Then answer the question: {{question.strip()}} + + Options: + + A. {{mc_answer1}} + + B. {{mc_answer2}} + + C. {{mc_answer3}} + + D. {{mc_answer4}} + + Please choose the correct option from the above list:' +include: belebele +task: belebele_fra_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_fuv.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_fuv.yaml new file mode 100644 index 0000000000000000000000000000000000000000..35103b5c19223a0a11e6a595b3f28fa3635455e9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_fuv.yaml @@ -0,0 +1,19 @@ +# Generated by utils.py +dataset_name: fuv_Latn +doc_to_text: 'Read the passage: {{flores_passage}} + + Then answer the question: {{question.strip()}} + + Options: + + A. {{mc_answer1}} + + B. {{mc_answer2}} + + C. {{mc_answer3}} + + D. {{mc_answer4}} + + Please choose the correct option from the above list:' +include: belebele +task: belebele_fuv_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_gaz.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_gaz.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3822a5886fc0a179e95140208ba7751306c09619 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_gaz.yaml @@ -0,0 +1,19 @@ +# Generated by utils.py +dataset_name: gaz_Latn +doc_to_text: 'Read the passage: {{flores_passage}} + + Then answer the question: {{question.strip()}} + + Options: + + A. {{mc_answer1}} + + B. {{mc_answer2}} + + C. {{mc_answer3}} + + D. {{mc_answer4}} + + Please choose the correct option from the above list:' +include: belebele +task: belebele_gaz_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9a0a53114456c6223050d8a444052e2a15b3e2aa --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_hau.yaml @@ -0,0 +1,19 @@ +# Generated by utils.py +dataset_name: hau_Latn +doc_to_text: 'Read the passage: {{flores_passage}} + + Then answer the question: {{question.strip()}} + + Options: + + A. {{mc_answer1}} + + B. {{mc_answer2}} + + C. {{mc_answer3}} + + D. {{mc_answer4}} + + Please choose the correct option from the above list:' +include: belebele +task: belebele_hau_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f5a8e29bcd708791b528111da1b3301586825561 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_ibo.yaml @@ -0,0 +1,19 @@ +# Generated by utils.py +dataset_name: ibo_Latn +doc_to_text: 'Read the passage: {{flores_passage}} + + Then answer the question: {{question.strip()}} + + Options: + + A. {{mc_answer1}} + + B. {{mc_answer2}} + + C. {{mc_answer3}} + + D. {{mc_answer4}} + + Please choose the correct option from the above list:' +include: belebele +task: belebele_ibo_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_kea.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_kea.yaml new file mode 100644 index 0000000000000000000000000000000000000000..45fb47ad9854a75127428774a10dbd6eb12d83a9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_kea.yaml @@ -0,0 +1,19 @@ +# Generated by utils.py +dataset_name: kea_Latn +doc_to_text: 'Read the passage: {{flores_passage}} + + Then answer the question: {{question.strip()}} + + Options: + + A. {{mc_answer1}} + + B. {{mc_answer2}} + + C. {{mc_answer3}} + + D. {{mc_answer4}} + + Please choose the correct option from the above list:' +include: belebele +task: belebele_kea_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8bd9a07b8853165e8b5022a2d110cd450f2708ce --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_kin.yaml @@ -0,0 +1,19 @@ +# Generated by utils.py +dataset_name: kin_Latn +doc_to_text: 'Read the passage: {{flores_passage}} + + Then answer the question: {{question.strip()}} + + Options: + + A. {{mc_answer1}} + + B. {{mc_answer2}} + + C. {{mc_answer3}} + + D. {{mc_answer4}} + + Please choose the correct option from the above list:' +include: belebele +task: belebele_kin_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1b64c68ba1b3a8aa1e026cb191dcf96873285d40 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_lug.yaml @@ -0,0 +1,19 @@ +# Generated by utils.py +dataset_name: lug_Latn +doc_to_text: 'Read the passage: {{flores_passage}} + + Then answer the question: {{question.strip()}} + + Options: + + A. {{mc_answer1}} + + B. {{mc_answer2}} + + C. {{mc_answer3}} + + D. {{mc_answer4}} + + Please choose the correct option from the above list:' +include: belebele +task: belebele_lug_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_nya.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_nya.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c957760af620a7bf10726fb9611baa5758d1a03d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_nya.yaml @@ -0,0 +1,19 @@ +# Generated by utils.py +dataset_name: nya_Latn +doc_to_text: 'Read the passage: {{flores_passage}} + + Then answer the question: {{question.strip()}} + + Options: + + A. {{mc_answer1}} + + B. {{mc_answer2}} + + C. {{mc_answer3}} + + D. {{mc_answer4}} + + Please choose the correct option from the above list:' +include: belebele +task: belebele_nya_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_plt.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_plt.yaml new file mode 100644 index 0000000000000000000000000000000000000000..baad68ab37cf0a5c5d34f9c16a0fdfaa473e5968 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_plt.yaml @@ -0,0 +1,19 @@ +# Generated by utils.py +dataset_name: plt_Latn +doc_to_text: 'Read the passage: {{flores_passage}} + + Then answer the question: {{question.strip()}} + + Options: + + A. {{mc_answer1}} + + B. {{mc_answer2}} + + C. {{mc_answer3}} + + D. {{mc_answer4}} + + Please choose the correct option from the above list:' +include: belebele +task: belebele_plt_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_por.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_por.yaml new file mode 100644 index 0000000000000000000000000000000000000000..13b4e63948d0bc6ba9c886490a8585d2339bd357 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_por.yaml @@ -0,0 +1,19 @@ +# Generated by utils.py +dataset_name: por_Latn +doc_to_text: 'Read the passage: {{flores_passage}} + + Then answer the question: {{question.strip()}} + + Options: + + A. {{mc_answer1}} + + B. {{mc_answer2}} + + C. {{mc_answer3}} + + D. {{mc_answer4}} + + Please choose the correct option from the above list:' +include: belebele +task: belebele_por_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_sna.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_sna.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fd4fc080074beb21801196d5df19ea59e6928b4a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_sna.yaml @@ -0,0 +1,19 @@ +# Generated by utils.py +dataset_name: sna_Latn +doc_to_text: 'Read the passage: {{flores_passage}} + + Then answer the question: {{question.strip()}} + + Options: + + A. {{mc_answer1}} + + B. {{mc_answer2}} + + C. {{mc_answer3}} + + D. {{mc_answer4}} + + Please choose the correct option from the above list:' +include: belebele +task: belebele_sna_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_som.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_som.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3dfa40665cc2874333872503f880c4c373c03b52 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_som.yaml @@ -0,0 +1,19 @@ +# Generated by utils.py +dataset_name: som_Latn +doc_to_text: 'Read the passage: {{flores_passage}} + + Then answer the question: {{question.strip()}} + + Options: + + A. {{mc_answer1}} + + B. {{mc_answer2}} + + C. {{mc_answer3}} + + D. {{mc_answer4}} + + Please choose the correct option from the above list:' +include: belebele +task: belebele_som_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_sot.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_sot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c78c862a08e3e6e6968d9e3e039a6cd67c99e978 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_sot.yaml @@ -0,0 +1,19 @@ +# Generated by utils.py +dataset_name: sot_Latn +doc_to_text: 'Read the passage: {{flores_passage}} + + Then answer the question: {{question.strip()}} + + Options: + + A. {{mc_answer1}} + + B. {{mc_answer2}} + + C. {{mc_answer3}} + + D. {{mc_answer4}} + + Please choose the correct option from the above list:' +include: belebele +task: belebele_sot_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_tir.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_tir.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4ef9af2ade11b49af64195a00811be7bf69b34d7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_tir.yaml @@ -0,0 +1,19 @@ +# Generated by utils.py +dataset_name: tir_Ethi +doc_to_text: 'Read the passage: {{flores_passage}} + + Then answer the question: {{question.strip()}} + + Options: + + A. {{mc_answer1}} + + B. {{mc_answer2}} + + C. {{mc_answer3}} + + D. {{mc_answer4}} + + Please choose the correct option from the above list:' +include: belebele +task: belebele_tir_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_tsn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_tsn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0de5669b2a6031c7a5960bc174951d99cdc02502 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_tsn.yaml @@ -0,0 +1,19 @@ +# Generated by utils.py +dataset_name: tsn_Latn +doc_to_text: 'Read the passage: {{flores_passage}} + + Then answer the question: {{question.strip()}} + + Options: + + A. {{mc_answer1}} + + B. {{mc_answer2}} + + C. {{mc_answer3}} + + D. {{mc_answer4}} + + Please choose the correct option from the above list:' +include: belebele +task: belebele_tsn_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_tso.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_tso.yaml new file mode 100644 index 0000000000000000000000000000000000000000..92def0f429f0c7517c8904a8a2ae86cc2f534653 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_tso.yaml @@ -0,0 +1,19 @@ +# Generated by utils.py +dataset_name: tso_Latn +doc_to_text: 'Read the passage: {{flores_passage}} + + Then answer the question: {{question.strip()}} + + Options: + + A. {{mc_answer1}} + + B. {{mc_answer2}} + + C. {{mc_answer3}} + + D. {{mc_answer4}} + + Please choose the correct option from the above list:' +include: belebele +task: belebele_tso_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..10192b8a69a44faf3b155a16287d3da15b0c4c5f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_wol.yaml @@ -0,0 +1,19 @@ +# Generated by utils.py +dataset_name: wol_Latn +doc_to_text: 'Read the passage: {{flores_passage}} + + Then answer the question: {{question.strip()}} + + Options: + + A. {{mc_answer1}} + + B. {{mc_answer2}} + + C. {{mc_answer3}} + + D. {{mc_answer4}} + + Please choose the correct option from the above list:' +include: belebele +task: belebele_wol_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9ea12584e13ae4b4e19ac5b9c1fcc3b54ae0f950 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_xho.yaml @@ -0,0 +1,19 @@ +# Generated by utils.py +dataset_name: xho_Latn +doc_to_text: 'Read the passage: {{flores_passage}} + + Then answer the question: {{question.strip()}} + + Options: + + A. {{mc_answer1}} + + B. {{mc_answer2}} + + C. {{mc_answer3}} + + D. {{mc_answer4}} + + Please choose the correct option from the above list:' +include: belebele +task: belebele_xho_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c69e05cee5d26e91892e9303ad09f06856c254d1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_yor.yaml @@ -0,0 +1,19 @@ +# Generated by utils.py +dataset_name: yor_Latn +doc_to_text: 'Read the passage: {{flores_passage}} + + Then answer the question: {{question.strip()}} + + Options: + + A. {{mc_answer1}} + + B. {{mc_answer2}} + + C. {{mc_answer3}} + + D. {{mc_answer4}} + + Please choose the correct option from the above list:' +include: belebele +task: belebele_yor_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c3c6905f9803bba171b95b48c540f730d16158bb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_zul.yaml @@ -0,0 +1,19 @@ +# Generated by utils.py +dataset_name: zul_Latn +doc_to_text: 'Read the passage: {{flores_passage}} + + Then answer the question: {{question.strip()}} + + Options: + + A. {{mc_answer1}} + + B. {{mc_answer2}} + + C. {{mc_answer3}} + + D. {{mc_answer4}} + + Please choose the correct option from the above list:' +include: belebele +task: belebele_zul_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/README.md b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/README.md new file mode 100644 index 0000000000000000000000000000000000000000..ccf433a9f884576ef412148ea67e1a07c86bea30 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/README.md @@ -0,0 +1,31 @@ +# + +## Paper +Title: `The FLORES-200 Evaluation Benchmark for Low-Resource and Multilingual Machine Translation` + +Paper Link: https://arxiv.org/abs/2207.04672 + +HomePage: https://huggingface.co/datasets/facebook/flores + +### Citation + +``` +@article{nllb2022, + author = {NLLB Team, Marta R. Costa-jussà, James Cross, Onur Çelebi, Maha Elbayad, Kenneth Heafield, Kevin Heffernan, Elahe Kalbassi, Janice Lam, Daniel Licht, Jean Maillard, Anna Sun, Skyler Wang, Guillaume Wenzek, Al Youngblood, Bapi Akula, Loic Barrault, Gabriel Mejia Gonzalez, Prangthip Hansanti, John Hoffman, Semarley Jarrett, Kaushik Ram Sadagopan, Dirk Rowe, Shannon Spruit, Chau Tran, Pierre Andrews, Necip Fazil Ayan, Shruti Bhosale, Sergey Edunov, Angela Fan, Cynthia Gao, Vedanuj Goswami, Francisco Guzmán, Philipp Koehn, Alexandre Mourachko, Christophe Ropers, Safiyyah Saleem, Holger Schwenk, Jeff Wang}, + title = {No Language Left Behind: Scaling Human-Centered Machine Translation}, + year = {2022} +} + +@inproceedings{, + title={The FLORES-101 Evaluation Benchmark for Low-Resource and Multilingual Machine Translation}, + author={Goyal, Naman and Gao, Cynthia and Chaudhary, Vishrav and Chen, Peng-Jen and Wenzek, Guillaume and Ju, Da and Krishnan, Sanjana and Ranzato, Marc'Aurelio and Guzm\'{a}n, Francisco and Fan, Angela}, + year={2021} +} + +@inproceedings{, + title={Two New Evaluation Datasets for Low-Resource Machine Translation: Nepali-English and Sinhala-English}, + author={Guzm\'{a}n, Francisco and Chen, Peng-Jen and Ott, Myle and Pino, Juan and Lample, Guillaume and Koehn, Philipp and Chaudhary, Vishrav and Ranzato, Marc'Aurelio}, + journal={arXiv preprint arXiv:1902.01382}, + year={2019} +} +``` diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/gen_utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/gen_utils.py new file mode 100644 index 0000000000000000000000000000000000000000..37e22e13d6b024976e9198df78dfa7ae81845e8a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/gen_utils.py @@ -0,0 +1,202 @@ +import argparse +import os + +import yaml + + +class FunctionTag: + def __init__(self, value): + self.value = value + + +def prompt_func(mode, lang, lang_dict): + language_column_name = f"sentence_{lang}" + prompt_map = { + "prompt_1": f"{lang_dict[lang]}: {{{{{language_column_name}}}}} \nEnglish: ", + "prompt_1_reverse": f"English: {{{{sentence_eng_Latn}}}} \n{lang_dict[lang]}: ", + "prompt_2": f"You are a translation expert. Translate the following {lang_dict[lang]} sentences to English \n" + f"{lang_dict[lang]}: {{{{{language_column_name}}}}}\nEnglish: ", + "prompt_2_reverse": f"You are a translation expert. Translate the following English sentences to " + f"{lang_dict[lang]} " + "\nEnglish: {{sentence_eng_Latn}} " + f"\n{lang_dict[lang]}: ", + "prompt_3": f"As a {lang_dict[lang]} and English linguist, translate the following {lang_dict[lang]} sentences " + f"to English \n{lang_dict[lang]}: {{{{{language_column_name}}}}}\nEnglish: ", + "prompt_3_reverse": f"As a {lang_dict[lang]} and English linguist, translate the following English sentences to " + f"{lang_dict[lang]} " + "\nEnglish: {{sentence_eng_Latn}} " + f"\n{lang_dict[lang]}: ", + } + return prompt_map[mode] + + +def gen_lang_yamls(output_dir: str, overwrite: bool, mode: str, reverse: bool) -> None: + """ + Generate a yaml file for each language. + + :param output_dir: The directory to output the files to. + :param overwrite: Whether to overwrite files if they already exist. + """ + err = [] + languages = { + "ace_Latn": "Acehnese (Latin script)", + "ace_Arab": "Acehnese (Arabic script)", + "acq_Arab": "Ta’izzi-Adeni Arabic", + "aeb_Arab": "Tunisian Arabic", + "afr_Latn": "Afrikaans", + "aka_Latn": "Akan", + "amh_Ethi": "Amharic", + "ary_Arab": "Moroccan Arabic", + "arz_Arab": "Egyptian Arabic", + "bam_Latn": "Bambara", + "ban_Latn": "Balinese", + "bem_Latn": "Bemba", + "cjk_Latn": "Chokwe", + "dik_Latn": "Southwestern Dinka", + "dyu_Latn": "Dyula", + "ewe_Latn": "Ewe", + "fon_Latn": "Fon", + "fra_Latn": "French", + "fuv_Latn": "Nigerian Fulfulde", + "hau_Latn": "Hausa", + "ibo_Latn": "Igbo", + "kab_Latn": "Kabyle", + "kam_Latn": "Kamba", + "knc_Arab": "Central Kanuri (Arabic script)", + "knc_Latn": "Central Kanuri (Latin script)", + "kbp_Latn": "Kabiyè", + "kea_Latn": "Kabuverdianu", + "kik_Latn": "Kikuyu", + "kin_Latn": "Kinyarwanda", + "kmb_Latn": "Kimbundu", + "kon_Latn": "Kikongo", + "lin_Latn": "Lingala", + "lua_Latn": "Luba-Kasai", + "lug_Latn": "Luganda", + "luo_Latn": "Luo", + "plt_Latn": "Plateau Malagasy", + "mos_Latn": "Mossi", + "nso_Latn": "Northern Sotho", + "nus_Latn": "Nuer", + "nya_Latn": "Nyanja", + "gaz_Latn": "Oromo", + "run_Latn": "Rundi", + "sag_Latn": "Sango", + "sna_Latn": "Shona", + "som_Latn": "Somali", + "sot_Latn": "Southern Sotho", + "ssw_Latn": "Swati", + "sun_Latn": "Sundanese", + "swh_Latn": "Swahili", + "tir_Ethi": "Tigrinya", + "taq_Latn": "Tamasheq", + "taq_Tfng": "Tamasheq (Tifinagh script)", + "tsn_Latn": "Setswana", + "tso_Latn": "Tsonga", + "tum_Latn": "Tumbuka", + "twi_Latn": "Twi", + "tzm_Tfng": "Central Atlas Tamazight", + "umb_Latn": "Umbundu", + "wol_Latn": "Wolof", + "xho_Latn": "Xhosa", + "yor_Latn": "Yoruba", + "zul_Latn": "Zulu", + } + + for lang in languages.keys(): + try: + if not reverse: + file_name = f"flores_{lang}-eng_Latn.yaml" + task_name = f"flores_{lang}-eng_Latn_{mode}" + yaml_template = "flores" + yaml_details = { + "include": yaml_template, + "task": task_name, + "dataset_name": f"{lang}-eng_Latn", + "doc_to_target": "sentence_eng_Latn", + "doc_to_text": prompt_func(mode, lang, languages), + } + os.makedirs(f"{output_dir}/{mode}/african-english", exist_ok=True) + with open( + f"{output_dir}/{mode}/african-english/{file_name}", + "w" if overwrite else "x", + encoding="utf8", + ) as f: + f.write("# Generated by utils.py\n") + yaml.dump( + yaml_details, + f, + allow_unicode=True, + ) + else: + file_name = f"flores_eng_Latn-{lang}.yaml" + task_name = f"flores_eng_Latn-{lang}_{mode}" + yaml_template = "flores" + # mode_reverse = f"{mode}_reverse" + yaml_details = { + "include": yaml_template, + "task": task_name, + "dataset_name": f"eng_Latn-{lang}", + "doc_to_target": f"sentence_{lang}", + "doc_to_text": prompt_func(f"{mode}_reverse", lang, languages), + } + os.makedirs(f"{output_dir}/{mode}/english-african", exist_ok=True) + with open( + f"{output_dir}/{mode}/english-african/{file_name}", + "w" if overwrite else "x", + encoding="utf8", + ) as f: + f.write("# Generated by utils.py\n") + yaml.dump( + yaml_details, + f, + allow_unicode=True, + ) + except FileExistsError: + err.append(file_name) + + if len(err) > 0: + raise FileExistsError( + "Files were not created because they already exist (use --overwrite flag):" + f" {', '.join(err)}" + ) + + +def main() -> None: + """Parse CLI args and generate language-specific yaml files.""" + parser = argparse.ArgumentParser() + parser.add_argument( + "--overwrite", + default=True, + action="store_true", + help="Overwrite files if they already exist", + ) + parser.add_argument( + "--output-dir", + default="./", + help="Directory to write yaml files to", + ) + parser.add_argument( + "--mode", + default="prompt_1", + choices=["prompt_1", "prompt_2", "prompt_3"], + help="Prompt number", + ) + parser.add_argument( + "--reverse", + default=True, + choices=[True, False], + help="Reverse the translation direction", + ) + args = parser.parse_args() + + gen_lang_yamls( + output_dir=args.output_dir, + overwrite=args.overwrite, + mode=args.mode, + reverse=args.reverse, + ) + + +if __name__ == "__main__": + main() diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_acq_Arab-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_acq_Arab-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3634e7a66c7b11be9a0450f6f5ab953707897eb1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_acq_Arab-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: acq_Arab-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Ta’izzi-Adeni Arabic: {{sentence_acq_Arab}} \nEnglish: " +include: flores +task: flores_acq_Arab-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_aeb_Arab-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_aeb_Arab-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..53636d7c01d92e87b80da4bd6656b7740d0f11a6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_aeb_Arab-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: aeb_Arab-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Tunisian Arabic: {{sentence_aeb_Arab}} \nEnglish: " +include: flores +task: flores_aeb_Arab-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_afr_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_afr_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2ac14a0c04b362f925e578a30a9eb615a6bc1fed --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_afr_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: afr_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Afrikaans: {{sentence_afr_Latn}} \nEnglish: " +include: flores +task: flores_afr_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_aka_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_aka_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c3caf192676374f70a707c77d84bb6eac1deacb5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_aka_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: aka_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Akan: {{sentence_aka_Latn}} \nEnglish: " +include: flores +task: flores_aka_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_amh_Ethi-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_amh_Ethi-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6c0be0828a5110df25911b503c0db29b2fe6dbb3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_amh_Ethi-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: amh_Ethi-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Amharic: {{sentence_amh_Ethi}} \nEnglish: " +include: flores +task: flores_amh_Ethi-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_ary_Arab-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_ary_Arab-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8bcd452d6e86cf669bcbc97b7216d72dbeb37ffd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_ary_Arab-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: ary_Arab-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Moroccan Arabic: {{sentence_ary_Arab}} \nEnglish: " +include: flores +task: flores_ary_Arab-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_arz_Arab-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_arz_Arab-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..72552bab1ce9b04feb4619a28e4a424b1bdb99d3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_arz_Arab-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: arz_Arab-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Egyptian Arabic: {{sentence_arz_Arab}} \nEnglish: " +include: flores +task: flores_arz_Arab-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_bam_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_bam_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..14e8a1c74fb7f9bcdd47703121bc127420d5cf3a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_bam_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: bam_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Bambara: {{sentence_bam_Latn}} \nEnglish: " +include: flores +task: flores_bam_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_ban_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_ban_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..54a582446ec44263f7020676a7e5fa3eee88e780 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_ban_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: ban_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Balinese: {{sentence_ban_Latn}} \nEnglish: " +include: flores +task: flores_ban_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_bem_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_bem_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..53bbe221d7ec3a98f4eedeb9fdcfd49a2d872198 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_bem_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: bem_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Bemba: {{sentence_bem_Latn}} \nEnglish: " +include: flores +task: flores_bem_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_cjk_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_cjk_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..63994d04d0dfc9ca7d8418835dcd47abf79d5031 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_cjk_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: cjk_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Chokwe: {{sentence_cjk_Latn}} \nEnglish: " +include: flores +task: flores_cjk_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_dik_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_dik_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fd9022b53f0325804b07bf5fb8c222a37c5eccde --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_dik_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: dik_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Southwestern Dinka: {{sentence_dik_Latn}} \nEnglish: " +include: flores +task: flores_dik_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_dyu_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_dyu_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e25e23d89d7090ec09c68af8c83705dd6a43d7d7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_dyu_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: dyu_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Dyula: {{sentence_dyu_Latn}} \nEnglish: " +include: flores +task: flores_dyu_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_ewe_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_ewe_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fffa31fcd93a02581b8e70e20d2b2fd84803365c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_ewe_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: ewe_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Ewe: {{sentence_ewe_Latn}} \nEnglish: " +include: flores +task: flores_ewe_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_fon_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_fon_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..70c9bfbe0f59666124b03c55432d4981472da9d8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_fon_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: fon_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Fon: {{sentence_fon_Latn}} \nEnglish: " +include: flores +task: flores_fon_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_fra_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_fra_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2c515a8f6adff914e6c237a1d637d8a589bc976f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_fra_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: fra_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "French: {{sentence_fra_Latn}} \nEnglish: " +include: flores +task: flores_fra_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_fuv_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_fuv_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4a162567753f9b6ee34d52f5ac06233b54973eac --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_fuv_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: fuv_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Nigerian Fulfulde: {{sentence_fuv_Latn}} \nEnglish: " +include: flores +task: flores_fuv_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_gaz_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_gaz_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ec443459d6b9ac698106c6dc2c500d2b897557f1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_gaz_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: gaz_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Oromo: {{sentence_gaz_Latn}} \nEnglish: " +include: flores +task: flores_gaz_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_hau_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_hau_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8d518b5122fa8ebf09188fd1eca8f6f5e2e23983 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_hau_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: hau_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Hausa: {{sentence_hau_Latn}} \nEnglish: " +include: flores +task: flores_hau_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_ibo_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_ibo_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9c121ae73d95a5acffa72c27af04c9c1b16a7b43 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_ibo_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: ibo_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Igbo: {{sentence_ibo_Latn}} \nEnglish: " +include: flores +task: flores_ibo_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_kab_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_kab_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..42c625488a60ed854f9769636ddc208d8d7d2e0c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_kab_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: kab_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Kabyle: {{sentence_kab_Latn}} \nEnglish: " +include: flores +task: flores_kab_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_kam_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_kam_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d7d10cc570d7c338e7928a36be4a3026cffaf159 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_kam_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: kam_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Kamba: {{sentence_kam_Latn}} \nEnglish: " +include: flores +task: flores_kam_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_kbp_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_kbp_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..43cc5e32a272d14bc8549d28f6f8784ab1e968dd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_kbp_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: kbp_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Kabiyè: {{sentence_kbp_Latn}} \nEnglish: " +include: flores +task: flores_kbp_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_kea_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_kea_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4c894681ef571f72d6edfa953a9d0aeaafb5dcc9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_kea_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: kea_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Kabuverdianu: {{sentence_kea_Latn}} \nEnglish: " +include: flores +task: flores_kea_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_kik_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_kik_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dbdff8e215e247fbc4b0061154f3c66903aad80c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_kik_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: kik_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Kikuyu: {{sentence_kik_Latn}} \nEnglish: " +include: flores +task: flores_kik_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_kin_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_kin_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b11194a98eacd84551995aa1506993a9c8a52bf6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_kin_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: kin_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Kinyarwanda: {{sentence_kin_Latn}} \nEnglish: " +include: flores +task: flores_kin_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_kmb_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_kmb_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..258b847d28294196b7c4d7455320e24d0ad2a59a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_kmb_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: kmb_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Kimbundu: {{sentence_kmb_Latn}} \nEnglish: " +include: flores +task: flores_kmb_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_knc_Arab-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_knc_Arab-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..642dfc6f891572f65037299b8f3a8381f51f0421 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_knc_Arab-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: knc_Arab-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Central Kanuri (Arabic script): {{sentence_knc_Arab}} \nEnglish: " +include: flores +task: flores_knc_Arab-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_knc_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_knc_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7f904da712bda335eab0ccc0e7036b8190caf93e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_knc_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: knc_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Central Kanuri (Latin script): {{sentence_knc_Latn}} \nEnglish: " +include: flores +task: flores_knc_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_kon_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_kon_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..54fce1f8da44e9b2c889e71e8a8d9f5eea4b3ef7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_kon_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: kon_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Kikongo: {{sentence_kon_Latn}} \nEnglish: " +include: flores +task: flores_kon_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_lin_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_lin_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..41494a7263855a44fd217ac6d7cee38e714a8597 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_lin_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: lin_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Lingala: {{sentence_lin_Latn}} \nEnglish: " +include: flores +task: flores_lin_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_lua_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_lua_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9d54350a45f7838492555c4268e552dc609f0e12 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_lua_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: lua_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Luba-Kasai: {{sentence_lua_Latn}} \nEnglish: " +include: flores +task: flores_lua_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_lug_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_lug_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..35d8e31b1331a8e478bc6960c262a8d5eb5630df --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_lug_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: lug_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Luganda: {{sentence_lug_Latn}} \nEnglish: " +include: flores +task: flores_lug_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_luo_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_luo_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7a22ec7db9d0cf3d3497ab0367c32a2aef602513 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_luo_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: luo_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Luo: {{sentence_luo_Latn}} \nEnglish: " +include: flores +task: flores_luo_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_mos_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_mos_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4a4c1009c46290faafcb15930cb98217612d5c14 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_mos_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: mos_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Mossi: {{sentence_mos_Latn}} \nEnglish: " +include: flores +task: flores_mos_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_nso_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_nso_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2409753c8a86d873fc33b68cbaba493b154fd947 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_nso_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: nso_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Northern Sotho: {{sentence_nso_Latn}} \nEnglish: " +include: flores +task: flores_nso_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_nus_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_nus_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f77380957e4c1b61cd9a277e8e2d831fa7d9a0da --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_nus_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: nus_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Nuer: {{sentence_nus_Latn}} \nEnglish: " +include: flores +task: flores_nus_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_nya_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_nya_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..def5625dd7a4b6dd4288a796681fe6fbfc40c6ac --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_nya_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: nya_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Nyanja: {{sentence_nya_Latn}} \nEnglish: " +include: flores +task: flores_nya_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_plt_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_plt_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f877a307254dfe00e645ea544bc1a7fb64411162 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_plt_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: plt_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Plateau Malagasy: {{sentence_plt_Latn}} \nEnglish: " +include: flores +task: flores_plt_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_run_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_run_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e00eb85718ffff0eb4fd84a1ce50fc4ff92c9988 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_run_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: run_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Rundi: {{sentence_run_Latn}} \nEnglish: " +include: flores +task: flores_run_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_sag_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_sag_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e7f43c6b6cf91d52311d4f0981086afcd833d8b3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_sag_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: sag_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Sango: {{sentence_sag_Latn}} \nEnglish: " +include: flores +task: flores_sag_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_som_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_som_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f625c559f3c1dd1b0490d42c29515dfeaef28d68 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_som_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: som_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Somali: {{sentence_som_Latn}} \nEnglish: " +include: flores +task: flores_som_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_sot_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_sot_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..11653e6059e51d71b48e722abd1c519ddd956d00 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_sot_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: sot_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Southern Sotho: {{sentence_sot_Latn}} \nEnglish: " +include: flores +task: flores_sot_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores new file mode 100644 index 0000000000000000000000000000000000000000..e6f4d051431159f4360115226ea58dec2487c0c2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores @@ -0,0 +1,27 @@ +tag: +- african_flores_tasks +- flores_eng-afr +- flores_eng-afr_prompt_1 +- afrobench_MT_tasks +dataset_path: facebook/flores +dataset_kwargs: {trust_remote_code: True} +output_type: generate_until +validation_split: dev +fewshot_split: dev +test_split: devtest +metric_list: + - metric: bleu + aggregation: bleu + higher_is_better: true + - metric: chrf + aggregation: chrf + higher_is_better: true +generation_kwargs: + until: + - "**" + - + do_sample: false + temperature: 0.0 +repeats: 1 +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-ace_Arab.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-ace_Arab.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a9da06483bc1e3f19e10636cdf1509ad899832ca --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-ace_Arab.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-ace_Arab +doc_to_target: sentence_ace_Arab +doc_to_text: "English: {{sentence_eng_Latn}} \nAcehnese (Arabic script): " +include: flores +task: flores_eng_Latn-ace_Arab_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-ace_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-ace_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d2ed60660bd3f565484d16440ce9fb2d82f6a555 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-ace_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-ace_Latn +doc_to_target: sentence_ace_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nAcehnese (Latin script): " +include: flores +task: flores_eng_Latn-ace_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-acq_Arab.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-acq_Arab.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e61bb2472b427de012ce3d47122906df99f14089 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-acq_Arab.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-acq_Arab +doc_to_target: sentence_acq_Arab +doc_to_text: "English: {{sentence_eng_Latn}} \nTa’izzi-Adeni Arabic: " +include: flores +task: flores_eng_Latn-acq_Arab_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-aeb_Arab.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-aeb_Arab.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d59000626aa9ef7f6cfcd6bc6a315cbb25a90142 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-aeb_Arab.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-aeb_Arab +doc_to_target: sentence_aeb_Arab +doc_to_text: "English: {{sentence_eng_Latn}} \nTunisian Arabic: " +include: flores +task: flores_eng_Latn-aeb_Arab_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-afr_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-afr_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3b4c4d46b432d78f6e9947dbcd26868885560c53 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-afr_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-afr_Latn +doc_to_target: sentence_afr_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nAfrikaans: " +include: flores +task: flores_eng_Latn-afr_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-aka_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-aka_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d66a637f75d19c72d3846819d249f9a67989e04c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-aka_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-aka_Latn +doc_to_target: sentence_aka_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nAkan: " +include: flores +task: flores_eng_Latn-aka_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-amh_Ethi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-amh_Ethi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e648d33270ae5944609c8ce50c2dd3e92bbfeb97 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-amh_Ethi.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-amh_Ethi +doc_to_target: sentence_amh_Ethi +doc_to_text: "English: {{sentence_eng_Latn}} \nAmharic: " +include: flores +task: flores_eng_Latn-amh_Ethi_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-ary_Arab.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-ary_Arab.yaml new file mode 100644 index 0000000000000000000000000000000000000000..54f9a2ad67ac390ba4cc4a7a6db6a1d2e5061a54 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-ary_Arab.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-ary_Arab +doc_to_target: sentence_ary_Arab +doc_to_text: "English: {{sentence_eng_Latn}} \nMoroccan Arabic: " +include: flores +task: flores_eng_Latn-ary_Arab_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-arz_Arab.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-arz_Arab.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a42fa079b4501f6402eebd3241c26f14b1e5af6e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-arz_Arab.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-arz_Arab +doc_to_target: sentence_arz_Arab +doc_to_text: "English: {{sentence_eng_Latn}} \nEgyptian Arabic: " +include: flores +task: flores_eng_Latn-arz_Arab_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-bam_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-bam_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4c85b7db9394d3c309b3e5c5b196a0e5451c4d0b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-bam_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-bam_Latn +doc_to_target: sentence_bam_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nBambara: " +include: flores +task: flores_eng_Latn-bam_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-ban_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-ban_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f43a4b71131da9cf555964b79a6258ce7f36c2ef --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-ban_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-ban_Latn +doc_to_target: sentence_ban_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nBalinese: " +include: flores +task: flores_eng_Latn-ban_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-dik_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-dik_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..36dea9d3a9dc3371a576315540f439af9e38b4e5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-dik_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-dik_Latn +doc_to_target: sentence_dik_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nSouthwestern Dinka: " +include: flores +task: flores_eng_Latn-dik_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-dyu_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-dyu_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c32be8ac93c7cecfbc171eb898232c7296cf6886 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-dyu_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-dyu_Latn +doc_to_target: sentence_dyu_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nDyula: " +include: flores +task: flores_eng_Latn-dyu_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-ewe_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-ewe_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0a71b4556a077b260bfb340a3c0c289ae79ac88b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-ewe_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-ewe_Latn +doc_to_target: sentence_ewe_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nEwe: " +include: flores +task: flores_eng_Latn-ewe_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-fon_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-fon_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1000e13ad1f6864f002c741b8074d06073cb3dc8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-fon_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-fon_Latn +doc_to_target: sentence_fon_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nFon: " +include: flores +task: flores_eng_Latn-fon_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-fuv_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-fuv_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8855378737fa985bf35a840c78f81d34f8542305 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-fuv_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-fuv_Latn +doc_to_target: sentence_fuv_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nNigerian Fulfulde: " +include: flores +task: flores_eng_Latn-fuv_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-ibo_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-ibo_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ebf8f517c3e96d64716db741688161d520bd04a5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-ibo_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-ibo_Latn +doc_to_target: sentence_ibo_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nIgbo: " +include: flores +task: flores_eng_Latn-ibo_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-kab_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-kab_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fd22cb7de77e624bea297d7011aa18aab3408b10 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-kab_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-kab_Latn +doc_to_target: sentence_kab_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nKabyle: " +include: flores +task: flores_eng_Latn-kab_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-kbp_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-kbp_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9cc1afd5e9af72a8a7f73a6789cb2dc0af1e9c39 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-kbp_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-kbp_Latn +doc_to_target: sentence_kbp_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nKabiyè: " +include: flores +task: flores_eng_Latn-kbp_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-kea_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-kea_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..55d3e7767c8533eb9f0f94c37a33fdb628c2b27a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-kea_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-kea_Latn +doc_to_target: sentence_kea_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nKabuverdianu: " +include: flores +task: flores_eng_Latn-kea_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-nus_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-nus_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7ae375058b9393357df2aca5f52c69d6b6fde744 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-nus_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-nus_Latn +doc_to_target: sentence_nus_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Nuer \nEnglish: {{sentence_eng_Latn}} \nNuer: " +include: flores +task: flores_eng_Latn-nus_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-plt_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-plt_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..faa85197438e9ff31744de7683957736b3ad34bc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-plt_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-plt_Latn +doc_to_target: sentence_plt_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Plateau Malagasy \nEnglish: {{sentence_eng_Latn}} \nPlateau Malagasy: " +include: flores +task: flores_eng_Latn-plt_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-run_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-run_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8b670e3f7146cec72c15c9be17ad0df6b30a1a4b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-run_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-run_Latn +doc_to_target: sentence_run_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Rundi \nEnglish: {{sentence_eng_Latn}} \nRundi: " +include: flores +task: flores_eng_Latn-run_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-som_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-som_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f87466dc875c1b2402ccd195f164949c94aa3e5e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-som_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-som_Latn +doc_to_target: sentence_som_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Somali \nEnglish: {{sentence_eng_Latn}} \nSomali: " +include: flores +task: flores_eng_Latn-som_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-sun_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-sun_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9f51ced5e6ce5353da73945815adbaab9e9c0d94 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-sun_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-sun_Latn +doc_to_target: sentence_sun_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Sundanese \nEnglish: {{sentence_eng_Latn}} \nSundanese: " +include: flores +task: flores_eng_Latn-sun_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-swh_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-swh_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1558af98e0011ddf66f5ec63bcde425414a539a2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-swh_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-swh_Latn +doc_to_target: sentence_swh_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Swahili \nEnglish: {{sentence_eng_Latn}} \nSwahili: " +include: flores +task: flores_eng_Latn-swh_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-taq_Tfng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-taq_Tfng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b69f1dbd4dd96f81e055b758bfc103a81f7c116c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-taq_Tfng.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: eng_Latn-taq_Tfng +doc_to_target: sentence_taq_Tfng +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Tamasheq (Tifinagh script) \nEnglish: {{sentence_eng_Latn}} \nTamasheq (Tifinagh\ + \ script): " +include: flores +task: flores_eng_Latn-taq_Tfng_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-tir_Ethi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-tir_Ethi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4340591d8e397caa4525832e1225b364574c4664 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-tir_Ethi.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-tir_Ethi +doc_to_target: sentence_tir_Ethi +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Tigrinya \nEnglish: {{sentence_eng_Latn}} \nTigrinya: " +include: flores +task: flores_eng_Latn-tir_Ethi_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-tsn_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-tsn_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1e592366ebb36a4c621f05e31326fbf36125025b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-tsn_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-tsn_Latn +doc_to_target: sentence_tsn_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Setswana \nEnglish: {{sentence_eng_Latn}} \nSetswana: " +include: flores +task: flores_eng_Latn-tsn_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-tum_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-tum_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1accaeaf4bd9cbbfb5c46c6341a3bb81663767be --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-tum_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-tum_Latn +doc_to_target: sentence_tum_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Tumbuka \nEnglish: {{sentence_eng_Latn}} \nTumbuka: " +include: flores +task: flores_eng_Latn-tum_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-twi_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-twi_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4a45df82e6c60141396deafa19cd7882b2edb689 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-twi_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-twi_Latn +doc_to_target: sentence_twi_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Twi \nEnglish: {{sentence_eng_Latn}} \nTwi: " +include: flores +task: flores_eng_Latn-twi_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-tzm_Tfng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-tzm_Tfng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6a3faa15d24df79d839a226cc374ace410133a12 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-tzm_Tfng.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-tzm_Tfng +doc_to_target: sentence_tzm_Tfng +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Central Atlas Tamazight \nEnglish: {{sentence_eng_Latn}} \nCentral Atlas Tamazight: " +include: flores +task: flores_eng_Latn-tzm_Tfng_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-umb_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-umb_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3f21c6fe1939f288d56b6bd1229ce6755babb807 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-umb_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-umb_Latn +doc_to_target: sentence_umb_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Umbundu \nEnglish: {{sentence_eng_Latn}} \nUmbundu: " +include: flores +task: flores_eng_Latn-umb_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-wol_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-wol_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..263ded277f0e1a596eac0bf1af5ab1858cf6cd42 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-wol_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-wol_Latn +doc_to_target: sentence_wol_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Wolof \nEnglish: {{sentence_eng_Latn}} \nWolof: " +include: flores +task: flores_eng_Latn-wol_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-xho_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-xho_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a92e46f996b3dfacb06bd5d8d589d43763800e1a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-xho_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-xho_Latn +doc_to_target: sentence_xho_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Xhosa \nEnglish: {{sentence_eng_Latn}} \nXhosa: " +include: flores +task: flores_eng_Latn-xho_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-yor_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-yor_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..80ec895c70fc2f09261ec6df48f2e6bc9755f479 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-yor_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-yor_Latn +doc_to_target: sentence_yor_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Yoruba \nEnglish: {{sentence_eng_Latn}} \nYoruba: " +include: flores +task: flores_eng_Latn-yor_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-zul_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-zul_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..593cdfe3c7c6878bef06e9af476476fdcbbfdfd6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-zul_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-zul_Latn +doc_to_target: sentence_zul_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Zulu \nEnglish: {{sentence_eng_Latn}} \nZulu: " +include: flores +task: flores_eng_Latn-zul_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_kik_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_kik_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..14a6be5dfd5c86c44c8f3acb23af0f36d5445ead --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_kik_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: kik_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Kikuyu and English linguist, translate the following Kikuyu sentences\ + \ to English \nKikuyu: {{sentence_kik_Latn}}\nEnglish: " +include: flores +task: flores_kik_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_kin_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_kin_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8bb14aedf4e95de93a3d4da32a07b19bf854b697 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_kin_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: kin_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Kinyarwanda and English linguist, translate the following Kinyarwanda\ + \ sentences to English \nKinyarwanda: {{sentence_kin_Latn}}\nEnglish: " +include: flores +task: flores_kin_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_knc_Arab-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_knc_Arab-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c8c7f8095e11d407d75360ee5e4794e45fc17eeb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_knc_Arab-eng_Latn.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: knc_Arab-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Central Kanuri (Arabic script) and English linguist, translate\ + \ the following Central Kanuri (Arabic script) sentences to English \nCentral Kanuri\ + \ (Arabic script): {{sentence_knc_Arab}}\nEnglish: " +include: flores +task: flores_knc_Arab-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_knc_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_knc_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9621de73d33ea37feeb3a4face5dbe50812bf9ca --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_knc_Latn-eng_Latn.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: knc_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Central Kanuri (Latin script) and English linguist, translate the\ + \ following Central Kanuri (Latin script) sentences to English \nCentral Kanuri\ + \ (Latin script): {{sentence_knc_Latn}}\nEnglish: " +include: flores +task: flores_knc_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_lin_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_lin_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ea7e736d949726bf2296091e4309a12d493dbe4f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_lin_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: lin_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Lingala and English linguist, translate the following Lingala sentences\ + \ to English \nLingala: {{sentence_lin_Latn}}\nEnglish: " +include: flores +task: flores_lin_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_lua_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_lua_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..327f014489f7502ac062537c72d4846a7099b64d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_lua_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: lua_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Luba-Kasai and English linguist, translate the following Luba-Kasai\ + \ sentences to English \nLuba-Kasai: {{sentence_lua_Latn}}\nEnglish: " +include: flores +task: flores_lua_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_lug_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_lug_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9bfa92fa280f98278f7735634231cb99d44bc71f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_lug_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: lug_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Luganda and English linguist, translate the following Luganda sentences\ + \ to English \nLuganda: {{sentence_lug_Latn}}\nEnglish: " +include: flores +task: flores_lug_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_luo_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_luo_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a66fded383a914454aed5e904278e60bf85d1e62 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_luo_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: luo_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Luo and English linguist, translate the following Luo sentences\ + \ to English \nLuo: {{sentence_luo_Latn}}\nEnglish: " +include: flores +task: flores_luo_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_mos_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_mos_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e428853bf22859e37bbd10dcd4e113188b7519ef --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_mos_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: mos_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Mossi and English linguist, translate the following Mossi sentences\ + \ to English \nMossi: {{sentence_mos_Latn}}\nEnglish: " +include: flores +task: flores_mos_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_nso_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_nso_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..054aa409b729cc70d70b1d53a49945f92096f4b9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_nso_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: nso_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Northern Sotho and English linguist, translate the following Northern\ + \ Sotho sentences to English \nNorthern Sotho: {{sentence_nso_Latn}}\nEnglish: " +include: flores +task: flores_nso_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_nus_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_nus_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a3e0d1e3ac8ff35a2b0c89f37cb2697a6d15299a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_nus_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: nus_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Nuer and English linguist, translate the following Nuer sentences\ + \ to English \nNuer: {{sentence_nus_Latn}}\nEnglish: " +include: flores +task: flores_nus_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_nya_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_nya_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e23c57c6807991e1b509c0a03a7c93a23eac3015 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_nya_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: nya_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Nyanja and English linguist, translate the following Nyanja sentences\ + \ to English \nNyanja: {{sentence_nya_Latn}}\nEnglish: " +include: flores +task: flores_nya_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_plt_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_plt_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3ddfd864c3830d4921e1fcda79155c733809b305 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_plt_Latn-eng_Latn.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: plt_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Plateau Malagasy and English linguist, translate the following\ + \ Plateau Malagasy sentences to English \nPlateau Malagasy: {{sentence_plt_Latn}}\n\ + English: " +include: flores +task: flores_plt_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_run_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_run_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..64a82f716b71950711e6b055a6e30f45356a082c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_run_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: run_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Rundi and English linguist, translate the following Rundi sentences\ + \ to English \nRundi: {{sentence_run_Latn}}\nEnglish: " +include: flores +task: flores_run_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_sag_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_sag_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..48408f94054fee78fe7c3be6460de563e9e60f0a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_sag_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: sag_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Sango and English linguist, translate the following Sango sentences\ + \ to English \nSango: {{sentence_sag_Latn}}\nEnglish: " +include: flores +task: flores_sag_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_sna_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_sna_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ff1626419b69a8e94349bfd63093ad33914bbcec --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_sna_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: sna_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Shona and English linguist, translate the following Shona sentences\ + \ to English \nShona: {{sentence_sna_Latn}}\nEnglish: " +include: flores +task: flores_sna_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_som_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_som_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7e27e2a5b3d1754f4c47755a1b92c7ac95938e67 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_som_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: som_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Somali and English linguist, translate the following Somali sentences\ + \ to English \nSomali: {{sentence_som_Latn}}\nEnglish: " +include: flores +task: flores_som_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_sot_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_sot_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cc70b6f62b317e27cd8c00d95d89e103945028dc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_sot_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: sot_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Southern Sotho and English linguist, translate the following Southern\ + \ Sotho sentences to English \nSouthern Sotho: {{sentence_sot_Latn}}\nEnglish: " +include: flores +task: flores_sot_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_ssw_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_ssw_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0cd61ae8e8ceb6b9a0f985457f26d25961272036 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_ssw_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: ssw_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Swati and English linguist, translate the following Swati sentences\ + \ to English \nSwati: {{sentence_ssw_Latn}}\nEnglish: " +include: flores +task: flores_ssw_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_sun_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_sun_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..000108f77ea1a0b578c1ca980ed0d74490f24fdf --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_sun_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: sun_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Sundanese and English linguist, translate the following Sundanese\ + \ sentences to English \nSundanese: {{sentence_sun_Latn}}\nEnglish: " +include: flores +task: flores_sun_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_swh_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_swh_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1c81805c1f22b9e6c6bd55ba925fa7dfb80f0cf1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_swh_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: swh_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Swahili and English linguist, translate the following Swahili sentences\ + \ to English \nSwahili: {{sentence_swh_Latn}}\nEnglish: " +include: flores +task: flores_swh_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_taq_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_taq_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6febb3004bc3f39f94287ac41a9883d95f056fe1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_taq_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: taq_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Tamasheq and English linguist, translate the following Tamasheq\ + \ sentences to English \nTamasheq: {{sentence_taq_Latn}}\nEnglish: " +include: flores +task: flores_taq_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_taq_Tfng-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_taq_Tfng-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6290ab94d3be2e52750f9af5900d7c27f27cb5af --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_taq_Tfng-eng_Latn.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: taq_Tfng-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Tamasheq (Tifinagh script) and English linguist, translate the\ + \ following Tamasheq (Tifinagh script) sentences to English \nTamasheq (Tifinagh\ + \ script): {{sentence_taq_Tfng}}\nEnglish: " +include: flores +task: flores_taq_Tfng-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_tir_Ethi-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_tir_Ethi-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..60133a3b735a0b1d91901bdd7f1ef2122b0f0f03 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_tir_Ethi-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: tir_Ethi-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Tigrinya and English linguist, translate the following Tigrinya\ + \ sentences to English \nTigrinya: {{sentence_tir_Ethi}}\nEnglish: " +include: flores +task: flores_tir_Ethi-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_tsn_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_tsn_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..40417bde77b4a0ede2e9c87c56668be101579f3b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_tsn_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: tsn_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Setswana and English linguist, translate the following Setswana\ + \ sentences to English \nSetswana: {{sentence_tsn_Latn}}\nEnglish: " +include: flores +task: flores_tsn_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_tso_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_tso_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..56d4632500b86964d0d665f1827cd129bc508d63 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_tso_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: tso_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Tsonga and English linguist, translate the following Tsonga sentences\ + \ to English \nTsonga: {{sentence_tso_Latn}}\nEnglish: " +include: flores +task: flores_tso_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_tum_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_tum_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cc4bb541f7a5692a067ad75f5e3a86490487cc70 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_tum_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: tum_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Tumbuka and English linguist, translate the following Tumbuka sentences\ + \ to English \nTumbuka: {{sentence_tum_Latn}}\nEnglish: " +include: flores +task: flores_tum_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_twi_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_twi_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4cc0d674c8ced1f005eef5a94c87a886a6f176aa --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_twi_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: twi_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Twi and English linguist, translate the following Twi sentences\ + \ to English \nTwi: {{sentence_twi_Latn}}\nEnglish: " +include: flores +task: flores_twi_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_tzm_Tfng-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_tzm_Tfng-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d3575ccb2a766a722cbc88880f34656af8cdb3a4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_tzm_Tfng-eng_Latn.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: tzm_Tfng-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Central Atlas Tamazight and English linguist, translate the following\ + \ Central Atlas Tamazight sentences to English \nCentral Atlas Tamazight: {{sentence_tzm_Tfng}}\n\ + English: " +include: flores +task: flores_tzm_Tfng-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_umb_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_umb_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e7df76cf07cb4bf8f772721136fd4d92280b3820 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_umb_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: umb_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Umbundu and English linguist, translate the following Umbundu sentences\ + \ to English \nUmbundu: {{sentence_umb_Latn}}\nEnglish: " +include: flores +task: flores_umb_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_wol_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_wol_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..22275ca15cd1829db481c0c77363a10649be101f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_wol_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: wol_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Wolof and English linguist, translate the following Wolof sentences\ + \ to English \nWolof: {{sentence_wol_Latn}}\nEnglish: " +include: flores +task: flores_wol_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_xho_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_xho_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..85ae368b6efa8dab3a8a2b110e438b043ef3c74f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_xho_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: xho_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Xhosa and English linguist, translate the following Xhosa sentences\ + \ to English \nXhosa: {{sentence_xho_Latn}}\nEnglish: " +include: flores +task: flores_xho_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_yor_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_yor_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5bbd8eb967ea1f4c6afe09bef65e379c4fed9c25 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_yor_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: yor_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Yoruba and English linguist, translate the following Yoruba sentences\ + \ to English \nYoruba: {{sentence_yor_Latn}}\nEnglish: " +include: flores +task: flores_yor_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_zul_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_zul_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ea2c2edb8439fffb452187b86ed1690501b20b3d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_zul_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: zul_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Zulu and English linguist, translate the following Zulu sentences\ + \ to English \nZulu: {{sentence_zul_Latn}}\nEnglish: " +include: flores +task: flores_zul_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-ace_Arab.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-ace_Arab.yaml new file mode 100644 index 0000000000000000000000000000000000000000..53cf711fa19132b8668d1c4a6024e1b96f54751b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-ace_Arab.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: eng_Latn-ace_Arab +doc_to_target: sentence_ace_Arab +doc_to_text: "As a Acehnese (Arabic script) and English linguist, translate the following\ + \ English sentences to Acehnese (Arabic script) \nEnglish: {{sentence_eng_Latn}}\ + \ \nAcehnese (Arabic script): " +include: flores +task: flores_eng_Latn-ace_Arab_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-ace_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-ace_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..766b7c30061e8adfd3e4827052fba7160483ae4a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-ace_Latn.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: eng_Latn-ace_Latn +doc_to_target: sentence_ace_Latn +doc_to_text: "As a Acehnese (Latin script) and English linguist, translate the following\ + \ English sentences to Acehnese (Latin script) \nEnglish: {{sentence_eng_Latn}}\ + \ \nAcehnese (Latin script): " +include: flores +task: flores_eng_Latn-ace_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-acq_Arab.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-acq_Arab.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e809c866eb602e76defc6c3fca983e02bc213a52 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-acq_Arab.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: eng_Latn-acq_Arab +doc_to_target: sentence_acq_Arab +doc_to_text: "As a Ta’izzi-Adeni Arabic and English linguist, translate the following\ + \ English sentences to Ta’izzi-Adeni Arabic \nEnglish: {{sentence_eng_Latn}} \n\ + Ta’izzi-Adeni Arabic: " +include: flores +task: flores_eng_Latn-acq_Arab_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-aeb_Arab.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-aeb_Arab.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9e8263fe6af0b65e5c935c7d56146b6940c6850b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-aeb_Arab.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-aeb_Arab +doc_to_target: sentence_aeb_Arab +doc_to_text: "As a Tunisian Arabic and English linguist, translate the following English\ + \ sentences to Tunisian Arabic \nEnglish: {{sentence_eng_Latn}} \nTunisian Arabic: " +include: flores +task: flores_eng_Latn-aeb_Arab_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-afr_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-afr_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..86421c268959192fae2dcbc19a1a4b935d6bff29 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-afr_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-afr_Latn +doc_to_target: sentence_afr_Latn +doc_to_text: "As a Afrikaans and English linguist, translate the following English\ + \ sentences to Afrikaans \nEnglish: {{sentence_eng_Latn}} \nAfrikaans: " +include: flores +task: flores_eng_Latn-afr_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-aka_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-aka_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3373390566a317a7438e216cea67b926d5dd20fa --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-aka_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-aka_Latn +doc_to_target: sentence_aka_Latn +doc_to_text: "As a Akan and English linguist, translate the following English sentences\ + \ to Akan \nEnglish: {{sentence_eng_Latn}} \nAkan: " +include: flores +task: flores_eng_Latn-aka_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-amh_Ethi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-amh_Ethi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ba3e0116586dfb106bc57103dc685ddd8856570d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-amh_Ethi.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-amh_Ethi +doc_to_target: sentence_amh_Ethi +doc_to_text: "As a Amharic and English linguist, translate the following English sentences\ + \ to Amharic \nEnglish: {{sentence_eng_Latn}} \nAmharic: " +include: flores +task: flores_eng_Latn-amh_Ethi_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-ary_Arab.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-ary_Arab.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c732756a2ea06e33114c117c630f9b3fccab32fe --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-ary_Arab.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-ary_Arab +doc_to_target: sentence_ary_Arab +doc_to_text: "As a Moroccan Arabic and English linguist, translate the following English\ + \ sentences to Moroccan Arabic \nEnglish: {{sentence_eng_Latn}} \nMoroccan Arabic: " +include: flores +task: flores_eng_Latn-ary_Arab_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-arz_Arab.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-arz_Arab.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f11bc38a2dc1c1cc565979a47b51b5aad9bc830e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-arz_Arab.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-arz_Arab +doc_to_target: sentence_arz_Arab +doc_to_text: "As a Egyptian Arabic and English linguist, translate the following English\ + \ sentences to Egyptian Arabic \nEnglish: {{sentence_eng_Latn}} \nEgyptian Arabic: " +include: flores +task: flores_eng_Latn-arz_Arab_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-bam_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-bam_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c762962832885fe21c75986f7ce006789217dbd4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-bam_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-bam_Latn +doc_to_target: sentence_bam_Latn +doc_to_text: "As a Bambara and English linguist, translate the following English sentences\ + \ to Bambara \nEnglish: {{sentence_eng_Latn}} \nBambara: " +include: flores +task: flores_eng_Latn-bam_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-ban_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-ban_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..601aecf5cebebdb6572fadf8f82d2963b9b87d5c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-ban_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-ban_Latn +doc_to_target: sentence_ban_Latn +doc_to_text: "As a Balinese and English linguist, translate the following English\ + \ sentences to Balinese \nEnglish: {{sentence_eng_Latn}} \nBalinese: " +include: flores +task: flores_eng_Latn-ban_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-bem_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-bem_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fadabdb9356f28fda88a68e69646a0fd60141e9c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-bem_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-bem_Latn +doc_to_target: sentence_bem_Latn +doc_to_text: "As a Bemba and English linguist, translate the following English sentences\ + \ to Bemba \nEnglish: {{sentence_eng_Latn}} \nBemba: " +include: flores +task: flores_eng_Latn-bem_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-cjk_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-cjk_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c522831373d25e05103918ba43f36183106cc509 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-cjk_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-cjk_Latn +doc_to_target: sentence_cjk_Latn +doc_to_text: "As a Chokwe and English linguist, translate the following English sentences\ + \ to Chokwe \nEnglish: {{sentence_eng_Latn}} \nChokwe: " +include: flores +task: flores_eng_Latn-cjk_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-dik_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-dik_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..acfeb83ad758573c632ca1b3e9e08f190b86fa30 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-dik_Latn.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: eng_Latn-dik_Latn +doc_to_target: sentence_dik_Latn +doc_to_text: "As a Southwestern Dinka and English linguist, translate the following\ + \ English sentences to Southwestern Dinka \nEnglish: {{sentence_eng_Latn}} \nSouthwestern\ + \ Dinka: " +include: flores +task: flores_eng_Latn-dik_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-dyu_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-dyu_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..796dc6d2f633c5baba22d2cce8592f0f01e3fe42 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-dyu_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-dyu_Latn +doc_to_target: sentence_dyu_Latn +doc_to_text: "As a Dyula and English linguist, translate the following English sentences\ + \ to Dyula \nEnglish: {{sentence_eng_Latn}} \nDyula: " +include: flores +task: flores_eng_Latn-dyu_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-ewe_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-ewe_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..31a07891820793360f26b2d093e98b5982816ea6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-ewe_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-ewe_Latn +doc_to_target: sentence_ewe_Latn +doc_to_text: "As a Ewe and English linguist, translate the following English sentences\ + \ to Ewe \nEnglish: {{sentence_eng_Latn}} \nEwe: " +include: flores +task: flores_eng_Latn-ewe_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-fon_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-fon_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6cdc7308d63891ed0fc65e779e394b71eafb3bb1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-fon_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-fon_Latn +doc_to_target: sentence_fon_Latn +doc_to_text: "As a Fon and English linguist, translate the following English sentences\ + \ to Fon \nEnglish: {{sentence_eng_Latn}} \nFon: " +include: flores +task: flores_eng_Latn-fon_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-fra_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-fra_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3896879db152bc1e583fd24f5823321d0f6eda4d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-fra_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-fra_Latn +doc_to_target: sentence_fra_Latn +doc_to_text: "As a French and English linguist, translate the following English sentences\ + \ to French \nEnglish: {{sentence_eng_Latn}} \nFrench: " +include: flores +task: flores_eng_Latn-fra_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-fuv_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-fuv_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6b63249be8c1e81e837e9a024dd19ecd822f748b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-fuv_Latn.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: eng_Latn-fuv_Latn +doc_to_target: sentence_fuv_Latn +doc_to_text: "As a Nigerian Fulfulde and English linguist, translate the following\ + \ English sentences to Nigerian Fulfulde \nEnglish: {{sentence_eng_Latn}} \nNigerian\ + \ Fulfulde: " +include: flores +task: flores_eng_Latn-fuv_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-gaz_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-gaz_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..95cde87c38c66448967d595f60709c2f908af5f0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-gaz_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-gaz_Latn +doc_to_target: sentence_gaz_Latn +doc_to_text: "As a Oromo and English linguist, translate the following English sentences\ + \ to Oromo \nEnglish: {{sentence_eng_Latn}} \nOromo: " +include: flores +task: flores_eng_Latn-gaz_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-hau_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-hau_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..eec82e34503bb64fcab1c90cf507499760f95a15 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-hau_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-hau_Latn +doc_to_target: sentence_hau_Latn +doc_to_text: "As a Hausa and English linguist, translate the following English sentences\ + \ to Hausa \nEnglish: {{sentence_eng_Latn}} \nHausa: " +include: flores +task: flores_eng_Latn-hau_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-ibo_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-ibo_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..838990b364097652e9ba4ed68726147e4424d05e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-ibo_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-ibo_Latn +doc_to_target: sentence_ibo_Latn +doc_to_text: "As a Igbo and English linguist, translate the following English sentences\ + \ to Igbo \nEnglish: {{sentence_eng_Latn}} \nIgbo: " +include: flores +task: flores_eng_Latn-ibo_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-kab_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-kab_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..16888ad8f28b64a7bb9715fdf8f193e18ce06072 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-kab_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-kab_Latn +doc_to_target: sentence_kab_Latn +doc_to_text: "As a Kabyle and English linguist, translate the following English sentences\ + \ to Kabyle \nEnglish: {{sentence_eng_Latn}} \nKabyle: " +include: flores +task: flores_eng_Latn-kab_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-kam_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-kam_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d48c52d16017b2e1241afd154539148d3f0d0ae4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-kam_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-kam_Latn +doc_to_target: sentence_kam_Latn +doc_to_text: "As a Kamba and English linguist, translate the following English sentences\ + \ to Kamba \nEnglish: {{sentence_eng_Latn}} \nKamba: " +include: flores +task: flores_eng_Latn-kam_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-kbp_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-kbp_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c992a28f7e168e3753e378631fa6ce716e7ee69e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-kbp_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-kbp_Latn +doc_to_target: sentence_kbp_Latn +doc_to_text: "As a Kabiyè and English linguist, translate the following English sentences\ + \ to Kabiyè \nEnglish: {{sentence_eng_Latn}} \nKabiyè: " +include: flores +task: flores_eng_Latn-kbp_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-kea_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-kea_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d8ce1b502edea3c9f00fe89dbf9dc382010e4bff --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-kea_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-kea_Latn +doc_to_target: sentence_kea_Latn +doc_to_text: "As a Kabuverdianu and English linguist, translate the following English\ + \ sentences to Kabuverdianu \nEnglish: {{sentence_eng_Latn}} \nKabuverdianu: " +include: flores +task: flores_eng_Latn-kea_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-kik_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-kik_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fc7975c2b23bb486ead2962f28064a5fcab6102f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-kik_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-kik_Latn +doc_to_target: sentence_kik_Latn +doc_to_text: "As a Kikuyu and English linguist, translate the following English sentences\ + \ to Kikuyu \nEnglish: {{sentence_eng_Latn}} \nKikuyu: " +include: flores +task: flores_eng_Latn-kik_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-kin_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-kin_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1e2b91d461378cb7c8ff098d237037eefdcacc03 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-kin_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-kin_Latn +doc_to_target: sentence_kin_Latn +doc_to_text: "As a Kinyarwanda and English linguist, translate the following English\ + \ sentences to Kinyarwanda \nEnglish: {{sentence_eng_Latn}} \nKinyarwanda: " +include: flores +task: flores_eng_Latn-kin_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-kmb_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-kmb_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..270f29b629e6f1f06da31ba154d977b0281fd63b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-kmb_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-kmb_Latn +doc_to_target: sentence_kmb_Latn +doc_to_text: "As a Kimbundu and English linguist, translate the following English\ + \ sentences to Kimbundu \nEnglish: {{sentence_eng_Latn}} \nKimbundu: " +include: flores +task: flores_eng_Latn-kmb_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-knc_Arab.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-knc_Arab.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bd2994d36152fc1dcb4a7a2561cc41982dd6fed1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-knc_Arab.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: eng_Latn-knc_Arab +doc_to_target: sentence_knc_Arab +doc_to_text: "As a Central Kanuri (Arabic script) and English linguist, translate\ + \ the following English sentences to Central Kanuri (Arabic script) \nEnglish: {{sentence_eng_Latn}}\ + \ \nCentral Kanuri (Arabic script): " +include: flores +task: flores_eng_Latn-knc_Arab_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-knc_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-knc_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..262d0c1f3b8efc51c35e7154f27a9a4e6ed1405f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-knc_Latn.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: eng_Latn-knc_Latn +doc_to_target: sentence_knc_Latn +doc_to_text: "As a Central Kanuri (Latin script) and English linguist, translate the\ + \ following English sentences to Central Kanuri (Latin script) \nEnglish: {{sentence_eng_Latn}}\ + \ \nCentral Kanuri (Latin script): " +include: flores +task: flores_eng_Latn-knc_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-kon_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-kon_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ae9e1201808061f32c0e9d9260b8d2900f7bd7d4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-kon_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-kon_Latn +doc_to_target: sentence_kon_Latn +doc_to_text: "As a Kikongo and English linguist, translate the following English sentences\ + \ to Kikongo \nEnglish: {{sentence_eng_Latn}} \nKikongo: " +include: flores +task: flores_eng_Latn-kon_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-lin_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-lin_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0945c697c27b39ed91cff296dd162735f0629f4a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-lin_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-lin_Latn +doc_to_target: sentence_lin_Latn +doc_to_text: "As a Lingala and English linguist, translate the following English sentences\ + \ to Lingala \nEnglish: {{sentence_eng_Latn}} \nLingala: " +include: flores +task: flores_eng_Latn-lin_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-lua_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-lua_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ff92a2cf381a3a4c94a5543901be39a647d24eb7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-lua_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-lua_Latn +doc_to_target: sentence_lua_Latn +doc_to_text: "As a Luba-Kasai and English linguist, translate the following English\ + \ sentences to Luba-Kasai \nEnglish: {{sentence_eng_Latn}} \nLuba-Kasai: " +include: flores +task: flores_eng_Latn-lua_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-lug_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-lug_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4dfc626b9fdbcde3de0383b5d512365137da00b5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-lug_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-lug_Latn +doc_to_target: sentence_lug_Latn +doc_to_text: "As a Luganda and English linguist, translate the following English sentences\ + \ to Luganda \nEnglish: {{sentence_eng_Latn}} \nLuganda: " +include: flores +task: flores_eng_Latn-lug_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-luo_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-luo_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..803ed75d8b732c81f859285f974cd216afb86784 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-luo_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-luo_Latn +doc_to_target: sentence_luo_Latn +doc_to_text: "As a Luo and English linguist, translate the following English sentences\ + \ to Luo \nEnglish: {{sentence_eng_Latn}} \nLuo: " +include: flores +task: flores_eng_Latn-luo_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-mos_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-mos_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0e959db1653eff6ca0054ec5032144a96c2c5713 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-mos_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-mos_Latn +doc_to_target: sentence_mos_Latn +doc_to_text: "As a Mossi and English linguist, translate the following English sentences\ + \ to Mossi \nEnglish: {{sentence_eng_Latn}} \nMossi: " +include: flores +task: flores_eng_Latn-mos_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-nso_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-nso_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..44839d82cac5af78f74b2f382d36cbd74f93baa6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-nso_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-nso_Latn +doc_to_target: sentence_nso_Latn +doc_to_text: "As a Northern Sotho and English linguist, translate the following English\ + \ sentences to Northern Sotho \nEnglish: {{sentence_eng_Latn}} \nNorthern Sotho: " +include: flores +task: flores_eng_Latn-nso_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-nus_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-nus_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..387e4341f0761727d3e07a8748291aac574c727f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-nus_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-nus_Latn +doc_to_target: sentence_nus_Latn +doc_to_text: "As a Nuer and English linguist, translate the following English sentences\ + \ to Nuer \nEnglish: {{sentence_eng_Latn}} \nNuer: " +include: flores +task: flores_eng_Latn-nus_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-nya_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-nya_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9311e264e1f7e617a22c52e7ac969b1001f7c5e7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-nya_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-nya_Latn +doc_to_target: sentence_nya_Latn +doc_to_text: "As a Nyanja and English linguist, translate the following English sentences\ + \ to Nyanja \nEnglish: {{sentence_eng_Latn}} \nNyanja: " +include: flores +task: flores_eng_Latn-nya_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-plt_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-plt_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..afc81158cbed0e268746ee51c1d4e1071f4315e0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-plt_Latn.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: eng_Latn-plt_Latn +doc_to_target: sentence_plt_Latn +doc_to_text: "As a Plateau Malagasy and English linguist, translate the following\ + \ English sentences to Plateau Malagasy \nEnglish: {{sentence_eng_Latn}} \nPlateau\ + \ Malagasy: " +include: flores +task: flores_eng_Latn-plt_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-run_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-run_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..519700cd32de76f039a6f1d3ce16a4c539278334 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-run_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-run_Latn +doc_to_target: sentence_run_Latn +doc_to_text: "As a Rundi and English linguist, translate the following English sentences\ + \ to Rundi \nEnglish: {{sentence_eng_Latn}} \nRundi: " +include: flores +task: flores_eng_Latn-run_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-sag_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-sag_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fa99b16137861e1e9fb4f19669dbb71977fd3cc1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-sag_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-sag_Latn +doc_to_target: sentence_sag_Latn +doc_to_text: "As a Sango and English linguist, translate the following English sentences\ + \ to Sango \nEnglish: {{sentence_eng_Latn}} \nSango: " +include: flores +task: flores_eng_Latn-sag_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-sna_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-sna_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fd7ac49ac5854133223a599569981f4c27d19a21 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-sna_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-sna_Latn +doc_to_target: sentence_sna_Latn +doc_to_text: "As a Shona and English linguist, translate the following English sentences\ + \ to Shona \nEnglish: {{sentence_eng_Latn}} \nShona: " +include: flores +task: flores_eng_Latn-sna_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-som_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-som_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..17870addf00ed2c4f0c4e126fc094a06aabc8027 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-som_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-som_Latn +doc_to_target: sentence_som_Latn +doc_to_text: "As a Somali and English linguist, translate the following English sentences\ + \ to Somali \nEnglish: {{sentence_eng_Latn}} \nSomali: " +include: flores +task: flores_eng_Latn-som_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-sot_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-sot_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a45cf383057f37f504f799d7cb241ec61274fd83 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-sot_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-sot_Latn +doc_to_target: sentence_sot_Latn +doc_to_text: "As a Southern Sotho and English linguist, translate the following English\ + \ sentences to Southern Sotho \nEnglish: {{sentence_eng_Latn}} \nSouthern Sotho: " +include: flores +task: flores_eng_Latn-sot_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-ssw_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-ssw_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0dbd162772b8938aef4f05da00ac3da0ce3be530 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-ssw_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-ssw_Latn +doc_to_target: sentence_ssw_Latn +doc_to_text: "As a Swati and English linguist, translate the following English sentences\ + \ to Swati \nEnglish: {{sentence_eng_Latn}} \nSwati: " +include: flores +task: flores_eng_Latn-ssw_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-sun_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-sun_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0f8f6339450e8af7318e570290370a21037ba98d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-sun_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-sun_Latn +doc_to_target: sentence_sun_Latn +doc_to_text: "As a Sundanese and English linguist, translate the following English\ + \ sentences to Sundanese \nEnglish: {{sentence_eng_Latn}} \nSundanese: " +include: flores +task: flores_eng_Latn-sun_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-swh_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-swh_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..20971c5383cbcce97b3743262adc35c9d2dfadcf --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-swh_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-swh_Latn +doc_to_target: sentence_swh_Latn +doc_to_text: "As a Swahili and English linguist, translate the following English sentences\ + \ to Swahili \nEnglish: {{sentence_eng_Latn}} \nSwahili: " +include: flores +task: flores_eng_Latn-swh_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-taq_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-taq_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bdb06f77b78b9ca296eb960fc62a63a94e990fd8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-taq_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-taq_Latn +doc_to_target: sentence_taq_Latn +doc_to_text: "As a Tamasheq and English linguist, translate the following English\ + \ sentences to Tamasheq \nEnglish: {{sentence_eng_Latn}} \nTamasheq: " +include: flores +task: flores_eng_Latn-taq_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-taq_Tfng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-taq_Tfng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d690651ddb21eaf3022475e98dc8f5ddf72f073b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-taq_Tfng.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: eng_Latn-taq_Tfng +doc_to_target: sentence_taq_Tfng +doc_to_text: "As a Tamasheq (Tifinagh script) and English linguist, translate the\ + \ following English sentences to Tamasheq (Tifinagh script) \nEnglish: {{sentence_eng_Latn}}\ + \ \nTamasheq (Tifinagh script): " +include: flores +task: flores_eng_Latn-taq_Tfng_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-tir_Ethi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-tir_Ethi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c6b3ba347ab935d9c87a48262a9cafb259985ea0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-tir_Ethi.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-tir_Ethi +doc_to_target: sentence_tir_Ethi +doc_to_text: "As a Tigrinya and English linguist, translate the following English\ + \ sentences to Tigrinya \nEnglish: {{sentence_eng_Latn}} \nTigrinya: " +include: flores +task: flores_eng_Latn-tir_Ethi_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-tsn_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-tsn_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..845626f5f347436a0fcd2c04fe3446cb806da44e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-tsn_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-tsn_Latn +doc_to_target: sentence_tsn_Latn +doc_to_text: "As a Setswana and English linguist, translate the following English\ + \ sentences to Setswana \nEnglish: {{sentence_eng_Latn}} \nSetswana: " +include: flores +task: flores_eng_Latn-tsn_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-tso_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-tso_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..958411f89ea39c77cc329ce5f14795761ec1f1a9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-tso_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-tso_Latn +doc_to_target: sentence_tso_Latn +doc_to_text: "As a Tsonga and English linguist, translate the following English sentences\ + \ to Tsonga \nEnglish: {{sentence_eng_Latn}} \nTsonga: " +include: flores +task: flores_eng_Latn-tso_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-tum_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-tum_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..95e6efa7dbdd8ea987bd680954e69e97742c452d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-tum_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-tum_Latn +doc_to_target: sentence_tum_Latn +doc_to_text: "As a Tumbuka and English linguist, translate the following English sentences\ + \ to Tumbuka \nEnglish: {{sentence_eng_Latn}} \nTumbuka: " +include: flores +task: flores_eng_Latn-tum_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-twi_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-twi_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0dcb20543ccff51461bc2af582fcdfaa5855d25f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-twi_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-twi_Latn +doc_to_target: sentence_twi_Latn +doc_to_text: "As a Twi and English linguist, translate the following English sentences\ + \ to Twi \nEnglish: {{sentence_eng_Latn}} \nTwi: " +include: flores +task: flores_eng_Latn-twi_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-umb_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-umb_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b8c4adc0139f6f2d9ce4fcd910e3f558b96df78d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-umb_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-umb_Latn +doc_to_target: sentence_umb_Latn +doc_to_text: "As a Umbundu and English linguist, translate the following English sentences\ + \ to Umbundu \nEnglish: {{sentence_eng_Latn}} \nUmbundu: " +include: flores +task: flores_eng_Latn-umb_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-wol_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-wol_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..66ad25794e67d54123f2a63ffad5e98d12c6ce59 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-wol_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-wol_Latn +doc_to_target: sentence_wol_Latn +doc_to_text: "As a Wolof and English linguist, translate the following English sentences\ + \ to Wolof \nEnglish: {{sentence_eng_Latn}} \nWolof: " +include: flores +task: flores_eng_Latn-wol_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-xho_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-xho_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8cd2fe08ec7bcd000182a7a9f080e739b6d82289 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-xho_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-xho_Latn +doc_to_target: sentence_xho_Latn +doc_to_text: "As a Xhosa and English linguist, translate the following English sentences\ + \ to Xhosa \nEnglish: {{sentence_eng_Latn}} \nXhosa: " +include: flores +task: flores_eng_Latn-xho_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-yor_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-yor_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..09562458138acb1d146e5319816b01e2205351ce --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-yor_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-yor_Latn +doc_to_target: sentence_yor_Latn +doc_to_text: "As a Yoruba and English linguist, translate the following English sentences\ + \ to Yoruba \nEnglish: {{sentence_eng_Latn}} \nYoruba: " +include: flores +task: flores_eng_Latn-yor_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/README.md b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/README.md new file mode 100644 index 0000000000000000000000000000000000000000..641877cb7c01a5b19791b20c95a246753ddee75a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/README.md @@ -0,0 +1,23 @@ +# + +## Paper +Title: `INJONGO: A Multicultural Intent Detection and Slot-filling Dataset for 16 African Languages` + +Paper Link: https://arxiv.org/abs/2502.09814 + +## Abstract +>Slot-filling and intent detection are well-established tasks in Conversational AI. However, current large-scale benchmarks for these tasks often exclude evaluations of low-resource languages and rely on translations from English benchmarks, thereby predominantly reflecting Western-centric concepts. In this paper, we introduce Injongo -- a multicultural, open-source benchmark dataset for 16 African languages with utterances generated by native speakers across diverse domains, including banking, travel, home, and dining. Through extensive experiments, we benchmark the fine-tuning multilingual transformer models and the prompting large language models (LLMs), and show the advantage of leveraging African-cultural utterances over Western-centric utterances for improving cross-lingual transfer from the English language. Experimental results reveal that current LLMs struggle with the slot-filling task, with GPT-4o achieving an average performance of 26 F1-score. In contrast, intent detection performance is notably better, with an average accuracy of 70.6%, though it still falls behind the fine-tuning baselines. Compared to the English language, GPT-4o and fine-tuning baselines perform similarly on intent detection, achieving an accuracy of approximately 81%. Our findings suggest that the performance of LLMs is still behind for many low-resource African languages, and more work is needed to further improve their downstream performance. + +### Citation + +``` +@misc{yu2025injongomulticulturalintentdetection, + title={INJONGO: A Multicultural Intent Detection and Slot-filling Dataset for 16 African Languages}, + author={Hao Yu and Jesujoba O. Alabi and Andiswa Bukula and Jian Yun Zhuang and En-Shiun Annie Lee and Tadesse Kebede Guge and Israel Abebe Azime and Happy Buzaaba and Blessing Kudzaishe Sibanda and Godson K. Kalipe and Jonathan Mukiibi and Salomon Kabongo Kabenamualu and Mmasibidi Setaka and Lolwethu Ndolela and Nkiruka Odu and Rooweither Mabuya and Shamsuddeen Hassan Muhammad and Salomey Osei and Sokhar Samb and Juliet W. Murage and Dietrich Klakow and David Ifeoluwa Adelani}, + year={2025}, + eprint={2502.09814}, + archivePrefix={arXiv}, + primaryClass={cs.CL}, + url={https://arxiv.org/abs/2502.09814}, +} +``` diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/gen_utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/gen_utils.py new file mode 100644 index 0000000000000000000000000000000000000000..112041999df20d26a31becc30633720a16457b18 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/gen_utils.py @@ -0,0 +1,159 @@ +import argparse +import os + +import yaml + + +class FunctionTag: + def __init__(self, value): + self.value = value + + +def prompt_func(mode, lang, intent): + prompt_map = { + "prompt_1": "Given the text: '{{text}}', determine the correct intent from the following list: " + f"[{', '.join(intent)}]. Only output one intent from the list.", + "prompt_2": "Analyze the text: '{{text}}'. Choose the most appropriate intent from these options: " + f"[{', '.join(intent)}]. Respond with only the selected intent.", + "prompt_3": "You are a linguistic analyst trained to understand user intent. Based on the text: '{{text}}', " + f"choose the intent that best matches from this list: [{', '.join(intent)}]. Return only the intent.", + "prompt_4": f"You are a {lang} linguistic analyst trained to understand {lang} user intent. Based on the {lang}" + "text: '{{text}}', choose the intent that best matches from this list: " + f"[{', '.join(intent)}]. Return only the intent.", + "prompt_5": f"The following text is in {lang}: '{{{{text}}}}'. Given the list of intents: [{', '.join(intent)}], " + "identify the intent expressed in the text. Return only the identified intent.", + } + return prompt_map[mode] + + +def gen_lang_yamls(output_dir: str, overwrite: bool, mode: str) -> None: + """ + Generate a yaml file for each language. + + :param output_dir: The directory to output the files to. + :param overwrite: Whether to overwrite files if they already exist. + """ + err = [] + languages = { + "amh": "Amharic", + "ewe": "Ewe", + "hau": "Hausa", + "ibo": "Igbo", + "kin": "Kinyarwanda", + "lin": "Lingala", + "lug": "Luganda", + "orm": "Oromo", + "sna": "Shona", + "sot": "Sotho", + "swa": "Swahili", + "twi": "Twi", + "wol": "Wolof", + "xho": "Xhosa", + "yor": "Yoruba", + "zul": "Zulu", + "eng": "English", + } + + intents = [ + "alarm", + "balance", + "bill_balance", + "book_flight", + "book_hotel", + "calendar_update", + "cancel_reservation", + "car_rental", + "confirm_reservation", + "cook_time", + "exchange_rate", + "food_last", + "freeze_account", + "ingredients_list", + "interest_rate", + "international_visa", + "make_call", + "meal_suggestion", + "min_payment", + "pay_bill", + "pin_change", + "play_music", + "plug_type", + "recipe", + "restaurant_reservation", + "restaurant_reviews", + "restaurant_suggestion", + "share_location", + "shopping_list_update", + "spending_history", + "text", + "time", + "timezone", + "transactions", + "transfer", + "translate", + "travel_notification", + "travel_suggestion", + "update_playlist", + "weather", + ] + + for lang in languages.keys(): + try: + file_name = f"injongointent_{lang}.yaml" + task_name = f"injongointent_{lang}_{mode}" + yaml_template = "injongointent" + yaml_details = { + "include": yaml_template, + "task": task_name, + "dataset_name": lang, + "doc_to_text": prompt_func(mode, languages[lang], intents), + } + os.makedirs(f"{output_dir}/{mode}", exist_ok=True) + with open( + f"{output_dir}/{mode}/{file_name}", + "w" if overwrite else "x", + encoding="utf8", + ) as f: + f.write("# Generated by utils.py\n") + yaml.dump( + yaml_details, + f, + allow_unicode=True, + ) + except FileExistsError: + err.append(file_name) + + if len(err) > 0: + raise FileExistsError( + "Files were not created because they already exist (use --overwrite flag):" + f" {', '.join(err)}" + ) + + +def main() -> None: + """Parse CLI args and generate language-specific yaml files.""" + parser = argparse.ArgumentParser() + parser.add_argument( + "--overwrite", + default=True, + action="store_true", + help="Overwrite files if they already exist", + ) + parser.add_argument( + "--output-dir", + default="./", + help="Directory to write yaml files to", + ) + parser.add_argument( + "--mode", + default="prompt_3", + choices=["prompt_1", "prompt_2", "prompt_3", "prompt_4", "prompt_5"], + help="Prompt number", + ) + args = parser.parse_args() + + gen_lang_yamls(output_dir=args.output_dir, overwrite=args.overwrite, mode=args.mode) + + +if __name__ == "__main__": + main() diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/injongointent.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/injongointent.yaml new file mode 100644 index 0000000000000000000000000000000000000000..220f4c514f0afb8ec9105d54c90f834b4fd57780 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/injongointent.yaml @@ -0,0 +1,13 @@ +group: injongointent +task: + - injongointent_prompt_1 + - injongointent_prompt_2 + - injongointent_prompt_3 + - injongointent_prompt_4 + - injongointent_prompt_5 +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent new file mode 100644 index 0000000000000000000000000000000000000000..a77bc5c95941392779b960df6ad26ebebe5ba96d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent @@ -0,0 +1,75 @@ +tag: +- injongointent_tasks +- injongointent_prompt_1 +dataset_path: masakhane/InjongoIntent +dataset_name: null +output_type: multiple_choice +validation_split: validation +test_split: test +fewshot_split: validation +doc_to_target: intent +doc_to_choice: + - alarm + - balance + - bill_balance + - book_flight + - book_hotel + - calendar_update + - cancel_reservation + - car_rental + - confirm_reservation + - cook_time + - exchange_rate + - food_last + - freeze_account + - ingredients_list + - interest_rate + - international_visa + - make_call + - meal_suggestion + - min_payment + - pay_bill + - pin_change + - play_music + - plug_type + - recipe + - restaurant_reservation + - restaurant_reviews + - restaurant_suggestion + - share_location + - shopping_list_update + - spending_history + - text + - time + - timezone + - transactions + - transfer + - translate + - travel_notification + - travel_suggestion + - update_playlist + - weather +should_decontaminate: true +doc_to_decontamination_query: text +metric_list: + - metric: f1 + aggregation: !function utils.weighted_f1_score + # aggregation: mean + average: weighted + hf_evaluate: true + higher_is_better: True + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" + - metric: acc + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0b3a3ee270683d5cc57a6e6ce81a3fe971f6c04e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_amh.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: amh +doc_to_text: 'Given the text: ''{{text}}'', determine the correct intent from the + following list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Only + output one intent from the list.' +include: injongointent +task: injongointent_amh_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_eng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..240c37d5f1cd4197314c51532ab25ab2e915e2ad --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_eng.yaml @@ -0,0 +1,16 @@ +# Generated by utils.py +dataset_name: eng +validation_split: train +test_split: test +fewshot_split: train +doc_to_text: 'Given the text: ''{{text}}'', determine the correct intent from the + following list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Only + output one intent from the list.' +include: injongointent +task: injongointent_eng_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_ewe.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_ewe.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c08d8bb0c151a812b1cd6d5131e4a0d1664f8725 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_ewe.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: ewe +doc_to_text: 'Given the text: ''{{text}}'', determine the correct intent from the + following list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Only + output one intent from the list.' +include: injongointent +task: injongointent_ewe_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9e1338c72cda04bbbbca713f83dd6aa715f19014 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_hau.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: hau +doc_to_text: 'Given the text: ''{{text}}'', determine the correct intent from the + following list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Only + output one intent from the list.' +include: injongointent +task: injongointent_hau_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e4a956d23c414332d353f9c7d04ac3ce876831ed --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_ibo.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: ibo +doc_to_text: 'Given the text: ''{{text}}'', determine the correct intent from the + following list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Only + output one intent from the list.' +include: injongointent +task: injongointent_ibo_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f55d787a3f4952afcbb3f70d9432bfc7dbf0a84e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_kin.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: kin +doc_to_text: 'Given the text: ''{{text}}'', determine the correct intent from the + following list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Only + output one intent from the list.' +include: injongointent +task: injongointent_kin_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_lin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_lin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2cc08df484bf58f9eaf2d498074eb1ac5dc72338 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_lin.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: lin +doc_to_text: 'Given the text: ''{{text}}'', determine the correct intent from the + following list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Only + output one intent from the list.' +include: injongointent +task: injongointent_lin_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b1a421577bbe8ff9ce34b00da45fdb3efcb22e9d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_lug.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: lug +doc_to_text: 'Given the text: ''{{text}}'', determine the correct intent from the + following list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Only + output one intent from the list.' +include: injongointent +task: injongointent_lug_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_orm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_orm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2b95a4e7b9cb3afe8c745cbd62a94c3fad6a5314 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_orm.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: orm +doc_to_text: 'Given the text: ''{{text}}'', determine the correct intent from the + following list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Only + output one intent from the list.' +include: injongointent +task: injongointent_orm_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_sna.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_sna.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6cbf0105abe8b3157df4c6898e3873fb25beba28 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_sna.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: sna +doc_to_text: 'Given the text: ''{{text}}'', determine the correct intent from the + following list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Only + output one intent from the list.' +include: injongointent +task: injongointent_sna_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_sot.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_sot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ad3b4497f8dbbb678048dc9dfa8aa8894dd241fe --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_sot.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: sot +doc_to_text: 'Given the text: ''{{text}}'', determine the correct intent from the + following list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Only + output one intent from the list.' +include: injongointent +task: injongointent_sot_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fc3d797c05c018baf599dcee28bccc0dd5c5ab72 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_swa.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: swa +doc_to_text: 'Given the text: ''{{text}}'', determine the correct intent from the + following list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Only + output one intent from the list.' +include: injongointent +task: injongointent_swa_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..73fc61c7e6ab11fbf71d8842818f020f147b5443 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_twi.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: twi +doc_to_text: 'Given the text: ''{{text}}'', determine the correct intent from the + following list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Only + output one intent from the list.' +include: injongointent +task: injongointent_twi_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7d359d2f8e4cb91028705cc8c76842c9b5e78c3c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_wol.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: wol +doc_to_text: 'Given the text: ''{{text}}'', determine the correct intent from the + following list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Only + output one intent from the list.' +include: injongointent +task: injongointent_wol_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4d9c173aac2832665724358697d59d3bf8f38e56 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_xho.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: xho +doc_to_text: 'Given the text: ''{{text}}'', determine the correct intent from the + following list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Only + output one intent from the list.' +include: injongointent +task: injongointent_xho_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..682e01c12972c8b9a98e53711b307ebbf62676fa --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_yor.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: yor +doc_to_text: 'Given the text: ''{{text}}'', determine the correct intent from the + following list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Only + output one intent from the list.' +include: injongointent +task: injongointent_yor_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1d38a78141e1912ff995f6d87c0753f317ec6ad0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_zul.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: zul +doc_to_text: 'Given the text: ''{{text}}'', determine the correct intent from the + following list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Only + output one intent from the list.' +include: injongointent +task: injongointent_zul_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..3e735e2deb1f9c53152c072615aebe8ba3acb90b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/utils.py @@ -0,0 +1 @@ +from lm_eval.utils import weighted_f1_score diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent new file mode 100644 index 0000000000000000000000000000000000000000..dfcb82678a61524c08bfd2d7e2d2ec0a50330f27 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent @@ -0,0 +1,75 @@ +tag: +- injongointent_tasks +- injongointent_prompt_2 +dataset_path: masakhane/InjongoIntent +dataset_name: null +output_type: multiple_choice +validation_split: validation +test_split: test +fewshot_split: validation +doc_to_target: intent +doc_to_choice: + - alarm + - balance + - bill_balance + - book_flight + - book_hotel + - calendar_update + - cancel_reservation + - car_rental + - confirm_reservation + - cook_time + - exchange_rate + - food_last + - freeze_account + - ingredients_list + - interest_rate + - international_visa + - make_call + - meal_suggestion + - min_payment + - pay_bill + - pin_change + - play_music + - plug_type + - recipe + - restaurant_reservation + - restaurant_reviews + - restaurant_suggestion + - share_location + - shopping_list_update + - spending_history + - text + - time + - timezone + - transactions + - transfer + - translate + - travel_notification + - travel_suggestion + - update_playlist + - weather +should_decontaminate: true +doc_to_decontamination_query: text +metric_list: + - metric: f1 + aggregation: !function utils.weighted_f1_score + # aggregation: mean + average: weighted + hf_evaluate: true + higher_is_better: True + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" + - metric: acc + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7c1b2189ac768bef8c6263ccc41a003cd00ee6d8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_amh.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: amh +doc_to_text: 'Analyze the text: ''{{text}}''. Choose the most appropriate intent from + these options: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Respond + with only the selected intent.' +include: injongointent +task: injongointent_amh_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_eng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cc03705d5868b6c67b9973937f4bba59f08bd8c7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_eng.yaml @@ -0,0 +1,16 @@ +# Generated by utils.py +dataset_name: eng +validation_split: train +test_split: test +fewshot_split: train +doc_to_text: 'Analyze the text: ''{{text}}''. Choose the most appropriate intent from + these options: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Respond + with only the selected intent.' +include: injongointent +task: injongointent_eng_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_ewe.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_ewe.yaml new file mode 100644 index 0000000000000000000000000000000000000000..58eb914472609d29836edb92f5c676deac50166a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_ewe.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: ewe +doc_to_text: 'Analyze the text: ''{{text}}''. Choose the most appropriate intent from + these options: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Respond + with only the selected intent.' +include: injongointent +task: injongointent_ewe_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e7745369ac6af719ea71e9f4e3032bcc7b66a8ac --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_hau.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: hau +doc_to_text: 'Analyze the text: ''{{text}}''. Choose the most appropriate intent from + these options: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Respond + with only the selected intent.' +include: injongointent +task: injongointent_hau_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b47052d71829c170849a07957a0412b8f21bddb2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_ibo.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: ibo +doc_to_text: 'Analyze the text: ''{{text}}''. Choose the most appropriate intent from + these options: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Respond + with only the selected intent.' +include: injongointent +task: injongointent_ibo_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..457935674bd84bb39e96d8ba000497430963f983 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_kin.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: kin +doc_to_text: 'Analyze the text: ''{{text}}''. Choose the most appropriate intent from + these options: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Respond + with only the selected intent.' +include: injongointent +task: injongointent_kin_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_lin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_lin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..54e7fcb71bb91ce4272dbd4723b7e2059da71c12 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_lin.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: lin +doc_to_text: 'Analyze the text: ''{{text}}''. Choose the most appropriate intent from + these options: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Respond + with only the selected intent.' +include: injongointent +task: injongointent_lin_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..96aa42fc9e2bed859ca386091267ca48b9566399 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_lug.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: lug +doc_to_text: 'Analyze the text: ''{{text}}''. Choose the most appropriate intent from + these options: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Respond + with only the selected intent.' +include: injongointent +task: injongointent_lug_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_orm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_orm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..872f96c542cd697bef3b0f92294e9247cedb458a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_orm.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: orm +doc_to_text: 'Analyze the text: ''{{text}}''. Choose the most appropriate intent from + these options: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Respond + with only the selected intent.' +include: injongointent +task: injongointent_orm_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_sna.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_sna.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a62dfe34a4611319a56bc06f7bb01c447ac7ad6f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_sna.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: sna +doc_to_text: 'Analyze the text: ''{{text}}''. Choose the most appropriate intent from + these options: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Respond + with only the selected intent.' +include: injongointent +task: injongointent_sna_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_sot.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_sot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e9ca6a5675529c7bb3701d332c20eb1c2af19d53 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_sot.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: sot +doc_to_text: 'Analyze the text: ''{{text}}''. Choose the most appropriate intent from + these options: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Respond + with only the selected intent.' +include: injongointent +task: injongointent_sot_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..339f66ac8d569245f6e8cab2f3456fcea24713d4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_swa.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: swa +doc_to_text: 'Analyze the text: ''{{text}}''. Choose the most appropriate intent from + these options: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Respond + with only the selected intent.' +include: injongointent +task: injongointent_swa_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b758bce0493c722c0d995cdbf8cc5e4b409e4af9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_twi.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: twi +doc_to_text: 'Analyze the text: ''{{text}}''. Choose the most appropriate intent from + these options: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Respond + with only the selected intent.' +include: injongointent +task: injongointent_twi_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9b573a444b22b2e34ed10ee9f75a914f16177bfc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_wol.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: wol +doc_to_text: 'Analyze the text: ''{{text}}''. Choose the most appropriate intent from + these options: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Respond + with only the selected intent.' +include: injongointent +task: injongointent_wol_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f2c02205fdb7b9701096089c26d6813fff802dd7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_xho.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: xho +doc_to_text: 'Analyze the text: ''{{text}}''. Choose the most appropriate intent from + these options: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Respond + with only the selected intent.' +include: injongointent +task: injongointent_xho_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4c821736809b4a37d94dedc82e6943594585ce35 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_yor.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: yor +doc_to_text: 'Analyze the text: ''{{text}}''. Choose the most appropriate intent from + these options: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Respond + with only the selected intent.' +include: injongointent +task: injongointent_yor_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c8a541b66bfb14f692c1230871f2482452eb7347 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_zul.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: zul +doc_to_text: 'Analyze the text: ''{{text}}''. Choose the most appropriate intent from + these options: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Respond + with only the selected intent.' +include: injongointent +task: injongointent_zul_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..3e735e2deb1f9c53152c072615aebe8ba3acb90b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/utils.py @@ -0,0 +1 @@ +from lm_eval.utils import weighted_f1_score diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent new file mode 100644 index 0000000000000000000000000000000000000000..afdf43cfc10b75238debbd5dbab36ac493872025 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent @@ -0,0 +1,75 @@ +tag: +- injongointent_tasks +- injongointent_prompt_3 +dataset_path: masakhane/InjongoIntent +dataset_name: null +output_type: multiple_choice +validation_split: validation +test_split: test +fewshot_split: validation +doc_to_target: intent +doc_to_choice: + - alarm + - balance + - bill_balance + - book_flight + - book_hotel + - calendar_update + - cancel_reservation + - car_rental + - confirm_reservation + - cook_time + - exchange_rate + - food_last + - freeze_account + - ingredients_list + - interest_rate + - international_visa + - make_call + - meal_suggestion + - min_payment + - pay_bill + - pin_change + - play_music + - plug_type + - recipe + - restaurant_reservation + - restaurant_reviews + - restaurant_suggestion + - share_location + - shopping_list_update + - spending_history + - text + - time + - timezone + - transactions + - transfer + - translate + - travel_notification + - travel_suggestion + - update_playlist + - weather +should_decontaminate: true +doc_to_decontamination_query: text +metric_list: + - metric: f1 + aggregation: !function utils.weighted_f1_score + # aggregation: mean + average: weighted + hf_evaluate: true + higher_is_better: True + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" + - metric: acc + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7bd62c5b00eedcfb1cd254ae679426955cacb8a0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_amh.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: amh +doc_to_text: 'You are a linguistic analyst trained to understand user intent. Based + on the text: ''{{text}}'', choose the intent that best matches from this list: [alarm, + balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather]. Return only the intent.' +include: injongointent +task: injongointent_amh_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_eng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..258f0cfab4a4dc9ee8f7f8baceb2c2bac35c0c02 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_eng.yaml @@ -0,0 +1,16 @@ +# Generated by utils.py +dataset_name: eng +validation_split: train +test_split: test +fewshot_split: train +doc_to_text: 'You are a linguistic analyst trained to understand user intent. Based + on the text: ''{{text}}'', choose the intent that best matches from this list: [alarm, + balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather]. Return only the intent.' +include: injongointent +task: injongointent_eng_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_ewe.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_ewe.yaml new file mode 100644 index 0000000000000000000000000000000000000000..12688cc9d97dcacc50419a91707ac2f5fa47363b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_ewe.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: ewe +doc_to_text: 'You are a linguistic analyst trained to understand user intent. Based + on the text: ''{{text}}'', choose the intent that best matches from this list: [alarm, + balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather]. Return only the intent.' +include: injongointent +task: injongointent_ewe_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8414a09bae932411fd51cfbcf9d5efbbb2d93a45 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_hau.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: hau +doc_to_text: 'You are a linguistic analyst trained to understand user intent. Based + on the text: ''{{text}}'', choose the intent that best matches from this list: [alarm, + balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather]. Return only the intent.' +include: injongointent +task: injongointent_hau_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8f254438388e95b551fd9344758851b0a9fc8768 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_ibo.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: ibo +doc_to_text: 'You are a linguistic analyst trained to understand user intent. Based + on the text: ''{{text}}'', choose the intent that best matches from this list: [alarm, + balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather]. Return only the intent.' +include: injongointent +task: injongointent_ibo_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b946cf004f775642ad8f0b9d5995820b97187f0a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_kin.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: kin +doc_to_text: 'You are a linguistic analyst trained to understand user intent. Based + on the text: ''{{text}}'', choose the intent that best matches from this list: [alarm, + balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather]. Return only the intent.' +include: injongointent +task: injongointent_kin_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_lin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_lin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4a485d5ce807f0f9200ccdf535d5427697561bbd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_lin.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: lin +doc_to_text: 'You are a linguistic analyst trained to understand user intent. Based + on the text: ''{{text}}'', choose the intent that best matches from this list: [alarm, + balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather]. Return only the intent.' +include: injongointent +task: injongointent_lin_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..71376ec3f8a3ae742cc694ea0d19c57ad9187b0b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_lug.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: lug +doc_to_text: 'You are a linguistic analyst trained to understand user intent. Based + on the text: ''{{text}}'', choose the intent that best matches from this list: [alarm, + balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather]. Return only the intent.' +include: injongointent +task: injongointent_lug_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_orm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_orm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..706f3a908a221dd3bc876adae4ddc40b6b4cb6ea --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_orm.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: orm +doc_to_text: 'You are a linguistic analyst trained to understand user intent. Based + on the text: ''{{text}}'', choose the intent that best matches from this list: [alarm, + balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather]. Return only the intent.' +include: injongointent +task: injongointent_orm_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_sna.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_sna.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f4aca73782fbcc8516983e588271bb336f674f2b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_sna.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: sna +doc_to_text: 'You are a linguistic analyst trained to understand user intent. Based + on the text: ''{{text}}'', choose the intent that best matches from this list: [alarm, + balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather]. Return only the intent.' +include: injongointent +task: injongointent_sna_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_sot.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_sot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..57e27afab1edb9509d4f9901bea7fc114f249a09 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_sot.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: sot +doc_to_text: 'You are a linguistic analyst trained to understand user intent. Based + on the text: ''{{text}}'', choose the intent that best matches from this list: [alarm, + balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather]. Return only the intent.' +include: injongointent +task: injongointent_sot_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6cb4886d1f1d9ba9c87cbc78a436dad83d72487e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_swa.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: swa +doc_to_text: 'You are a linguistic analyst trained to understand user intent. Based + on the text: ''{{text}}'', choose the intent that best matches from this list: [alarm, + balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather]. Return only the intent.' +include: injongointent +task: injongointent_swa_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e8623bf33bcc89d2b25390b487617159263327f6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_twi.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: twi +doc_to_text: 'You are a linguistic analyst trained to understand user intent. Based + on the text: ''{{text}}'', choose the intent that best matches from this list: [alarm, + balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather]. Return only the intent.' +include: injongointent +task: injongointent_twi_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..afc3cf4a907143eb7a13dca4c4783a0c715e1524 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_wol.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: wol +doc_to_text: 'You are a linguistic analyst trained to understand user intent. Based + on the text: ''{{text}}'', choose the intent that best matches from this list: [alarm, + balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather]. Return only the intent.' +include: injongointent +task: injongointent_wol_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9f41aa561fbd6a6da39e9f90d93c4fa909ffbf72 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_xho.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: xho +doc_to_text: 'You are a linguistic analyst trained to understand user intent. Based + on the text: ''{{text}}'', choose the intent that best matches from this list: [alarm, + balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather]. Return only the intent.' +include: injongointent +task: injongointent_xho_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3a5d5686de20ff32a1a4207ea2711d2176464df5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_yor.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: yor +doc_to_text: 'You are a linguistic analyst trained to understand user intent. Based + on the text: ''{{text}}'', choose the intent that best matches from this list: [alarm, + balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather]. Return only the intent.' +include: injongointent +task: injongointent_yor_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f857ff065b13b0eb050107b884a7b57f795dc779 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/injongointent_zul.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: zul +doc_to_text: 'You are a linguistic analyst trained to understand user intent. Based + on the text: ''{{text}}'', choose the intent that best matches from this list: [alarm, + balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather]. Return only the intent.' +include: injongointent +task: injongointent_zul_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..3e735e2deb1f9c53152c072615aebe8ba3acb90b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_3/utils.py @@ -0,0 +1 @@ +from lm_eval.utils import weighted_f1_score diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent new file mode 100644 index 0000000000000000000000000000000000000000..5d5c05ae113bdcef17764decefc59f335ddb3ba3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent @@ -0,0 +1,75 @@ +tag: +- injongointent_tasks +- injongointent_prompt_4 +dataset_path: masakhane/InjongoIntent +dataset_name: null +output_type: multiple_choice +validation_split: validation +test_split: test +fewshot_split: validation +doc_to_target: intent +doc_to_choice: + - alarm + - balance + - bill_balance + - book_flight + - book_hotel + - calendar_update + - cancel_reservation + - car_rental + - confirm_reservation + - cook_time + - exchange_rate + - food_last + - freeze_account + - ingredients_list + - interest_rate + - international_visa + - make_call + - meal_suggestion + - min_payment + - pay_bill + - pin_change + - play_music + - plug_type + - recipe + - restaurant_reservation + - restaurant_reviews + - restaurant_suggestion + - share_location + - shopping_list_update + - spending_history + - text + - time + - timezone + - transactions + - transfer + - translate + - travel_notification + - travel_suggestion + - update_playlist + - weather +should_decontaminate: true +doc_to_decontamination_query: text +metric_list: + - metric: f1 + aggregation: !function utils.weighted_f1_score + # aggregation: mean + average: weighted + hf_evaluate: true + higher_is_better: True + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" + - metric: acc + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..aa14ee5b178f7577c036039b089678bcfa697a04 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_amh.yaml @@ -0,0 +1,14 @@ +# Generated by utils.py +dataset_name: amh +doc_to_text: 'You are a Amharic linguistic analyst trained to understand Amharic user + intent. Based on the Amharictext: ''{{text}}'', choose the intent that best matches + from this list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Return + only the intent.' +include: injongointent +task: injongointent_amh_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_eng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..853e64965251e37e59efe72554557b2a378e358f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_eng.yaml @@ -0,0 +1,17 @@ +# Generated by utils.py +dataset_name: eng +validation_split: train +test_split: test +fewshot_split: train +doc_to_text: 'You are a English linguistic analyst trained to understand English user + intent. Based on the English text: ''{{text}}'', choose the intent that best matches + from this list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Return + only the intent.' +include: injongointent +task: injongointent_eng_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_ewe.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_ewe.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f61a3db57d2bcf2af8cf30dc65ac08b896013fbe --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_ewe.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: ewe +doc_to_text: 'You are a Ewe linguistic analyst trained to understand Ewe user intent. + Based on the Ewetext: ''{{text}}'', choose the intent that best matches from this + list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather]. Return only the intent.' +include: injongointent +task: injongointent_ewe_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fdef34cb847fbc8008ed496826bc2cb79361a546 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_hau.yaml @@ -0,0 +1,14 @@ +# Generated by utils.py +dataset_name: hau +doc_to_text: 'You are a Hausa linguistic analyst trained to understand Hausa user + intent. Based on the Hausatext: ''{{text}}'', choose the intent that best matches + from this list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Return + only the intent.' +include: injongointent +task: injongointent_hau_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..23b59831ed97b56ed983d26871d155b1f72a176b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_ibo.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: ibo +doc_to_text: 'You are a Igbo linguistic analyst trained to understand Igbo user intent. + Based on the Igbotext: ''{{text}}'', choose the intent that best matches from this + list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather]. Return only the intent.' +include: injongointent +task: injongointent_ibo_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..28f05aeb00423bb80f2f070763f3f22345ae4776 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_kin.yaml @@ -0,0 +1,14 @@ +# Generated by utils.py +dataset_name: kin +doc_to_text: 'You are a Kinyarwanda linguistic analyst trained to understand Kinyarwanda + user intent. Based on the Kinyarwandatext: ''{{text}}'', choose the intent that + best matches from this list: [alarm, balance, bill_balance, book_flight, book_hotel, + calendar_update, cancel_reservation, car_rental, confirm_reservation, cook_time, + exchange_rate, food_last, freeze_account, ingredients_list, interest_rate, international_visa, + make_call, meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, + recipe, restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Return + only the intent.' +include: injongointent +task: injongointent_kin_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_lin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_lin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..df991d89146112468bae83c7c5fe87eef307dc49 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_lin.yaml @@ -0,0 +1,14 @@ +# Generated by utils.py +dataset_name: lin +doc_to_text: 'You are a Lingala linguistic analyst trained to understand Lingala user + intent. Based on the Lingalatext: ''{{text}}'', choose the intent that best matches + from this list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Return + only the intent.' +include: injongointent +task: injongointent_lin_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c1abb66edb31a6a76e987e665f91f630fe8d3416 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_lug.yaml @@ -0,0 +1,14 @@ +# Generated by utils.py +dataset_name: lug +doc_to_text: 'You are a Luganda linguistic analyst trained to understand Luganda user + intent. Based on the Lugandatext: ''{{text}}'', choose the intent that best matches + from this list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Return + only the intent.' +include: injongointent +task: injongointent_lug_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_orm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_orm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..195ff4a232e782d38bae93b33648535934ff07e4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_orm.yaml @@ -0,0 +1,14 @@ +# Generated by utils.py +dataset_name: orm +doc_to_text: 'You are a Oromo linguistic analyst trained to understand Oromo user + intent. Based on the Oromotext: ''{{text}}'', choose the intent that best matches + from this list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Return + only the intent.' +include: injongointent +task: injongointent_orm_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_sna.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_sna.yaml new file mode 100644 index 0000000000000000000000000000000000000000..23d066c3d8c8b41e7184f867ba492d58a8736d82 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_sna.yaml @@ -0,0 +1,14 @@ +# Generated by utils.py +dataset_name: sna +doc_to_text: 'You are a Shona linguistic analyst trained to understand Shona user + intent. Based on the Shonatext: ''{{text}}'', choose the intent that best matches + from this list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Return + only the intent.' +include: injongointent +task: injongointent_sna_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_sot.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_sot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..82102a21e2e7bb0a3d57a6fff4dd678b587d1d95 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_sot.yaml @@ -0,0 +1,14 @@ +# Generated by utils.py +dataset_name: sot +doc_to_text: 'You are a Sotho linguistic analyst trained to understand Sotho user + intent. Based on the Sothotext: ''{{text}}'', choose the intent that best matches + from this list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Return + only the intent.' +include: injongointent +task: injongointent_sot_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..031ffbb40ceba3233e5f396e38a523f2bca81ad9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_swa.yaml @@ -0,0 +1,14 @@ +# Generated by utils.py +dataset_name: swa +doc_to_text: 'You are a Swahili linguistic analyst trained to understand Swahili user + intent. Based on the Swahilitext: ''{{text}}'', choose the intent that best matches + from this list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Return + only the intent.' +include: injongointent +task: injongointent_swa_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d773a1756e17eb5cd21b3bc550b5847acaae671c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_xho.yaml @@ -0,0 +1,14 @@ +# Generated by utils.py +dataset_name: xho +doc_to_text: 'You are a Xhosa linguistic analyst trained to understand Xhosa user + intent. Based on the Xhosatext: ''{{text}}'', choose the intent that best matches + from this list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Return + only the intent.' +include: injongointent +task: injongointent_xho_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..af01d9f3e8efc1d57003194dd0201dd76bd76fbf --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/injongointent_yor.yaml @@ -0,0 +1,14 @@ +# Generated by utils.py +dataset_name: yor +doc_to_text: 'You are a Yoruba linguistic analyst trained to understand Yoruba user + intent. Based on the Yorubatext: ''{{text}}'', choose the intent that best matches + from this list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Return + only the intent.' +include: injongointent +task: injongointent_yor_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..3e735e2deb1f9c53152c072615aebe8ba3acb90b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_4/utils.py @@ -0,0 +1 @@ +from lm_eval.utils import weighted_f1_score diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent new file mode 100644 index 0000000000000000000000000000000000000000..0012857bdaa787ad8bf9ba345330844c8b266a8e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent @@ -0,0 +1,75 @@ +tag: +- injongointent_tasks +- injongointent_prompt_5 +dataset_path: masakhane/InjongoIntent +dataset_name: null +output_type: multiple_choice +validation_split: validation +test_split: test +fewshot_split: validation +doc_to_target: intent +doc_to_choice: + - alarm + - balance + - bill_balance + - book_flight + - book_hotel + - calendar_update + - cancel_reservation + - car_rental + - confirm_reservation + - cook_time + - exchange_rate + - food_last + - freeze_account + - ingredients_list + - interest_rate + - international_visa + - make_call + - meal_suggestion + - min_payment + - pay_bill + - pin_change + - play_music + - plug_type + - recipe + - restaurant_reservation + - restaurant_reviews + - restaurant_suggestion + - share_location + - shopping_list_update + - spending_history + - text + - time + - timezone + - transactions + - transfer + - translate + - travel_notification + - travel_suggestion + - update_playlist + - weather +should_decontaminate: true +doc_to_decontamination_query: text +metric_list: + - metric: f1 + aggregation: !function utils.weighted_f1_score + # aggregation: mean + average: weighted + hf_evaluate: true + higher_is_better: True + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" + - metric: acc + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2a6623a9387a98d2d21791f51d0a8690b756c9a4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_amh.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: amh +doc_to_text: 'The following text is in Amharic: ''{{text}}''. Given the list of intents: + [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather], identify the intent expressed in the + text. Return only the identified intent.' +include: injongointent +task: injongointent_amh_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_eng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4dcbebbbd00cf8a3abb7f5e848d3a6b8520d46a3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_eng.yaml @@ -0,0 +1,16 @@ +# Generated by utils.py +dataset_name: eng +validation_split: train +test_split: test +fewshot_split: train +doc_to_text: 'The following text is in English: ''{{text}}''. Given the list of intents: + [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather], identify the intent expressed in the + text. Return only the identified intent.' +include: injongointent +task: injongointent_eng_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_ewe.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_ewe.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0cab84252dcbfdc4d1eefcd2f4aa36b031edd7ba --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_ewe.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: ewe +doc_to_text: 'The following text is in Ewe: ''{{text}}''. Given the list of intents: + [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather], identify the intent expressed in the + text. Return only the identified intent.' +include: injongointent +task: injongointent_ewe_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b6275db8383eddada828f7cb1963d554b4f8d658 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_hau.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: hau +doc_to_text: 'The following text is in Hausa: ''{{text}}''. Given the list of intents: + [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather], identify the intent expressed in the + text. Return only the identified intent.' +include: injongointent +task: injongointent_hau_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..348535c679af3c08da162c812fbfd700557e8326 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_kin.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: kin +doc_to_text: 'The following text is in Kinyarwanda: ''{{text}}''. Given the list of + intents: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather], identify + the intent expressed in the text. Return only the identified intent.' +include: injongointent +task: injongointent_kin_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_lin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_lin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..75bbf4ec5935505c554d5c7b58934188d1591532 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_lin.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: lin +doc_to_text: 'The following text is in Lingala: ''{{text}}''. Given the list of intents: + [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather], identify the intent expressed in the + text. Return only the identified intent.' +include: injongointent +task: injongointent_lin_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..49b7f6faddd5ad59c079de1a1986e33ef4a29311 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_lug.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: lug +doc_to_text: 'The following text is in Luganda: ''{{text}}''. Given the list of intents: + [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather], identify the intent expressed in the + text. Return only the identified intent.' +include: injongointent +task: injongointent_lug_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_sna.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_sna.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8931b65ce15999ea1215cf420a671baed32a51c2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_5/injongointent_sna.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: sna +doc_to_text: 'The following text is in Shona: ''{{text}}''. Given the list of intents: + [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, cancel_reservation, + car_rental, confirm_reservation, cook_time, exchange_rate, food_last, freeze_account, + ingredients_list, interest_rate, international_visa, make_call, meal_suggestion, + min_payment, pay_bill, pin_change, play_music, plug_type, recipe, restaurant_reservation, + restaurant_reviews, restaurant_suggestion, share_location, shopping_list_update, + spending_history, text, time, timezone, transactions, transfer, translate, travel_notification, + travel_suggestion, update_playlist, weather], identify the intent expressed in the + text. Return only the identified intent.' +include: injongointent +task: injongointent_sna_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/README.md b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/README.md new file mode 100644 index 0000000000000000000000000000000000000000..9e7eea17598d29defff07bb37c4f47efaa446547 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/README.md @@ -0,0 +1,73 @@ +# + +## Paper +Title: `A Few Thousand Translations Go a Long Way! Leveraging Pre-trained Models for African News Translation` + +Paper Link: https://aclanthology.org/2022.naacl-main.223/ + +## Abstract +>Recent advances in the pre-training of language models leverage large-scale datasets to create multilingual models. However, low-resource languages are mostly left out in these datasets. This is primarily because many widely spoken languages are not well represented on the web and therefore excluded from the large-scale crawls used to create datasets. Furthermore, downstream users of these models are restricted to the selection of languages originally chosen for pre-training. This work investigates how to optimally leverage existing pre-trained models to create low-resource translation systems for 16 African languages. We focus on two questions: 1) How can pre-trained models be used for languages not included in the initial pre-training? and 2) How can the resulting translation models effectively transfer to new domains? To answer these questions, we create a new African news corpus covering 16 languages, of which eight languages are not part of any existing evaluation dataset. We demonstrate that the most effective strategy for transferring both to additional languages and to additional domains is to fine-tune large pre-trained models on small quantities of high-quality translation data. + +HomePage: https://github.com/masakhane-io/lafand-mt + +### Citation + +``` +@inproceedings{adelani-etal-2022-thousand, + title = "A Few Thousand Translations Go a Long Way! Leveraging Pre-trained Models for {A}frican News Translation", + author = "Adelani, David and + Alabi, Jesujoba and + Fan, Angela and + Kreutzer, Julia and + Shen, Xiaoyu and + Reid, Machel and + Ruiter, Dana and + Klakow, Dietrich and + Nabende, Peter and + Chang, Ernie and + Gwadabe, Tajuddeen and + Sackey, Freshia and + Dossou, Bonaventure F. P. and + Emezue, Chris and + Leong, Colin and + Beukman, Michael and + Muhammad, Shamsuddeen and + Jarso, Guyo and + Yousuf, Oreen and + Niyongabo Rubungo, Andre and + Hacheme, Gilles and + Wairagala, Eric Peter and + Nasir, Muhammad Umair and + Ajibade, Benjamin and + Ajayi, Tunde and + Gitau, Yvonne and + Abbott, Jade and + Ahmed, Mohamed and + Ochieng, Millicent and + Aremu, Anuoluwapo and + Ogayo, Perez and + Mukiibi, Jonathan and + Ouoba Kabore, Fatoumata and + Kalipe, Godson and + Mbaye, Derguene and + Tapo, Allahsera Auguste and + Memdjokam Koagne, Victoire and + Munkoh-Buabeng, Edwin and + Wagner, Valencia and + Abdulmumin, Idris and + Awokoya, Ayodele and + Buzaaba, Happy and + Sibanda, Blessing and + Bukula, Andiswa and + Manthalu, Sam", + booktitle = "Proceedings of the 2022 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies", + month = jul, + year = "2022", + address = "Seattle, United States", + publisher = "Association for Computational Linguistics", + url = "https://aclanthology.org/2022.naacl-main.223", + doi = "10.18653/v1/2022.naacl-main.223", + pages = "3053--3070", + abstract = "Recent advances in the pre-training for language models leverage large-scale datasets to create multilingual models. However, low-resource languages are mostly left out in these datasets. This is primarily because many widely spoken languages that are not well represented on the web and therefore excluded from the large-scale crawls for datasets. Furthermore, downstream users of these models are restricted to the selection of languages originally chosen for pre-training. This work investigates how to optimally leverage existing pre-trained models to create low-resource translation systems for 16 African languages. We focus on two questions: 1) How can pre-trained models be used for languages not included in the initial pretraining? and 2) How can the resulting translation models effectively transfer to new domains? To answer these questions, we create a novel African news corpus covering 16 languages, of which eight languages are not part of any existing evaluation dataset. We demonstrate that the most effective strategy for transferring both additional languages and additional domains is to leverage small quantities of high-quality translation data to fine-tune large pre-trained models.", +} +``` diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/gen_utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/gen_utils.py new file mode 100644 index 0000000000000000000000000000000000000000..c260a321a419b3013545b738850af5f796a1bc32 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/gen_utils.py @@ -0,0 +1,147 @@ +import argparse +import os + +import yaml + + +class FunctionTag: + def __init__(self, value): + self.value = value + + +def prompt_func(mode, lang, lang_dict): + language_column_name = f"{lang}_text" + prompt_map = { + "prompt_1": "You are an advanced Translator, a specialized assistant designed to translate documents from " + f"{lang_dict[lang]} into English. \nYour main goal is to ensure translations are grammatically " + f"correct and human-oriented. \n{lang_dict[lang]}: {{{{{language_column_name}}}}} \nEnglish: ", + "prompt_1_reverse": "You are an advanced Translator, a specialized assistant designed to translate documents " + f"from English into {lang_dict[lang]}. \nYour main goal is to ensure translations are " + f"grammatically correct and human-oriented. " + f"\nEnglish: {{eng_text}} \n{lang_dict[lang]}: ", + "prompt_2": f"{lang_dict[lang]} sentence: {{{{{language_column_name}}}}} \nEnglish sentence: ", + "prompt_2_reverse": "English sentence: {{eng_text}} " + f"\n{lang_dict[lang]} sentence: ", + "prompt_3": f"You are a translation expert. Translate the following {lang_dict[lang]} sentences to English \n" + f"{lang_dict[lang]} sentence: {{{{{language_column_name}}}}}\nEnglish sentence: ", + "prompt_3_reverse": f"You are a translation expert. Translate the following English sentences to " + f"{lang_dict[lang]} " + "\nEnglish sentence: {{eng_text}} " + f"\n{lang_dict[lang]} sentence: ", + } + return prompt_map[mode] + + +def gen_lang_yamls(output_dir: str, overwrite: bool, mode: str, reverse: bool) -> None: + """ + Generate a yaml file for each language. + + :param output_dir: The directory to output the files to. + :param overwrite: Whether to overwrite files if they already exist. + """ + err = [] + languages = { + "amh": "Amharic", + "bam": "Bambara", + "bbj": "Gbomala", + "ewe": "Ewe", + "fon": "Fon", + "hau": "Hausa", + "ibo": "Igbo", + "kin": "Kinyarwanda", + "lug": "Luganda", + "luo": "Luo", + "mos": "Mossi", + "nya": "Chichewa", + "pcm": "Nigerian Pidgin", + "sna": "Shona", + "swa": "Swahili", + "tsn": "Setswana", + "twi": "Twi", + "wol": "Wolof", + "xho": "Xhosa", + "yor": "Yoruba", + "zul": "Zulu", + } + + french_langs = ["bam", "bbj", "ewe", "fon", "wol", "mos"] + + for lang in languages.keys(): + try: + norm_lang = f"{lang}-en" if lang not in french_langs else f"{lang}-fr" + reverse_lang = f"en-{lang}" if lang not in french_langs else f"fr-{lang}" + dataset_name = norm_lang if reverse else reverse_lang + file_name = f"mafand_{dataset_name}.yaml" + task_name = f"mafand_{dataset_name}_{mode}" + yaml_template = "mafand" + yaml_details = { + "include": yaml_template, + "task": task_name, + "dataset_name": reverse_lang, + } + file_dir = ( + f"{output_dir}/{mode}/african-english" + if reverse + else f"{output_dir}/{mode}/english-african" + ) + os.makedirs(file_dir, exist_ok=True) + with open( + f"{file_dir}/{file_name}", + "w" if overwrite else "x", + encoding="utf8", + ) as f: + f.write("# Generated by utils.py\n") + yaml.dump( + yaml_details, + f, + allow_unicode=True, + ) + + except FileExistsError: + err.append(file_name) + + if len(err) > 0: + raise FileExistsError( + "Files were not created because they already exist (use --overwrite flag):" + f" {', '.join(err)}" + ) + + +def main() -> None: + """Parse CLI args and generate language-specific yaml files.""" + parser = argparse.ArgumentParser() + parser.add_argument( + "--overwrite", + default=True, + action="store_true", + help="Overwrite files if they already exist", + ) + parser.add_argument( + "--output-dir", + default="./", + help="Directory to write yaml files to", + ) + parser.add_argument( + "--mode", + default="prompt_3", + choices=["prompt_1", "prompt_2", "prompt_3"], + help="Prompt number", + ) + parser.add_argument( + "--reverse", + default=True, + choices=[True, False], + help="Reverse the translation direction", + ) + args = parser.parse_args() + + gen_lang_yamls( + output_dir=args.output_dir, + overwrite=args.overwrite, + mode=args.mode, + reverse=args.reverse, + ) + + +if __name__ == "__main__": + main() diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_bam-fr.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_bam-fr.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dbc612ac327d46f46c4df459d558c8429d2089dd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_bam-fr.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fr-bam +include: mafand +task: mafand_bam-fr_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_hau-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_hau-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3b84d9cecabfd92350a9ab63585f38fcfff328d7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_1/african-english/mafand_hau-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-hau +include: mafand +task: mafand_hau-en_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_xho-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_xho-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fd1960d0efbd42ba25132c938944692fbf63b92f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_xho-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-xho +include: mafand +task: mafand_xho-en_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_yor-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_yor-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cb7241ad2cc2fc2f889bab56c4ad3e233a2d2165 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_yor-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-yor +include: mafand +task: mafand_yor-en_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-pcm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-pcm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..73229c4f18ac24014cf15f161454910a921e1d02 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-pcm.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-pcm +include: mafand +task: mafand_en-pcm_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-sna.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-sna.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ac37187170a032226e1b33dd87fb09c7b9952cf1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-sna.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-sna +include: mafand +task: mafand_en-sna_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5502ffa4e5547433d7cba66191456178c5d3377e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-twi.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-twi +include: mafand +task: mafand_en-twi_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..89c070c7c5e9f6d53fde695f156408f37038242d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-yor.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-yor +include: mafand +task: mafand_en-yor_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e54725404ba481e668156618a97f0b11d1a1fb31 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-zul.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-zul +include: mafand +task: mafand_en-zul_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_fr-ewe.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_fr-ewe.yaml new file mode 100644 index 0000000000000000000000000000000000000000..29d4214cf433e56e8f2d479299cf413e2d211d34 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_fr-ewe.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fr-ewe +include: mafand +task: mafand_fr-ewe_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_fr-fon.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_fr-fon.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9710db5b341b91663a782945da57a3a21ae3c1ae --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_fr-fon.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fr-fon +include: mafand +task: mafand_fr-fon_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_fr-wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_fr-wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3740ca9b73557e472f867b7b5c33131a441e18fa --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_fr-wol.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fr-wol +include: mafand +task: mafand_fr-wol_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..0df3a329824d44fa94eb830ae943fa30dd32bab7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/utils.py @@ -0,0 +1,121 @@ +languages = { + "amh": "Amharic", + "bam": "Bambara", + "bbj": "Gbomala", + "ewe": "Ewe", + "fon": "Fon", + "hau": "Hausa", + "ibo": "Igbo", + "kin": "Kinyarwanda", + "lug": "Luganda", + "luo": "Luo", + "mos": "Mossi", + "nya": "Chichewa", + "pcm": "Nigerian Pidgin", + "sna": "Shona", + "swa": "Swahili", + "tsn": "Setswana", + "twi": "Twi", + "wol": "Wolof", + "xho": "Xhosa", + "yor": "Yoruba", + "zul": "Zulu", +} + + +def get_target(doc): + target = ( + doc["translation"]["en"] + if "en" in doc["translation"].keys() + else doc["translation"]["fr"] + ) + return target + + +def get_target_reverse(doc): + target_key = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + target = doc["translation"][target_key] + return target + + +def create_text_prompt_1(doc): + source_key = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + source_sentence = doc["translation"][source_key] + source_lang = "English" if "en" in doc["translation"].keys() else "French" + prompt = ( + "You are an advanced Translator, a specialized assistant designed to translate documents from " + f"{languages[source_key]} into {source_lang}. \nYour main goal is to ensure translations are grammatically " + f"correct and human-oriented. \n{languages[source_key]}: {source_sentence} \n{source_lang}: " + ) + return prompt + + +def create_reverse_prompt_1(doc): + target_lang = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + source_key = "en" if "en" in doc["translation"].keys() else "fr" + source_lang = "English" if source_key == "en" else "French" + source_sentence = doc["translation"][source_key] + prompt = ( + "You are an advanced Translator, a specialized assistant designed to translate documents from " + f"{source_lang} into {languages[target_lang]}. \nYour main goal is to ensure translations are " + f"grammatically correct and human-oriented. \n{source_lang}: {source_sentence} \n{languages[target_lang]}: " + ) + return prompt + + +def create_text_prompt_2(doc): + source_key = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + source_sentence = doc["translation"][source_key] + source_lang = "English" if "en" in doc["translation"].keys() else "French" + prompt = ( + f"{languages[source_key]} sentence: {source_sentence} \n{source_lang} sentence: ", + ) + return prompt + + +def create_reverse_prompt_2(doc): + target_lang = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + source_key = "en" if "en" in doc["translation"].keys() else "fr" + source_lang = "English" if source_key == "en" else "French" + source_sentence = doc["translation"][source_key] + prompt = ( + f"{source_lang} sentence: {source_sentence} \n{languages[target_lang]} sentence: \n", + ) + return prompt + + +def create_text_prompt_3(doc): + source_key = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + source_sentence = doc["translation"][source_key] + source_lang = "English" if "en" in doc["translation"].keys() else "French" + prompt = ( + f"You are a translation expert. Translate the following {languages[source_key]} sentences " + f"to {source_lang}. \n{languages[source_key]} sentence: {source_sentence}\n{source_lang} sentence: " + ) + return prompt + + +def create_reverse_prompt_3(doc): + target_lang = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + source_key = "en" if "en" in doc["translation"].keys() else "fr" + source_lang = "English" if source_key == "en" else "French" + source_sentence = doc["translation"][source_key] + prompt = ( + f"You are a translation expert. Translate the following {source_lang} sentence into {languages[target_lang]}\n" + f"{source_lang} sentence: {source_sentence}\n{languages[target_lang]} sentence: " + ) + return prompt diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_amh-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_amh-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..856318b5ad1b19fde25dd12ee3a2fc712b053b1a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_amh-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-amh +include: mafand +task: mafand_amh-en_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_bbj-fr.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_bbj-fr.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1170c266f4b14550499e7ddf930c50714560ec66 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_bbj-fr.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fr-bbj +include: mafand +task: mafand_bbj-fr_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_ewe-fr.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_ewe-fr.yaml new file mode 100644 index 0000000000000000000000000000000000000000..39a345cb6b6b86684e67978bd52c0e3071302a04 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_ewe-fr.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fr-ewe +include: mafand +task: mafand_ewe-fr_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_fon-fr.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_fon-fr.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1b464bb913f4888d825a2ab2aa2d566ef5d422d4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_fon-fr.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fr-fon +include: mafand +task: mafand_fon-fr_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_hau-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_hau-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9c0b0f15fe011d52ff4bbd167e23453a621e2928 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_hau-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-hau +include: mafand +task: mafand_hau-en_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_kin-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_kin-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..254b22be3883110224aee429313cf2636993e675 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_kin-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-kin +include: mafand +task: mafand_kin-en_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_lug-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_lug-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ad19b3c85e6aae28a01c7e3476f8492e804c6d83 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_lug-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-lug +include: mafand +task: mafand_lug-en_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_mos-fr.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_mos-fr.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4ea419312925c06760ddc6f6efbf343594f2b932 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_mos-fr.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fr-mos +include: mafand +task: mafand_mos-fr_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_nya-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_nya-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..de9ec930a1b738673b91ed89916c17109b159e66 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_nya-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-nya +include: mafand +task: mafand_nya-en_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_sna-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_sna-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2d86ccc3ad64220d354d4e9e230ee585e556b32c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_sna-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-sna +include: mafand +task: mafand_sna-en_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_swa-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_swa-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3c70f2e3e77f4ee94edd757311e8b37816911104 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_swa-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-swa +include: mafand +task: mafand_swa-en_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_tsn-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_tsn-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0ee8f4152a80d5af1bfd4b792167021d81e37284 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_tsn-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-tsn +include: mafand +task: mafand_tsn-en_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_twi-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_twi-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a37d2395a4721c6553a878f6693221cbac6a22a6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_twi-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-twi +include: mafand +task: mafand_twi-en_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_wol-fr.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_wol-fr.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ed778cbe6690aafa7abd33fad50ef41fc25dbaea --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_wol-fr.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fr-wol +include: mafand +task: mafand_wol-fr_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_xho-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_xho-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..93e9e2fee5b39b755b9e060697e509a5f63d58e0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_xho-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-xho +include: mafand +task: mafand_xho-en_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_yor-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_yor-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..78301f7e658cd489eb6433f3ac5a10a0f0cde49b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_yor-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-yor +include: mafand +task: mafand_yor-en_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_zul-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_zul-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..06177d14ec829fe74a12030d88e06a5fee7bc9a0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_zul-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-zul +include: mafand +task: mafand_zul-en_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..0df3a329824d44fa94eb830ae943fa30dd32bab7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/utils.py @@ -0,0 +1,121 @@ +languages = { + "amh": "Amharic", + "bam": "Bambara", + "bbj": "Gbomala", + "ewe": "Ewe", + "fon": "Fon", + "hau": "Hausa", + "ibo": "Igbo", + "kin": "Kinyarwanda", + "lug": "Luganda", + "luo": "Luo", + "mos": "Mossi", + "nya": "Chichewa", + "pcm": "Nigerian Pidgin", + "sna": "Shona", + "swa": "Swahili", + "tsn": "Setswana", + "twi": "Twi", + "wol": "Wolof", + "xho": "Xhosa", + "yor": "Yoruba", + "zul": "Zulu", +} + + +def get_target(doc): + target = ( + doc["translation"]["en"] + if "en" in doc["translation"].keys() + else doc["translation"]["fr"] + ) + return target + + +def get_target_reverse(doc): + target_key = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + target = doc["translation"][target_key] + return target + + +def create_text_prompt_1(doc): + source_key = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + source_sentence = doc["translation"][source_key] + source_lang = "English" if "en" in doc["translation"].keys() else "French" + prompt = ( + "You are an advanced Translator, a specialized assistant designed to translate documents from " + f"{languages[source_key]} into {source_lang}. \nYour main goal is to ensure translations are grammatically " + f"correct and human-oriented. \n{languages[source_key]}: {source_sentence} \n{source_lang}: " + ) + return prompt + + +def create_reverse_prompt_1(doc): + target_lang = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + source_key = "en" if "en" in doc["translation"].keys() else "fr" + source_lang = "English" if source_key == "en" else "French" + source_sentence = doc["translation"][source_key] + prompt = ( + "You are an advanced Translator, a specialized assistant designed to translate documents from " + f"{source_lang} into {languages[target_lang]}. \nYour main goal is to ensure translations are " + f"grammatically correct and human-oriented. \n{source_lang}: {source_sentence} \n{languages[target_lang]}: " + ) + return prompt + + +def create_text_prompt_2(doc): + source_key = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + source_sentence = doc["translation"][source_key] + source_lang = "English" if "en" in doc["translation"].keys() else "French" + prompt = ( + f"{languages[source_key]} sentence: {source_sentence} \n{source_lang} sentence: ", + ) + return prompt + + +def create_reverse_prompt_2(doc): + target_lang = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + source_key = "en" if "en" in doc["translation"].keys() else "fr" + source_lang = "English" if source_key == "en" else "French" + source_sentence = doc["translation"][source_key] + prompt = ( + f"{source_lang} sentence: {source_sentence} \n{languages[target_lang]} sentence: \n", + ) + return prompt + + +def create_text_prompt_3(doc): + source_key = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + source_sentence = doc["translation"][source_key] + source_lang = "English" if "en" in doc["translation"].keys() else "French" + prompt = ( + f"You are a translation expert. Translate the following {languages[source_key]} sentences " + f"to {source_lang}. \n{languages[source_key]} sentence: {source_sentence}\n{source_lang} sentence: " + ) + return prompt + + +def create_reverse_prompt_3(doc): + target_lang = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + source_key = "en" if "en" in doc["translation"].keys() else "fr" + source_lang = "English" if source_key == "en" else "French" + source_sentence = doc["translation"][source_key] + prompt = ( + f"You are a translation expert. Translate the following {source_lang} sentence into {languages[target_lang]}\n" + f"{source_lang} sentence: {source_sentence}\n{languages[target_lang]} sentence: " + ) + return prompt diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand new file mode 100644 index 0000000000000000000000000000000000000000..9a59654e4feba09f57277b539633c2b0efda291e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand @@ -0,0 +1,28 @@ +tag: +- mafand_tasks +- mafand_eng-afr +- mafand_eng-afr_prompt_3 +- afrobench_MT_tasks +dataset_path: masakhane/mafand +dataset_kwargs: {trust_remote_code: True} +output_type: generate_until +validation_split: validation +fewshot_split: validation +test_split: test +doc_to_target: !function utils.get_target_reverse +doc_to_text: !function utils.create_reverse_prompt_3 +metric_list: + - metric: bleu + aggregation: bleu + higher_is_better: true + - metric: chrf + aggregation: chrf + higher_is_better: true +generation_kwargs: + until: + - "\n" + do_sample: false + temperature: 0.0 +repeats: 1 +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..10872430688882c53e5f2dba5aea70d1194c6020 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-amh.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-amh +include: mafand +task: mafand_en-amh_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f64e68162a76b1a0cc33f44d62a0f66b5dc099e0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-hau.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-hau +include: mafand +task: mafand_en-hau_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..72df05e551444a9e82de940171c646797f1c18d6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-ibo.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-ibo +include: mafand +task: mafand_en-ibo_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..44c48678e89c3dd7e72d310fd050b5a1d3bc6092 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-kin.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-kin +include: mafand +task: mafand_en-kin_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2beae91b569db6fc361da97e0879e854af005e4d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-lug.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-lug +include: mafand +task: mafand_en-lug_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-luo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-luo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b4c1aa8becf693cb16d4d1be2820707e521a1052 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-luo.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-luo +include: mafand +task: mafand_en-luo_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-nya.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-nya.yaml new file mode 100644 index 0000000000000000000000000000000000000000..eee7af0ce8ed105fa08769793ad816c2f4d17318 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-nya.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-nya +include: mafand +task: mafand_en-nya_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-pcm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-pcm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6e60642562403f3b86d2e13fb3ea48368dc84883 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-pcm.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-pcm +include: mafand +task: mafand_en-pcm_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-sna.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-sna.yaml new file mode 100644 index 0000000000000000000000000000000000000000..82abd862535426ee078cd45f057c792643062981 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-sna.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-sna +include: mafand +task: mafand_en-sna_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8a7135ff6921556928338e52e3dbcc8afa731025 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-swa.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-swa +include: mafand +task: mafand_en-swa_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-tsn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-tsn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b976b5fd24007928854afe3b533ffd8b66ea0851 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-tsn.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-tsn +include: mafand +task: mafand_en-tsn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..53345a2668eccb93b11568c22f6d218598f20ba2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-twi.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-twi +include: mafand +task: mafand_en-twi_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4eba7f6994b577b4769b6b74a2848ecc0b3b0fb0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-xho.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-xho +include: mafand +task: mafand_en-xho_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9b20e9f920deb657630ef0aa5aa6a443934f0519 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-yor.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-yor +include: mafand +task: mafand_en-yor_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cb5280b995b0754ae6a4f6cd33bfc82350d0e8cb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-zul.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-zul +include: mafand +task: mafand_en-zul_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_fr-bam.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_fr-bam.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3e94be00dea4c013238a543cc3ceeb2982b92ce4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_fr-bam.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fr-bam +include: mafand +task: mafand_fr-bam_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_fr-bbj.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_fr-bbj.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9170a6b500239357dacf736b06044dd17aff5b25 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_fr-bbj.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fr-bbj +include: mafand +task: mafand_fr-bbj_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_fr-ewe.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_fr-ewe.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7139c81fd0c991bf9ee6a24013a34a8b0b700efc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_fr-ewe.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fr-ewe +include: mafand +task: mafand_fr-ewe_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_fr-fon.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_fr-fon.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b42292ce56e6abc7127e9988e5619e0eee3d56ab --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_fr-fon.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fr-fon +include: mafand +task: mafand_fr-fon_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_fr-mos.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_fr-mos.yaml new file mode 100644 index 0000000000000000000000000000000000000000..044047c346abcb945443b2b500eef7bd32f2caad --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_fr-mos.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fr-mos +include: mafand +task: mafand_fr-mos_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_fr-wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_fr-wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9fc1bca3b94f64489d26632c70c022558e7793b3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_fr-wol.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fr-wol +include: mafand +task: mafand_fr-wol_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..0df3a329824d44fa94eb830ae943fa30dd32bab7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/utils.py @@ -0,0 +1,121 @@ +languages = { + "amh": "Amharic", + "bam": "Bambara", + "bbj": "Gbomala", + "ewe": "Ewe", + "fon": "Fon", + "hau": "Hausa", + "ibo": "Igbo", + "kin": "Kinyarwanda", + "lug": "Luganda", + "luo": "Luo", + "mos": "Mossi", + "nya": "Chichewa", + "pcm": "Nigerian Pidgin", + "sna": "Shona", + "swa": "Swahili", + "tsn": "Setswana", + "twi": "Twi", + "wol": "Wolof", + "xho": "Xhosa", + "yor": "Yoruba", + "zul": "Zulu", +} + + +def get_target(doc): + target = ( + doc["translation"]["en"] + if "en" in doc["translation"].keys() + else doc["translation"]["fr"] + ) + return target + + +def get_target_reverse(doc): + target_key = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + target = doc["translation"][target_key] + return target + + +def create_text_prompt_1(doc): + source_key = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + source_sentence = doc["translation"][source_key] + source_lang = "English" if "en" in doc["translation"].keys() else "French" + prompt = ( + "You are an advanced Translator, a specialized assistant designed to translate documents from " + f"{languages[source_key]} into {source_lang}. \nYour main goal is to ensure translations are grammatically " + f"correct and human-oriented. \n{languages[source_key]}: {source_sentence} \n{source_lang}: " + ) + return prompt + + +def create_reverse_prompt_1(doc): + target_lang = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + source_key = "en" if "en" in doc["translation"].keys() else "fr" + source_lang = "English" if source_key == "en" else "French" + source_sentence = doc["translation"][source_key] + prompt = ( + "You are an advanced Translator, a specialized assistant designed to translate documents from " + f"{source_lang} into {languages[target_lang]}. \nYour main goal is to ensure translations are " + f"grammatically correct and human-oriented. \n{source_lang}: {source_sentence} \n{languages[target_lang]}: " + ) + return prompt + + +def create_text_prompt_2(doc): + source_key = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + source_sentence = doc["translation"][source_key] + source_lang = "English" if "en" in doc["translation"].keys() else "French" + prompt = ( + f"{languages[source_key]} sentence: {source_sentence} \n{source_lang} sentence: ", + ) + return prompt + + +def create_reverse_prompt_2(doc): + target_lang = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + source_key = "en" if "en" in doc["translation"].keys() else "fr" + source_lang = "English" if source_key == "en" else "French" + source_sentence = doc["translation"][source_key] + prompt = ( + f"{source_lang} sentence: {source_sentence} \n{languages[target_lang]} sentence: \n", + ) + return prompt + + +def create_text_prompt_3(doc): + source_key = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + source_sentence = doc["translation"][source_key] + source_lang = "English" if "en" in doc["translation"].keys() else "French" + prompt = ( + f"You are a translation expert. Translate the following {languages[source_key]} sentences " + f"to {source_lang}. \n{languages[source_key]} sentence: {source_sentence}\n{source_lang} sentence: " + ) + return prompt + + +def create_reverse_prompt_3(doc): + target_lang = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + source_key = "en" if "en" in doc["translation"].keys() else "fr" + source_lang = "English" if source_key == "en" else "French" + source_sentence = doc["translation"][source_key] + prompt = ( + f"You are a translation expert. Translate the following {source_lang} sentence into {languages[target_lang]}\n" + f"{source_lang} sentence: {source_sentence}\n{languages[target_lang]} sentence: " + ) + return prompt diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/README.md b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/README.md new file mode 100644 index 0000000000000000000000000000000000000000..ca96648e07bb2c54fbf0c79d968b2ef4cb6aba75 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/README.md @@ -0,0 +1,76 @@ +# + +## Paper +Title: `MasakhaNER 2.0: Africa-centric Transfer Learning for Named Entity Recognition` + +Paper Link: https://aclanthology.org/2022.emnlp-main.298/ + +## Abstract +>African languages are spoken by over a billion people, but they are under-represented in NLP research and development. Multiple challenges exist, including the limited availability of annotated training and evaluation datasets as well as the lack of understanding of which settings, languages, and recently proposed methods like cross-lingual transfer will be effective. In this paper, we aim to move towards solutions for these challenges, focusing on the task of named entity recognition (NER). We present the creation of the largest to-date human-annotated NER dataset for 20 African languages. We study the behaviour of state-of-the-art cross-lingual transfer methods in an Africa-centric setting, empirically demonstrating that the choice of source transfer language significantly affects performance. While much previous work defaults to using English as the source language, our results show that choosing the best transfer language improves zero-shot F1 scores by an average of 14% over 20 languages as compared to using English. + +HomePage: https://github.com/masakhane-io/masakhane-ner + +### Citation + +``` +@inproceedings{adelani-etal-2022-masakhaner, + title = "{M}asakha{NER} 2.0: {A}frica-centric Transfer Learning for Named Entity Recognition", + author = "Adelani, David Ifeoluwa and + Neubig, Graham and + Ruder, Sebastian and + Rijhwani, Shruti and + Beukman, Michael and + Palen-Michel, Chester and + Lignos, Constantine and + Alabi, Jesujoba O. and + Muhammad, Shamsuddeen H. and + Nabende, Peter and + Dione, Cheikh M. Bamba and + Bukula, Andiswa and + Mabuya, Rooweither and + Dossou, Bonaventure F. P. and + Sibanda, Blessing and + Buzaaba, Happy and + Mukiibi, Jonathan and + Kalipe, Godson and + Mbaye, Derguene and + Taylor, Amelia and + Kabore, Fatoumata and + Emezue, Chris Chinenye and + Aremu, Anuoluwapo and + Ogayo, Perez and + Gitau, Catherine and + Munkoh-Buabeng, Edwin and + Memdjokam Koagne, Victoire and + Tapo, Allahsera Auguste and + Macucwa, Tebogo and + Marivate, Vukosi and + Mboning, Elvis and + Gwadabe, Tajuddeen and + Adewumi, Tosin and + Ahia, Orevaoghene and + Nakatumba-Nabende, Joyce and + Mokono, Neo L. and + Ezeani, Ignatius and + Chukwuneke, Chiamaka and + Adeyemi, Mofetoluwa and + Hacheme, Gilles Q. and + Abdulmumim, Idris and + Ogundepo, Odunayo and + Yousuf, Oreen and + Moteu Ngoli, Tatiana and + Klakow, Dietrich", + editor = "Goldberg, Yoav and + Kozareva, Zornitsa and + Zhang, Yue", + booktitle = "Proceedings of the 2022 Conference on Empirical Methods in Natural Language Processing", + month = dec, + year = "2022", + address = "Abu Dhabi, United Arab Emirates", + publisher = "Association for Computational Linguistics", + url = "https://aclanthology.org/2022.emnlp-main.298/", + doi = "10.18653/v1/2022.emnlp-main.298", + pages = "4488--4508", + abstract = "African languages are spoken by over a billion people, but they are under-represented in NLP research and development. Multiple challenges exist, including the limited availability of annotated training and evaluation datasets as well as the lack of understanding of which settings, languages, and recently proposed methods like cross-lingual transfer will be effective. In this paper, we aim to move towards solutions for these challenges, focusing on the task of named entity recognition (NER). We present the creation of the largest to-date human-annotated NER dataset for 20 African languages. We study the behaviour of state-of-the-art cross-lingual transfer methods in an Africa-centric setting, empirically demonstrating that the choice of source transfer language significantly affects performance. While much previous work defaults to using English as the source language, our results show that choosing the best transfer language improves zero-shot F1 scores by an average of 14{\%} over 20 languages as compared to using English." +} +``` diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/gen_utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/gen_utils.py new file mode 100644 index 0000000000000000000000000000000000000000..4d1012021f567ab02ccdd6259788e00ea1f759e9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/gen_utils.py @@ -0,0 +1,138 @@ +import argparse +import os + +import yaml + + +class FunctionTag: + def __init__(self, value): + self.value = value + + +def prompt_func(mode, lang): + prompt_map = { + "prompt_1": "Named entities refers to names of location, organisation and personal name. \n For example, " + "'David is an employee of Amazon and he is visiting New York next week to see Esther' will be \n" + "PERSON: David $ ORGANIZATION: Amazon $ LOCATION: New York $ PERSON: Esther \n\n" + "Ensure the output strictly follows the format: label: entity $ label: entity, with each unique " + "entity on a separate label line, avoiding grouped entities (e.g., avoid LOC: entity, entity) or " + "irrelevant entries like none. \n\nText: {{text}} \n" + "Return only the output", + "prompt_2": "You are working as a named entity recognition expert and your task is to label a given text " + "with named entity labels. Your task is to identify and label any named entities present in the " + "text. The named entity labels that you will be using are PER (person), LOC (location), " + "ORG (organization) and DATE (date). Label multi-word entities as a single named entity. " + "For words which are not part of any named entity, do not return any value for it. \n" + "Ensure the output strictly follows the format: label: entity $$ label: entity, with each unique " + "entity on a separate label line, avoiding grouped entities (e.g., avoid LOC: entity, entity) or " + "irrelevant entries like none. Return only the output \n\nText: {{text}}", + "prompt_3": f"You are a Named Entity Recognition expert in {lang} language. \nExtract all named entities from " + f"the following {lang} text and categorize them into PERSON, LOCATION, ORGANIZATION, or DATE. " + f"Ensure the output strictly follows the format: label: entity $$ label: entity, with each unique " + "entity on a separate label line, avoiding grouped entities (e.g., avoid LOC: entity, entity) or " + "irrelevant entries like none. Return only the output \n\nText: {{text}}", + "prompt_4": f"As a {lang} linguist, label all named entities in the {lang} text below with the categories: " + "PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the output strictly follows the format: label: " + "entity $$ label: entity, with each unique entity on a separate label line, avoiding grouped " + "entities (e.g., avoid LOC: entity, entity) or irrelevant entries like none. Return only the " + "output. \n\nText: {{text}}", + "prompt_5": "Provide a concise list of named entities in the text below. Use the following labels: " + "PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the output strictly follows the format: label: " + "entity $$ label: entity, with each unique entity on a separate label line, avoiding grouped " + "entities (e.g., avoid LOC: entity, entity) or irrelevant entries like none. Return only the " + "output. \n\nText: {{text}}", + } + return prompt_map[mode] + + +def gen_lang_yamls(output_dir: str, overwrite: bool, mode: str) -> None: + """ + Generate a yaml file for each language. + + :param output_dir: The directory to output the files to. + :param overwrite: Whether to overwrite files if they already exist. + """ + err = [] + languages = { + "am": "Amharic", + "bm": "Bambara", + "bbj": "Ghomala", + "ee": "Ewe", + "ha": "Hausa", + "ig": "Igbo", + "rw": "Kinyarwanda", + "lg": "Luganda", + "luo": "Luo", + "mos": "Mossi", + "ny": "Chichewa", + "pcm": "Nigerian Pidgin", + "sn": "chiShona", + "sw": "Kiswahili", + "tn": "Setswana", + "tw": "Twi", + "wo": "Wolof", + "xh": "isiXhosa", + "yo": "Yoruba", + "zu": "isiZulu", + } + + for lang in languages.keys(): + try: + file_name = f"masakhaner_{lang}.yaml" + task_name = f"masakhaner_{lang}_{mode}" + yaml_template = "masakhaner" + yaml_details = { + "include": yaml_template, + "task": task_name, + "dataset_name": lang, + "doc_to_text": prompt_func(mode, languages[lang]), + } + os.makedirs(f"{output_dir}/{mode}", exist_ok=True) + with open( + f"{output_dir}/{mode}/{file_name}", + "w" if overwrite else "x", + encoding="utf8", + ) as f: + f.write("# Generated by utils.py\n") + yaml.dump( + yaml_details, + f, + allow_unicode=True, + ) + except FileExistsError: + err.append(file_name) + + if len(err) > 0: + raise FileExistsError( + "Files were not created because they already exist (use --overwrite flag):" + f" {', '.join(err)}" + ) + + +def main() -> None: + """Parse CLI args and generate language-specific yaml files.""" + parser = argparse.ArgumentParser() + parser.add_argument( + "--overwrite", + default=True, + action="store_true", + help="Overwrite files if they already exist", + ) + parser.add_argument( + "--output-dir", + default="./", + help="Directory to write yaml files to", + ) + parser.add_argument( + "--mode", + default="prompt_1", + choices=["prompt_1", "prompt_2", "prompt_3", "prompt_4", "prompt_5"], + help="Prompt number", + ) + args = parser.parse_args() + + gen_lang_yamls(output_dir=args.output_dir, overwrite=args.overwrite, mode=args.mode) + + +if __name__ == "__main__": + main() diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/masakhaner.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/masakhaner.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b0d374e80c43cc8831d167887e386f3500773b48 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/masakhaner.yaml @@ -0,0 +1,13 @@ +group: masakhaner +task: + - masakhaner_prompt_1 + - masakhaner_prompt_2 + - masakhaner_prompt_3 + - masakhaner_prompt_4 + - masakhaner_prompt_5 +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner new file mode 100644 index 0000000000000000000000000000000000000000..706eb36644524b2aaa10b686ce120048e6322390 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner @@ -0,0 +1,26 @@ +tag: +- masakhaner_tasks +- masakhaner_prompt_1 +dataset_path: masakhane/masakhaner-x +dataset_name: null +dataset_kwargs: {trust_remote_code: True} +output_type: generate_until +generation_kwargs: + do_sample: false + until: + - + - <|im_end|> +validation_split: validation +test_split: test +fewshot_split: train +doc_to_target: target +filter_list: + - name: flexible-extract + filter: + - function: format_span +metric_list: + - metric: f1 + aggregation: !function utils.span_f1_agg + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_am.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_am.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b2128752f754eb8c46f7608c89dbefd7a3800480 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_am.yaml @@ -0,0 +1,11 @@ +# Generated by utils.py +dataset_name: am +doc_to_text: "Named entities refers to names of location, organisation and personal\ + \ name. \n For example, 'David is an employee of Amazon and he is visiting New York\ + \ next week to see Esther' will be \nPERSON: David $ ORGANIZATION: Amazon $ LOCATION:\ + \ New York $ PERSON: Esther \n\nEnsure the output strictly follows the format: label:\ + \ entity $ label: entity, with each unique entity on a separate label line, avoiding\ + \ grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries like\ + \ none. \n\nText: {{text}} \nReturn only the output" +include: masakhaner +task: masakhaner_am_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_bbj.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_bbj.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3f3a72bdc0257aedee8787af75a70c14560bfc53 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_bbj.yaml @@ -0,0 +1,11 @@ +# Generated by utils.py +dataset_name: bbj +doc_to_text: "Named entities refers to names of location, organisation and personal\ + \ name. \n For example, 'David is an employee of Amazon and he is visiting New York\ + \ next week to see Esther' will be \nPERSON: David $ ORGANIZATION: Amazon $ LOCATION:\ + \ New York $ PERSON: Esther \n\nEnsure the output strictly follows the format: label:\ + \ entity $ label: entity, with each unique entity on a separate label line, avoiding\ + \ grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries like\ + \ none. \n\nText: {{text}} \nReturn only the output" +include: masakhaner +task: masakhaner_bbj_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_bm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_bm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c38bdee947c2d34a4a7797eae1251fc476be0f53 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_bm.yaml @@ -0,0 +1,11 @@ +# Generated by utils.py +dataset_name: bm +doc_to_text: "Named entities refers to names of location, organisation and personal\ + \ name. \n For example, 'David is an employee of Amazon and he is visiting New York\ + \ next week to see Esther' will be \nPERSON: David $ ORGANIZATION: Amazon $ LOCATION:\ + \ New York $ PERSON: Esther \n\nEnsure the output strictly follows the format: label:\ + \ entity $ label: entity, with each unique entity on a separate label line, avoiding\ + \ grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries like\ + \ none. \n\nText: {{text}} \nReturn only the output" +include: masakhaner +task: masakhaner_bm_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_ee.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_ee.yaml new file mode 100644 index 0000000000000000000000000000000000000000..97903908e30cfe376626f7c668d5d9862593d73e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_ee.yaml @@ -0,0 +1,11 @@ +# Generated by utils.py +dataset_name: ee +doc_to_text: "Named entities refers to names of location, organisation and personal\ + \ name. \n For example, 'David is an employee of Amazon and he is visiting New York\ + \ next week to see Esther' will be \nPERSON: David $ ORGANIZATION: Amazon $ LOCATION:\ + \ New York $ PERSON: Esther \n\nEnsure the output strictly follows the format: label:\ + \ entity $ label: entity, with each unique entity on a separate label line, avoiding\ + \ grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries like\ + \ none. \n\nText: {{text}} \nReturn only the output" +include: masakhaner +task: masakhaner_ee_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_ha.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_ha.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ad11710407bda79b9561cd05725c21eb945ca292 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_ha.yaml @@ -0,0 +1,11 @@ +# Generated by utils.py +dataset_name: ha +doc_to_text: "Named entities refers to names of location, organisation and personal\ + \ name. \n For example, 'David is an employee of Amazon and he is visiting New York\ + \ next week to see Esther' will be \nPERSON: David $ ORGANIZATION: Amazon $ LOCATION:\ + \ New York $ PERSON: Esther \n\nEnsure the output strictly follows the format: label:\ + \ entity $ label: entity, with each unique entity on a separate label line, avoiding\ + \ grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries like\ + \ none. \n\nText: {{text}} \nReturn only the output" +include: masakhaner +task: masakhaner_ha_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_ig.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_ig.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0f06c0655595ae81b985266e4e629b080d49130c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_ig.yaml @@ -0,0 +1,11 @@ +# Generated by utils.py +dataset_name: ig +doc_to_text: "Named entities refers to names of location, organisation and personal\ + \ name. \n For example, 'David is an employee of Amazon and he is visiting New York\ + \ next week to see Esther' will be \nPERSON: David $ ORGANIZATION: Amazon $ LOCATION:\ + \ New York $ PERSON: Esther \n\nEnsure the output strictly follows the format: label:\ + \ entity $ label: entity, with each unique entity on a separate label line, avoiding\ + \ grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries like\ + \ none. \n\nText: {{text}} \nReturn only the output" +include: masakhaner +task: masakhaner_ig_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_lg.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_lg.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1823b20f63e1e4d2bcf5edc1427be758c3d16a62 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_lg.yaml @@ -0,0 +1,11 @@ +# Generated by utils.py +dataset_name: lg +doc_to_text: "Named entities refers to names of location, organisation and personal\ + \ name. \n For example, 'David is an employee of Amazon and he is visiting New York\ + \ next week to see Esther' will be \nPERSON: David $ ORGANIZATION: Amazon $ LOCATION:\ + \ New York $ PERSON: Esther \n\nEnsure the output strictly follows the format: label:\ + \ entity $ label: entity, with each unique entity on a separate label line, avoiding\ + \ grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries like\ + \ none. \n\nText: {{text}} \nReturn only the output" +include: masakhaner +task: masakhaner_lg_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_luo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_luo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..55b6d82968ed9a09ef43a12bd5cba91ea4ab5c87 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_luo.yaml @@ -0,0 +1,11 @@ +# Generated by utils.py +dataset_name: luo +doc_to_text: "Named entities refers to names of location, organisation and personal\ + \ name. \n For example, 'David is an employee of Amazon and he is visiting New York\ + \ next week to see Esther' will be \nPERSON: David $ ORGANIZATION: Amazon $ LOCATION:\ + \ New York $ PERSON: Esther \n\nEnsure the output strictly follows the format: label:\ + \ entity $ label: entity, with each unique entity on a separate label line, avoiding\ + \ grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries like\ + \ none. \n\nText: {{text}} \nReturn only the output" +include: masakhaner +task: masakhaner_luo_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_mos.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_mos.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ac5ddf43cd0eefd5fbe85785c7b4687135924938 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_mos.yaml @@ -0,0 +1,11 @@ +# Generated by utils.py +dataset_name: mos +doc_to_text: "Named entities refers to names of location, organisation and personal\ + \ name. \n For example, 'David is an employee of Amazon and he is visiting New York\ + \ next week to see Esther' will be \nPERSON: David $ ORGANIZATION: Amazon $ LOCATION:\ + \ New York $ PERSON: Esther \n\nEnsure the output strictly follows the format: label:\ + \ entity $ label: entity, with each unique entity on a separate label line, avoiding\ + \ grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries like\ + \ none. \n\nText: {{text}} \nReturn only the output" +include: masakhaner +task: masakhaner_mos_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_ny.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_ny.yaml new file mode 100644 index 0000000000000000000000000000000000000000..36d12ad2c00c0101aa2405af5824b6cbf310d132 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_ny.yaml @@ -0,0 +1,11 @@ +# Generated by utils.py +dataset_name: ny +doc_to_text: "Named entities refers to names of location, organisation and personal\ + \ name. \n For example, 'David is an employee of Amazon and he is visiting New York\ + \ next week to see Esther' will be \nPERSON: David $ ORGANIZATION: Amazon $ LOCATION:\ + \ New York $ PERSON: Esther \n\nEnsure the output strictly follows the format: label:\ + \ entity $ label: entity, with each unique entity on a separate label line, avoiding\ + \ grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries like\ + \ none. \n\nText: {{text}} \nReturn only the output" +include: masakhaner +task: masakhaner_ny_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_pcm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_pcm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8c09bf44c682758e53deb9262e4aef393d8bbc8e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_pcm.yaml @@ -0,0 +1,11 @@ +# Generated by utils.py +dataset_name: pcm +doc_to_text: "Named entities refers to names of location, organisation and personal\ + \ name. \n For example, 'David is an employee of Amazon and he is visiting New York\ + \ next week to see Esther' will be \nPERSON: David $ ORGANIZATION: Amazon $ LOCATION:\ + \ New York $ PERSON: Esther \n\nEnsure the output strictly follows the format: label:\ + \ entity $ label: entity, with each unique entity on a separate label line, avoiding\ + \ grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries like\ + \ none. \n\nText: {{text}} \nReturn only the output" +include: masakhaner +task: masakhaner_pcm_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_rw.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_rw.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7398e5fbe7b55f837b900158b7fdc4a3b7e2ac92 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_rw.yaml @@ -0,0 +1,11 @@ +# Generated by utils.py +dataset_name: rw +doc_to_text: "Named entities refers to names of location, organisation and personal\ + \ name. \n For example, 'David is an employee of Amazon and he is visiting New York\ + \ next week to see Esther' will be \nPERSON: David $ ORGANIZATION: Amazon $ LOCATION:\ + \ New York $ PERSON: Esther \n\nEnsure the output strictly follows the format: label:\ + \ entity $ label: entity, with each unique entity on a separate label line, avoiding\ + \ grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries like\ + \ none. \n\nText: {{text}} \nReturn only the output" +include: masakhaner +task: masakhaner_rw_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_sn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_sn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ecdd3260fcb8b0fd147dfba666aa0a42e4687323 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_sn.yaml @@ -0,0 +1,11 @@ +# Generated by utils.py +dataset_name: sn +doc_to_text: "Named entities refers to names of location, organisation and personal\ + \ name. \n For example, 'David is an employee of Amazon and he is visiting New York\ + \ next week to see Esther' will be \nPERSON: David $ ORGANIZATION: Amazon $ LOCATION:\ + \ New York $ PERSON: Esther \n\nEnsure the output strictly follows the format: label:\ + \ entity $ label: entity, with each unique entity on a separate label line, avoiding\ + \ grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries like\ + \ none. \n\nText: {{text}} \nReturn only the output" +include: masakhaner +task: masakhaner_sn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_sw.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_sw.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f2bd3379c3b3508067ae500c5cad947ba7006b74 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_sw.yaml @@ -0,0 +1,11 @@ +# Generated by utils.py +dataset_name: sw +doc_to_text: "Named entities refers to names of location, organisation and personal\ + \ name. \n For example, 'David is an employee of Amazon and he is visiting New York\ + \ next week to see Esther' will be \nPERSON: David $ ORGANIZATION: Amazon $ LOCATION:\ + \ New York $ PERSON: Esther \n\nEnsure the output strictly follows the format: label:\ + \ entity $ label: entity, with each unique entity on a separate label line, avoiding\ + \ grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries like\ + \ none. \n\nText: {{text}} \nReturn only the output" +include: masakhaner +task: masakhaner_sw_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_tn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_tn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..50d80dcb79ab8f2bb0f423d42159dd1d56fd262f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_tn.yaml @@ -0,0 +1,11 @@ +# Generated by utils.py +dataset_name: tn +doc_to_text: "Named entities refers to names of location, organisation and personal\ + \ name. \n For example, 'David is an employee of Amazon and he is visiting New York\ + \ next week to see Esther' will be \nPERSON: David $ ORGANIZATION: Amazon $ LOCATION:\ + \ New York $ PERSON: Esther \n\nEnsure the output strictly follows the format: label:\ + \ entity $ label: entity, with each unique entity on a separate label line, avoiding\ + \ grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries like\ + \ none. \n\nText: {{text}} \nReturn only the output" +include: masakhaner +task: masakhaner_tn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_tw.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_tw.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6c8a8d40575c4867cc0ec33a96343f1eb9c29f7b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_tw.yaml @@ -0,0 +1,11 @@ +# Generated by utils.py +dataset_name: tw +doc_to_text: "Named entities refers to names of location, organisation and personal\ + \ name. \n For example, 'David is an employee of Amazon and he is visiting New York\ + \ next week to see Esther' will be \nPERSON: David $ ORGANIZATION: Amazon $ LOCATION:\ + \ New York $ PERSON: Esther \n\nEnsure the output strictly follows the format: label:\ + \ entity $ label: entity, with each unique entity on a separate label line, avoiding\ + \ grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries like\ + \ none. \n\nText: {{text}} \nReturn only the output" +include: masakhaner +task: masakhaner_tw_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_wo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_wo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5e5f6eeaecb9ad56ee2cf035cf1390f489d8ba98 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_wo.yaml @@ -0,0 +1,11 @@ +# Generated by utils.py +dataset_name: wo +doc_to_text: "Named entities refers to names of location, organisation and personal\ + \ name. \n For example, 'David is an employee of Amazon and he is visiting New York\ + \ next week to see Esther' will be \nPERSON: David $ ORGANIZATION: Amazon $ LOCATION:\ + \ New York $ PERSON: Esther \n\nEnsure the output strictly follows the format: label:\ + \ entity $ label: entity, with each unique entity on a separate label line, avoiding\ + \ grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries like\ + \ none. \n\nText: {{text}} \nReturn only the output" +include: masakhaner +task: masakhaner_wo_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_xh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_xh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8b27051f5df77d39561c0fcee81033a357a9220d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_xh.yaml @@ -0,0 +1,11 @@ +# Generated by utils.py +dataset_name: xh +doc_to_text: "Named entities refers to names of location, organisation and personal\ + \ name. \n For example, 'David is an employee of Amazon and he is visiting New York\ + \ next week to see Esther' will be \nPERSON: David $ ORGANIZATION: Amazon $ LOCATION:\ + \ New York $ PERSON: Esther \n\nEnsure the output strictly follows the format: label:\ + \ entity $ label: entity, with each unique entity on a separate label line, avoiding\ + \ grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries like\ + \ none. \n\nText: {{text}} \nReturn only the output" +include: masakhaner +task: masakhaner_xh_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_yo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_yo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2fdb71aa53d4eacdb5cfcdada220423909f9515d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_yo.yaml @@ -0,0 +1,11 @@ +# Generated by utils.py +dataset_name: yo +doc_to_text: "Named entities refers to names of location, organisation and personal\ + \ name. \n For example, 'David is an employee of Amazon and he is visiting New York\ + \ next week to see Esther' will be \nPERSON: David $ ORGANIZATION: Amazon $ LOCATION:\ + \ New York $ PERSON: Esther \n\nEnsure the output strictly follows the format: label:\ + \ entity $ label: entity, with each unique entity on a separate label line, avoiding\ + \ grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries like\ + \ none. \n\nText: {{text}} \nReturn only the output" +include: masakhaner +task: masakhaner_yo_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_zu.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_zu.yaml new file mode 100644 index 0000000000000000000000000000000000000000..83b9d4b0fd5ef3ab2d9f62639d49a9cea1e3a1a1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_zu.yaml @@ -0,0 +1,11 @@ +# Generated by utils.py +dataset_name: zu +doc_to_text: "Named entities refers to names of location, organisation and personal\ + \ name. \n For example, 'David is an employee of Amazon and he is visiting New York\ + \ next week to see Esther' will be \nPERSON: David $ ORGANIZATION: Amazon $ LOCATION:\ + \ New York $ PERSON: Esther \n\nEnsure the output strictly follows the format: label:\ + \ entity $ label: entity, with each unique entity on a separate label line, avoiding\ + \ grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries like\ + \ none. \n\nText: {{text}} \nReturn only the output" +include: masakhaner +task: masakhaner_zu_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..76909044e7f35948156f8bb506ce2fce563ec689 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/utils.py @@ -0,0 +1,146 @@ +import collections +import re + +from lm_eval.utils import weighted_f1_score + + +def doc_to_target(doc): + return transform_text(doc["ner_tags"]) + + +def transform_text(text): + entities = [] + current_entity = "" + current_tag = "" + + for pair in text.split("\n"): + if pair: # Check if the line is not empty + word, tag = pair.strip().split(": ") + tag = tag.upper() + word = word.lower() + word = word.strip(",.").strip() + + if tag.startswith("B-"): + if current_entity: + entities.append(f"{current_tag}: {current_entity}") + current_tag = tag.split("-")[1] + current_entity = word + elif tag.startswith("I-") and tag.split("-")[1] == current_tag: + current_entity += word + else: + if current_entity: + entities.append(f"{current_tag}: {current_entity}") + current_entity = "" + current_tag = "" + if current_entity: + entities.append(f"{current_tag}: {current_entity}") + + # Join all the transformed output lines with $$ as separator + return " $$ ".join(entities) + + +def span_f1_agg(items): + """Computes Span based F1 score. + + This function is copied from + https://github.com/google-research/multilingual-t5/blob/master/multilingual_t5/evaluation/metrics.py + + Args: + targets: list of strings or list of list of strings if multiple references + are present. + predictions: list of strings + + Returns: + span f1 across all targets and predictions (Based on CoNLL script) + """ + unzipped_list = list(zip(*items)) + targets = unzipped_list[0] + predictions = unzipped_list[1] + + true_positives = collections.defaultdict(int) + false_positives = collections.defaultdict(int) + false_negatives = collections.defaultdict(int) + + def normalize_text(strings): + def get_blank_spaces_pattern(): + return re.compile(r"\s{3,}|\t") + + def remove_blank_spaces(text): + text = re.sub(pattern=get_blank_spaces_pattern(), repl="", string=text) + text = re.sub("\s+", " ", text) + return text + + def remove_punctuation(text): + my_punctuation = '!"$%&\'()*+,-./:;<=>?[\\]^_`{|}~•@.""-,`' + text = re.sub( + "[" + my_punctuation + "]+", " ", str(text) + ) # strip punctuation + return text + + def remove_articles(text): + regex = re.compile(r"\b(a|an|the)\b", re.UNICODE) + return re.sub(regex, " ", text) + + def lowercase(text): + text = text.lower() + return text + + strings = remove_punctuation(strings) + strings = remove_articles(strings) + strings = remove_blank_spaces(strings) + strings = lowercase(strings) + + return strings + + def tags_to_spans(tag_sequence, delimiter="$$"): + """Extract spans from IOB1 or BIO tags.""" + if isinstance(tag_sequence, list): + tag_sequence = " ".join(i.strip() for i in tag_sequence) + tag_sequence_split = [ + item.strip() + for sub in tag_sequence.strip().split(delimiter) + for item in sub.split("$") + if item + ] + tag_sequence_split = [ + item.strip() + for value in tag_sequence_split + for sub in value.split(". ") + for item in sub.split(", ") + ] + tags_entities = [] + for tag_entity in tag_sequence_split: + tag_entity_split = tag_entity.split(": ") + if len(tag_entity_split) != 2: + continue + tag = normalize_text(tag_entity_split[0].strip()) + entity = normalize_text(tag_entity_split[1].rstrip().lstrip()) + tags_entities.append((tag, entity)) + return tags_entities + + def compute_f1_metrics(true_positive, false_positive, false_negative): + precision = float(true_positive) / float(true_positive + false_positive + 1e-13) + recall = float(true_positive) / float(true_positive + false_negative + 1e-13) + f1_measures = 2.0 * ((precision * recall) / (precision + recall + 1e-13)) + return precision, recall, f1_measures + + for target, pred in zip(targets, predictions): + gold_spans = tags_to_spans(target) + predicted_spans = tags_to_spans(pred) + + for span in predicted_spans: + if span in gold_spans: + true_positives[span[0]] += 1 + gold_spans.remove(span) + else: + false_positives[span[0]] += 1 + # These spans weren't predicted. + for span in gold_spans: + false_negatives[span[0]] += 1 + + _, _, f1_measure = compute_f1_metrics( + sum(true_positives.values()), + sum(false_positives.values()), + sum(false_negatives.values()), + ) + return f1_measure diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner new file mode 100644 index 0000000000000000000000000000000000000000..2fd5eb829ce60a5a16d970dbf6b0078e42135cf4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner @@ -0,0 +1,26 @@ +tag: +- masakhaner_tasks +- masakhaner_prompt_2 +dataset_path: masakhane/masakhaner-x +dataset_name: null +dataset_kwargs: {trust_remote_code: True} +output_type: generate_until +generation_kwargs: + do_sample: false + until: + - + - <|im_end|> +validation_split: validation +test_split: test +fewshot_split: train +doc_to_target: target +filter_list: + - name: flexible-extract + filter: + - function: format_span +metric_list: + - metric: f1 + aggregation: !function utils.span_f1_agg + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_am.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_am.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bd1bd33551e152f6f29ff9d248d76ec236be75d4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_am.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: am +doc_to_text: "You are working as a named entity recognition expert and your task is\ + \ to label a given text with named entity labels. Your task is to identify and label\ + \ any named entities present in the text. The named entity labels that you will\ + \ be using are PER (person), LOC (location), ORG (organization) and DATE (date).\ + \ Label multi-word entities as a single named entity. For words which are not part\ + \ of any named entity, do not return any value for it. \nEnsure the output strictly\ + \ follows the format: label: entity $$ label: entity, with each unique entity on\ + \ a separate label line, avoiding grouped entities (e.g., avoid LOC: entity, entity)\ + \ or irrelevant entries like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_am_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_bbj.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_bbj.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5d817ecbe3596b40fad80fe4b95af7ace7bd9e35 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_bbj.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: bbj +doc_to_text: "You are working as a named entity recognition expert and your task is\ + \ to label a given text with named entity labels. Your task is to identify and label\ + \ any named entities present in the text. The named entity labels that you will\ + \ be using are PER (person), LOC (location), ORG (organization) and DATE (date).\ + \ Label multi-word entities as a single named entity. For words which are not part\ + \ of any named entity, do not return any value for it. \nEnsure the output strictly\ + \ follows the format: label: entity $$ label: entity, with each unique entity on\ + \ a separate label line, avoiding grouped entities (e.g., avoid LOC: entity, entity)\ + \ or irrelevant entries like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_bbj_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_bm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_bm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f99a03c7486f8f7170d19b1935e223210915c137 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_bm.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: bm +doc_to_text: "You are working as a named entity recognition expert and your task is\ + \ to label a given text with named entity labels. Your task is to identify and label\ + \ any named entities present in the text. The named entity labels that you will\ + \ be using are PER (person), LOC (location), ORG (organization) and DATE (date).\ + \ Label multi-word entities as a single named entity. For words which are not part\ + \ of any named entity, do not return any value for it. \nEnsure the output strictly\ + \ follows the format: label: entity $$ label: entity, with each unique entity on\ + \ a separate label line, avoiding grouped entities (e.g., avoid LOC: entity, entity)\ + \ or irrelevant entries like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_bm_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_ee.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_ee.yaml new file mode 100644 index 0000000000000000000000000000000000000000..da31685e7d9fb0fa462403e0347347600c705d41 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_ee.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: ee +doc_to_text: "You are working as a named entity recognition expert and your task is\ + \ to label a given text with named entity labels. Your task is to identify and label\ + \ any named entities present in the text. The named entity labels that you will\ + \ be using are PER (person), LOC (location), ORG (organization) and DATE (date).\ + \ Label multi-word entities as a single named entity. For words which are not part\ + \ of any named entity, do not return any value for it. \nEnsure the output strictly\ + \ follows the format: label: entity $$ label: entity, with each unique entity on\ + \ a separate label line, avoiding grouped entities (e.g., avoid LOC: entity, entity)\ + \ or irrelevant entries like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_ee_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_ha.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_ha.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8075046a92f4c88ce8a75db0416732b4f0e96f45 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_ha.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: ha +doc_to_text: "You are working as a named entity recognition expert and your task is\ + \ to label a given text with named entity labels. Your task is to identify and label\ + \ any named entities present in the text. The named entity labels that you will\ + \ be using are PER (person), LOC (location), ORG (organization) and DATE (date).\ + \ Label multi-word entities as a single named entity. For words which are not part\ + \ of any named entity, do not return any value for it. \nEnsure the output strictly\ + \ follows the format: label: entity $$ label: entity, with each unique entity on\ + \ a separate label line, avoiding grouped entities (e.g., avoid LOC: entity, entity)\ + \ or irrelevant entries like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_ha_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_ig.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_ig.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c8771f510a5de2a7ff56acd593b4de3b2309dd07 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_ig.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: ig +doc_to_text: "You are working as a named entity recognition expert and your task is\ + \ to label a given text with named entity labels. Your task is to identify and label\ + \ any named entities present in the text. The named entity labels that you will\ + \ be using are PER (person), LOC (location), ORG (organization) and DATE (date).\ + \ Label multi-word entities as a single named entity. For words which are not part\ + \ of any named entity, do not return any value for it. \nEnsure the output strictly\ + \ follows the format: label: entity $$ label: entity, with each unique entity on\ + \ a separate label line, avoiding grouped entities (e.g., avoid LOC: entity, entity)\ + \ or irrelevant entries like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_ig_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_lg.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_lg.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5c6729e368b3cf53900035c5730d9b729a9e2baa --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_lg.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: lg +doc_to_text: "You are working as a named entity recognition expert and your task is\ + \ to label a given text with named entity labels. Your task is to identify and label\ + \ any named entities present in the text. The named entity labels that you will\ + \ be using are PER (person), LOC (location), ORG (organization) and DATE (date).\ + \ Label multi-word entities as a single named entity. For words which are not part\ + \ of any named entity, do not return any value for it. \nEnsure the output strictly\ + \ follows the format: label: entity $$ label: entity, with each unique entity on\ + \ a separate label line, avoiding grouped entities (e.g., avoid LOC: entity, entity)\ + \ or irrelevant entries like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_lg_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_luo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_luo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a458235f101ae430c77f9da1124a0b8d9fdcda38 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_luo.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: luo +doc_to_text: "You are working as a named entity recognition expert and your task is\ + \ to label a given text with named entity labels. Your task is to identify and label\ + \ any named entities present in the text. The named entity labels that you will\ + \ be using are PER (person), LOC (location), ORG (organization) and DATE (date).\ + \ Label multi-word entities as a single named entity. For words which are not part\ + \ of any named entity, do not return any value for it. \nEnsure the output strictly\ + \ follows the format: label: entity $$ label: entity, with each unique entity on\ + \ a separate label line, avoiding grouped entities (e.g., avoid LOC: entity, entity)\ + \ or irrelevant entries like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_luo_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_mos.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_mos.yaml new file mode 100644 index 0000000000000000000000000000000000000000..816b9bdedc4578c6d2ade8cb05258a4ebc7280de --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_mos.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: mos +doc_to_text: "You are working as a named entity recognition expert and your task is\ + \ to label a given text with named entity labels. Your task is to identify and label\ + \ any named entities present in the text. The named entity labels that you will\ + \ be using are PER (person), LOC (location), ORG (organization) and DATE (date).\ + \ Label multi-word entities as a single named entity. For words which are not part\ + \ of any named entity, do not return any value for it. \nEnsure the output strictly\ + \ follows the format: label: entity $$ label: entity, with each unique entity on\ + \ a separate label line, avoiding grouped entities (e.g., avoid LOC: entity, entity)\ + \ or irrelevant entries like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_mos_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_ny.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_ny.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2f8c4c13c89495b4da3b07ad432b3969310037f8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_ny.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: ny +doc_to_text: "You are working as a named entity recognition expert and your task is\ + \ to label a given text with named entity labels. Your task is to identify and label\ + \ any named entities present in the text. The named entity labels that you will\ + \ be using are PER (person), LOC (location), ORG (organization) and DATE (date).\ + \ Label multi-word entities as a single named entity. For words which are not part\ + \ of any named entity, do not return any value for it. \nEnsure the output strictly\ + \ follows the format: label: entity $$ label: entity, with each unique entity on\ + \ a separate label line, avoiding grouped entities (e.g., avoid LOC: entity, entity)\ + \ or irrelevant entries like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_ny_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_pcm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_pcm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..75dc6ec048cab2dc550d8174dbcc44d966a6ff8e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_pcm.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: pcm +doc_to_text: "You are working as a named entity recognition expert and your task is\ + \ to label a given text with named entity labels. Your task is to identify and label\ + \ any named entities present in the text. The named entity labels that you will\ + \ be using are PER (person), LOC (location), ORG (organization) and DATE (date).\ + \ Label multi-word entities as a single named entity. For words which are not part\ + \ of any named entity, do not return any value for it. \nEnsure the output strictly\ + \ follows the format: label: entity $$ label: entity, with each unique entity on\ + \ a separate label line, avoiding grouped entities (e.g., avoid LOC: entity, entity)\ + \ or irrelevant entries like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_pcm_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_rw.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_rw.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fb93e2d4b24c68bc46b0f141d4c43a51b39ea41e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_rw.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: rw +doc_to_text: "You are working as a named entity recognition expert and your task is\ + \ to label a given text with named entity labels. Your task is to identify and label\ + \ any named entities present in the text. The named entity labels that you will\ + \ be using are PER (person), LOC (location), ORG (organization) and DATE (date).\ + \ Label multi-word entities as a single named entity. For words which are not part\ + \ of any named entity, do not return any value for it. \nEnsure the output strictly\ + \ follows the format: label: entity $$ label: entity, with each unique entity on\ + \ a separate label line, avoiding grouped entities (e.g., avoid LOC: entity, entity)\ + \ or irrelevant entries like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_rw_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_sn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_sn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..60380a512424253dc84f826922fc1fd2f9e75d72 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_sn.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: sn +doc_to_text: "You are working as a named entity recognition expert and your task is\ + \ to label a given text with named entity labels. Your task is to identify and label\ + \ any named entities present in the text. The named entity labels that you will\ + \ be using are PER (person), LOC (location), ORG (organization) and DATE (date).\ + \ Label multi-word entities as a single named entity. For words which are not part\ + \ of any named entity, do not return any value for it. \nEnsure the output strictly\ + \ follows the format: label: entity $$ label: entity, with each unique entity on\ + \ a separate label line, avoiding grouped entities (e.g., avoid LOC: entity, entity)\ + \ or irrelevant entries like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_sn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_sw.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_sw.yaml new file mode 100644 index 0000000000000000000000000000000000000000..82cf74ae26a0cec7d9d659713c7d0203ab829a1f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_sw.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: sw +doc_to_text: "You are working as a named entity recognition expert and your task is\ + \ to label a given text with named entity labels. Your task is to identify and label\ + \ any named entities present in the text. The named entity labels that you will\ + \ be using are PER (person), LOC (location), ORG (organization) and DATE (date).\ + \ Label multi-word entities as a single named entity. For words which are not part\ + \ of any named entity, do not return any value for it. \nEnsure the output strictly\ + \ follows the format: label: entity $$ label: entity, with each unique entity on\ + \ a separate label line, avoiding grouped entities (e.g., avoid LOC: entity, entity)\ + \ or irrelevant entries like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_sw_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_tn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_tn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1852ebe9ae79b51e586fed461b88e7b03fc8557c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_tn.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: tn +doc_to_text: "You are working as a named entity recognition expert and your task is\ + \ to label a given text with named entity labels. Your task is to identify and label\ + \ any named entities present in the text. The named entity labels that you will\ + \ be using are PER (person), LOC (location), ORG (organization) and DATE (date).\ + \ Label multi-word entities as a single named entity. For words which are not part\ + \ of any named entity, do not return any value for it. \nEnsure the output strictly\ + \ follows the format: label: entity $$ label: entity, with each unique entity on\ + \ a separate label line, avoiding grouped entities (e.g., avoid LOC: entity, entity)\ + \ or irrelevant entries like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_tn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_tw.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_tw.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ea354958bcf1590a41fc4e48426d403cae4f9454 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_tw.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: tw +doc_to_text: "You are working as a named entity recognition expert and your task is\ + \ to label a given text with named entity labels. Your task is to identify and label\ + \ any named entities present in the text. The named entity labels that you will\ + \ be using are PER (person), LOC (location), ORG (organization) and DATE (date).\ + \ Label multi-word entities as a single named entity. For words which are not part\ + \ of any named entity, do not return any value for it. \nEnsure the output strictly\ + \ follows the format: label: entity $$ label: entity, with each unique entity on\ + \ a separate label line, avoiding grouped entities (e.g., avoid LOC: entity, entity)\ + \ or irrelevant entries like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_tw_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_wo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_wo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e7cd0d754be2d6fc09054f50d18d29fa07a8551a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_wo.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: wo +doc_to_text: "You are working as a named entity recognition expert and your task is\ + \ to label a given text with named entity labels. Your task is to identify and label\ + \ any named entities present in the text. The named entity labels that you will\ + \ be using are PER (person), LOC (location), ORG (organization) and DATE (date).\ + \ Label multi-word entities as a single named entity. For words which are not part\ + \ of any named entity, do not return any value for it. \nEnsure the output strictly\ + \ follows the format: label: entity $$ label: entity, with each unique entity on\ + \ a separate label line, avoiding grouped entities (e.g., avoid LOC: entity, entity)\ + \ or irrelevant entries like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_wo_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_xh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_xh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9451f0edd121337f8d4dd316284c05a8ea73ff6b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_xh.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: xh +doc_to_text: "You are working as a named entity recognition expert and your task is\ + \ to label a given text with named entity labels. Your task is to identify and label\ + \ any named entities present in the text. The named entity labels that you will\ + \ be using are PER (person), LOC (location), ORG (organization) and DATE (date).\ + \ Label multi-word entities as a single named entity. For words which are not part\ + \ of any named entity, do not return any value for it. \nEnsure the output strictly\ + \ follows the format: label: entity $$ label: entity, with each unique entity on\ + \ a separate label line, avoiding grouped entities (e.g., avoid LOC: entity, entity)\ + \ or irrelevant entries like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_xh_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_yo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_yo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fc0d92c50ed1cf7182b55ed192b2b440d58317e9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_yo.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: yo +doc_to_text: "You are working as a named entity recognition expert and your task is\ + \ to label a given text with named entity labels. Your task is to identify and label\ + \ any named entities present in the text. The named entity labels that you will\ + \ be using are PER (person), LOC (location), ORG (organization) and DATE (date).\ + \ Label multi-word entities as a single named entity. For words which are not part\ + \ of any named entity, do not return any value for it. \nEnsure the output strictly\ + \ follows the format: label: entity $$ label: entity, with each unique entity on\ + \ a separate label line, avoiding grouped entities (e.g., avoid LOC: entity, entity)\ + \ or irrelevant entries like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_yo_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_zu.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_zu.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e06bf3cef883c5ae3f9bf12d7abf5c27618d37e8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_zu.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: zu +doc_to_text: "You are working as a named entity recognition expert and your task is\ + \ to label a given text with named entity labels. Your task is to identify and label\ + \ any named entities present in the text. The named entity labels that you will\ + \ be using are PER (person), LOC (location), ORG (organization) and DATE (date).\ + \ Label multi-word entities as a single named entity. For words which are not part\ + \ of any named entity, do not return any value for it. \nEnsure the output strictly\ + \ follows the format: label: entity $$ label: entity, with each unique entity on\ + \ a separate label line, avoiding grouped entities (e.g., avoid LOC: entity, entity)\ + \ or irrelevant entries like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_zu_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..76909044e7f35948156f8bb506ce2fce563ec689 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/utils.py @@ -0,0 +1,146 @@ +import collections +import re + +from lm_eval.utils import weighted_f1_score + + +def doc_to_target(doc): + return transform_text(doc["ner_tags"]) + + +def transform_text(text): + entities = [] + current_entity = "" + current_tag = "" + + for pair in text.split("\n"): + if pair: # Check if the line is not empty + word, tag = pair.strip().split(": ") + tag = tag.upper() + word = word.lower() + word = word.strip(",.").strip() + + if tag.startswith("B-"): + if current_entity: + entities.append(f"{current_tag}: {current_entity}") + current_tag = tag.split("-")[1] + current_entity = word + elif tag.startswith("I-") and tag.split("-")[1] == current_tag: + current_entity += word + else: + if current_entity: + entities.append(f"{current_tag}: {current_entity}") + current_entity = "" + current_tag = "" + if current_entity: + entities.append(f"{current_tag}: {current_entity}") + + # Join all the transformed output lines with $$ as separator + return " $$ ".join(entities) + + +def span_f1_agg(items): + """Computes Span based F1 score. + + This function is copied from + https://github.com/google-research/multilingual-t5/blob/master/multilingual_t5/evaluation/metrics.py + + Args: + targets: list of strings or list of list of strings if multiple references + are present. + predictions: list of strings + + Returns: + span f1 across all targets and predictions (Based on CoNLL script) + """ + unzipped_list = list(zip(*items)) + targets = unzipped_list[0] + predictions = unzipped_list[1] + + true_positives = collections.defaultdict(int) + false_positives = collections.defaultdict(int) + false_negatives = collections.defaultdict(int) + + def normalize_text(strings): + def get_blank_spaces_pattern(): + return re.compile(r"\s{3,}|\t") + + def remove_blank_spaces(text): + text = re.sub(pattern=get_blank_spaces_pattern(), repl="", string=text) + text = re.sub("\s+", " ", text) + return text + + def remove_punctuation(text): + my_punctuation = '!"$%&\'()*+,-./:;<=>?[\\]^_`{|}~•@.""-,`' + text = re.sub( + "[" + my_punctuation + "]+", " ", str(text) + ) # strip punctuation + return text + + def remove_articles(text): + regex = re.compile(r"\b(a|an|the)\b", re.UNICODE) + return re.sub(regex, " ", text) + + def lowercase(text): + text = text.lower() + return text + + strings = remove_punctuation(strings) + strings = remove_articles(strings) + strings = remove_blank_spaces(strings) + strings = lowercase(strings) + + return strings + + def tags_to_spans(tag_sequence, delimiter="$$"): + """Extract spans from IOB1 or BIO tags.""" + if isinstance(tag_sequence, list): + tag_sequence = " ".join(i.strip() for i in tag_sequence) + tag_sequence_split = [ + item.strip() + for sub in tag_sequence.strip().split(delimiter) + for item in sub.split("$") + if item + ] + tag_sequence_split = [ + item.strip() + for value in tag_sequence_split + for sub in value.split(". ") + for item in sub.split(", ") + ] + tags_entities = [] + for tag_entity in tag_sequence_split: + tag_entity_split = tag_entity.split(": ") + if len(tag_entity_split) != 2: + continue + tag = normalize_text(tag_entity_split[0].strip()) + entity = normalize_text(tag_entity_split[1].rstrip().lstrip()) + tags_entities.append((tag, entity)) + return tags_entities + + def compute_f1_metrics(true_positive, false_positive, false_negative): + precision = float(true_positive) / float(true_positive + false_positive + 1e-13) + recall = float(true_positive) / float(true_positive + false_negative + 1e-13) + f1_measures = 2.0 * ((precision * recall) / (precision + recall + 1e-13)) + return precision, recall, f1_measures + + for target, pred in zip(targets, predictions): + gold_spans = tags_to_spans(target) + predicted_spans = tags_to_spans(pred) + + for span in predicted_spans: + if span in gold_spans: + true_positives[span[0]] += 1 + gold_spans.remove(span) + else: + false_positives[span[0]] += 1 + # These spans weren't predicted. + for span in gold_spans: + false_negatives[span[0]] += 1 + + _, _, f1_measure = compute_f1_metrics( + sum(true_positives.values()), + sum(false_positives.values()), + sum(false_negatives.values()), + ) + return f1_measure diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_am.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_am.yaml new file mode 100644 index 0000000000000000000000000000000000000000..54ad8b54111743ecf392d944c6201bfc56e5362c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_am.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: am +doc_to_text: "You are a Named Entity Recognition expert in Amharic language. \nExtract\ + \ all named entities from the following Amharic text and categorize them into PERSON,\ + \ LOCATION, ORGANIZATION, or DATE. Ensure the output strictly follows the format:\ + \ label: entity $$ label: entity, with each unique entity on a separate label line,\ + \ avoiding grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries\ + \ like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_am_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_bbj.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_bbj.yaml new file mode 100644 index 0000000000000000000000000000000000000000..23e724f424a862308d1947b176cc24b5eb040d47 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_bbj.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: bbj +doc_to_text: "You are a Named Entity Recognition expert in Ghomala language. \nExtract\ + \ all named entities from the following Ghomala text and categorize them into PERSON,\ + \ LOCATION, ORGANIZATION, or DATE. Ensure the output strictly follows the format:\ + \ label: entity $$ label: entity, with each unique entity on a separate label line,\ + \ avoiding grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries\ + \ like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_bbj_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_bm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_bm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..62b5b80e7c26335f304f7ed9673dd9f1c94ef970 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_bm.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: bm +doc_to_text: "You are a Named Entity Recognition expert in Bambara language. \nExtract\ + \ all named entities from the following Bambara text and categorize them into PERSON,\ + \ LOCATION, ORGANIZATION, or DATE. Ensure the output strictly follows the format:\ + \ label: entity $$ label: entity, with each unique entity on a separate label line,\ + \ avoiding grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries\ + \ like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_bm_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_ee.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_ee.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2cdadd27559ab964cacfd396be6fb29a3ae392e8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_ee.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: ee +doc_to_text: "You are a Named Entity Recognition expert in Ewe language. \nExtract\ + \ all named entities from the following Ewe text and categorize them into PERSON,\ + \ LOCATION, ORGANIZATION, or DATE. Ensure the output strictly follows the format:\ + \ label: entity $$ label: entity, with each unique entity on a separate label line,\ + \ avoiding grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries\ + \ like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_ee_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_ha.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_ha.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9d19d26f67447f8916f74b5e26d1bccc0e65bc57 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_ha.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: ha +doc_to_text: "You are a Named Entity Recognition expert in Hausa language. \nExtract\ + \ all named entities from the following Hausa text and categorize them into PERSON,\ + \ LOCATION, ORGANIZATION, or DATE. Ensure the output strictly follows the format:\ + \ label: entity $$ label: entity, with each unique entity on a separate label line,\ + \ avoiding grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries\ + \ like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_ha_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_ig.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_ig.yaml new file mode 100644 index 0000000000000000000000000000000000000000..edf6119689b51760de8a77b70c7dd4461f41b99d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_ig.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: ig +doc_to_text: "You are a Named Entity Recognition expert in Igbo language. \nExtract\ + \ all named entities from the following Igbo text and categorize them into PERSON,\ + \ LOCATION, ORGANIZATION, or DATE. Ensure the output strictly follows the format:\ + \ label: entity $$ label: entity, with each unique entity on a separate label line,\ + \ avoiding grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries\ + \ like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_ig_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_lg.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_lg.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9318a78207c0f3f9e6d2de954e6707100143258e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_lg.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: lg +doc_to_text: "You are a Named Entity Recognition expert in Luganda language. \nExtract\ + \ all named entities from the following Luganda text and categorize them into PERSON,\ + \ LOCATION, ORGANIZATION, or DATE. Ensure the output strictly follows the format:\ + \ label: entity $$ label: entity, with each unique entity on a separate label line,\ + \ avoiding grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries\ + \ like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_lg_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_luo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_luo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..61254fc358a2bd8f2ed2a5dd6aca3d09f88ae232 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_luo.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: luo +doc_to_text: "You are a Named Entity Recognition expert in Luo language. \nExtract\ + \ all named entities from the following Luo text and categorize them into PERSON,\ + \ LOCATION, ORGANIZATION, or DATE. Ensure the output strictly follows the format:\ + \ label: entity $$ label: entity, with each unique entity on a separate label line,\ + \ avoiding grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries\ + \ like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_luo_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_mos.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_mos.yaml new file mode 100644 index 0000000000000000000000000000000000000000..84ff6b24aaffaf14564de139bde92b1c850bcddd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_mos.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: mos +doc_to_text: "You are a Named Entity Recognition expert in Mossi language. \nExtract\ + \ all named entities from the following Mossi text and categorize them into PERSON,\ + \ LOCATION, ORGANIZATION, or DATE. Ensure the output strictly follows the format:\ + \ label: entity $$ label: entity, with each unique entity on a separate label line,\ + \ avoiding grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries\ + \ like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_mos_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_ny.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_ny.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bd592c5b93e39a488d5c5d909137e1caada54840 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_ny.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: ny +doc_to_text: "You are a Named Entity Recognition expert in Chichewa language. \nExtract\ + \ all named entities from the following Chichewa text and categorize them into PERSON,\ + \ LOCATION, ORGANIZATION, or DATE. Ensure the output strictly follows the format:\ + \ label: entity $$ label: entity, with each unique entity on a separate label line,\ + \ avoiding grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries\ + \ like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_ny_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_pcm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_pcm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b448b244b8d6580c0ebc53817060de633dd39efb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_pcm.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: pcm +doc_to_text: "You are a Named Entity Recognition expert in Nigerian Pidgin language.\ + \ \nExtract all named entities from the following Nigerian Pidgin text and categorize\ + \ them into PERSON, LOCATION, ORGANIZATION, or DATE. Ensure the output strictly\ + \ follows the format: label: entity $$ label: entity, with each unique entity on\ + \ a separate label line, avoiding grouped entities (e.g., avoid LOC: entity, entity)\ + \ or irrelevant entries like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_pcm_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_rw.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_rw.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5356ce8b011c53b77ab18302512e30b52c727e37 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_rw.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: rw +doc_to_text: "You are a Named Entity Recognition expert in Kinyarwanda language. \n\ + Extract all named entities from the following Kinyarwanda text and categorize them\ + \ into PERSON, LOCATION, ORGANIZATION, or DATE. Ensure the output strictly follows\ + \ the format: label: entity $$ label: entity, with each unique entity on a separate\ + \ label line, avoiding grouped entities (e.g., avoid LOC: entity, entity) or irrelevant\ + \ entries like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_rw_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_sn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_sn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ab356ae80061b8c49348513b8f6fa05b2dea9473 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_sn.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: sn +doc_to_text: "You are a Named Entity Recognition expert in chiShona language. \nExtract\ + \ all named entities from the following chiShona text and categorize them into PERSON,\ + \ LOCATION, ORGANIZATION, or DATE. Ensure the output strictly follows the format:\ + \ label: entity $$ label: entity, with each unique entity on a separate label line,\ + \ avoiding grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries\ + \ like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_sn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_tn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_tn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d42d164ad81853e92855fd51e73bd0e276d4c761 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_tn.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: tn +doc_to_text: "You are a Named Entity Recognition expert in Setswana language. \nExtract\ + \ all named entities from the following Setswana text and categorize them into PERSON,\ + \ LOCATION, ORGANIZATION, or DATE. Ensure the output strictly follows the format:\ + \ label: entity $$ label: entity, with each unique entity on a separate label line,\ + \ avoiding grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries\ + \ like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_tn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_tw.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_tw.yaml new file mode 100644 index 0000000000000000000000000000000000000000..62b4e2af7c4916e851c834a27c8357a91140d45b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_tw.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: tw +doc_to_text: "You are a Named Entity Recognition expert in Twi language. \nExtract\ + \ all named entities from the following Twi text and categorize them into PERSON,\ + \ LOCATION, ORGANIZATION, or DATE. Ensure the output strictly follows the format:\ + \ label: entity $$ label: entity, with each unique entity on a separate label line,\ + \ avoiding grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries\ + \ like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_tw_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_xh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_xh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6a697b274e71c54293edab7d6dbefabd719843ea --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_xh.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: xh +doc_to_text: "You are a Named Entity Recognition expert in isiXhosa language. \nExtract\ + \ all named entities from the following isiXhosa text and categorize them into PERSON,\ + \ LOCATION, ORGANIZATION, or DATE. Ensure the output strictly follows the format:\ + \ label: entity $$ label: entity, with each unique entity on a separate label line,\ + \ avoiding grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries\ + \ like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_xh_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..76909044e7f35948156f8bb506ce2fce563ec689 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/utils.py @@ -0,0 +1,146 @@ +import collections +import re + +from lm_eval.utils import weighted_f1_score + + +def doc_to_target(doc): + return transform_text(doc["ner_tags"]) + + +def transform_text(text): + entities = [] + current_entity = "" + current_tag = "" + + for pair in text.split("\n"): + if pair: # Check if the line is not empty + word, tag = pair.strip().split(": ") + tag = tag.upper() + word = word.lower() + word = word.strip(",.").strip() + + if tag.startswith("B-"): + if current_entity: + entities.append(f"{current_tag}: {current_entity}") + current_tag = tag.split("-")[1] + current_entity = word + elif tag.startswith("I-") and tag.split("-")[1] == current_tag: + current_entity += word + else: + if current_entity: + entities.append(f"{current_tag}: {current_entity}") + current_entity = "" + current_tag = "" + if current_entity: + entities.append(f"{current_tag}: {current_entity}") + + # Join all the transformed output lines with $$ as separator + return " $$ ".join(entities) + + +def span_f1_agg(items): + """Computes Span based F1 score. + + This function is copied from + https://github.com/google-research/multilingual-t5/blob/master/multilingual_t5/evaluation/metrics.py + + Args: + targets: list of strings or list of list of strings if multiple references + are present. + predictions: list of strings + + Returns: + span f1 across all targets and predictions (Based on CoNLL script) + """ + unzipped_list = list(zip(*items)) + targets = unzipped_list[0] + predictions = unzipped_list[1] + + true_positives = collections.defaultdict(int) + false_positives = collections.defaultdict(int) + false_negatives = collections.defaultdict(int) + + def normalize_text(strings): + def get_blank_spaces_pattern(): + return re.compile(r"\s{3,}|\t") + + def remove_blank_spaces(text): + text = re.sub(pattern=get_blank_spaces_pattern(), repl="", string=text) + text = re.sub("\s+", " ", text) + return text + + def remove_punctuation(text): + my_punctuation = '!"$%&\'()*+,-./:;<=>?[\\]^_`{|}~•@.""-,`' + text = re.sub( + "[" + my_punctuation + "]+", " ", str(text) + ) # strip punctuation + return text + + def remove_articles(text): + regex = re.compile(r"\b(a|an|the)\b", re.UNICODE) + return re.sub(regex, " ", text) + + def lowercase(text): + text = text.lower() + return text + + strings = remove_punctuation(strings) + strings = remove_articles(strings) + strings = remove_blank_spaces(strings) + strings = lowercase(strings) + + return strings + + def tags_to_spans(tag_sequence, delimiter="$$"): + """Extract spans from IOB1 or BIO tags.""" + if isinstance(tag_sequence, list): + tag_sequence = " ".join(i.strip() for i in tag_sequence) + tag_sequence_split = [ + item.strip() + for sub in tag_sequence.strip().split(delimiter) + for item in sub.split("$") + if item + ] + tag_sequence_split = [ + item.strip() + for value in tag_sequence_split + for sub in value.split(". ") + for item in sub.split(", ") + ] + tags_entities = [] + for tag_entity in tag_sequence_split: + tag_entity_split = tag_entity.split(": ") + if len(tag_entity_split) != 2: + continue + tag = normalize_text(tag_entity_split[0].strip()) + entity = normalize_text(tag_entity_split[1].rstrip().lstrip()) + tags_entities.append((tag, entity)) + return tags_entities + + def compute_f1_metrics(true_positive, false_positive, false_negative): + precision = float(true_positive) / float(true_positive + false_positive + 1e-13) + recall = float(true_positive) / float(true_positive + false_negative + 1e-13) + f1_measures = 2.0 * ((precision * recall) / (precision + recall + 1e-13)) + return precision, recall, f1_measures + + for target, pred in zip(targets, predictions): + gold_spans = tags_to_spans(target) + predicted_spans = tags_to_spans(pred) + + for span in predicted_spans: + if span in gold_spans: + true_positives[span[0]] += 1 + gold_spans.remove(span) + else: + false_positives[span[0]] += 1 + # These spans weren't predicted. + for span in gold_spans: + false_negatives[span[0]] += 1 + + _, _, f1_measure = compute_f1_metrics( + sum(true_positives.values()), + sum(false_positives.values()), + sum(false_negatives.values()), + ) + return f1_measure diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner new file mode 100644 index 0000000000000000000000000000000000000000..5c0ae52e62da27b202b794918ac568747195de34 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner @@ -0,0 +1,26 @@ +tag: +- masakhaner_tasks +- masakhaner_prompt_4 +dataset_path: masakhane/masakhaner-x +dataset_name: null +dataset_kwargs: {trust_remote_code: True} +output_type: generate_until +generation_kwargs: + do_sample: false + until: + - + - <|im_end|> +validation_split: validation +test_split: test +fewshot_split: train +doc_to_target: target +filter_list: + - name: flexible-extract + filter: + - function: format_span +metric_list: + - metric: f1 + aggregation: !function utils.span_f1_agg + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_am.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_am.yaml new file mode 100644 index 0000000000000000000000000000000000000000..19b06221f2366dbeb7418961ecbf00f6b1146f1d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_am.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: am +doc_to_text: "As a Amharic linguist, label all named entities in the Amharic text\ + \ below with the categories: PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the\ + \ output strictly follows the format: label: entity $$ label: entity, with each\ + \ unique entity on a separate label line, avoiding grouped entities (e.g., avoid\ + \ LOC: entity, entity) or irrelevant entries like none. Return only the output.\ + \ \n\nText: {{text}}" +include: masakhaner +task: masakhaner_am_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_bbj.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_bbj.yaml new file mode 100644 index 0000000000000000000000000000000000000000..03ed5210a03e4ce5bc9afde45166519152a153c7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_bbj.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: bbj +doc_to_text: "As a Ghomala linguist, label all named entities in the Ghomala text\ + \ below with the categories: PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the\ + \ output strictly follows the format: label: entity $$ label: entity, with each\ + \ unique entity on a separate label line, avoiding grouped entities (e.g., avoid\ + \ LOC: entity, entity) or irrelevant entries like none. Return only the output.\ + \ \n\nText: {{text}}" +include: masakhaner +task: masakhaner_bbj_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_bm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_bm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e719db9ac9f7ffb2011e7af1f12bf6319cb1b9cc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_bm.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: bm +doc_to_text: "As a Bambara linguist, label all named entities in the Bambara text\ + \ below with the categories: PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the\ + \ output strictly follows the format: label: entity $$ label: entity, with each\ + \ unique entity on a separate label line, avoiding grouped entities (e.g., avoid\ + \ LOC: entity, entity) or irrelevant entries like none. Return only the output.\ + \ \n\nText: {{text}}" +include: masakhaner +task: masakhaner_bm_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_ee.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_ee.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fe5fc75d28eef016bac579673cf1db875ee0a9f3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_ee.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: ee +doc_to_text: "As a Ewe linguist, label all named entities in the Ewe text below with\ + \ the categories: PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the output strictly\ + \ follows the format: label: entity $$ label: entity, with each unique entity on\ + \ a separate label line, avoiding grouped entities (e.g., avoid LOC: entity, entity)\ + \ or irrelevant entries like none. Return only the output. \n\nText: {{text}}" +include: masakhaner +task: masakhaner_ee_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_ha.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_ha.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9f88b9d19d4545c1131573a2303c6014421ed54b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_ha.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: ha +doc_to_text: "As a Hausa linguist, label all named entities in the Hausa text below\ + \ with the categories: PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the output\ + \ strictly follows the format: label: entity $$ label: entity, with each unique\ + \ entity on a separate label line, avoiding grouped entities (e.g., avoid LOC: entity,\ + \ entity) or irrelevant entries like none. Return only the output. \n\nText: {{text}}" +include: masakhaner +task: masakhaner_ha_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_ig.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_ig.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d4712d7e8edd36a87c59c9f0bc759f7b884ec830 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_ig.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: ig +doc_to_text: "As a Igbo linguist, label all named entities in the Igbo text below\ + \ with the categories: PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the output\ + \ strictly follows the format: label: entity $$ label: entity, with each unique\ + \ entity on a separate label line, avoiding grouped entities (e.g., avoid LOC: entity,\ + \ entity) or irrelevant entries like none. Return only the output. \n\nText: {{text}}" +include: masakhaner +task: masakhaner_ig_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_lg.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_lg.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cd7bde4a6f098bc8e946d83b556a00e38de43a5b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_lg.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: lg +doc_to_text: "As a Luganda linguist, label all named entities in the Luganda text\ + \ below with the categories: PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the\ + \ output strictly follows the format: label: entity $$ label: entity, with each\ + \ unique entity on a separate label line, avoiding grouped entities (e.g., avoid\ + \ LOC: entity, entity) or irrelevant entries like none. Return only the output.\ + \ \n\nText: {{text}}" +include: masakhaner +task: masakhaner_lg_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_luo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_luo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..92c0ddfa2c58c90a9da84e3dd3e002f9eb8c1098 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_luo.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: luo +doc_to_text: "As a Luo linguist, label all named entities in the Luo text below with\ + \ the categories: PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the output strictly\ + \ follows the format: label: entity $$ label: entity, with each unique entity on\ + \ a separate label line, avoiding grouped entities (e.g., avoid LOC: entity, entity)\ + \ or irrelevant entries like none. Return only the output. \n\nText: {{text}}" +include: masakhaner +task: masakhaner_luo_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_ny.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_ny.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e8cb8218aff23a34f48455b2cbb897112a407f06 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_ny.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: ny +doc_to_text: "As a Chichewa linguist, label all named entities in the Chichewa text\ + \ below with the categories: PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the\ + \ output strictly follows the format: label: entity $$ label: entity, with each\ + \ unique entity on a separate label line, avoiding grouped entities (e.g., avoid\ + \ LOC: entity, entity) or irrelevant entries like none. Return only the output.\ + \ \n\nText: {{text}}" +include: masakhaner +task: masakhaner_ny_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_sn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_sn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..40230fb1cebb4f1aa9d9001360389ef1cdfda64e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_sn.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: sn +doc_to_text: "As a chiShona linguist, label all named entities in the chiShona text\ + \ below with the categories: PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the\ + \ output strictly follows the format: label: entity $$ label: entity, with each\ + \ unique entity on a separate label line, avoiding grouped entities (e.g., avoid\ + \ LOC: entity, entity) or irrelevant entries like none. Return only the output.\ + \ \n\nText: {{text}}" +include: masakhaner +task: masakhaner_sn_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_sw.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_sw.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7b27554ddd1faa7a796b5249c340b4204fdfaa5a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_sw.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: sw +doc_to_text: "As a Kiswahili linguist, label all named entities in the Kiswahili text\ + \ below with the categories: PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the\ + \ output strictly follows the format: label: entity $$ label: entity, with each\ + \ unique entity on a separate label line, avoiding grouped entities (e.g., avoid\ + \ LOC: entity, entity) or irrelevant entries like none. Return only the output.\ + \ \n\nText: {{text}}" +include: masakhaner +task: masakhaner_sw_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_tn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_tn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..88080456ba554e20a63f3c68638908b31d5294cd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_tn.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: tn +doc_to_text: "As a Setswana linguist, label all named entities in the Setswana text\ + \ below with the categories: PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the\ + \ output strictly follows the format: label: entity $$ label: entity, with each\ + \ unique entity on a separate label line, avoiding grouped entities (e.g., avoid\ + \ LOC: entity, entity) or irrelevant entries like none. Return only the output.\ + \ \n\nText: {{text}}" +include: masakhaner +task: masakhaner_tn_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_xh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_xh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4b29fda3f444dd526ee7cc94f6579e74b6f63b97 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_xh.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: xh +doc_to_text: "As a isiXhosa linguist, label all named entities in the isiXhosa text\ + \ below with the categories: PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the\ + \ output strictly follows the format: label: entity $$ label: entity, with each\ + \ unique entity on a separate label line, avoiding grouped entities (e.g., avoid\ + \ LOC: entity, entity) or irrelevant entries like none. Return only the output.\ + \ \n\nText: {{text}}" +include: masakhaner +task: masakhaner_xh_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner new file mode 100644 index 0000000000000000000000000000000000000000..09cd77e13106cca8862dcfa31b86c7742b97985a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner @@ -0,0 +1,26 @@ +tag: +- masakhaner_tasks +- masakhaner_prompt_5 +dataset_path: masakhane/masakhaner-x +dataset_name: null +dataset_kwargs: {trust_remote_code: True} +output_type: generate_until +generation_kwargs: + do_sample: false + until: + - + - <|im_end|> +validation_split: validation +test_split: test +fewshot_split: train +doc_to_target: target +filter_list: + - name: flexible-extract + filter: + - function: format_span +metric_list: + - metric: f1 + aggregation: !function utils.span_f1_agg + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_bm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_bm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c97e0c22a609ceb20a40b5185a58baad850e4e81 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_bm.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: bm +doc_to_text: "Provide a concise list of named entities in the text below. Use the\ + \ following labels: PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the output\ + \ strictly follows the format: label: entity $$ label: entity, with each unique\ + \ entity on a separate label line, avoiding grouped entities (e.g., avoid LOC: entity,\ + \ entity) or irrelevant entries like none. Return only the output. \n\nText: {{text}}" +include: masakhaner +task: masakhaner_bm_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_ee.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_ee.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6371649d2db361bfda7fabe8ed353ca529324375 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_ee.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: ee +doc_to_text: "Provide a concise list of named entities in the text below. Use the\ + \ following labels: PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the output\ + \ strictly follows the format: label: entity $$ label: entity, with each unique\ + \ entity on a separate label line, avoiding grouped entities (e.g., avoid LOC: entity,\ + \ entity) or irrelevant entries like none. Return only the output. \n\nText: {{text}}" +include: masakhaner +task: masakhaner_ee_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_ha.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_ha.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1d68c7eed339d51ba9d72863624a2ee9842be64e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_ha.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: ha +doc_to_text: "Provide a concise list of named entities in the text below. Use the\ + \ following labels: PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the output\ + \ strictly follows the format: label: entity $$ label: entity, with each unique\ + \ entity on a separate label line, avoiding grouped entities (e.g., avoid LOC: entity,\ + \ entity) or irrelevant entries like none. Return only the output. \n\nText: {{text}}" +include: masakhaner +task: masakhaner_ha_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_ny.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_ny.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e400f10e186f0e4ae6119fa622065a23da73c680 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_ny.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: ny +doc_to_text: "Provide a concise list of named entities in the text below. Use the\ + \ following labels: PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the output\ + \ strictly follows the format: label: entity $$ label: entity, with each unique\ + \ entity on a separate label line, avoiding grouped entities (e.g., avoid LOC: entity,\ + \ entity) or irrelevant entries like none. Return only the output. \n\nText: {{text}}" +include: masakhaner +task: masakhaner_ny_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/masakhanews b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/masakhanews new file mode 100644 index 0000000000000000000000000000000000000000..282a38422e526b9f8ce8731f950ae04e2a6cbf08 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/masakhanews @@ -0,0 +1,43 @@ +tag: +- masakhanews_tasks +- masakhanews_prompt_1 +- afrobench_TC_tasks +dataset_path: masakhane/masakhanews +dataset_name: null +output_type: multiple_choice +validation_split: validation +test_split: test +fewshot_split: validation +doc_to_target: label +doc_to_choice: + - "business" + - "entertainment" + - "health" + - "politics" + - "religion" + - "sports" + - "technology" +should_decontaminate: true +doc_to_decontamination_query: headline_text +metric_list: + - metric: f1 + aggregation: !function utils.weighted_f1_score + # aggregation: mean + average: weighted + hf_evaluate: true + higher_is_better: True + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" + - metric: acc + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" +metadata: + version: 1.0