diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fdbbec809a0b21cabc58defd6cbac15d0ea29ff2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_3/afrixnli_translate_zul.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: zul +doc_to_text: "Given the following premise and hypothesis in Zulu, identify if the\ + \ premise entails, contradicts, or is neutral towards the hypothesis. Please respond\ + \ with exact 'entailment', 'contradiction', or 'neutral'. \n\nPremise: {{premise}}\ + \ \nHypothesis: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_zul_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..107c663428d39e3eaa565315a40e4aa5f4b53201 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrixnli/translate/prompt_5/afrixnli_translate_yor.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: yor +doc_to_text: "Based on the given statement, is the following claim 'true', 'false',\ + \ or 'inconclusive'. \nStatement: {{premise}} \nClaim: {{hypothesis}}" +include: afrixnli_translate_yaml +task: afrixnli_translate_yor_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_1/afridiacritics_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_1/afridiacritics_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..983bc3914c421f46fa1adbdd67c15c433996a584 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_1/afridiacritics_ibo.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: ibo +doc_to_text: 'Please restore the missing diacritics in the following sentence: {{text}}. + Return output sentence only' +include: afridiacritics_yaml +task: afridiacritics_ibo_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_1/afridiacritics_wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_1/afridiacritics_wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9067770a3dac5fc4aab23cdad2d15ede76b82de4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_1/afridiacritics_wol.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: wol +doc_to_text: 'Please restore the missing diacritics in the following sentence: {{text}}. + Return output sentence only' +include: afridiacritics_yaml +task: afridiacritics_wol_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_2/afridiacritics_bbj.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_2/afridiacritics_bbj.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0f054eea4c29da978b830d3a5eb2571af364f920 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_2/afridiacritics_bbj.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: bbj +doc_to_text: "Given a sentence without diacritics, add the appropriate diacritics\ + \ to make it grammatically and semantically correct. \nSentence: {{text}}. Return\ + \ output sentence only" +include: afridiacritics_yaml +task: afridiacritics_bbj_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_2/afridiacritics_fon.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_2/afridiacritics_fon.yaml new file mode 100644 index 0000000000000000000000000000000000000000..07f7114649ff8f362a0a2072995724290c5224bd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_2/afridiacritics_fon.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: fon +doc_to_text: "Given a sentence without diacritics, add the appropriate diacritics\ + \ to make it grammatically and semantically correct. \nSentence: {{text}}. Return\ + \ output sentence only" +include: afridiacritics_yaml +task: afridiacritics_fon_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_3/afridiacritics_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_3/afridiacritics_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..04d1df0e1f07ac7c082bd75b1ce93959e0e0d56d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_3/afridiacritics_ibo.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: ibo +doc_to_text: 'This text is in Igbo. Restore all diacritical marks to their proper + places in the following sentence: {{text}}. Return output sentence only' +include: afridiacritics_yaml +task: afridiacritics_ibo_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_3/afridiacritics_wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_3/afridiacritics_wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..576e0845188b523be7ec2f342a440174aa496263 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_3/afridiacritics_wol.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: wol +doc_to_text: 'This text is in Wolof. Restore all diacritical marks to their proper + places in the following sentence: {{text}}. Return output sentence only' +include: afridiacritics_yaml +task: afridiacritics_wol_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_3/afridiacritics_yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_3/afridiacritics_yaml new file mode 100644 index 0000000000000000000000000000000000000000..0a27eeef2d37880527c7b99f1fa9296f843b72a0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_3/afridiacritics_yaml @@ -0,0 +1,25 @@ +tag: +- adr_tasks +- adr_prompt_3 +dataset_path: masakhane/diacritics-restoration +dataset_kwargs: {trust_remote_code: True} +doc_to_target: target +output_type: generate_until +fewshot_split: dev +test_split: test +training_split: train +metric_list: + - metric: bleu + aggregation: bleu + higher_is_better: true + - metric: chrf + aggregation: chrf + higher_is_better: true +generation_kwargs: + do_sample: false + until: + - '' + - + - <|im_end|> +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_3/afridiacritics_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_3/afridiacritics_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..169c110872d6fdc2d2d41b6472fe30d93934f5df --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_3/afridiacritics_yor.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: yor +doc_to_text: 'This text is in Yoruba. Restore all diacritical marks to their proper + places in the following sentence: {{text}}. Return output sentence only' +include: afridiacritics_yaml +task: afridiacritics_yor_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_4/afridiacritics_bbj.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_4/afridiacritics_bbj.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7a807b09ee3000f022374e31e61cdb2f2e091f0e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_4/afridiacritics_bbj.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: bbj +doc_to_text: 'You are a linguist specializing in diacritical marks for Gbomala. Add + the appropriate diacritics to this Gbomala sentence: {{text}}. Return output sentence + only' +include: afridiacritics_yaml +task: afridiacritics_bbj_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_4/afridiacritics_fon.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_4/afridiacritics_fon.yaml new file mode 100644 index 0000000000000000000000000000000000000000..11076e685ae5f6a4435a486d3f35db269ded8f51 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_4/afridiacritics_fon.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: fon +doc_to_text: 'You are a linguist specializing in diacritical marks for Fon. Add the + appropriate diacritics to this Fon sentence: {{text}}. Return output sentence only' +include: afridiacritics_yaml +task: afridiacritics_fon_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_4/afridiacritics_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_4/afridiacritics_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..367e387ae7456f57d604b1fe3ac032084b16fb98 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_4/afridiacritics_ibo.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: ibo +doc_to_text: 'You are a linguist specializing in diacritical marks for Igbo. Add the + appropriate diacritics to this Igbo sentence: {{text}}. Return output sentence only' +include: afridiacritics_yaml +task: afridiacritics_ibo_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_4/afridiacritics_wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_4/afridiacritics_wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..23fb81e754445e8745d5d67720707af7d502e3df --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_4/afridiacritics_wol.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: wol +doc_to_text: 'You are a linguist specializing in diacritical marks for Wolof. Add + the appropriate diacritics to this Wolof sentence: {{text}}. Return output sentence + only' +include: afridiacritics_yaml +task: afridiacritics_wol_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_4/afridiacritics_yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_4/afridiacritics_yaml new file mode 100644 index 0000000000000000000000000000000000000000..6ae62e9d3384d3ee1bff044dbfd1cb23275ae517 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_4/afridiacritics_yaml @@ -0,0 +1,25 @@ +tag: +- adr_tasks +- adr_prompt_4 +dataset_path: masakhane/diacritics-restoration +dataset_kwargs: {trust_remote_code: True} +doc_to_target: target +output_type: generate_until +fewshot_split: dev +test_split: test +training_split: train +metric_list: + - metric: bleu + aggregation: bleu + higher_is_better: true + - metric: chrf + aggregation: chrf + higher_is_better: true +generation_kwargs: + do_sample: false + until: + - '' + - + - <|im_end|> +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_5/afridiacritics_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_5/afridiacritics_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6cc9865dca7a2d8889df88d73abfb54b615089f4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_5/afridiacritics_ibo.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: ibo +doc_to_text: 'You are a linguist specializing in diacritical marks for Igbo. Diacritics + are essential for proper pronunciation and meaning in Igbo. You are tasked with + converting Igbo sentences without diacritics into their correctly accented forms. + Here''s the input: {{text}}. Return output sentence only' +include: afridiacritics_yaml +task: afridiacritics_ibo_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_5/afridiacritics_wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_5/afridiacritics_wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fed10a7031ac71a948720b15eff1677df411934c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_5/afridiacritics_wol.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: wol +doc_to_text: 'You are a linguist specializing in diacritical marks for Wolof. Diacritics + are essential for proper pronunciation and meaning in Wolof. You are tasked with + converting Wolof sentences without diacritics into their correctly accented forms. + Here''s the input: {{text}}. Return output sentence only' +include: afridiacritics_yaml +task: afridiacritics_wol_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_5/afridiacritics_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_5/afridiacritics_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bd1c9007a394de95a19c1a09b398614366537a1f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/adr/prompt_5/afridiacritics_yor.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: yor +doc_to_text: 'You are a linguist specializing in diacritical marks for Yoruba. Diacritics + are essential for proper pronunciation and meaning in Yoruba. You are tasked with + converting Yoruba sentences without diacritics into their correctly accented forms. + Here''s the input: {{text}}. Return output sentence only' +include: afridiacritics_yaml +task: afridiacritics_yor_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/README.md b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/README.md new file mode 100644 index 0000000000000000000000000000000000000000..8730d7c8d8d68b6b83dfad3d4f584534b048d111 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/README.md @@ -0,0 +1,24 @@ +# + +## Paper +Title: `AfriQA: Cross-lingual Open-Retrieval Question Answering for African Languages` + +Paper Link: https://arxiv.org/abs/2305.06897 + +## Abstract +>AfriQA is the first cross-lingual question answering (QA) dataset with a focus on African languages. The dataset includes over 12,000 XOR QA examples across 10 African languages, making it an invaluable resource for developing more equitable QA technology. African languages have historically been underserved in the digital landscape, with far less in-language content available online. This makes it difficult for QA systems to provide accurate information to users in their native language. However, cross-lingual open-retrieval question answering (XOR QA) systems can help fill this gap by retrieving answer content from other languages. AfriQA focuses specifically on African languages where cross-lingual answer content is the only high-coverage source of information. Previous datasets have primarily focused on languages where cross-lingual QA augments coverage from the target language, but AfriQA highlights the importance of African languages as a realistic use case for XOR QA. + +HomePage: https://github.com/masakhane-io/afriqa + +### Citation + +``` +@misc{ogundepo2023afriqa, + title={AfriQA: Cross-lingual Open-Retrieval Question Answering for African Languages}, + author={Odunayo Ogundepo and Tajuddeen R. Gwadabe and Clara E. Rivera and Jonathan H. Clark and Sebastian Ruder and David Ifeoluwa Adelani and Bonaventure F. P. Dossou and Abdou Aziz DIOP and Claytone Sikasote and Gilles Hacheme and Happy Buzaaba and Ignatius Ezeani and Rooweither Mabuya and Salomey Osei and Chris Emezue and Albert Njoroge Kahira and Shamsuddeen H. Muhammad and Akintunde Oladipo and Abraham Toluwase Owodunni and Atnafu Lambebo Tonja and Iyanuoluwa Shode and Akari Asai and Tunde Oluwaseyi Ajayi and Clemencia Siro and Steven Arthur and Mofetoluwa Adeyemi and Orevaoghene Ahia and Aremu Anuoluwapo and Oyinkansola Awosan and Chiamaka Chukwuneke and Bernard Opoku and Awokoya Ayodele and Verrah Otiende and Christine Mwase and Boyd Sinkala and Andre Niyongabo Rubungo and Daniel A. Ajisafe and Emeka Felix Onwuegbuzia and Habib Mbow and Emile Niyomutabazi and Eunice Mukonde and Falalu Ibrahim Lawan and Ibrahim Said Ahmad and Jesujoba O. Alabi and Martin Namukombo and Mbonu Chinedu and Mofya Phiri and Neo Putini and Ndumiso Mngoma and Priscilla A. Amuok and Ruqayya Nasir Iro and Sonia Adhiambo}, + year={2023}, + eprint={2305.06897}, + archivePrefix={arXiv}, + primaryClass={cs.CL} +} +``` diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/afriqa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/afriqa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..80810ca4c1195f281b6eaa9581bf420ff4582291 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/afriqa.yaml @@ -0,0 +1,13 @@ +group: afriqa +task: + - afriqa_prompt_1 + - afriqa_prompt_2 + - afriqa_prompt_3 + - afriqa_prompt_4 + - afriqa_prompt_5 +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_1/afriqa b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_1/afriqa new file mode 100644 index 0000000000000000000000000000000000000000..d9b6218e766a57309804e9514cf9d9682cf49131 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_1/afriqa @@ -0,0 +1,42 @@ +tag: + - afrobench_xqa_tasks + - afriqa_prompt_1 +dataset_kwargs: {trust_remote_code: True} +dataset_path: masakhane/afriqa-gold-passages +dataset_name: null +output_type: generate_until +test_split: test +fewshot_split: train +doc_to_target: answer_pivot +should_decontaminate: true +doc_to_decontamination_query: question_lang +generation_kwargs: + until: + - "\n" + do_sample: false + temperature: 0.0 +filter_list: + - name: remove_whitespace + filter: + - function: remove_whitespace + - function: take_first +target_delimiter: " " +metric_list: + - metric: exact_match + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true + - "." + - "," + - "\\$" + - metric: f1 + aggregation: !function utils.f1 + higher_is_better: true + ignore_case: true + ignore_punctuation: true + - "." + - "," + - "\\$" +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_1/afriqa_bem.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_1/afriqa_bem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a3b639a833ee4e8fb32d992538b747ab92b1f360 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_1/afriqa_bem.yaml @@ -0,0 +1,12 @@ +# Generated by utils.py +dataset_name: bem +doc_to_text: 'Your task is to answer a qestion given a context.Make sure you respond + with the shortest span containing the answer in the context. + + Question: {{question_lang}} + + Context: {{context}} + + Answer:' +include: afriqa +task: afriqa_bem_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_1/afriqa_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_1/afriqa_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0536590ac3486f815ac35d5333b8a6a268fd851a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_1/afriqa_hau.yaml @@ -0,0 +1,12 @@ +# Generated by utils.py +dataset_name: hau +doc_to_text: 'Your task is to answer a qestion given a context.Make sure you respond + with the shortest span containing the answer in the context. + + Question: {{question_lang}} + + Context: {{context}} + + Answer:' +include: afriqa +task: afriqa_hau_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_1/afriqa_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_1/afriqa_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..62eb71160c6cf546b274b8724cd8955c0b0e86c8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_1/afriqa_ibo.yaml @@ -0,0 +1,12 @@ +# Generated by utils.py +dataset_name: ibo +doc_to_text: 'Your task is to answer a qestion given a context.Make sure you respond + with the shortest span containing the answer in the context. + + Question: {{question_lang}} + + Context: {{context}} + + Answer:' +include: afriqa +task: afriqa_ibo_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_1/afriqa_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_1/afriqa_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e632c4beef739a3e299280071229521db78cbf21 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_1/afriqa_kin.yaml @@ -0,0 +1,12 @@ +# Generated by utils.py +dataset_name: kin +doc_to_text: 'Your task is to answer a qestion given a context.Make sure you respond + with the shortest span containing the answer in the context. + + Question: {{question_lang}} + + Context: {{context}} + + Answer:' +include: afriqa +task: afriqa_kin_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_1/afriqa_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_1/afriqa_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..67ba171569e34842e2c5af86875d2495f4715421 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_1/afriqa_twi.yaml @@ -0,0 +1,12 @@ +# Generated by utils.py +dataset_name: twi +doc_to_text: 'Your task is to answer a qestion given a context.Make sure you respond + with the shortest span containing the answer in the context. + + Question: {{question_lang}} + + Context: {{context}} + + Answer:' +include: afriqa +task: afriqa_twi_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_1/afriqa_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_1/afriqa_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..51d20e43e0b4dcb6d5da0e61c22ad827085b253f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_1/afriqa_yor.yaml @@ -0,0 +1,12 @@ +# Generated by utils.py +dataset_name: yor +doc_to_text: 'Your task is to answer a qestion given a context.Make sure you respond + with the shortest span containing the answer in the context. + + Question: {{question_lang}} + + Context: {{context}} + + Answer:' +include: afriqa +task: afriqa_yor_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_1/afriqa_zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_1/afriqa_zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1c254b96e565e01750c6a838b427f926ed3f40d0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_1/afriqa_zul.yaml @@ -0,0 +1,12 @@ +# Generated by utils.py +dataset_name: zul +doc_to_text: 'Your task is to answer a qestion given a context.Make sure you respond + with the shortest span containing the answer in the context. + + Question: {{question_lang}} + + Context: {{context}} + + Answer:' +include: afriqa +task: afriqa_zul_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_1/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_1/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..eae1d885037da14892b39715c49e4d3aac61f06f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_1/utils.py @@ -0,0 +1,53 @@ +import re +import string +from collections import Counter + + +def normalize_answer(s): + """ + Taken from the official evaluation script for v1.1 of the SQuAD dataset. + Lower text and remove punctuation, articles and extra whitespace. + """ + + def remove_articles(text): + return re.sub(r"\b(a|an|the)\b", " ", text) + + def white_space_fix(text): + return " ".join(text.split()) + + def remove_punc(text): + exclude = set(string.punctuation) + return "".join(ch for ch in text if ch not in exclude) + + def lower(text): + return text.lower() + + return white_space_fix(remove_articles(remove_punc(lower(s)))) + + +def f1(items): + """ + Taken from the official evaluation script for v1.1 of the SQuAD dataset. + """ + + unzipped_list = list(zip(*items)) + golds = unzipped_list[0] + preds = unzipped_list[1] + + f1_list = [] + + for i in range(len(golds)): + prediction_tokens = normalize_answer(preds[i]).split() + references_tokens = normalize_answer(golds[i]).split() + common = Counter(prediction_tokens) & Counter(references_tokens) + num_same = sum(common.values()) + if num_same == 0: + f1_score = 0 + else: + precision = 1.0 * num_same / len(prediction_tokens) + recall = 1.0 * num_same / len(references_tokens) + f1_score = (2 * precision * recall) / (precision + recall) + + f1_list.append(f1_score) + + return sum(f1_list) / len(f1_list) diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_2/afriqa b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_2/afriqa new file mode 100644 index 0000000000000000000000000000000000000000..d53ce05b168b8ffaf1325167aaee6537b9b2dbbe --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_2/afriqa @@ -0,0 +1,42 @@ +tag: + - afrobench_xqa_tasks + - afriqa_prompt_2 +dataset_kwargs: {trust_remote_code: True} +dataset_path: masakhane/afriqa-gold-passages +dataset_name: null +output_type: generate_until +test_split: test +fewshot_split: train +doc_to_target: answer_pivot +should_decontaminate: true +doc_to_decontamination_query: question_lang +generation_kwargs: + until: + - "\n" + do_sample: false + temperature: 0.0 +filter_list: + - name: remove_whitespace + filter: + - function: remove_whitespace + - function: take_first +target_delimiter: " " +metric_list: + - metric: exact_match + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true + - "." + - "," + - "\\$" + - metric: f1 + aggregation: !function utils.f1 + higher_is_better: true + ignore_case: true + ignore_punctuation: true + - "." + - "," + - "\\$" +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_2/afriqa_bem.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_2/afriqa_bem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2469c7f434e133f0b94c41940917b17368566dfe --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_2/afriqa_bem.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: bem +doc_to_text: 'Your task is to answer a question given a context. The question is in + Bemba, while the context is in English or French.Make sure you respond with the + shortest span in the context that contains the answer. + + Question: {{question_lang}} + + Context: {{context}} + + Answer:' +include: afriqa +task: afriqa_bem_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_2/afriqa_fon.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_2/afriqa_fon.yaml new file mode 100644 index 0000000000000000000000000000000000000000..384db44987a074c88615164e9da85b71fad2da37 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_2/afriqa_fon.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: fon +doc_to_text: 'Your task is to answer a question given a context. The question is in + Fon, while the context is in English or French.Make sure you respond with the shortest + span in the context that contains the answer. + + Question: {{question_lang}} + + Context: {{context}} + + Answer:' +include: afriqa +task: afriqa_fon_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_2/afriqa_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_2/afriqa_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..40c942eced4451a620541f0cce6fe87d8f82e5cb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_2/afriqa_hau.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: hau +doc_to_text: 'Your task is to answer a question given a context. The question is in + Hausa, while the context is in English or French.Make sure you respond with the + shortest span in the context that contains the answer. + + Question: {{question_lang}} + + Context: {{context}} + + Answer:' +include: afriqa +task: afriqa_hau_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_2/afriqa_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_2/afriqa_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8198795d2d1d79855d45d1c00800e56c8ad5742b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_2/afriqa_ibo.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: ibo +doc_to_text: 'Your task is to answer a question given a context. The question is in + Igbo, while the context is in English or French.Make sure you respond with the shortest + span in the context that contains the answer. + + Question: {{question_lang}} + + Context: {{context}} + + Answer:' +include: afriqa +task: afriqa_ibo_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_2/afriqa_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_2/afriqa_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7a238ae5cc6f687aeb84d88808f994b66464d8bd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_2/afriqa_kin.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: kin +doc_to_text: 'Your task is to answer a question given a context. The question is in + Kinyarwanda, while the context is in English or French.Make sure you respond with + the shortest span in the context that contains the answer. + + Question: {{question_lang}} + + Context: {{context}} + + Answer:' +include: afriqa +task: afriqa_kin_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_2/afriqa_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_2/afriqa_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4be94d07d9ed58745b6e87d38bfe927d20116137 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_2/afriqa_swa.yaml @@ -0,0 +1,16 @@ +# Generated by utils.py +dataset_name: swa +doc_to_text: 'Your task is to answer a question given a context. The question is in + Swahili, while the context is in English or French.Make sure you respond with the + shortest span in the context that contains the answer. + + Question: {{question_lang}} + + Context: {{context}} + + Answer:' +include: afriqa +fewshot_split: test +fewshot_config: + sampler: first_n +task: afriqa_swa_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_2/afriqa_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_2/afriqa_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f08487d0c539b519b2707cc671f7ceda3b50387d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_2/afriqa_twi.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: twi +doc_to_text: 'Your task is to answer a question given a context. The question is in + Twi, while the context is in English or French.Make sure you respond with the shortest + span in the context that contains the answer. + + Question: {{question_lang}} + + Context: {{context}} + + Answer:' +include: afriqa +task: afriqa_twi_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_2/afriqa_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_2/afriqa_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..44aee11a143976b845c8cb3a9c6a1d8cf01ccf17 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_2/afriqa_yor.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: yor +doc_to_text: 'Your task is to answer a question given a context. The question is in + Yoruba, while the context is in English or French.Make sure you respond with the + shortest span in the context that contains the answer. + + Question: {{question_lang}} + + Context: {{context}} + + Answer:' +include: afriqa +task: afriqa_yor_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_2/afriqa_zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_2/afriqa_zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..99c5b18fa243c47f843862950f07e1211745f8b5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_2/afriqa_zul.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: zul +doc_to_text: 'Your task is to answer a question given a context. The question is in + Zulu, while the context is in English or French.Make sure you respond with the shortest + span in the context that contains the answer. + + Question: {{question_lang}} + + Context: {{context}} + + Answer:' +include: afriqa +task: afriqa_zul_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_2/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_2/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..eae1d885037da14892b39715c49e4d3aac61f06f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_2/utils.py @@ -0,0 +1,53 @@ +import re +import string +from collections import Counter + + +def normalize_answer(s): + """ + Taken from the official evaluation script for v1.1 of the SQuAD dataset. + Lower text and remove punctuation, articles and extra whitespace. + """ + + def remove_articles(text): + return re.sub(r"\b(a|an|the)\b", " ", text) + + def white_space_fix(text): + return " ".join(text.split()) + + def remove_punc(text): + exclude = set(string.punctuation) + return "".join(ch for ch in text if ch not in exclude) + + def lower(text): + return text.lower() + + return white_space_fix(remove_articles(remove_punc(lower(s)))) + + +def f1(items): + """ + Taken from the official evaluation script for v1.1 of the SQuAD dataset. + """ + + unzipped_list = list(zip(*items)) + golds = unzipped_list[0] + preds = unzipped_list[1] + + f1_list = [] + + for i in range(len(golds)): + prediction_tokens = normalize_answer(preds[i]).split() + references_tokens = normalize_answer(golds[i]).split() + common = Counter(prediction_tokens) & Counter(references_tokens) + num_same = sum(common.values()) + if num_same == 0: + f1_score = 0 + else: + precision = 1.0 * num_same / len(prediction_tokens) + recall = 1.0 * num_same / len(references_tokens) + f1_score = (2 * precision * recall) / (precision + recall) + + f1_list.append(f1_score) + + return sum(f1_list) / len(f1_list) diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_3/afriqa b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_3/afriqa new file mode 100644 index 0000000000000000000000000000000000000000..79a923b1b30075d31407e804641b71339e9bedb0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_3/afriqa @@ -0,0 +1,42 @@ +tag: + - afrobench_xqa_tasks + - afriqa_prompt_3 +dataset_kwargs: {trust_remote_code: True} +dataset_path: masakhane/afriqa-gold-passages +dataset_name: null +output_type: generate_until +test_split: test +fewshot_split: train +doc_to_target: answer_pivot +should_decontaminate: true +doc_to_decontamination_query: question_lang +generation_kwargs: + until: + - "\n" + do_sample: false + temperature: 0.0 +filter_list: + - name: remove_whitespace + filter: + - function: remove_whitespace + - function: take_first +target_delimiter: " " +metric_list: + - metric: exact_match + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true + - "." + - "," + - "\\$" + - metric: f1 + aggregation: !function utils.f1 + higher_is_better: true + ignore_case: true + ignore_punctuation: true + - "." + - "," + - "\\$" +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_3/afriqa_bem.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_3/afriqa_bem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3af92f5a4abc656a862c701757d056b836506582 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_3/afriqa_bem.yaml @@ -0,0 +1,12 @@ +# Generated by utils.py +dataset_name: bem +doc_to_text: 'Given the context, provide the answer to the following question.Ensure + your response is concise and directly from the context. + + Question: {{question_lang}} + + Context: {{context}} + + Answer:' +include: afriqa +task: afriqa_bem_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_3/afriqa_fon.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_3/afriqa_fon.yaml new file mode 100644 index 0000000000000000000000000000000000000000..73c12439632863ee0cc49a84459abd0dbe4ea985 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_3/afriqa_fon.yaml @@ -0,0 +1,12 @@ +# Generated by utils.py +dataset_name: fon +doc_to_text: 'Given the context, provide the answer to the following question.Ensure + your response is concise and directly from the context. + + Question: {{question_lang}} + + Context: {{context}} + + Answer:' +include: afriqa +task: afriqa_fon_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_3/afriqa_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_3/afriqa_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ff08d081971aa54fd645c9e626231e82341ecabb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_3/afriqa_hau.yaml @@ -0,0 +1,12 @@ +# Generated by utils.py +dataset_name: hau +doc_to_text: 'Given the context, provide the answer to the following question.Ensure + your response is concise and directly from the context. + + Question: {{question_lang}} + + Context: {{context}} + + Answer:' +include: afriqa +task: afriqa_hau_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_3/afriqa_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_3/afriqa_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..12f18a0bff19f69cbfaa0fce86cf9d61ecd08136 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_3/afriqa_ibo.yaml @@ -0,0 +1,12 @@ +# Generated by utils.py +dataset_name: ibo +doc_to_text: 'Given the context, provide the answer to the following question.Ensure + your response is concise and directly from the context. + + Question: {{question_lang}} + + Context: {{context}} + + Answer:' +include: afriqa +task: afriqa_ibo_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_3/afriqa_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_3/afriqa_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e92dec41c227b0d14e2a646893b7fe4dc55e5425 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_3/afriqa_kin.yaml @@ -0,0 +1,12 @@ +# Generated by utils.py +dataset_name: kin +doc_to_text: 'Given the context, provide the answer to the following question.Ensure + your response is concise and directly from the context. + + Question: {{question_lang}} + + Context: {{context}} + + Answer:' +include: afriqa +task: afriqa_kin_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_3/afriqa_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_3/afriqa_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..30c574e5fa77ad65dc91d2670720945cdb67c032 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_3/afriqa_swa.yaml @@ -0,0 +1,15 @@ +# Generated by utils.py +dataset_name: swa +doc_to_text: 'Given the context, provide the answer to the following question.Ensure + your response is concise and directly from the context. + + Question: {{question_lang}} + + Context: {{context}} + + Answer:' +include: afriqa +fewshot_split: test +fewshot_config: + sampler: first_n +task: afriqa_swa_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_3/afriqa_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_3/afriqa_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b08534d9bb98b2603a405aa7d8888a8d4230a52c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_3/afriqa_twi.yaml @@ -0,0 +1,12 @@ +# Generated by utils.py +dataset_name: twi +doc_to_text: 'Given the context, provide the answer to the following question.Ensure + your response is concise and directly from the context. + + Question: {{question_lang}} + + Context: {{context}} + + Answer:' +include: afriqa +task: afriqa_twi_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_3/afriqa_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_3/afriqa_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d3c74ce7c3a86d7df40572096bdddf75ee5321c5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_3/afriqa_yor.yaml @@ -0,0 +1,12 @@ +# Generated by utils.py +dataset_name: yor +doc_to_text: 'Given the context, provide the answer to the following question.Ensure + your response is concise and directly from the context. + + Question: {{question_lang}} + + Context: {{context}} + + Answer:' +include: afriqa +task: afriqa_yor_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_3/afriqa_zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_3/afriqa_zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c54b0bd7f054dc8111c4c72e8fca55d7e7f0a4e6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_3/afriqa_zul.yaml @@ -0,0 +1,12 @@ +# Generated by utils.py +dataset_name: zul +doc_to_text: 'Given the context, provide the answer to the following question.Ensure + your response is concise and directly from the context. + + Question: {{question_lang}} + + Context: {{context}} + + Answer:' +include: afriqa +task: afriqa_zul_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_3/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_3/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..eae1d885037da14892b39715c49e4d3aac61f06f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_3/utils.py @@ -0,0 +1,53 @@ +import re +import string +from collections import Counter + + +def normalize_answer(s): + """ + Taken from the official evaluation script for v1.1 of the SQuAD dataset. + Lower text and remove punctuation, articles and extra whitespace. + """ + + def remove_articles(text): + return re.sub(r"\b(a|an|the)\b", " ", text) + + def white_space_fix(text): + return " ".join(text.split()) + + def remove_punc(text): + exclude = set(string.punctuation) + return "".join(ch for ch in text if ch not in exclude) + + def lower(text): + return text.lower() + + return white_space_fix(remove_articles(remove_punc(lower(s)))) + + +def f1(items): + """ + Taken from the official evaluation script for v1.1 of the SQuAD dataset. + """ + + unzipped_list = list(zip(*items)) + golds = unzipped_list[0] + preds = unzipped_list[1] + + f1_list = [] + + for i in range(len(golds)): + prediction_tokens = normalize_answer(preds[i]).split() + references_tokens = normalize_answer(golds[i]).split() + common = Counter(prediction_tokens) & Counter(references_tokens) + num_same = sum(common.values()) + if num_same == 0: + f1_score = 0 + else: + precision = 1.0 * num_same / len(prediction_tokens) + recall = 1.0 * num_same / len(references_tokens) + f1_score = (2 * precision * recall) / (precision + recall) + + f1_list.append(f1_score) + + return sum(f1_list) / len(f1_list) diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_4/afriqa b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_4/afriqa new file mode 100644 index 0000000000000000000000000000000000000000..e251f1e27fab773d7fd54364ebfc870819df5d55 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_4/afriqa @@ -0,0 +1,42 @@ +tag: + - afrobench_xqa_tasks + - afriqa_prompt_4 +dataset_kwargs: {trust_remote_code: True} +dataset_path: masakhane/afriqa-gold-passages +dataset_name: null +output_type: generate_until +test_split: test +fewshot_split: train +doc_to_target: answer_pivot +should_decontaminate: true +doc_to_decontamination_query: question_lang +generation_kwargs: + until: + - "\n" + do_sample: false + temperature: 0.0 +filter_list: + - name: remove_whitespace + filter: + - function: remove_whitespace + - function: take_first +target_delimiter: " " +metric_list: + - metric: exact_match + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true + - "." + - "," + - "\\$" + - metric: f1 + aggregation: !function utils.f1 + higher_is_better: true + ignore_case: true + ignore_punctuation: true + - "." + - "," + - "\\$" +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_4/afriqa_bem.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_4/afriqa_bem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..db3d1c2a142ae6b6b1bd86678027df58c8715785 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_4/afriqa_bem.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: bem +doc_to_text: 'You are an AI assistant and your task is to answer the question based + on the provided context.Your answer should be the shortest span that contains the + answer within the context. + + Question: {{question_lang}} + + Context: {{context}} + + Answer:' +include: afriqa +task: afriqa_bem_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_4/afriqa_fon.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_4/afriqa_fon.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0c65dd07265d54a8c2aad56c563073fab7b38b49 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_4/afriqa_fon.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: fon +doc_to_text: 'You are an AI assistant and your task is to answer the question based + on the provided context.Your answer should be the shortest span that contains the + answer within the context. + + Question: {{question_lang}} + + Context: {{context}} + + Answer:' +include: afriqa +task: afriqa_fon_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_4/afriqa_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_4/afriqa_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..baeaf020b05b04e025024ac7dd4d3f6bf86caaa9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_4/afriqa_hau.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: hau +doc_to_text: 'You are an AI assistant and your task is to answer the question based + on the provided context.Your answer should be the shortest span that contains the + answer within the context. + + Question: {{question_lang}} + + Context: {{context}} + + Answer:' +include: afriqa +task: afriqa_hau_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_4/afriqa_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_4/afriqa_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6db1cc71614a9e316eeb3b12a3f0740d9cb6e671 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_4/afriqa_ibo.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: ibo +doc_to_text: 'You are an AI assistant and your task is to answer the question based + on the provided context.Your answer should be the shortest span that contains the + answer within the context. + + Question: {{question_lang}} + + Context: {{context}} + + Answer:' +include: afriqa +task: afriqa_ibo_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_4/afriqa_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_4/afriqa_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dc8f3678207cdbc1dab076573ed7aeea02b19b92 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_4/afriqa_kin.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: kin +doc_to_text: 'You are an AI assistant and your task is to answer the question based + on the provided context.Your answer should be the shortest span that contains the + answer within the context. + + Question: {{question_lang}} + + Context: {{context}} + + Answer:' +include: afriqa +task: afriqa_kin_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_4/afriqa_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_4/afriqa_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4fe8fbcdf2a298f9b8c58047da94c24cddc72fdc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_4/afriqa_swa.yaml @@ -0,0 +1,16 @@ +# Generated by utils.py +dataset_name: swa +doc_to_text: 'You are an AI assistant and your task is to answer the question based + on the provided context.Your answer should be the shortest span that contains the + answer within the context. + + Question: {{question_lang}} + + Context: {{context}} + + Answer:' +include: afriqa +fewshot_split: test +fewshot_config: + sampler: first_n +task: afriqa_swa_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_4/afriqa_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_4/afriqa_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d679cd0bb6173696e7555158356b067f2eea895c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_4/afriqa_twi.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: twi +doc_to_text: 'You are an AI assistant and your task is to answer the question based + on the provided context.Your answer should be the shortest span that contains the + answer within the context. + + Question: {{question_lang}} + + Context: {{context}} + + Answer:' +include: afriqa +task: afriqa_twi_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_4/afriqa_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_4/afriqa_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6011dc3313ff35fe55ed204529c5d0c1503d468a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_4/afriqa_yor.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: yor +doc_to_text: 'You are an AI assistant and your task is to answer the question based + on the provided context.Your answer should be the shortest span that contains the + answer within the context. + + Question: {{question_lang}} + + Context: {{context}} + + Answer:' +include: afriqa +task: afriqa_yor_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_4/afriqa_zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_4/afriqa_zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..26a6ccad3efe93ea094a64189b5efbe1dddb9734 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_4/afriqa_zul.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: zul +doc_to_text: 'You are an AI assistant and your task is to answer the question based + on the provided context.Your answer should be the shortest span that contains the + answer within the context. + + Question: {{question_lang}} + + Context: {{context}} + + Answer:' +include: afriqa +task: afriqa_zul_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_4/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_4/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..eae1d885037da14892b39715c49e4d3aac61f06f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_4/utils.py @@ -0,0 +1,53 @@ +import re +import string +from collections import Counter + + +def normalize_answer(s): + """ + Taken from the official evaluation script for v1.1 of the SQuAD dataset. + Lower text and remove punctuation, articles and extra whitespace. + """ + + def remove_articles(text): + return re.sub(r"\b(a|an|the)\b", " ", text) + + def white_space_fix(text): + return " ".join(text.split()) + + def remove_punc(text): + exclude = set(string.punctuation) + return "".join(ch for ch in text if ch not in exclude) + + def lower(text): + return text.lower() + + return white_space_fix(remove_articles(remove_punc(lower(s)))) + + +def f1(items): + """ + Taken from the official evaluation script for v1.1 of the SQuAD dataset. + """ + + unzipped_list = list(zip(*items)) + golds = unzipped_list[0] + preds = unzipped_list[1] + + f1_list = [] + + for i in range(len(golds)): + prediction_tokens = normalize_answer(preds[i]).split() + references_tokens = normalize_answer(golds[i]).split() + common = Counter(prediction_tokens) & Counter(references_tokens) + num_same = sum(common.values()) + if num_same == 0: + f1_score = 0 + else: + precision = 1.0 * num_same / len(prediction_tokens) + recall = 1.0 * num_same / len(references_tokens) + f1_score = (2 * precision * recall) / (precision + recall) + + f1_list.append(f1_score) + + return sum(f1_list) / len(f1_list) diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_5/afriqa_bem.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_5/afriqa_bem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4288845d30d2e3ec1620dd1a226bf17d385322be --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_5/afriqa_bem.yaml @@ -0,0 +1,12 @@ +# Generated by utils.py +dataset_name: bem +doc_to_text: 'Using the context, find the answer to the question.Respond with the + briefest span that includes the answer from the context. + + Question: {{question_lang}} + + Context: {{context}} + + Answer:' +include: afriqa +task: afriqa_bem_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_5/afriqa_fon.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_5/afriqa_fon.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c234e944783b0c456b1d0a0599d43a1d569ad18f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_5/afriqa_fon.yaml @@ -0,0 +1,12 @@ +# Generated by utils.py +dataset_name: fon +doc_to_text: 'Using the context, find the answer to the question.Respond with the + briefest span that includes the answer from the context. + + Question: {{question_lang}} + + Context: {{context}} + + Answer:' +include: afriqa +task: afriqa_fon_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_5/afriqa_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_5/afriqa_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..34823c9e47d2e8d614df530559c638973a02d956 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_5/afriqa_hau.yaml @@ -0,0 +1,12 @@ +# Generated by utils.py +dataset_name: hau +doc_to_text: 'Using the context, find the answer to the question.Respond with the + briefest span that includes the answer from the context. + + Question: {{question_lang}} + + Context: {{context}} + + Answer:' +include: afriqa +task: afriqa_hau_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_5/afriqa_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_5/afriqa_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6582d2d56632e96a3d14c646a47dd7d4b55b1652 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_5/afriqa_ibo.yaml @@ -0,0 +1,12 @@ +# Generated by utils.py +dataset_name: ibo +doc_to_text: 'Using the context, find the answer to the question.Respond with the + briefest span that includes the answer from the context. + + Question: {{question_lang}} + + Context: {{context}} + + Answer:' +include: afriqa +task: afriqa_ibo_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_5/afriqa_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_5/afriqa_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ed9d6517878d8ba4e29342c915648fdb48fdd45e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_5/afriqa_kin.yaml @@ -0,0 +1,12 @@ +# Generated by utils.py +dataset_name: kin +doc_to_text: 'Using the context, find the answer to the question.Respond with the + briefest span that includes the answer from the context. + + Question: {{question_lang}} + + Context: {{context}} + + Answer:' +include: afriqa +task: afriqa_kin_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_5/afriqa_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_5/afriqa_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dfcfb147f8d7789d89e0521129ba1b01f1725384 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_5/afriqa_swa.yaml @@ -0,0 +1,15 @@ +# Generated by utils.py +dataset_name: swa +doc_to_text: 'Using the context, find the answer to the question.Respond with the + briefest span that includes the answer from the context. + + Question: {{question_lang}} + + Context: {{context}} + + Answer:' +include: afriqa +fewshot_split: test +fewshot_config: + sampler: first_n +task: afriqa_swa_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_5/afriqa_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_5/afriqa_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cde555cf760b159378dbc137a4759ab3c87a6b4a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_5/afriqa_twi.yaml @@ -0,0 +1,12 @@ +# Generated by utils.py +dataset_name: twi +doc_to_text: 'Using the context, find the answer to the question.Respond with the + briefest span that includes the answer from the context. + + Question: {{question_lang}} + + Context: {{context}} + + Answer:' +include: afriqa +task: afriqa_twi_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_5/afriqa_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_5/afriqa_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c9fa17e82c271118309bf6769efff0635cf94230 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_5/afriqa_yor.yaml @@ -0,0 +1,12 @@ +# Generated by utils.py +dataset_name: yor +doc_to_text: 'Using the context, find the answer to the question.Respond with the + briefest span that includes the answer from the context. + + Question: {{question_lang}} + + Context: {{context}} + + Answer:' +include: afriqa +task: afriqa_yor_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_5/afriqa_zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_5/afriqa_zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..427e7217f4b113e28ad3efc568b35ab093eed463 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_5/afriqa_zul.yaml @@ -0,0 +1,12 @@ +# Generated by utils.py +dataset_name: zul +doc_to_text: 'Using the context, find the answer to the question.Respond with the + briefest span that includes the answer from the context. + + Question: {{question_lang}} + + Context: {{context}} + + Answer:' +include: afriqa +task: afriqa_zul_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_5/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_5/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..eae1d885037da14892b39715c49e4d3aac61f06f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/prompt_5/utils.py @@ -0,0 +1,53 @@ +import re +import string +from collections import Counter + + +def normalize_answer(s): + """ + Taken from the official evaluation script for v1.1 of the SQuAD dataset. + Lower text and remove punctuation, articles and extra whitespace. + """ + + def remove_articles(text): + return re.sub(r"\b(a|an|the)\b", " ", text) + + def white_space_fix(text): + return " ".join(text.split()) + + def remove_punc(text): + exclude = set(string.punctuation) + return "".join(ch for ch in text if ch not in exclude) + + def lower(text): + return text.lower() + + return white_space_fix(remove_articles(remove_punc(lower(s)))) + + +def f1(items): + """ + Taken from the official evaluation script for v1.1 of the SQuAD dataset. + """ + + unzipped_list = list(zip(*items)) + golds = unzipped_list[0] + preds = unzipped_list[1] + + f1_list = [] + + for i in range(len(golds)): + prediction_tokens = normalize_answer(preds[i]).split() + references_tokens = normalize_answer(golds[i]).split() + common = Counter(prediction_tokens) & Counter(references_tokens) + num_same = sum(common.values()) + if num_same == 0: + f1_score = 0 + else: + precision = 1.0 * num_same / len(prediction_tokens) + recall = 1.0 * num_same / len(references_tokens) + f1_score = (2 * precision * recall) / (precision + recall) + + f1_list.append(f1_score) + + return sum(f1_list) / len(f1_list) diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..5fef58f013ff9d31da0a952e1315cc09b53c2e74 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afriqa/utils.py @@ -0,0 +1,125 @@ +import argparse +import os + +import yaml + + +class FunctionTag: + def __init__(self, value): + self.value = value + + +def prompt_func(mode, lang): + prompt_map = { + "prompt_1": "Your task is to answer a question given a context." + "Make sure you respond with the shortest span containing the answer in the context.\n" + "Question: {{question_lang}}\n" + "Context: {{context}}\n" + "Answer:", + "prompt_2": f"Your task is to answer a question given a context. The question is in {lang}, while the context is in English or French." + "Make sure you respond with the shortest span in the context that contains the answer.\n" + "Question: {{question_lang}}\n" + "Context: {{context}}\n" + "Answer:", + "prompt_3": "Given the context, provide the answer to the following question." + "Ensure your response is concise and directly from the context.\n" + "Question: {{question_lang}}\n" + "Context: {{context}}\n" + "Answer:", + "prompt_4": "You are an AI assistant and your task is to answer the question based on the provided context." + "Your answer should be the shortest span that contains the answer within the context.\n" + "Question: {{question_lang}}\n" + "Context: {{context}}\n" + "Answer:", + "prompt_5": "Using the context, find the answer to the question." + "Respond with the briefest span that includes the answer from the context.\n" + "Question: {{question_lang}}\n" + "Context: {{context}}\n" + "Answer:", + } + return prompt_map[mode] + + +def gen_lang_yamls(output_dir: str, overwrite: bool, mode: str) -> None: + """ + Generate a yaml file for each language. + + :param output_dir: The directory to output the files to. + :param overwrite: Whether to overwrite files if they already exist. + """ + err = [] + languages = { + "bem": "Bemba", + "fon": "Fon", + "hau": "Hausa", + "ibo": "Igbo", + "kin": "Kinyarwanda", + "swa": "Swahili", + "twi": "Twi", + "wol": "Wolof", + "yor": "Yoruba", + "zul": "Zulu", + } + + for lang in languages.keys(): + try: + file_name = f"afriqa_{lang}.yaml" + task_name = f"afriqa_{lang}_{mode}" + yaml_template = "afriqa" + yaml_details = { + "include": yaml_template, + "task": task_name, + "dataset_name": lang, + "doc_to_text": prompt_func(mode, languages[lang]), + } + file_path = os.path.join(output_dir, mode) + os.makedirs(file_path, exist_ok=True) + + with open( + f"{output_dir}/{mode}/{file_name}", + "w" if overwrite else "x", + encoding="utf8", + ) as f: + f.write("# Generated by utils.py\n") + yaml.dump( + yaml_details, + f, + allow_unicode=True, + ) + except FileExistsError: + err.append(file_name) + + if len(err) > 0: + raise FileExistsError( + "Files were not created because they already exist (use --overwrite flag):" + f" {', '.join(err)}" + ) + + +def main() -> None: + """Parse CLI args and generate language-specific yaml files.""" + parser = argparse.ArgumentParser() + parser.add_argument( + "--overwrite", + default=True, + action="store_true", + help="Overwrite files if they already exist", + ) + parser.add_argument( + "--output-dir", + default="./", + help="Directory to write yaml files to", + ) + parser.add_argument( + "--mode", + default="prompt_1", + choices=["prompt_1", "prompt_2", "prompt_3", "prompt_4", "prompt_5"], + help="Prompt number", + ) + args = parser.parse_args() + + gen_lang_yamls(output_dir=args.output_dir, overwrite=args.overwrite, mode=args.mode) + + +if __name__ == "__main__": + main() diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/README.md b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/README.md new file mode 100644 index 0000000000000000000000000000000000000000..99bd489e3eb2cd99eb888a0c2903a4c6259668df --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/README.md @@ -0,0 +1,58 @@ +# + +## Paper +Title: `AfriSenti: A Twitter Sentiment Analysis Benchmark for African Languages` + +Paper Link: https://aclanthology.org/2023.emnlp-main.862/ + +## Abstract +>Africa is home to over 2,000 languages from over six language families and has the highest linguistic diversity among all continents. This includes 75 languages with at least one million speakers each. Yet, there is little NLP research conducted on African languages. Crucial in enabling such research is the availability of high-quality annotated datasets. In this paper, we introduce AfriSenti, a sentiment analysis benchmark that contains a total of >110,000 tweets in 14 African languages (Amharic, Algerian Arabic, Hausa, Igbo, Kinyarwanda, Moroccan Arabic, Mozambican Portuguese, Nigerian Pidgin, Oromo, Swahili, Tigrinya, Twi, Xitsonga, and Yoruba) from four language families. The tweets were annotated by native speakers and used in the AfriSenti-SemEval shared task (with over 200 participants, see website: https://afrisenti-semeval.github.io). We describe the data collection methodology, annotation process, and the challenges we dealt with when curating each dataset. We further report baseline experiments conducted on the AfriSenti datasets and discuss their usefulness. + +HomePage: https://github.com/afrisenti-semeval/afrisent-semeval-2023 + +### Citation + +``` +@inproceedings{muhammad-etal-2023-afrisenti, + title = "{A}fri{S}enti: A {T}witter Sentiment Analysis Benchmark for {A}frican Languages", + author = "Muhammad, Shamsuddeen Hassan and + Abdulmumin, Idris and + Ayele, Abinew Ali and + Ousidhoum, Nedjma and + Adelani, David Ifeoluwa and + Yimam, Seid Muhie and + Ahmad, Ibrahim Sa'id and + Beloucif, Meriem and + Mohammad, Saif M. and + Ruder, Sebastian and + Hourrane, Oumaima and + Brazdil, Pavel and + Jorge, Alipio and + Ali, Felermino D{\'a}rio M{\'a}rio Ant{\'o}nio and + David, Davis and + Osei, Salomey and + Shehu Bello, Bello and + Ibrahim, Falalu and + Gwadabe, Tajuddeen and + Rutunda, Samuel and + Belay, Tadesse and + Messelle, Wendimu Baye and + Balcha, Hailu Beshada and + Chala, Sisay Adugna and + Gebremichael, Hagos Tesfahun and + Opoku, Bernard and + Arthur, Stephen", + editor = "Bouamor, Houda and + Pino, Juan and + Bali, Kalika", + booktitle = "Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing", + month = dec, + year = "2023", + address = "Singapore", + publisher = "Association for Computational Linguistics", + url = "https://aclanthology.org/2023.emnlp-main.862/", + doi = "10.18653/v1/2023.emnlp-main.862", + pages = "13968--13981", + abstract = "Africa is home to over 2,000 languages from over six language families and has the highest linguistic diversity among all continents. This includes 75 languages with at least one million speakers each. Yet, there is little NLP research conducted on African languages. Crucial in enabling such research is the availability of high-quality annotated datasets. In this paper, we introduce AfriSenti, a sentiment analysis benchmark that contains a total of {\ensuremath{>}}110,000 tweets in 14 African languages (Amharic, Algerian Arabic, Hausa, Igbo, Kinyarwanda, Moroccan Arabic, Mozambican Portuguese, Nigerian Pidgin, Oromo, Swahili, Tigrinya, Twi, Xitsonga, and Yoruba) from four language families. The tweets were annotated by native speakers and used in the AfriSenti-SemEval shared task (with over 200 participants, see website: https://afrisenti-semeval.github.io). We describe the data collection methodology, annotation process, and the challenges we dealt with when curating each dataset. We further report baseline experiments conducted on the AfriSenti datasets and discuss their usefulness." +} +``` diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/afrisenti.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/afrisenti.yaml new file mode 100644 index 0000000000000000000000000000000000000000..36a1efdb3033e70060251e346847b73fd9de2f60 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/afrisenti.yaml @@ -0,0 +1,13 @@ +group: afrisenti +task: + - afrisenti_prompt_1 + - afrisenti_prompt_2 + - afrisenti_prompt_3 + - afrisenti_prompt_4 + - afrisenti_prompt_5 +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/fewshot.sh b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/fewshot.sh new file mode 100644 index 0000000000000000000000000000000000000000..428d455b65ac917efee1810a68626f36e777e2d9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/fewshot.sh @@ -0,0 +1,109 @@ +lm_eval --model hf \ + --model_args pretrained=masakhane/African-ultrachat-alpaca \ + --tasks afrimmlu_direct_amh,afrimmlu_direct_eng,afrimmlu_direct_ewe,afrimmlu_direct_fra,afrimmlu_direct_hau,afrimmlu_direct_ibo,afrimmlu_direct_kin,afrimmlu_direct_lin,afrimmlu_direct_lug,afrimmlu_direct_orm,afrimmlu_direct_sna,afrimmlu_direct_sot,afrimmlu_direct_twi,afrimmlu_direct_wol,afrimmlu_direct_xho,afrimmlu_direct_yor,afrimmlu_direct_zul \ + --device cuda:0 \ + --batch_size 1 \ + --num_fewshot 0 \ + --verbosity DEBUG \ + --wandb_args project=afrimmlu + + +lm_eval --model hf \ + --model_args pretrained=bigscience/mt0-small,parallelize=true \ + --tasks afrisenti_amh_prompt_1,afrisenti_arq_prompt_1,afrisenti_ary_prompt_1,afrisenti_hau_prompt_1,afrisenti_ibo_prompt_1,afrisenti_kin_prompt_1,afrisenti_orm_prompt_1,afrisenti_pcm_prompt_1,afrisenti_por_prompt_1,afrisenti_swa_prompt_1,afrisenti_tir_prompt_1,afrisenti_tso_prompt_1,afrisenti_twi_prompt_1,afrisenti_yor_prompt_1\ + --device cuda:0 \ + --batch_size 1 \ + --num_fewshot 0 \ + --verbosity DEBUG \ + --limit 5 + + +lm_eval --model hf \ + --model_args pretrained=bigscience/mt0-xxl,parallelize=true \ + --tasks afrisenti_amh_prompt_1,afrisenti_arq_prompt_1,afrisenti_ary_prompt_1,afrisenti_hau_prompt_1,afrisenti_ibo_prompt_1,afrisenti_kin_prompt_1,afrisenti_orm_prompt_1,afrisenti_pcm_prompt_1,afrisenti_por_prompt_1,afrisenti_swa_prompt_1,afrisenti_tir_prompt_1,afrisenti_tso_prompt_1,afrisenti_twi_prompt_1,afrisenti_yor_prompt_1\ + --batch_size 128 \ + --num_fewshot 0 \ + --verbosity DEBUG + +lm_eval --model hf \ + --model_args pretrained=google/gemma-2-27b-it,parallelize=true,trust_remote_code=True \ + --tasks afriqa_wol_prompt_2\ + --batch_size 1 \ + --device 'cuda' \ + --num_fewshot 5 \ + --verbosity DEBUG \ + --output_path './afriqa_results/' \ + --log_samples + +lm_eval --model vllm \ + --model_args pretrained=meta-llama/Llama-2-7b-chat-hf,tensor_parallel_size=2,dtype='auto',gpu_memory_utilization=0.8,data_parallel_size=1 \ + --tasks masakhapos_pcm_prompt_1,masakhapos_pcm_prompt_2,masakhapos_pcm_prompt_3,masakhapos_pcm_prompt_4,masakhapos_pcm_prompt_5 \ + --batch_size 'auto' \ + --device 'cuda' \ + --num_fewshot 0 \ + --verbosity DEBUG \ + --limit 2 + + +lm_eval --model vllm \ + --model_args pretrained=meta-llama/Llama-2-7b-chat-hf,tensor_parallel_size=2,dtype='auto',gpu_memory_utilization=0.8,data_parallel_size=1 \ + --tasks masakhapos_pcm_prompt_1,masakhapos_pcm_prompt_2,masakhapos_pcm_prompt_3,masakhapos_bam_prompt_2,masakhapos_bbj_prompt_3 \ + --batch_size 'auto' \ + --device 'cuda' \ + --num_fewshot 0 \ + --verbosity DEBUG + +lm_eval --model vllm \ + --model_args pretrained=google/gemma-1.1-7b-it,tensor_parallel_size=2,dtype='auto',gpu_memory_utilization=0.8,data_parallel_size=1 \ + --tasks masakhaner_pcm_prompt_1\ + --batch_size 'auto' \ + --device 'cuda' \ + --num_fewshot 0 \ + --verbosity DEBUG \ + --limit 5 + +lm_eval --model vllm \ + --model_args pretrained=google/gemma-2-9b-it,tensor_parallel_size=2,dtype='auto',gpu_memory_utilization=0.8,data_parallel_size=1 \ + --tasks masakhaner_pcm_prompt_1,masakhaner_pcm_prompt_2,masakhaner_pcm_prompt_3,masakhaner_pcm_prompt_4,masakhaner_pcm_prompt_5\ + --batch_size 'auto' \ + --device 'cuda' \ + --num_fewshot 0 \ + --verbosity DEBUG \ + --limit 5 + +lm_eval --model vllm \ + --model_args pretrained=google/gemma-1.1-7b-it,tensor_parallel_size=2,dtype='auto',gpu_memory_utilization=0.8,data_parallel_size=1 \ + --tasks flores_eng_Latn-fuv_Latn_prompt_1,flores_eng_Latn-fuv_Latn_prompt_2,flores_eng_Latn-fuv_Latn_prompt_3,flores_fuv_Latn-eng_Latn_prompt_1,flores_fuv_Latn-eng_Latn_prompt_2,flores_fuv_Latn-eng_Latn_prompt_3 \ + --batch_size 'auto' \ + --device 'cuda' \ + --num_fewshot 0 \ + --verbosity DEBUG \ + --limit 2 + +lm_eval --model vllm \ + --model_args pretrained=google/gemma-2-27b-it,tensor_parallel_size=2,dtype='auto',gpu_memory_utilization=0.9,data_parallel_size=1 \ + --tasks masakhapos_twi_prompt_3,masakhapos_wol_prompt_3,masakhapos_xho_prompt_3,masakhapos_yor_prompt_3,masakhapos_zul_prompt_3\ + --batch_size 'auto' \ + --num_fewshot 5 \ + --verbosity DEBUG \ + --output_path './masakhapos_results/' \ + --log_samples + +lm_eval --model hf \ + --model_args pretrained=bigscience/mt0-small,parallelize=true \ + --tasks injongointent_amh_prompt_1,injongointent_eng_prompt_1,injongointent_yor_prompt_1,injongointent_ibo_prompt_1,injongointent_wol_prompt_1\ + --device 'mps' \ + --batch_size 1 \ + --num_fewshot 0 \ + --verbosity DEBUG \ + --limit 5 + +lm_eval --model hf \ + --model_args pretrained=google/gemma-3-27b-it,parallelize=true \ + --tasks afrobench_sentiment_tasks\ + --device 'cuda' \ + --batch_size 1 \ + --num_fewshot 0 \ + --verbosity DEBUG \ + --output_path './senti_results/' \ + --log_samples diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_1/afrisenti_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_1/afrisenti_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7eefbe867360070e0701a558c124ad4ad7da786a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_1/afrisenti_amh.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: amh +include: afrisenti +task: afrisenti_amh_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_1/afrisenti_arq.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_1/afrisenti_arq.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8b2e2522d94e2d0da8cb9efec72d225c7161f8e7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_1/afrisenti_arq.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: arq +include: afrisenti +task: afrisenti_arq_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_1/afrisenti_ary.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_1/afrisenti_ary.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8f9ef3f20d654a0e97e40a9dd4e3d6bd2e7d949b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_1/afrisenti_ary.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: ary +include: afrisenti +task: afrisenti_ary_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_1/afrisenti_orm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_1/afrisenti_orm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..65c63b06fbb721d59033e5f748c6899270e92831 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_1/afrisenti_orm.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: orm +include: afrisenti +task: afrisenti_orm_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_1/afrisenti_pcm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_1/afrisenti_pcm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0f24fe9fc01cc162294aa1b387676591450b2d39 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_1/afrisenti_pcm.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: pcm +include: afrisenti +task: afrisenti_pcm_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_1/afrisenti_por.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_1/afrisenti_por.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1e1b4cd60a1533ab6804624c8e967b46c37a69be --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_1/afrisenti_por.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: por +include: afrisenti +task: afrisenti_por_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_1/afrisenti_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_1/afrisenti_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3386948ccf5eef56d293cc39d0811ab06e1a5127 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_1/afrisenti_swa.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: swa +include: afrisenti +task: afrisenti_swa_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_1/afrisenti_tir.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_1/afrisenti_tir.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c4942628e8115f58a047d7819052b27cc50883e9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_1/afrisenti_tir.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: tir +include: afrisenti +task: afrisenti_tir_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_1/afrisenti_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_1/afrisenti_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a68bb23dcaeedcc6c97768c42a1e49c26b425e40 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_1/afrisenti_twi.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: twi +include: afrisenti +task: afrisenti_twi_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_1/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_1/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..3e735e2deb1f9c53152c072615aebe8ba3acb90b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_1/utils.py @@ -0,0 +1 @@ +from lm_eval.utils import weighted_f1_score diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_1/xx.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_1/xx.py new file mode 100644 index 0000000000000000000000000000000000000000..ca0e325e526e33dfd19ba03a93652244977fc119 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_1/xx.py @@ -0,0 +1,13 @@ +from datasets import load_dataset + + +# ['amh', 'hau', 'ibo', 'arq', 'ary', 'yor', 'por', 'twi', 'tso', 'tir', 'orm', 'pcm', 'kin', 'swa'] + +data = load_dataset("masakhane/afrisenti", "pcm", trust_remote_code=True) +print(data) +print(data["test"][:5]) +# +# ['Naija', 'Pipo', 'wey', 'dey', 'for', 'inside', 'social', 'Media', 'sef', 'don', 'put', 'hand', 'for', 'ear', 'give', +# 'federal', 'goment', 'and', 'polical', 'leader', 'dem', 'ova', 'di', 'kilin', '.'] +# +# [6, 0, 14, 17, 2, 2, 6, 0, 7, 17, 16, 0, 2, 0, 16, 0, 0, 9, 0, 0, 11, 2, 8, 0, 1] diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_2/afrisenti b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_2/afrisenti new file mode 100644 index 0000000000000000000000000000000000000000..879f2826c3f26025fcb5e41342f86ef3f9c6c677 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_2/afrisenti @@ -0,0 +1,39 @@ +tag: + - afrobench_sentiment_tasks + - afrisent_prompt_2 +dataset_path: masakhane/afrisenti +dataset_name: null +dataset_kwargs: {trust_remote_code: True} +output_type: multiple_choice +validation_split: validation +test_split: test +fewshot_split: train +doc_to_target: label +doc_to_choice: + - "negative" + - "positive" + - "neutral" +should_decontaminate: true +doc_to_decontamination_query: 'text: {{tweet}} \nlabel: ' +metric_list: + - metric: f1 + aggregation: !function utils.weighted_f1_score + # aggregation: mean + average: weighted + hf_evaluate: true + higher_is_better: True + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" + - metric: acc + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_2/afrisenti_arq.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_2/afrisenti_arq.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c61e310dee3a9f97557aaaa8d465ce329ba29610 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_2/afrisenti_arq.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: arq +doc_to_text: Does this Algerian Arabic statement; '{{tweet}}' have a Neutral, Positive + or Negative sentiment? Labels only +include: afrisenti +task: afrisenti_arq_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_2/afrisenti_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_2/afrisenti_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d4d6c6c8094bb5b41e5488c972f2702705c9f3d5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_2/afrisenti_ibo.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: ibo +doc_to_text: Does this Igbo statement; '{{tweet}}' have a Neutral, Positive or Negative + sentiment? Labels only +include: afrisenti +task: afrisenti_ibo_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_2/afrisenti_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_2/afrisenti_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5067b9fb75cd1579a8bc915a3a010696ca60b177 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_2/afrisenti_kin.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: kin +doc_to_text: Does this Kinyarwanda statement; '{{tweet}}' have a Neutral, Positive + or Negative sentiment? Labels only +include: afrisenti +task: afrisenti_kin_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_2/afrisenti_por.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_2/afrisenti_por.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4b8beecff946f4764becafb103f890ea924926a9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_2/afrisenti_por.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: por +doc_to_text: Does this Mozambique Portuguese statement; '{{tweet}}' have a Neutral, + Positive or Negative sentiment? Labels only +include: afrisenti +task: afrisenti_por_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_2/afrisenti_tso.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_2/afrisenti_tso.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b371b7479489bd5387d2d18cd7f2edab8496dc00 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_2/afrisenti_tso.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: tso +doc_to_text: Does this Xithonga statement; '{{tweet}}' have a Neutral, Positive or + Negative sentiment? Labels only +include: afrisenti +task: afrisenti_tso_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_2/afrisenti_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_2/afrisenti_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c985efc4d32d30ae7f18ed5aac13c97d4dbe112b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_2/afrisenti_twi.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: twi +doc_to_text: Does this Twi statement; '{{tweet}}' have a Neutral, Positive or Negative + sentiment? Labels only +include: afrisenti +task: afrisenti_twi_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_2/afrisenti_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_2/afrisenti_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..78932ed4cfe5f88bc91a6a0d26eb8c33a71c1ecb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_2/afrisenti_yor.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: yor +doc_to_text: Does this Yoruba statement; '{{tweet}}' have a Neutral, Positive or Negative + sentiment? Labels only +include: afrisenti +task: afrisenti_yor_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_2/xx.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_2/xx.py new file mode 100644 index 0000000000000000000000000000000000000000..4aa0db7af761fd7ea8858383b4564130a374f223 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_2/xx.py @@ -0,0 +1,5 @@ +from datasets import load_dataset + + +data = load_dataset("HausaNLP/AfriSenti-Twitter", "yor", trust_remote_code=True) +print(data) diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_3/afrisenti b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_3/afrisenti new file mode 100644 index 0000000000000000000000000000000000000000..53cb77771f2cc6622fa4c67ea5ea20485df761d6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_3/afrisenti @@ -0,0 +1,39 @@ +tag: + - afrobench_sentiment_tasks + - afrisenti_prompt_3 +dataset_path: masakhane/afrisenti +dataset_name: null +dataset_kwargs: {trust_remote_code: True} +output_type: multiple_choice +validation_split: validation +test_split: test +fewshot_split: train +doc_to_target: label +doc_to_choice: + - "negative" + - "positive" + - "neutral" +should_decontaminate: true +doc_to_decontamination_query: 'text: {{tweet}} \nlabel: ' +metric_list: + - metric: f1 + aggregation: !function utils.weighted_f1_score + # aggregation: mean + average: weighted + hf_evaluate: true + higher_is_better: True + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" + - metric: acc + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_3/afrisenti_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_3/afrisenti_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bbcc88d70447a6257d1acefa94b65336d6b330c1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_3/afrisenti_ibo.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: ibo +doc_to_text: "You are an assistant able to detect sentiments in tweets. \n\nGiven\ + \ the sentiment labels Neutral, Positive or Negative; what is the sentiment of the\ + \ Igbo statement below? Return only the labels. \n\ntext: {{tweet}} \nlabel:" +include: afrisenti +task: afrisenti_ibo_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_3/afrisenti_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_3/afrisenti_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..52d84b2684f9a071c0dc6b5889ad790e3116b0fc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_3/afrisenti_kin.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: kin +doc_to_text: "You are an assistant able to detect sentiments in tweets. \n\nGiven\ + \ the sentiment labels Neutral, Positive or Negative; what is the sentiment of the\ + \ Kinyarwanda statement below? Return only the labels. \n\ntext: {{tweet}} \nlabel:" +include: afrisenti +task: afrisenti_kin_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_3/afrisenti_orm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_3/afrisenti_orm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e2d524bfd781b37cb156ce09fd7fc5aae493392b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_3/afrisenti_orm.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: orm +doc_to_text: "You are an assistant able to detect sentiments in tweets. \n\nGiven\ + \ the sentiment labels Neutral, Positive or Negative; what is the sentiment of the\ + \ Oromo statement below? Return only the labels. \n\ntext: {{tweet}} \nlabel:" +include: afrisenti +task: afrisenti_orm_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_3/afrisenti_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_3/afrisenti_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d8e92842e01b61cc815fb48f7a390c6f13587e18 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_3/afrisenti_swa.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: swa +doc_to_text: "You are an assistant able to detect sentiments in tweets. \n\nGiven\ + \ the sentiment labels Neutral, Positive or Negative; what is the sentiment of the\ + \ Swahili statement below? Return only the labels. \n\ntext: {{tweet}} \nlabel:" +include: afrisenti +task: afrisenti_swa_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_3/afrisenti_tir.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_3/afrisenti_tir.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c0f96c24a290d33a3c16754f3c0412ac11cb285a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_3/afrisenti_tir.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: tir +doc_to_text: "You are an assistant able to detect sentiments in tweets. \n\nGiven\ + \ the sentiment labels Neutral, Positive or Negative; what is the sentiment of the\ + \ Tigrinya statement below? Return only the labels. \n\ntext: {{tweet}} \nlabel:" +include: afrisenti +task: afrisenti_tir_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_3/afrisenti_tso.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_3/afrisenti_tso.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8355035e963a13e2da541c65d4f778c5f0d46a58 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_3/afrisenti_tso.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: tso +doc_to_text: "You are an assistant able to detect sentiments in tweets. \n\nGiven\ + \ the sentiment labels Neutral, Positive or Negative; what is the sentiment of the\ + \ Xithonga statement below? Return only the labels. \n\ntext: {{tweet}} \nlabel:" +include: afrisenti +task: afrisenti_tso_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_3/afrisenti_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_3/afrisenti_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..98809176e9693cd65ec711fb81739fdbe0030e70 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_3/afrisenti_twi.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: twi +doc_to_text: "You are an assistant able to detect sentiments in tweets. \n\nGiven\ + \ the sentiment labels Neutral, Positive or Negative; what is the sentiment of the\ + \ Twi statement below? Return only the labels. \n\ntext: {{tweet}} \nlabel:" +include: afrisenti +task: afrisenti_twi_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_3/afrisenti_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_3/afrisenti_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9d1b7ac324b28b7a64220b6b318aabf9537594fc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_3/afrisenti_yor.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: yor +doc_to_text: "You are an assistant able to detect sentiments in tweets. \n\nGiven\ + \ the sentiment labels Neutral, Positive or Negative; what is the sentiment of the\ + \ Yoruba statement below? Return only the labels. \n\ntext: {{tweet}} \nlabel:" +include: afrisenti +task: afrisenti_yor_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_3/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_3/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..3e735e2deb1f9c53152c072615aebe8ba3acb90b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_3/utils.py @@ -0,0 +1 @@ +from lm_eval.utils import weighted_f1_score diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_3/xx.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_3/xx.py new file mode 100644 index 0000000000000000000000000000000000000000..2133cfa0139de116c3d54e6c3866c5e4c26bbc53 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_3/xx.py @@ -0,0 +1,5 @@ +from datasets import load_dataset + + +data = load_dataset("masakhane/afrisenti", "por", trust_remote_code=True) +print(data) diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_4/afrisenti b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_4/afrisenti new file mode 100644 index 0000000000000000000000000000000000000000..6464d7b21693a1565f8479757a89a650cf84ff0c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_4/afrisenti @@ -0,0 +1,39 @@ +tag: + - afrobench_sentiment_tasks + - afrisenti_prompt_4 +dataset_path: masakhane/afrisenti +dataset_name: null +dataset_kwargs: {trust_remote_code: True} +output_type: multiple_choice +validation_split: validation +test_split: test +fewshot_split: train +doc_to_target: label +doc_to_choice: + - "negative" + - "positive" + - "neutral" +should_decontaminate: true +doc_to_decontamination_query: 'text: {{tweet}} \nlabel: ' +metric_list: + - metric: f1 + aggregation: !function utils.weighted_f1_score + # aggregation: mean + average: weighted + hf_evaluate: true + higher_is_better: True + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" + - metric: acc + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_4/afrisenti_arq.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_4/afrisenti_arq.yaml new file mode 100644 index 0000000000000000000000000000000000000000..125771f5d6877585bb2b9a50121da7e5a56d7805 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_4/afrisenti_arq.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: arq +doc_to_text: "Label the following text as Neutral, Positive, or Negative. Provide\ + \ only the label as your response. \n\ntext: {{tweet}} \nlabel: " +include: afrisenti +task: afrisenti_arq_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_4/afrisenti_ary.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_4/afrisenti_ary.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7868fbf3e6739cd69a4732a09b80cc08359ccbe8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_4/afrisenti_ary.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: ary +doc_to_text: "Label the following text as Neutral, Positive, or Negative. Provide\ + \ only the label as your response. \n\ntext: {{tweet}} \nlabel: " +include: afrisenti +task: afrisenti_ary_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_4/afrisenti_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_4/afrisenti_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..686e16c29e4aad0ff13e160befeb31e5b25a7f54 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_4/afrisenti_ibo.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: ibo +doc_to_text: "Label the following text as Neutral, Positive, or Negative. Provide\ + \ only the label as your response. \n\ntext: {{tweet}} \nlabel: " +include: afrisenti +task: afrisenti_ibo_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_4/afrisenti_orm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_4/afrisenti_orm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8f745ebb28e843d400ccba5973de312d841a2592 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_4/afrisenti_orm.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: orm +doc_to_text: "Label the following text as Neutral, Positive, or Negative. Provide\ + \ only the label as your response. \n\ntext: {{tweet}} \nlabel: " +include: afrisenti +task: afrisenti_orm_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_4/afrisenti_pcm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_4/afrisenti_pcm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b5071134ac6d1645650edef34ee6dde2d3bdabce --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_4/afrisenti_pcm.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: pcm +doc_to_text: "Label the following text as Neutral, Positive, or Negative. Provide\ + \ only the label as your response. \n\ntext: {{tweet}} \nlabel: " +include: afrisenti +task: afrisenti_pcm_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_4/afrisenti_por.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_4/afrisenti_por.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f5196bcf58303558d5a214caffde460b8675d08f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_4/afrisenti_por.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: por +doc_to_text: "Label the following text as Neutral, Positive, or Negative. Provide\ + \ only the label as your response. \n\ntext: {{tweet}} \nlabel: " +include: afrisenti +task: afrisenti_por_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_4/afrisenti_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_4/afrisenti_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..97b9e4f1a2fc87433bf511e6e511bd9f54284d5e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_4/afrisenti_swa.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: swa +doc_to_text: "Label the following text as Neutral, Positive, or Negative. Provide\ + \ only the label as your response. \n\ntext: {{tweet}} \nlabel: " +include: afrisenti +task: afrisenti_swa_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_4/afrisenti_tir.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_4/afrisenti_tir.yaml new file mode 100644 index 0000000000000000000000000000000000000000..02dfca854e166ca3d96a570e8db126e646fbd5b5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_4/afrisenti_tir.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: tir +doc_to_text: "Label the following text as Neutral, Positive, or Negative. Provide\ + \ only the label as your response. \n\ntext: {{tweet}} \nlabel: " +include: afrisenti +task: afrisenti_tir_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_4/afrisenti_tso.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_4/afrisenti_tso.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fa83c1378d15b8c3cedd8582025aea8ea795bb07 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_4/afrisenti_tso.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: tso +doc_to_text: "Label the following text as Neutral, Positive, or Negative. Provide\ + \ only the label as your response. \n\ntext: {{tweet}} \nlabel: " +include: afrisenti +task: afrisenti_tso_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_4/xx.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_4/xx.py new file mode 100644 index 0000000000000000000000000000000000000000..4515053c4265ba0b6bc9afa9f876d20ef5fc5c2a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_4/xx.py @@ -0,0 +1,5 @@ +from datasets import load_dataset + + +data = load_dataset("masakhane/afrisenti", "orm", trust_remote_code=True) +print(data) diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_5/afrisenti_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_5/afrisenti_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..866ffbe9fcab9e67a1ba4a9781dddd1ef60e8043 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_5/afrisenti_amh.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: amh +doc_to_text: "You are tasked with performing sentiment classification on the following\ + \ Amharic text. For each input, classify the sentiment as positive, negative, or\ + \ neutral. Use the following guidelines: \n\n Positive: The text expresses happiness,\ + \ satisfaction, or optimism. \nNegative: The text conveys disappointment, dissatisfaction,\ + \ or pessimism. \nNeutral: The text is factual, objective, or without strong emotional\ + \ undertones. \n\nIf the text contains both positive and negative sentiments, choose\ + \ the dominant sentiment. For ambiguous or unclear sentiments, select the label\ + \ that best reflects the overall tone. Please provide a single classification for\ + \ each input.\n\ntext: {{tweet}} \nlabel: " +include: afrisenti +task: afrisenti_amh_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_5/afrisenti_arq.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_5/afrisenti_arq.yaml new file mode 100644 index 0000000000000000000000000000000000000000..783785c031aab91f8840df8e70316ab1d4e24606 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_5/afrisenti_arq.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: arq +doc_to_text: "You are tasked with performing sentiment classification on the following\ + \ Algerian Arabic text. For each input, classify the sentiment as positive, negative,\ + \ or neutral. Use the following guidelines: \n\n Positive: The text expresses happiness,\ + \ satisfaction, or optimism. \nNegative: The text conveys disappointment, dissatisfaction,\ + \ or pessimism. \nNeutral: The text is factual, objective, or without strong emotional\ + \ undertones. \n\nIf the text contains both positive and negative sentiments, choose\ + \ the dominant sentiment. For ambiguous or unclear sentiments, select the label\ + \ that best reflects the overall tone. Please provide a single classification for\ + \ each input.\n\ntext: {{tweet}} \nlabel: " +include: afrisenti +task: afrisenti_arq_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_5/afrisenti_ary.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_5/afrisenti_ary.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e601dc19ddd1bf960ee56e5345bbbcc8c6b84caf --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_5/afrisenti_ary.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: ary +doc_to_text: "You are tasked with performing sentiment classification on the following\ + \ Moroccan Arabic text. For each input, classify the sentiment as positive, negative,\ + \ or neutral. Use the following guidelines: \n\n Positive: The text expresses happiness,\ + \ satisfaction, or optimism. \nNegative: The text conveys disappointment, dissatisfaction,\ + \ or pessimism. \nNeutral: The text is factual, objective, or without strong emotional\ + \ undertones. \n\nIf the text contains both positive and negative sentiments, choose\ + \ the dominant sentiment. For ambiguous or unclear sentiments, select the label\ + \ that best reflects the overall tone. Please provide a single classification for\ + \ each input.\n\ntext: {{tweet}} \nlabel: " +include: afrisenti +task: afrisenti_ary_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_5/afrisenti_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_5/afrisenti_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2ab2adc1aaaf15ce9dced631d20ec2efb428ace5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_5/afrisenti_hau.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: hau +doc_to_text: "You are tasked with performing sentiment classification on the following\ + \ Hausa text. For each input, classify the sentiment as positive, negative, or neutral.\ + \ Use the following guidelines: \n\n Positive: The text expresses happiness, satisfaction,\ + \ or optimism. \nNegative: The text conveys disappointment, dissatisfaction, or\ + \ pessimism. \nNeutral: The text is factual, objective, or without strong emotional\ + \ undertones. \n\nIf the text contains both positive and negative sentiments, choose\ + \ the dominant sentiment. For ambiguous or unclear sentiments, select the label\ + \ that best reflects the overall tone. Please provide a single classification for\ + \ each input.\n\ntext: {{tweet}} \nlabel: " +include: afrisenti +task: afrisenti_hau_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_5/afrisenti_orm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_5/afrisenti_orm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c61ea75ee6b448a7dcf30ecdfcef7328201379ae --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_5/afrisenti_orm.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: orm +doc_to_text: "You are tasked with performing sentiment classification on the following\ + \ Oromo text. For each input, classify the sentiment as positive, negative, or neutral.\ + \ Use the following guidelines: \n\n Positive: The text expresses happiness, satisfaction,\ + \ or optimism. \nNegative: The text conveys disappointment, dissatisfaction, or\ + \ pessimism. \nNeutral: The text is factual, objective, or without strong emotional\ + \ undertones. \n\nIf the text contains both positive and negative sentiments, choose\ + \ the dominant sentiment. For ambiguous or unclear sentiments, select the label\ + \ that best reflects the overall tone. Please provide a single classification for\ + \ each input.\n\ntext: {{tweet}} \nlabel: " +include: afrisenti +task: afrisenti_orm_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_5/afrisenti_pcm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_5/afrisenti_pcm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6de78061a142a0bd80d69619a5eba4ce3756d616 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_5/afrisenti_pcm.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: pcm +doc_to_text: "You are tasked with performing sentiment classification on the following\ + \ Nigerian Pidgin text. For each input, classify the sentiment as positive, negative,\ + \ or neutral. Use the following guidelines: \n\n Positive: The text expresses happiness,\ + \ satisfaction, or optimism. \nNegative: The text conveys disappointment, dissatisfaction,\ + \ or pessimism. \nNeutral: The text is factual, objective, or without strong emotional\ + \ undertones. \n\nIf the text contains both positive and negative sentiments, choose\ + \ the dominant sentiment. For ambiguous or unclear sentiments, select the label\ + \ that best reflects the overall tone. Please provide a single classification for\ + \ each input.\n\ntext: {{tweet}} \nlabel: " +include: afrisenti +task: afrisenti_pcm_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_5/afrisenti_por.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_5/afrisenti_por.yaml new file mode 100644 index 0000000000000000000000000000000000000000..48b728d5dd38bb55632fd3fb22f37fbcbeb66eea --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_5/afrisenti_por.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: por +doc_to_text: "You are tasked with performing sentiment classification on the following\ + \ Mozambique Portuguese text. For each input, classify the sentiment as positive,\ + \ negative, or neutral. Use the following guidelines: \n\n Positive: The text expresses\ + \ happiness, satisfaction, or optimism. \nNegative: The text conveys disappointment,\ + \ dissatisfaction, or pessimism. \nNeutral: The text is factual, objective, or without\ + \ strong emotional undertones. \n\nIf the text contains both positive and negative\ + \ sentiments, choose the dominant sentiment. For ambiguous or unclear sentiments,\ + \ select the label that best reflects the overall tone. Please provide a single\ + \ classification for each input.\n\ntext: {{tweet}} \nlabel: " +include: afrisenti +task: afrisenti_por_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_5/afrisenti_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_5/afrisenti_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fee357ab47703dea57fe9b78a2d46da0f0212874 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_5/afrisenti_swa.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: swa +doc_to_text: "You are tasked with performing sentiment classification on the following\ + \ Swahili text. For each input, classify the sentiment as positive, negative, or\ + \ neutral. Use the following guidelines: \n\n Positive: The text expresses happiness,\ + \ satisfaction, or optimism. \nNegative: The text conveys disappointment, dissatisfaction,\ + \ or pessimism. \nNeutral: The text is factual, objective, or without strong emotional\ + \ undertones. \n\nIf the text contains both positive and negative sentiments, choose\ + \ the dominant sentiment. For ambiguous or unclear sentiments, select the label\ + \ that best reflects the overall tone. Please provide a single classification for\ + \ each input.\n\ntext: {{tweet}} \nlabel: " +include: afrisenti +task: afrisenti_swa_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_5/afrisenti_tso.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_5/afrisenti_tso.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5f5705285e81a7d998dd77b44c9221697aaea435 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_5/afrisenti_tso.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: tso +doc_to_text: "You are tasked with performing sentiment classification on the following\ + \ Xithonga text. For each input, classify the sentiment as positive, negative, or\ + \ neutral. Use the following guidelines: \n\n Positive: The text expresses happiness,\ + \ satisfaction, or optimism. \nNegative: The text conveys disappointment, dissatisfaction,\ + \ or pessimism. \nNeutral: The text is factual, objective, or without strong emotional\ + \ undertones. \n\nIf the text contains both positive and negative sentiments, choose\ + \ the dominant sentiment. For ambiguous or unclear sentiments, select the label\ + \ that best reflects the overall tone. Please provide a single classification for\ + \ each input.\n\ntext: {{tweet}} \nlabel: " +include: afrisenti +task: afrisenti_tso_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_5/afrisenti_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_5/afrisenti_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b96edb4104e85d01642b2ceca14fc59f2c296ecd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_5/afrisenti_yor.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: yor +doc_to_text: "You are tasked with performing sentiment classification on the following\ + \ Yoruba text. For each input, classify the sentiment as positive, negative, or\ + \ neutral. Use the following guidelines: \n\n Positive: The text expresses happiness,\ + \ satisfaction, or optimism. \nNegative: The text conveys disappointment, dissatisfaction,\ + \ or pessimism. \nNeutral: The text is factual, objective, or without strong emotional\ + \ undertones. \n\nIf the text contains both positive and negative sentiments, choose\ + \ the dominant sentiment. For ambiguous or unclear sentiments, select the label\ + \ that best reflects the overall tone. Please provide a single classification for\ + \ each input.\n\ntext: {{tweet}} \nlabel: " +include: afrisenti +task: afrisenti_yor_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_5/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_5/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..3e735e2deb1f9c53152c072615aebe8ba3acb90b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/prompt_5/utils.py @@ -0,0 +1 @@ +from lm_eval.utils import weighted_f1_score diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..b5f9b74e2eb12db6a985c8428830933f8adcc936 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrisenti/utils.py @@ -0,0 +1,124 @@ +import argparse + +import yaml + + +class FunctionTag: + def __init__(self, value): + self.value = value + + +def prompt_func(mode, lang): + prompt_map = { + "prompt_1": "Does this statement; {{tweet}} have a Neutral, Positive or Negative sentiment? Labels only", + "prompt_2": f"Does this {lang} statement; " + "'{{tweet}}' have a Neutral, Positive or Negative sentiment? Labels only", + "prompt_3": f"You are an assistant able to detect sentiments in tweets. \n\n" + f"Given the sentiment labels Neutral, Positive or Negative; what is " + f"the sentiment of the {lang} statement below? Return only the labels. " + "\n\ntext: {{tweet}} \nlabel:", + "prompt_4": "Label the following text as Neutral, Positive, or Negative. Provide only the label as your " + "response. \n\ntext: {{tweet}} \nlabel: ", + "prompt_5": f"You are tasked with performing sentiment classification on the following {lang} text. " + f"For each input, classify the sentiment as positive, negative, or neutral. " + f"Use the following guidelines: \n\n " + f"Positive: The text expresses happiness, satisfaction, or optimism. \n" + f"Negative: The text conveys disappointment, dissatisfaction, or pessimism. \n" + f"Neutral: The text is factual, objective, or without strong emotional undertones. \n\n" + f"If the text contains both positive and negative sentiments, choose the dominant sentiment. " + f"For ambiguous or unclear sentiments, select the label that best reflects the overall tone. " + "Please provide a single classification for each input.\n\ntext: {{tweet}} \nlabel: ", + } + return prompt_map[mode] + + +def gen_lang_yamls(output_dir: str, overwrite: bool, mode: str) -> None: + """ + Generate a yaml file for each language. + + :param output_dir: The directory to output the files to. + :param overwrite: Whether to overwrite files if they already exist. + """ + err = [] + languages = { + "amh": "Amharic", + "arq": "Algerian Arabic", + "ary": "Moroccan Arabic", + "hau": "Hausa", + "ibo": "Igbo", + "kin": "Kinyarwanda", + "orm": "Oromo", + "pcm": "Nigerian Pidgin", + "por": "Mozambique Portuguese", + "swa": "Swahili", + "tir": "Tigrinya", + "tso": "Xithonga", + "twi": "Twi", + "yor": "Yoruba", + } + for lang in languages.keys(): + try: + file_name = f"afrisenti_{lang}.yaml" + task_name = f"afrisenti_{lang}_{mode}" + yaml_template = "afrisenti" + if int(mode.split("_")[-1]) > 1: + yaml_details = { + "include": yaml_template, + "task": task_name, + "dataset_name": lang, + "doc_to_text": prompt_func(mode, languages[lang]), + } + else: + yaml_details = { + "include": yaml_template, + "task": task_name, + "dataset_name": lang, + } + with open( + f"{output_dir}/{mode}/{file_name}", + "w" if overwrite else "x", + encoding="utf8", + ) as f: + f.write("# Generated by utils.py\n") + yaml.dump( + yaml_details, + f, + allow_unicode=True, + ) + except FileExistsError: + err.append(file_name) + + if len(err) > 0: + raise FileExistsError( + "Files were not created because they already exist (use --overwrite flag):" + f" {', '.join(err)}" + ) + + +def main() -> None: + """Parse CLI args and generate language-specific yaml files.""" + parser = argparse.ArgumentParser() + parser.add_argument( + "--overwrite", + default=True, + action="store_true", + help="Overwrite files if they already exist", + ) + parser.add_argument( + "--output-dir", + default="./", + help="Directory to write yaml files to", + ) + parser.add_argument( + "--mode", + default="prompt_1", + choices=["prompt_1", "prompt_2", "prompt_3", "prompt_4", "prompt_5"], + help="Prompt number", + ) + args = parser.parse_args() + + gen_lang_yamls(output_dir=args.output_dir, overwrite=args.overwrite, mode=args.mode) + + +if __name__ == "__main__": + main() diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/afrobench.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrobench.yaml new file mode 100644 index 0000000000000000000000000000000000000000..52234bef5cde6b5695fa6510019bcf37502ddd40 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/afrobench.yaml @@ -0,0 +1,23 @@ +group: afrobench +task: +# - adr_tasks +## - afrihate_tasks #dataset not publicly available yet +# - afrimgsm_cot_tasks +# - afrixnli_tasks +# - afrobench_xqa_tasks +# - afrobench_sentiment_tasks + - afrobench_MT_tasks +# - afrobench_TC_tasks +# - afrobench_mmlu_tasks +# - injongointent_tasks +# - masakhaner_tasks +# - masakhapos_tasks +# - RC_tasks +# - uhura_arc_easy_tasks +# - xlsum_tasks +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/belebele.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/belebele.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5c7d3a9dc4450ffbb152abd6eac2655c2cf2199c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/belebele.yaml @@ -0,0 +1,13 @@ +group: belebele +task: + - belebele_prompt_1 + - belebele_prompt_2 + - belebele_prompt_3 + - belebele_prompt_4 + - belebele_prompt_5 +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true +metadata: + version: 2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_afr.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_afr.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e3a7c2b97208d5ac2ced4bbacc3192a290fa6ea2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_afr.yaml @@ -0,0 +1,17 @@ +# Generated by utils.py +dataset_name: afr_Latn +doc_to_text: 'P: {{flores_passage}} + + Q: {{question.strip()}} + + A: {{mc_answer1}} + + B: {{mc_answer2}} + + C: {{mc_answer3}} + + D: {{mc_answer4}} + + Please choose the correct answer from the options above:' +include: belebele +task: belebele_afr_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_ary.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_ary.yaml new file mode 100644 index 0000000000000000000000000000000000000000..82f0d5230d4d24a77d381be67a22dd270255fea1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_ary.yaml @@ -0,0 +1,17 @@ +# Generated by utils.py +dataset_name: ary_Arab +doc_to_text: 'P: {{flores_passage}} + + Q: {{question.strip()}} + + A: {{mc_answer1}} + + B: {{mc_answer2}} + + C: {{mc_answer3}} + + D: {{mc_answer4}} + + Please choose the correct answer from the options above:' +include: belebele +task: belebele_ary_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_arz.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_arz.yaml new file mode 100644 index 0000000000000000000000000000000000000000..38f8c3edc09bd33823c0a221bcfe8a6d7d758d91 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_arz.yaml @@ -0,0 +1,17 @@ +# Generated by utils.py +dataset_name: arz_Arab +doc_to_text: 'P: {{flores_passage}} + + Q: {{question.strip()}} + + A: {{mc_answer1}} + + B: {{mc_answer2}} + + C: {{mc_answer3}} + + D: {{mc_answer4}} + + Please choose the correct answer from the options above:' +include: belebele +task: belebele_arz_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_bam.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_bam.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f2bc2d49f78a094ad11711baee3df33923077c89 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_bam.yaml @@ -0,0 +1,17 @@ +# Generated by utils.py +dataset_name: bam_Latn +doc_to_text: 'P: {{flores_passage}} + + Q: {{question.strip()}} + + A: {{mc_answer1}} + + B: {{mc_answer2}} + + C: {{mc_answer3}} + + D: {{mc_answer4}} + + Please choose the correct answer from the options above:' +include: belebele +task: belebele_bam_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_eng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ef1f0463d4c1b7c6661c570f00f9beb855cfd534 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_eng.yaml @@ -0,0 +1,17 @@ +# Generated by utils.py +dataset_name: eng_Latn +doc_to_text: 'P: {{flores_passage}} + + Q: {{question.strip()}} + + A: {{mc_answer1}} + + B: {{mc_answer2}} + + C: {{mc_answer3}} + + D: {{mc_answer4}} + + Please choose the correct answer from the options above:' +include: belebele +task: belebele_eng_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_fra.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_fra.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5f2513826f2808d242ced3d0b7278878bb846649 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_fra.yaml @@ -0,0 +1,17 @@ +# Generated by utils.py +dataset_name: fra_Latn +doc_to_text: 'P: {{flores_passage}} + + Q: {{question.strip()}} + + A: {{mc_answer1}} + + B: {{mc_answer2}} + + C: {{mc_answer3}} + + D: {{mc_answer4}} + + Please choose the correct answer from the options above:' +include: belebele +task: belebele_fra_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_fuv.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_fuv.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b24422c03aa1f81b18bab14d8810c916afb8e367 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_fuv.yaml @@ -0,0 +1,17 @@ +# Generated by utils.py +dataset_name: fuv_Latn +doc_to_text: 'P: {{flores_passage}} + + Q: {{question.strip()}} + + A: {{mc_answer1}} + + B: {{mc_answer2}} + + C: {{mc_answer3}} + + D: {{mc_answer4}} + + Please choose the correct answer from the options above:' +include: belebele +task: belebele_fuv_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_gaz.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_gaz.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b999f4a85fab866f0cbc6ccab5bc78ddd7a65bc5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_gaz.yaml @@ -0,0 +1,17 @@ +# Generated by utils.py +dataset_name: gaz_Latn +doc_to_text: 'P: {{flores_passage}} + + Q: {{question.strip()}} + + A: {{mc_answer1}} + + B: {{mc_answer2}} + + C: {{mc_answer3}} + + D: {{mc_answer4}} + + Please choose the correct answer from the options above:' +include: belebele +task: belebele_gaz_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..933e90b50653a19529277c30e0c940130d656528 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_hau.yaml @@ -0,0 +1,17 @@ +# Generated by utils.py +dataset_name: hau_Latn +doc_to_text: 'P: {{flores_passage}} + + Q: {{question.strip()}} + + A: {{mc_answer1}} + + B: {{mc_answer2}} + + C: {{mc_answer3}} + + D: {{mc_answer4}} + + Please choose the correct answer from the options above:' +include: belebele +task: belebele_hau_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fa17935c3b331e347ddc13512b46c80e484e24d9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_ibo.yaml @@ -0,0 +1,17 @@ +# Generated by utils.py +dataset_name: ibo_Latn +doc_to_text: 'P: {{flores_passage}} + + Q: {{question.strip()}} + + A: {{mc_answer1}} + + B: {{mc_answer2}} + + C: {{mc_answer3}} + + D: {{mc_answer4}} + + Please choose the correct answer from the options above:' +include: belebele +task: belebele_ibo_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_kea.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_kea.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ad535d498447fde8ee74423a80858d42a85f558c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_kea.yaml @@ -0,0 +1,17 @@ +# Generated by utils.py +dataset_name: kea_Latn +doc_to_text: 'P: {{flores_passage}} + + Q: {{question.strip()}} + + A: {{mc_answer1}} + + B: {{mc_answer2}} + + C: {{mc_answer3}} + + D: {{mc_answer4}} + + Please choose the correct answer from the options above:' +include: belebele +task: belebele_kea_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..de957a59b3cb665760f8c36c5968ce75c5b4271c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_kin.yaml @@ -0,0 +1,17 @@ +# Generated by utils.py +dataset_name: kin_Latn +doc_to_text: 'P: {{flores_passage}} + + Q: {{question.strip()}} + + A: {{mc_answer1}} + + B: {{mc_answer2}} + + C: {{mc_answer3}} + + D: {{mc_answer4}} + + Please choose the correct answer from the options above:' +include: belebele +task: belebele_kin_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_lin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_lin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b3247f065c8f5137f52af8965f18a2a23e942b64 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_lin.yaml @@ -0,0 +1,17 @@ +# Generated by utils.py +dataset_name: lin_Latn +doc_to_text: 'P: {{flores_passage}} + + Q: {{question.strip()}} + + A: {{mc_answer1}} + + B: {{mc_answer2}} + + C: {{mc_answer3}} + + D: {{mc_answer4}} + + Please choose the correct answer from the options above:' +include: belebele +task: belebele_lin_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_plt.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_plt.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d5c286e8517e6348798dc9c7807b1aec687c9ad1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_plt.yaml @@ -0,0 +1,17 @@ +# Generated by utils.py +dataset_name: plt_Latn +doc_to_text: 'P: {{flores_passage}} + + Q: {{question.strip()}} + + A: {{mc_answer1}} + + B: {{mc_answer2}} + + C: {{mc_answer3}} + + D: {{mc_answer4}} + + Please choose the correct answer from the options above:' +include: belebele +task: belebele_plt_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_por.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_por.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ceba2310c5e483727eb5f4386058b666cd49620c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_por.yaml @@ -0,0 +1,17 @@ +# Generated by utils.py +dataset_name: por_Latn +doc_to_text: 'P: {{flores_passage}} + + Q: {{question.strip()}} + + A: {{mc_answer1}} + + B: {{mc_answer2}} + + C: {{mc_answer3}} + + D: {{mc_answer4}} + + Please choose the correct answer from the options above:' +include: belebele +task: belebele_por_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_som.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_som.yaml new file mode 100644 index 0000000000000000000000000000000000000000..24af555865acd5d2f8c48c5aa2814e401891ccd4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_som.yaml @@ -0,0 +1,17 @@ +# Generated by utils.py +dataset_name: som_Latn +doc_to_text: 'P: {{flores_passage}} + + Q: {{question.strip()}} + + A: {{mc_answer1}} + + B: {{mc_answer2}} + + C: {{mc_answer3}} + + D: {{mc_answer4}} + + Please choose the correct answer from the options above:' +include: belebele +task: belebele_som_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_ssw.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_ssw.yaml new file mode 100644 index 0000000000000000000000000000000000000000..032b4629b569fc340c12fc701dbc694fd8631d95 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_ssw.yaml @@ -0,0 +1,17 @@ +# Generated by utils.py +dataset_name: ssw_Latn +doc_to_text: 'P: {{flores_passage}} + + Q: {{question.strip()}} + + A: {{mc_answer1}} + + B: {{mc_answer2}} + + C: {{mc_answer3}} + + D: {{mc_answer4}} + + Please choose the correct answer from the options above:' +include: belebele +task: belebele_ssw_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_tso.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_tso.yaml new file mode 100644 index 0000000000000000000000000000000000000000..869c50153d289b01ee408656701d0fb14dcb679d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_tso.yaml @@ -0,0 +1,17 @@ +# Generated by utils.py +dataset_name: tso_Latn +doc_to_text: 'P: {{flores_passage}} + + Q: {{question.strip()}} + + A: {{mc_answer1}} + + B: {{mc_answer2}} + + C: {{mc_answer3}} + + D: {{mc_answer4}} + + Please choose the correct answer from the options above:' +include: belebele +task: belebele_tso_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..549560ac1d4cd2181750d6e587f355f743d52cef --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_xho.yaml @@ -0,0 +1,17 @@ +# Generated by utils.py +dataset_name: xho_Latn +doc_to_text: 'P: {{flores_passage}} + + Q: {{question.strip()}} + + A: {{mc_answer1}} + + B: {{mc_answer2}} + + C: {{mc_answer3}} + + D: {{mc_answer4}} + + Please choose the correct answer from the options above:' +include: belebele +task: belebele_xho_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..70c55eba1c99673595f00b297237391dd767c74c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_1/belebele_yor.yaml @@ -0,0 +1,17 @@ +# Generated by utils.py +dataset_name: yor_Latn +doc_to_text: 'P: {{flores_passage}} + + Q: {{question.strip()}} + + A: {{mc_answer1}} + + B: {{mc_answer2}} + + C: {{mc_answer3}} + + D: {{mc_answer4}} + + Please choose the correct answer from the options above:' +include: belebele +task: belebele_yor_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_2/belebele_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_2/belebele_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f8f95cac2a0f27d8a1699215855167bffe3b38a0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_2/belebele_amh.yaml @@ -0,0 +1,17 @@ +# Generated by utils.py +dataset_name: amh_Ethi +doc_to_text: 'Passage: {{flores_passage}} + + Question: {{question.strip()}} + + 1: {{mc_answer1}} + + 2: {{mc_answer2}} + + 3: {{mc_answer3}} + + 4: {{mc_answer4}} + + Please select the correct answer from the given choices:' +include: belebele +task: belebele_amh_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_2/belebele_ary.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_2/belebele_ary.yaml new file mode 100644 index 0000000000000000000000000000000000000000..12a784902ad61589942794fe4661375154909999 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_2/belebele_ary.yaml @@ -0,0 +1,17 @@ +# Generated by utils.py +dataset_name: ary_Arab +doc_to_text: 'Passage: {{flores_passage}} + + Question: {{question.strip()}} + + 1: {{mc_answer1}} + + 2: {{mc_answer2}} + + 3: {{mc_answer3}} + + 4: {{mc_answer4}} + + Please select the correct answer from the given choices:' +include: belebele +task: belebele_ary_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_2/belebele_bam.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_2/belebele_bam.yaml new file mode 100644 index 0000000000000000000000000000000000000000..814d32b5483290ef00380b30ce0797492f696497 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_2/belebele_bam.yaml @@ -0,0 +1,17 @@ +# Generated by utils.py +dataset_name: bam_Latn +doc_to_text: 'Passage: {{flores_passage}} + + Question: {{question.strip()}} + + 1: {{mc_answer1}} + + 2: {{mc_answer2}} + + 3: {{mc_answer3}} + + 4: {{mc_answer4}} + + Please select the correct answer from the given choices:' +include: belebele +task: belebele_bam_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_2/belebele_fra.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_2/belebele_fra.yaml new file mode 100644 index 0000000000000000000000000000000000000000..157433727099d0bc545ddb1e821aa80fbb37f3a1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_2/belebele_fra.yaml @@ -0,0 +1,17 @@ +# Generated by utils.py +dataset_name: fra_Latn +doc_to_text: 'Passage: {{flores_passage}} + + Question: {{question.strip()}} + + 1: {{mc_answer1}} + + 2: {{mc_answer2}} + + 3: {{mc_answer3}} + + 4: {{mc_answer4}} + + Please select the correct answer from the given choices:' +include: belebele +task: belebele_fra_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_2/belebele_gaz.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_2/belebele_gaz.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bc2b2704e382ea4eaa32276e369ad43b8fa298b1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_2/belebele_gaz.yaml @@ -0,0 +1,17 @@ +# Generated by utils.py +dataset_name: gaz_Latn +doc_to_text: 'Passage: {{flores_passage}} + + Question: {{question.strip()}} + + 1: {{mc_answer1}} + + 2: {{mc_answer2}} + + 3: {{mc_answer3}} + + 4: {{mc_answer4}} + + Please select the correct answer from the given choices:' +include: belebele +task: belebele_gaz_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_2/belebele_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_2/belebele_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e686e477dca84fe5fa1c666c553a0ac1c0efe1b5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_2/belebele_kin.yaml @@ -0,0 +1,17 @@ +# Generated by utils.py +dataset_name: kin_Latn +doc_to_text: 'Passage: {{flores_passage}} + + Question: {{question.strip()}} + + 1: {{mc_answer1}} + + 2: {{mc_answer2}} + + 3: {{mc_answer3}} + + 4: {{mc_answer4}} + + Please select the correct answer from the given choices:' +include: belebele +task: belebele_kin_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_2/belebele_lin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_2/belebele_lin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..544eb9ddec49e0fbca3c29d56a3108db039d8979 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_2/belebele_lin.yaml @@ -0,0 +1,17 @@ +# Generated by utils.py +dataset_name: lin_Latn +doc_to_text: 'Passage: {{flores_passage}} + + Question: {{question.strip()}} + + 1: {{mc_answer1}} + + 2: {{mc_answer2}} + + 3: {{mc_answer3}} + + 4: {{mc_answer4}} + + Please select the correct answer from the given choices:' +include: belebele +task: belebele_lin_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_2/belebele_luo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_2/belebele_luo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d7bdde48a01bd9a08b75961f43bff0438889228c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_2/belebele_luo.yaml @@ -0,0 +1,17 @@ +# Generated by utils.py +dataset_name: luo_Latn +doc_to_text: 'Passage: {{flores_passage}} + + Question: {{question.strip()}} + + 1: {{mc_answer1}} + + 2: {{mc_answer2}} + + 3: {{mc_answer3}} + + 4: {{mc_answer4}} + + Please select the correct answer from the given choices:' +include: belebele +task: belebele_luo_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_2/belebele_nya.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_2/belebele_nya.yaml new file mode 100644 index 0000000000000000000000000000000000000000..212c0635342943d90922467829854dd051c72298 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_2/belebele_nya.yaml @@ -0,0 +1,17 @@ +# Generated by utils.py +dataset_name: nya_Latn +doc_to_text: 'Passage: {{flores_passage}} + + Question: {{question.strip()}} + + 1: {{mc_answer1}} + + 2: {{mc_answer2}} + + 3: {{mc_answer3}} + + 4: {{mc_answer4}} + + Please select the correct answer from the given choices:' +include: belebele +task: belebele_nya_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_3/belebele_afr.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_3/belebele_afr.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8c296cc93f26c6bb91bb7a844b22c8827ebc57fe --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_3/belebele_afr.yaml @@ -0,0 +1,17 @@ +# Generated by utils.py +dataset_name: afr_Latn +doc_to_text: 'Context: {{flores_passage}} + + Query: {{question.strip()}} + + Option A: {{mc_answer1}} + + Option B: {{mc_answer2}} + + Option C: {{mc_answer3}} + + Option D: {{mc_answer4}} + + Please indicate the correct option from the list above:' +include: belebele +task: belebele_afr_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_3/belebele_arz.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_3/belebele_arz.yaml new file mode 100644 index 0000000000000000000000000000000000000000..97f13672f57f52613728a862805273451e78b5c7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_3/belebele_arz.yaml @@ -0,0 +1,17 @@ +# Generated by utils.py +dataset_name: arz_Arab +doc_to_text: 'Context: {{flores_passage}} + + Query: {{question.strip()}} + + Option A: {{mc_answer1}} + + Option B: {{mc_answer2}} + + Option C: {{mc_answer3}} + + Option D: {{mc_answer4}} + + Please indicate the correct option from the list above:' +include: belebele +task: belebele_arz_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_3/belebele_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_3/belebele_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..689724b4f8593af87018cef13c7c596868df7c2b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_3/belebele_hau.yaml @@ -0,0 +1,17 @@ +# Generated by utils.py +dataset_name: hau_Latn +doc_to_text: 'Context: {{flores_passage}} + + Query: {{question.strip()}} + + Option A: {{mc_answer1}} + + Option B: {{mc_answer2}} + + Option C: {{mc_answer3}} + + Option D: {{mc_answer4}} + + Please indicate the correct option from the list above:' +include: belebele +task: belebele_hau_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_afr.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_afr.yaml new file mode 100644 index 0000000000000000000000000000000000000000..325cb85f391fb709081167cb70ea97d5664d2c86 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_afr.yaml @@ -0,0 +1,21 @@ +# Generated by utils.py +dataset_name: afr_Latn +doc_to_text: '{{flores_passage}} + + Based on the above passage, answer the following question: + + {{question.strip()}} + + Choices: + + A) {{mc_answer1}} + + B) {{mc_answer2}} + + C) {{mc_answer3}} + + D) {{mc_answer4}} + + Please provide the correct answer from the choices given:' +include: belebele +task: belebele_afr_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4e7b6a20c34b08afb2d6f569a07a275203e48ae1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_lug.yaml @@ -0,0 +1,21 @@ +# Generated by utils.py +dataset_name: lug_Latn +doc_to_text: '{{flores_passage}} + + Based on the above passage, answer the following question: + + {{question.strip()}} + + Choices: + + A) {{mc_answer1}} + + B) {{mc_answer2}} + + C) {{mc_answer3}} + + D) {{mc_answer4}} + + Please provide the correct answer from the choices given:' +include: belebele +task: belebele_lug_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_nya.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_nya.yaml new file mode 100644 index 0000000000000000000000000000000000000000..26d2f699da8daad91726627cd37a6a8d92965faf --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_nya.yaml @@ -0,0 +1,21 @@ +# Generated by utils.py +dataset_name: nya_Latn +doc_to_text: '{{flores_passage}} + + Based on the above passage, answer the following question: + + {{question.strip()}} + + Choices: + + A) {{mc_answer1}} + + B) {{mc_answer2}} + + C) {{mc_answer3}} + + D) {{mc_answer4}} + + Please provide the correct answer from the choices given:' +include: belebele +task: belebele_nya_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_plt.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_plt.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ffdf1460a464cdc0a14459b8d48178b4246e6ed4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_plt.yaml @@ -0,0 +1,21 @@ +# Generated by utils.py +dataset_name: plt_Latn +doc_to_text: '{{flores_passage}} + + Based on the above passage, answer the following question: + + {{question.strip()}} + + Choices: + + A) {{mc_answer1}} + + B) {{mc_answer2}} + + C) {{mc_answer3}} + + D) {{mc_answer4}} + + Please provide the correct answer from the choices given:' +include: belebele +task: belebele_plt_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_por.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_por.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e8c06382b61eaec15b027533a5553c77d2974180 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_por.yaml @@ -0,0 +1,21 @@ +# Generated by utils.py +dataset_name: por_Latn +doc_to_text: '{{flores_passage}} + + Based on the above passage, answer the following question: + + {{question.strip()}} + + Choices: + + A) {{mc_answer1}} + + B) {{mc_answer2}} + + C) {{mc_answer3}} + + D) {{mc_answer4}} + + Please provide the correct answer from the choices given:' +include: belebele +task: belebele_por_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_som.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_som.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6d7be50ca6bd926924a79dc261156b7d00775966 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_som.yaml @@ -0,0 +1,21 @@ +# Generated by utils.py +dataset_name: som_Latn +doc_to_text: '{{flores_passage}} + + Based on the above passage, answer the following question: + + {{question.strip()}} + + Choices: + + A) {{mc_answer1}} + + B) {{mc_answer2}} + + C) {{mc_answer3}} + + D) {{mc_answer4}} + + Please provide the correct answer from the choices given:' +include: belebele +task: belebele_som_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..afbdcad238a03bf388dd4ddb070f159cf41782ab --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_swa.yaml @@ -0,0 +1,21 @@ +# Generated by utils.py +dataset_name: swh_Latn +doc_to_text: '{{flores_passage}} + + Based on the above passage, answer the following question: + + {{question.strip()}} + + Choices: + + A) {{mc_answer1}} + + B) {{mc_answer2}} + + C) {{mc_answer3}} + + D) {{mc_answer4}} + + Please provide the correct answer from the choices given:' +include: belebele +task: belebele_swa_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_tir.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_tir.yaml new file mode 100644 index 0000000000000000000000000000000000000000..827f1f3614be4b1945bc8495ef52729b6c0778c0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_tir.yaml @@ -0,0 +1,21 @@ +# Generated by utils.py +dataset_name: tir_Ethi +doc_to_text: '{{flores_passage}} + + Based on the above passage, answer the following question: + + {{question.strip()}} + + Choices: + + A) {{mc_answer1}} + + B) {{mc_answer2}} + + C) {{mc_answer3}} + + D) {{mc_answer4}} + + Please provide the correct answer from the choices given:' +include: belebele +task: belebele_tir_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_tso.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_tso.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4f1a87faa18a1af4729eca19d50b6a86bda83771 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_tso.yaml @@ -0,0 +1,21 @@ +# Generated by utils.py +dataset_name: tso_Latn +doc_to_text: '{{flores_passage}} + + Based on the above passage, answer the following question: + + {{question.strip()}} + + Choices: + + A) {{mc_answer1}} + + B) {{mc_answer2}} + + C) {{mc_answer3}} + + D) {{mc_answer4}} + + Please provide the correct answer from the choices given:' +include: belebele +task: belebele_tso_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0e0f6a629eea7632d7879943813fbd8964de5b8f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_wol.yaml @@ -0,0 +1,21 @@ +# Generated by utils.py +dataset_name: wol_Latn +doc_to_text: '{{flores_passage}} + + Based on the above passage, answer the following question: + + {{question.strip()}} + + Choices: + + A) {{mc_answer1}} + + B) {{mc_answer2}} + + C) {{mc_answer3}} + + D) {{mc_answer4}} + + Please provide the correct answer from the choices given:' +include: belebele +task: belebele_wol_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..526e24ef92ecfbada52046a0d72e878ef939e86c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_4/belebele_yor.yaml @@ -0,0 +1,21 @@ +# Generated by utils.py +dataset_name: yor_Latn +doc_to_text: '{{flores_passage}} + + Based on the above passage, answer the following question: + + {{question.strip()}} + + Choices: + + A) {{mc_answer1}} + + B) {{mc_answer2}} + + C) {{mc_answer3}} + + D) {{mc_answer4}} + + Please provide the correct answer from the choices given:' +include: belebele +task: belebele_yor_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele new file mode 100644 index 0000000000000000000000000000000000000000..0d85bf5172c2e8b9408448196191f3b7d40367a0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele @@ -0,0 +1,23 @@ +tag: + - belebele_tasks + - belebele_prompt_5 + - RC_tasks +dataset_path: facebook/belebele +dataset_name: null +output_type: multiple_choice +test_split: test +fewshot_config: + sampler: first_n +doc_to_target: "{{['1', '2', '3', '4'].index(correct_answer_num)}}" +should_decontaminate: true +doc_to_decontamination_query: "{{question}}" +doc_to_choice: ["A", "B", "C", "D"] +metric_list: + - metric: acc + aggregation: mean + weight_by_size: true + - metric: acc_norm + aggregation: mean + weight_by_size: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_ary.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_ary.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2cf68405132141da73c1c4c0085bffa47c6aab41 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_ary.yaml @@ -0,0 +1,19 @@ +# Generated by utils.py +dataset_name: ary_Arab +doc_to_text: 'Read the passage: {{flores_passage}} + + Then answer the question: {{question.strip()}} + + Options: + + A. {{mc_answer1}} + + B. {{mc_answer2}} + + C. {{mc_answer3}} + + D. {{mc_answer4}} + + Please choose the correct option from the above list:' +include: belebele +task: belebele_ary_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_arz.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_arz.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4c0314a96d451eb0cbbe04ef2036fba63d7927f5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_arz.yaml @@ -0,0 +1,19 @@ +# Generated by utils.py +dataset_name: arz_Arab +doc_to_text: 'Read the passage: {{flores_passage}} + + Then answer the question: {{question.strip()}} + + Options: + + A. {{mc_answer1}} + + B. {{mc_answer2}} + + C. {{mc_answer3}} + + D. {{mc_answer4}} + + Please choose the correct option from the above list:' +include: belebele +task: belebele_arz_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_eng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..62617bf1e3266c46897623097836dbc3337add03 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_eng.yaml @@ -0,0 +1,19 @@ +# Generated by utils.py +dataset_name: eng_Latn +doc_to_text: 'Read the passage: {{flores_passage}} + + Then answer the question: {{question.strip()}} + + Options: + + A. {{mc_answer1}} + + B. {{mc_answer2}} + + C. {{mc_answer3}} + + D. {{mc_answer4}} + + Please choose the correct option from the above list:' +include: belebele +task: belebele_eng_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_fra.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_fra.yaml new file mode 100644 index 0000000000000000000000000000000000000000..05131046414169281a6bdb6046cbfef3f461939f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_fra.yaml @@ -0,0 +1,19 @@ +# Generated by utils.py +dataset_name: fra_Latn +doc_to_text: 'Read the passage: {{flores_passage}} + + Then answer the question: {{question.strip()}} + + Options: + + A. {{mc_answer1}} + + B. {{mc_answer2}} + + C. {{mc_answer3}} + + D. {{mc_answer4}} + + Please choose the correct option from the above list:' +include: belebele +task: belebele_fra_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_fuv.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_fuv.yaml new file mode 100644 index 0000000000000000000000000000000000000000..35103b5c19223a0a11e6a595b3f28fa3635455e9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_fuv.yaml @@ -0,0 +1,19 @@ +# Generated by utils.py +dataset_name: fuv_Latn +doc_to_text: 'Read the passage: {{flores_passage}} + + Then answer the question: {{question.strip()}} + + Options: + + A. {{mc_answer1}} + + B. {{mc_answer2}} + + C. {{mc_answer3}} + + D. {{mc_answer4}} + + Please choose the correct option from the above list:' +include: belebele +task: belebele_fuv_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_gaz.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_gaz.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3822a5886fc0a179e95140208ba7751306c09619 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_gaz.yaml @@ -0,0 +1,19 @@ +# Generated by utils.py +dataset_name: gaz_Latn +doc_to_text: 'Read the passage: {{flores_passage}} + + Then answer the question: {{question.strip()}} + + Options: + + A. {{mc_answer1}} + + B. {{mc_answer2}} + + C. {{mc_answer3}} + + D. {{mc_answer4}} + + Please choose the correct option from the above list:' +include: belebele +task: belebele_gaz_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_kea.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_kea.yaml new file mode 100644 index 0000000000000000000000000000000000000000..45fb47ad9854a75127428774a10dbd6eb12d83a9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_kea.yaml @@ -0,0 +1,19 @@ +# Generated by utils.py +dataset_name: kea_Latn +doc_to_text: 'Read the passage: {{flores_passage}} + + Then answer the question: {{question.strip()}} + + Options: + + A. {{mc_answer1}} + + B. {{mc_answer2}} + + C. {{mc_answer3}} + + D. {{mc_answer4}} + + Please choose the correct option from the above list:' +include: belebele +task: belebele_kea_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_lin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_lin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ff6493b711b0126ed2b32ad0d7d7f668c5c71482 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_lin.yaml @@ -0,0 +1,19 @@ +# Generated by utils.py +dataset_name: lin_Latn +doc_to_text: 'Read the passage: {{flores_passage}} + + Then answer the question: {{question.strip()}} + + Options: + + A. {{mc_answer1}} + + B. {{mc_answer2}} + + C. {{mc_answer3}} + + D. {{mc_answer4}} + + Please choose the correct option from the above list:' +include: belebele +task: belebele_lin_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1b64c68ba1b3a8aa1e026cb191dcf96873285d40 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_lug.yaml @@ -0,0 +1,19 @@ +# Generated by utils.py +dataset_name: lug_Latn +doc_to_text: 'Read the passage: {{flores_passage}} + + Then answer the question: {{question.strip()}} + + Options: + + A. {{mc_answer1}} + + B. {{mc_answer2}} + + C. {{mc_answer3}} + + D. {{mc_answer4}} + + Please choose the correct option from the above list:' +include: belebele +task: belebele_lug_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_plt.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_plt.yaml new file mode 100644 index 0000000000000000000000000000000000000000..baad68ab37cf0a5c5d34f9c16a0fdfaa473e5968 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_plt.yaml @@ -0,0 +1,19 @@ +# Generated by utils.py +dataset_name: plt_Latn +doc_to_text: 'Read the passage: {{flores_passage}} + + Then answer the question: {{question.strip()}} + + Options: + + A. {{mc_answer1}} + + B. {{mc_answer2}} + + C. {{mc_answer3}} + + D. {{mc_answer4}} + + Please choose the correct option from the above list:' +include: belebele +task: belebele_plt_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_sna.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_sna.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fd4fc080074beb21801196d5df19ea59e6928b4a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_sna.yaml @@ -0,0 +1,19 @@ +# Generated by utils.py +dataset_name: sna_Latn +doc_to_text: 'Read the passage: {{flores_passage}} + + Then answer the question: {{question.strip()}} + + Options: + + A. {{mc_answer1}} + + B. {{mc_answer2}} + + C. {{mc_answer3}} + + D. {{mc_answer4}} + + Please choose the correct option from the above list:' +include: belebele +task: belebele_sna_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_som.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_som.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3dfa40665cc2874333872503f880c4c373c03b52 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_som.yaml @@ -0,0 +1,19 @@ +# Generated by utils.py +dataset_name: som_Latn +doc_to_text: 'Read the passage: {{flores_passage}} + + Then answer the question: {{question.strip()}} + + Options: + + A. {{mc_answer1}} + + B. {{mc_answer2}} + + C. {{mc_answer3}} + + D. {{mc_answer4}} + + Please choose the correct option from the above list:' +include: belebele +task: belebele_som_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_sot.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_sot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c78c862a08e3e6e6968d9e3e039a6cd67c99e978 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_sot.yaml @@ -0,0 +1,19 @@ +# Generated by utils.py +dataset_name: sot_Latn +doc_to_text: 'Read the passage: {{flores_passage}} + + Then answer the question: {{question.strip()}} + + Options: + + A. {{mc_answer1}} + + B. {{mc_answer2}} + + C. {{mc_answer3}} + + D. {{mc_answer4}} + + Please choose the correct option from the above list:' +include: belebele +task: belebele_sot_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_ssw.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_ssw.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d2e8b96f93a43c64e36d0f2b8199524344511b70 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_ssw.yaml @@ -0,0 +1,19 @@ +# Generated by utils.py +dataset_name: ssw_Latn +doc_to_text: 'Read the passage: {{flores_passage}} + + Then answer the question: {{question.strip()}} + + Options: + + A. {{mc_answer1}} + + B. {{mc_answer2}} + + C. {{mc_answer3}} + + D. {{mc_answer4}} + + Please choose the correct option from the above list:' +include: belebele +task: belebele_ssw_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5a44af344142293043db80d9f55140569b7fdebf --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_swa.yaml @@ -0,0 +1,19 @@ +# Generated by utils.py +dataset_name: swh_Latn +doc_to_text: 'Read the passage: {{flores_passage}} + + Then answer the question: {{question.strip()}} + + Options: + + A. {{mc_answer1}} + + B. {{mc_answer2}} + + C. {{mc_answer3}} + + D. {{mc_answer4}} + + Please choose the correct option from the above list:' +include: belebele +task: belebele_swa_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_tir.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_tir.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4ef9af2ade11b49af64195a00811be7bf69b34d7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_tir.yaml @@ -0,0 +1,19 @@ +# Generated by utils.py +dataset_name: tir_Ethi +doc_to_text: 'Read the passage: {{flores_passage}} + + Then answer the question: {{question.strip()}} + + Options: + + A. {{mc_answer1}} + + B. {{mc_answer2}} + + C. {{mc_answer3}} + + D. {{mc_answer4}} + + Please choose the correct option from the above list:' +include: belebele +task: belebele_tir_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_tsn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_tsn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0de5669b2a6031c7a5960bc174951d99cdc02502 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_tsn.yaml @@ -0,0 +1,19 @@ +# Generated by utils.py +dataset_name: tsn_Latn +doc_to_text: 'Read the passage: {{flores_passage}} + + Then answer the question: {{question.strip()}} + + Options: + + A. {{mc_answer1}} + + B. {{mc_answer2}} + + C. {{mc_answer3}} + + D. {{mc_answer4}} + + Please choose the correct option from the above list:' +include: belebele +task: belebele_tsn_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_tso.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_tso.yaml new file mode 100644 index 0000000000000000000000000000000000000000..92def0f429f0c7517c8904a8a2ae86cc2f534653 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_tso.yaml @@ -0,0 +1,19 @@ +# Generated by utils.py +dataset_name: tso_Latn +doc_to_text: 'Read the passage: {{flores_passage}} + + Then answer the question: {{question.strip()}} + + Options: + + A. {{mc_answer1}} + + B. {{mc_answer2}} + + C. {{mc_answer3}} + + D. {{mc_answer4}} + + Please choose the correct option from the above list:' +include: belebele +task: belebele_tso_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..10192b8a69a44faf3b155a16287d3da15b0c4c5f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_wol.yaml @@ -0,0 +1,19 @@ +# Generated by utils.py +dataset_name: wol_Latn +doc_to_text: 'Read the passage: {{flores_passage}} + + Then answer the question: {{question.strip()}} + + Options: + + A. {{mc_answer1}} + + B. {{mc_answer2}} + + C. {{mc_answer3}} + + D. {{mc_answer4}} + + Please choose the correct option from the above list:' +include: belebele +task: belebele_wol_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9ea12584e13ae4b4e19ac5b9c1fcc3b54ae0f950 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_xho.yaml @@ -0,0 +1,19 @@ +# Generated by utils.py +dataset_name: xho_Latn +doc_to_text: 'Read the passage: {{flores_passage}} + + Then answer the question: {{question.strip()}} + + Options: + + A. {{mc_answer1}} + + B. {{mc_answer2}} + + C. {{mc_answer3}} + + D. {{mc_answer4}} + + Please choose the correct option from the above list:' +include: belebele +task: belebele_xho_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c69e05cee5d26e91892e9303ad09f06856c254d1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_yor.yaml @@ -0,0 +1,19 @@ +# Generated by utils.py +dataset_name: yor_Latn +doc_to_text: 'Read the passage: {{flores_passage}} + + Then answer the question: {{question.strip()}} + + Options: + + A. {{mc_answer1}} + + B. {{mc_answer2}} + + C. {{mc_answer3}} + + D. {{mc_answer4}} + + Please choose the correct option from the above list:' +include: belebele +task: belebele_yor_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c3c6905f9803bba171b95b48c540f730d16158bb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/prompt_5/belebele_zul.yaml @@ -0,0 +1,19 @@ +# Generated by utils.py +dataset_name: zul_Latn +doc_to_text: 'Read the passage: {{flores_passage}} + + Then answer the question: {{question.strip()}} + + Options: + + A. {{mc_answer1}} + + B. {{mc_answer2}} + + C. {{mc_answer3}} + + D. {{mc_answer4}} + + Please choose the correct option from the above list:' +include: belebele +task: belebele_zul_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..7654a6cfe7974b352446e8c71c5740fb9e45f9f2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/belebele/utils.py @@ -0,0 +1,155 @@ +import argparse +import os + +import yaml + + +def prompt_func(mode, lang): + prompt_map = { + "prompt_1": "P: {{flores_passage}}\nQ: {{question.strip()}}\nA: {{mc_answer1}}\nB: {{mc_answer2}}\nC: {{mc_answer3}}\nD: {{mc_answer4}}\nPlease choose the correct answer from the options above:", + "prompt_2": "Passage: {{flores_passage}}\nQuestion: {{question.strip()}}\n1: {{mc_answer1}}\n2: {{mc_answer2}}\n3: {{mc_answer3}}\n4: {{mc_answer4}}\nPlease select the correct answer from the given choices:", + "prompt_3": "Context: {{flores_passage}}\nQuery: {{question.strip()}}\nOption A: {{mc_answer1}}\nOption B: {{mc_answer2}}\nOption C: {{mc_answer3}}\nOption D: {{mc_answer4}}\nPlease indicate the correct option from the list above:", + "prompt_4": "{{flores_passage}}\nBased on the above passage, answer the following question:\n{{question.strip()}}\nChoices:\nA) {{mc_answer1}}\nB) {{mc_answer2}}\nC) {{mc_answer3}}\nD) {{mc_answer4}}\nPlease provide the correct answer from the choices given:", + "prompt_5": "Read the passage: {{flores_passage}}\nThen answer the question: {{question.strip()}}\nOptions:\nA. {{mc_answer1}}\nB. {{mc_answer2}}\nC. {{mc_answer3}}\nD. {{mc_answer4}}\nPlease choose the correct option from the above list:", + } + return prompt_map[mode] + + +def gen_lang_yamls(output_dir: str, overwrite: bool, mode: str) -> None: + """ + Generate a yaml file for each language. + + :param output_dir: The directory to output the files to. + :param overwrite: Whether to overwrite files if they already exist. + """ + err = [] + languages = { + "afr": "Afrikaans", + "amh": "Amharic", + "ary": "Moroccan Arabic", + "arz": "Egyptian Arabic", + "bam": "Bambara", + "eng": "English", + "fra": "French", + "hau": "Hausa", + "ibo": "Igbo", + "lin": "Lingala", + "por": "Portuguese", + "sna": "Shona", + "swa": "Swahili", + "tir": "Tigrinya", + "tso": "Tsonga", + "tsn": "Tswana", + "wol": "Wolof", + "xho": "Xhosa", + "yor": "Yoruba", + "zul": "Zulu", + "ssw": "Swati", + "sot": "Southern Sotho", + "som": "Somali", + "plt": "Plateau Malagasy", + "nya": "Nyanja", + "luo": "Luo", + "lug": "Luganda", + "kin": "Kinyarwanda", + "kea": "Kabuverdianu", + "gaz": "Oromo", + "fuv": "Nigerian Fulfulde", + } + + lang_2_dataset_lang_code = { + "afr": "afr_Latn", + "amh": "amh_Ethi", + "ary": "ary_Arab", + "arz": "arz_Arab", + "bam": "bam_Latn", + "eng": "eng_Latn", + "fra": "fra_Latn", + "hau": "hau_Latn", + "ibo": "ibo_Latn", + "lin": "lin_Latn", + "por": "por_Latn", + "sna": "sna_Latn", + "swa": "swh_Latn", + "tir": "tir_Ethi", + "tso": "tso_Latn", + "tsn": "tsn_Latn", + "wol": "wol_Latn", + "xho": "xho_Latn", + "yor": "yor_Latn", + "zul": "zul_Latn", + "ssw": "ssw_Latn", + "sot": "sot_Latn", + "som": "som_Latn", + "plt": "plt_Latn", + "nya": "nya_Latn", + "luo": "luo_Latn", + "lug": "lug_Latn", + "kin": "kin_Latn", + "kea": "kea_Latn", + "gaz": "gaz_Latn", + "fuv": "fuv_Latn", + } + + for lang in languages.keys(): + try: + file_name = f"belebele_{lang}.yaml" + task_name = f"belebele_{lang}_{mode}" + yaml_template = "belebele" + yaml_details = { + "include": yaml_template, + "task": task_name, + "dataset_name": lang_2_dataset_lang_code[lang], + "doc_to_text": prompt_func(mode, languages[lang]), + } + file_path = os.path.join(output_dir, mode) + os.makedirs(file_path, exist_ok=True) + + with open( + f"{output_dir}/{mode}/{file_name}", + "w" if overwrite else "x", + encoding="utf8", + ) as f: + f.write("# Generated by utils.py\n") + yaml.dump( + yaml_details, + f, + allow_unicode=True, + ) + except FileExistsError: + err.append(file_name) + + if len(err) > 0: + raise FileExistsError( + "Files were not created because they already exist (use --overwrite flag):" + f" {', '.join(err)}" + ) + + +def main() -> None: + """Parse CLI args and generate language-specific yaml files.""" + parser = argparse.ArgumentParser() + parser.add_argument( + "--overwrite", + default=True, + action="store_true", + help="Overwrite files if they already exist", + ) + parser.add_argument( + "--output-dir", + default="./", + help="Directory to write yaml files to", + ) + parser.add_argument( + "--mode", + default="prompt_5", + choices=["prompt_1", "prompt_2", "prompt_3", "prompt_4", "prompt_5"], + help="Prompt number", + ) + args = parser.parse_args() + + gen_lang_yamls(output_dir=args.output_dir, overwrite=args.overwrite, mode=args.mode) + + +if __name__ == "__main__": + main() diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/flores.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/flores.yaml new file mode 100644 index 0000000000000000000000000000000000000000..09b6e39274a4a686d20f742927b9a2740c2ef59f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/flores.yaml @@ -0,0 +1,14 @@ +group: african_flores +task: + - flores_eng-afr_prompt_1 + - flores_eng-afr_prompt_2 + - flores_eng-afr_prompt_3 + - flores_afr-eng_prompt_1 + - flores_afr-eng_prompt_2 + - flores_afr-eng_prompt_3 +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_ace_Arab-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_ace_Arab-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c558249774e6182755078db81154e3d88db656c1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_ace_Arab-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: ace_Arab-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Acehnese (Arabic script): {{sentence_ace_Arab}} \nEnglish: " +include: flores +task: flores_ace_Arab-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_ace_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_ace_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1f0a6ee27cfc218e297ccaf05ab7d0bcef5da57b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_ace_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: ace_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Acehnese (Latin script): {{sentence_ace_Latn}} \nEnglish: " +include: flores +task: flores_ace_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_aeb_Arab-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_aeb_Arab-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..53636d7c01d92e87b80da4bd6656b7740d0f11a6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_aeb_Arab-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: aeb_Arab-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Tunisian Arabic: {{sentence_aeb_Arab}} \nEnglish: " +include: flores +task: flores_aeb_Arab-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_aka_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_aka_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c3caf192676374f70a707c77d84bb6eac1deacb5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_aka_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: aka_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Akan: {{sentence_aka_Latn}} \nEnglish: " +include: flores +task: flores_aka_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_amh_Ethi-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_amh_Ethi-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6c0be0828a5110df25911b503c0db29b2fe6dbb3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_amh_Ethi-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: amh_Ethi-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Amharic: {{sentence_amh_Ethi}} \nEnglish: " +include: flores +task: flores_amh_Ethi-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_ary_Arab-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_ary_Arab-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8bcd452d6e86cf669bcbc97b7216d72dbeb37ffd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_ary_Arab-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: ary_Arab-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Moroccan Arabic: {{sentence_ary_Arab}} \nEnglish: " +include: flores +task: flores_ary_Arab-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_bam_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_bam_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..14e8a1c74fb7f9bcdd47703121bc127420d5cf3a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_bam_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: bam_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Bambara: {{sentence_bam_Latn}} \nEnglish: " +include: flores +task: flores_bam_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_ban_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_ban_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..54a582446ec44263f7020676a7e5fa3eee88e780 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_ban_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: ban_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Balinese: {{sentence_ban_Latn}} \nEnglish: " +include: flores +task: flores_ban_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_bem_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_bem_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..53bbe221d7ec3a98f4eedeb9fdcfd49a2d872198 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_bem_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: bem_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Bemba: {{sentence_bem_Latn}} \nEnglish: " +include: flores +task: flores_bem_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_cjk_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_cjk_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..63994d04d0dfc9ca7d8418835dcd47abf79d5031 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_cjk_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: cjk_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Chokwe: {{sentence_cjk_Latn}} \nEnglish: " +include: flores +task: flores_cjk_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_dik_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_dik_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fd9022b53f0325804b07bf5fb8c222a37c5eccde --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_dik_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: dik_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Southwestern Dinka: {{sentence_dik_Latn}} \nEnglish: " +include: flores +task: flores_dik_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_dyu_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_dyu_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e25e23d89d7090ec09c68af8c83705dd6a43d7d7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_dyu_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: dyu_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Dyula: {{sentence_dyu_Latn}} \nEnglish: " +include: flores +task: flores_dyu_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_ewe_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_ewe_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fffa31fcd93a02581b8e70e20d2b2fd84803365c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_ewe_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: ewe_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Ewe: {{sentence_ewe_Latn}} \nEnglish: " +include: flores +task: flores_ewe_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_fon_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_fon_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..70c9bfbe0f59666124b03c55432d4981472da9d8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_fon_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: fon_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Fon: {{sentence_fon_Latn}} \nEnglish: " +include: flores +task: flores_fon_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_fra_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_fra_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2c515a8f6adff914e6c237a1d637d8a589bc976f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_fra_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: fra_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "French: {{sentence_fra_Latn}} \nEnglish: " +include: flores +task: flores_fra_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_fuv_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_fuv_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4a162567753f9b6ee34d52f5ac06233b54973eac --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_fuv_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: fuv_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Nigerian Fulfulde: {{sentence_fuv_Latn}} \nEnglish: " +include: flores +task: flores_fuv_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_gaz_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_gaz_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ec443459d6b9ac698106c6dc2c500d2b897557f1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_gaz_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: gaz_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Oromo: {{sentence_gaz_Latn}} \nEnglish: " +include: flores +task: flores_gaz_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_hau_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_hau_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8d518b5122fa8ebf09188fd1eca8f6f5e2e23983 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_hau_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: hau_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Hausa: {{sentence_hau_Latn}} \nEnglish: " +include: flores +task: flores_hau_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_ibo_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_ibo_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9c121ae73d95a5acffa72c27af04c9c1b16a7b43 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_ibo_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: ibo_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Igbo: {{sentence_ibo_Latn}} \nEnglish: " +include: flores +task: flores_ibo_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_kab_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_kab_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..42c625488a60ed854f9769636ddc208d8d7d2e0c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_kab_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: kab_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Kabyle: {{sentence_kab_Latn}} \nEnglish: " +include: flores +task: flores_kab_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_kam_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_kam_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d7d10cc570d7c338e7928a36be4a3026cffaf159 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_kam_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: kam_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Kamba: {{sentence_kam_Latn}} \nEnglish: " +include: flores +task: flores_kam_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_kbp_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_kbp_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..43cc5e32a272d14bc8549d28f6f8784ab1e968dd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_kbp_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: kbp_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Kabiyè: {{sentence_kbp_Latn}} \nEnglish: " +include: flores +task: flores_kbp_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_kea_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_kea_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4c894681ef571f72d6edfa953a9d0aeaafb5dcc9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_kea_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: kea_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Kabuverdianu: {{sentence_kea_Latn}} \nEnglish: " +include: flores +task: flores_kea_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_kik_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_kik_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dbdff8e215e247fbc4b0061154f3c66903aad80c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_kik_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: kik_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Kikuyu: {{sentence_kik_Latn}} \nEnglish: " +include: flores +task: flores_kik_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_kin_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_kin_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b11194a98eacd84551995aa1506993a9c8a52bf6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_kin_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: kin_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Kinyarwanda: {{sentence_kin_Latn}} \nEnglish: " +include: flores +task: flores_kin_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_kmb_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_kmb_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..258b847d28294196b7c4d7455320e24d0ad2a59a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_kmb_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: kmb_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Kimbundu: {{sentence_kmb_Latn}} \nEnglish: " +include: flores +task: flores_kmb_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_knc_Arab-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_knc_Arab-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..642dfc6f891572f65037299b8f3a8381f51f0421 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_knc_Arab-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: knc_Arab-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Central Kanuri (Arabic script): {{sentence_knc_Arab}} \nEnglish: " +include: flores +task: flores_knc_Arab-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_knc_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_knc_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7f904da712bda335eab0ccc0e7036b8190caf93e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_knc_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: knc_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Central Kanuri (Latin script): {{sentence_knc_Latn}} \nEnglish: " +include: flores +task: flores_knc_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_kon_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_kon_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..54fce1f8da44e9b2c889e71e8a8d9f5eea4b3ef7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_kon_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: kon_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Kikongo: {{sentence_kon_Latn}} \nEnglish: " +include: flores +task: flores_kon_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_lua_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_lua_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9d54350a45f7838492555c4268e552dc609f0e12 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_lua_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: lua_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Luba-Kasai: {{sentence_lua_Latn}} \nEnglish: " +include: flores +task: flores_lua_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_lug_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_lug_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..35d8e31b1331a8e478bc6960c262a8d5eb5630df --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_lug_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: lug_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Luganda: {{sentence_lug_Latn}} \nEnglish: " +include: flores +task: flores_lug_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_mos_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_mos_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4a4c1009c46290faafcb15930cb98217612d5c14 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_mos_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: mos_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Mossi: {{sentence_mos_Latn}} \nEnglish: " +include: flores +task: flores_mos_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_nus_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_nus_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f77380957e4c1b61cd9a277e8e2d831fa7d9a0da --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_nus_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: nus_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Nuer: {{sentence_nus_Latn}} \nEnglish: " +include: flores +task: flores_nus_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_nya_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_nya_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..def5625dd7a4b6dd4288a796681fe6fbfc40c6ac --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_nya_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: nya_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Nyanja: {{sentence_nya_Latn}} \nEnglish: " +include: flores +task: flores_nya_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_plt_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_plt_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f877a307254dfe00e645ea544bc1a7fb64411162 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_plt_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: plt_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Plateau Malagasy: {{sentence_plt_Latn}} \nEnglish: " +include: flores +task: flores_plt_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_sot_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_sot_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..11653e6059e51d71b48e722abd1c519ddd956d00 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_sot_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: sot_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Southern Sotho: {{sentence_sot_Latn}} \nEnglish: " +include: flores +task: flores_sot_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_ssw_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_ssw_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fe3ceb9a874c035a06f3c9fdc46e254a544bc563 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_ssw_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: ssw_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Swati: {{sentence_ssw_Latn}} \nEnglish: " +include: flores +task: flores_ssw_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_sun_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_sun_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c3f605f9400dacb782a0066b1f6559aaba6ed270 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_sun_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: sun_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Sundanese: {{sentence_sun_Latn}} \nEnglish: " +include: flores +task: flores_sun_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_swh_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_swh_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7651ac3159f59d94886dc97a8e854fb19e184115 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_swh_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: swh_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Swahili: {{sentence_swh_Latn}} \nEnglish: " +include: flores +task: flores_swh_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_taq_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_taq_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d3fca39004e66ae6c74858cea59e930527a41eff --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_taq_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: taq_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Tamasheq: {{sentence_taq_Latn}} \nEnglish: " +include: flores +task: flores_taq_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_taq_Tfng-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_taq_Tfng-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7152867ee6e32951f630f2d24d615ec248090fde --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/african-english/flores_taq_Tfng-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: taq_Tfng-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Tamasheq (Tifinagh script): {{sentence_taq_Tfng}} \nEnglish: " +include: flores +task: flores_taq_Tfng-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores new file mode 100644 index 0000000000000000000000000000000000000000..e6f4d051431159f4360115226ea58dec2487c0c2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores @@ -0,0 +1,27 @@ +tag: +- african_flores_tasks +- flores_eng-afr +- flores_eng-afr_prompt_1 +- afrobench_MT_tasks +dataset_path: facebook/flores +dataset_kwargs: {trust_remote_code: True} +output_type: generate_until +validation_split: dev +fewshot_split: dev +test_split: devtest +metric_list: + - metric: bleu + aggregation: bleu + higher_is_better: true + - metric: chrf + aggregation: chrf + higher_is_better: true +generation_kwargs: + until: + - "**" + - + do_sample: false + temperature: 0.0 +repeats: 1 +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-ace_Arab.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-ace_Arab.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a9da06483bc1e3f19e10636cdf1509ad899832ca --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-ace_Arab.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-ace_Arab +doc_to_target: sentence_ace_Arab +doc_to_text: "English: {{sentence_eng_Latn}} \nAcehnese (Arabic script): " +include: flores +task: flores_eng_Latn-ace_Arab_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-acq_Arab.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-acq_Arab.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e61bb2472b427de012ce3d47122906df99f14089 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-acq_Arab.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-acq_Arab +doc_to_target: sentence_acq_Arab +doc_to_text: "English: {{sentence_eng_Latn}} \nTa’izzi-Adeni Arabic: " +include: flores +task: flores_eng_Latn-acq_Arab_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-aeb_Arab.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-aeb_Arab.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d59000626aa9ef7f6cfcd6bc6a315cbb25a90142 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-aeb_Arab.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-aeb_Arab +doc_to_target: sentence_aeb_Arab +doc_to_text: "English: {{sentence_eng_Latn}} \nTunisian Arabic: " +include: flores +task: flores_eng_Latn-aeb_Arab_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-afr_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-afr_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3b4c4d46b432d78f6e9947dbcd26868885560c53 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-afr_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-afr_Latn +doc_to_target: sentence_afr_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nAfrikaans: " +include: flores +task: flores_eng_Latn-afr_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-aka_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-aka_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d66a637f75d19c72d3846819d249f9a67989e04c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-aka_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-aka_Latn +doc_to_target: sentence_aka_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nAkan: " +include: flores +task: flores_eng_Latn-aka_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-amh_Ethi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-amh_Ethi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e648d33270ae5944609c8ce50c2dd3e92bbfeb97 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-amh_Ethi.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-amh_Ethi +doc_to_target: sentence_amh_Ethi +doc_to_text: "English: {{sentence_eng_Latn}} \nAmharic: " +include: flores +task: flores_eng_Latn-amh_Ethi_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-ary_Arab.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-ary_Arab.yaml new file mode 100644 index 0000000000000000000000000000000000000000..54f9a2ad67ac390ba4cc4a7a6db6a1d2e5061a54 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-ary_Arab.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-ary_Arab +doc_to_target: sentence_ary_Arab +doc_to_text: "English: {{sentence_eng_Latn}} \nMoroccan Arabic: " +include: flores +task: flores_eng_Latn-ary_Arab_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-bam_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-bam_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4c85b7db9394d3c309b3e5c5b196a0e5451c4d0b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-bam_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-bam_Latn +doc_to_target: sentence_bam_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nBambara: " +include: flores +task: flores_eng_Latn-bam_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-cjk_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-cjk_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..eb4e3566b8383d7bce70a88bd3b663fa984c5154 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-cjk_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-cjk_Latn +doc_to_target: sentence_cjk_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nChokwe: " +include: flores +task: flores_eng_Latn-cjk_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-dyu_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-dyu_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c32be8ac93c7cecfbc171eb898232c7296cf6886 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-dyu_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-dyu_Latn +doc_to_target: sentence_dyu_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nDyula: " +include: flores +task: flores_eng_Latn-dyu_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-ewe_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-ewe_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0a71b4556a077b260bfb340a3c0c289ae79ac88b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-ewe_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-ewe_Latn +doc_to_target: sentence_ewe_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nEwe: " +include: flores +task: flores_eng_Latn-ewe_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-fon_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-fon_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1000e13ad1f6864f002c741b8074d06073cb3dc8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-fon_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-fon_Latn +doc_to_target: sentence_fon_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nFon: " +include: flores +task: flores_eng_Latn-fon_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-fra_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-fra_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..47b99a088c485bed46c51d1da0308ac569aaebd3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-fra_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-fra_Latn +doc_to_target: sentence_fra_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nFrench: " +include: flores +task: flores_eng_Latn-fra_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-gaz_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-gaz_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8e124ae153091ed617f87388afb7d6c4c980d754 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-gaz_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-gaz_Latn +doc_to_target: sentence_gaz_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nOromo: " +include: flores +task: flores_eng_Latn-gaz_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-hau_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-hau_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a9aaf537f1d1c491ae3de996d2180c9b32002647 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-hau_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-hau_Latn +doc_to_target: sentence_hau_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nHausa: " +include: flores +task: flores_eng_Latn-hau_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-ibo_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-ibo_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ebf8f517c3e96d64716db741688161d520bd04a5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-ibo_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-ibo_Latn +doc_to_target: sentence_ibo_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nIgbo: " +include: flores +task: flores_eng_Latn-ibo_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-kik_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-kik_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bee435fe5495364b08420772d1dfade8f9ac671d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-kik_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-kik_Latn +doc_to_target: sentence_kik_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nKikuyu: " +include: flores +task: flores_eng_Latn-kik_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-lua_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-lua_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..246fc1354a71eadbe1ea6e058387859fd5c018c1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-lua_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-lua_Latn +doc_to_target: sentence_lua_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nLuba-Kasai: " +include: flores +task: flores_eng_Latn-lua_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-tsn_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-tsn_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d782a2af5cde5ae3a006c205a0796ea1a15750d8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-tsn_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-tsn_Latn +doc_to_target: sentence_tsn_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nSetswana: " +include: flores +task: flores_eng_Latn-tsn_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-tum_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-tum_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a9036f3b7a1d2c4fa91a1f4278c1019cdf2bc68a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-tum_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-tum_Latn +doc_to_target: sentence_tum_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nTumbuka: " +include: flores +task: flores_eng_Latn-tum_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-twi_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-twi_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9658615983d1c44bfd74d88def6db73a465ce96d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-twi_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-twi_Latn +doc_to_target: sentence_twi_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nTwi: " +include: flores +task: flores_eng_Latn-twi_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-wol_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-wol_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..eb6965245032f6821b4ca413d6ead9e892bdb407 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-wol_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-wol_Latn +doc_to_target: sentence_wol_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nWolof: " +include: flores +task: flores_eng_Latn-wol_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-xho_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-xho_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..08480361c18c787ead563d02783982bd1ad8b8e8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-xho_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-xho_Latn +doc_to_target: sentence_xho_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nXhosa: " +include: flores +task: flores_eng_Latn-xho_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-yor_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-yor_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d29e9a9c859134f25abdc46ca44a256d473415a0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/english-african/flores_eng_Latn-yor_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: eng_Latn-yor_Latn +doc_to_target: sentence_yor_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nYoruba: " +include: flores +task: flores_eng_Latn-yor_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/flores b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/flores new file mode 100644 index 0000000000000000000000000000000000000000..74f9f33eb22662bec79709bd64d8d31f3fb8eae0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_1/flores @@ -0,0 +1,24 @@ +tag: +- flores_tasks +- flores_afr-eng +dataset_path: facebook/flores +dataset_kwargs: {trust_remote_code: True} +output_type: generate_until +validation_split: dev +fewshot_split: dev +test_split: devtest +metric_list: + - metric: bleu + aggregation: bleu + higher_is_better: true + - metric: chrf + aggregation: chrf + higher_is_better: true +generation_kwargs: + until: + - "\n" + do_sample: false + temperature: 0.0 +repeats: 1 +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores new file mode 100644 index 0000000000000000000000000000000000000000..e0fa69a2a441116ef15a4158cc366792d841f304 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores @@ -0,0 +1,27 @@ +tag: +- african_flores_tasks +- flores_afr-eng +- flores_afr-eng_prompt_2 +- afrobench_MT_tasks +dataset_path: facebook/flores +dataset_kwargs: {trust_remote_code: True} +output_type: generate_until +validation_split: dev +fewshot_split: dev +test_split: devtest +metric_list: + - metric: bleu + aggregation: bleu + higher_is_better: true + - metric: chrf + aggregation: chrf + higher_is_better: true +generation_kwargs: + until: + - "**" + - + do_sample: false + temperature: 0.0 +repeats: 1 +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_amh_Ethi-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_amh_Ethi-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d615bfc3d6f9ed19dce78c024fdaa45a53fdac8b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_amh_Ethi-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: amh_Ethi-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Amharic sentences\ + \ to English \nAmharic: {{sentence_amh_Ethi}}\nEnglish: " +include: flores +task: flores_amh_Ethi-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_kbp_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_kbp_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7e20af3149dc1baad3b0edca477444a98ba078c9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_kbp_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: kbp_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Kabiyè sentences\ + \ to English \nKabiyè: {{sentence_kbp_Latn}}\nEnglish: " +include: flores +task: flores_kbp_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_knc_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_knc_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1d60408cd39dc640f7db077186340de97aa4702f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_knc_Latn-eng_Latn.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: knc_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Central Kanuri\ + \ (Latin script) sentences to English \nCentral Kanuri (Latin script): {{sentence_knc_Latn}}\n\ + English: " +include: flores +task: flores_knc_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_mos_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_mos_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b72acf36fdc39990bc8d6a91a13e1194ce3d42df --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_mos_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: mos_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Mossi sentences\ + \ to English \nMossi: {{sentence_mos_Latn}}\nEnglish: " +include: flores +task: flores_mos_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_nus_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_nus_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e1f9ca54df695d83cc607b4409522ab459c28a99 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_nus_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: nus_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Nuer sentences\ + \ to English \nNuer: {{sentence_nus_Latn}}\nEnglish: " +include: flores +task: flores_nus_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_nya_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_nya_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a5ceb01789f8d71651931a85a2b3580381895d97 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_nya_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: nya_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Nyanja sentences\ + \ to English \nNyanja: {{sentence_nya_Latn}}\nEnglish: " +include: flores +task: flores_nya_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_plt_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_plt_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a2cdace5ed128379cd6093e6da5fee9345ef44c3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_plt_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: plt_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Plateau Malagasy\ + \ sentences to English \nPlateau Malagasy: {{sentence_plt_Latn}}\nEnglish: " +include: flores +task: flores_plt_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_run_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_run_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..aa4b5bc968c230b50942903d989e30e80cb51f8b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_run_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: run_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Rundi sentences\ + \ to English \nRundi: {{sentence_run_Latn}}\nEnglish: " +include: flores +task: flores_run_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_sag_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_sag_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b20eef56654fac2f2f086dcf6e0deea8a59c345d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_sag_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: sag_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Sango sentences\ + \ to English \nSango: {{sentence_sag_Latn}}\nEnglish: " +include: flores +task: flores_sag_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_som_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_som_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b862c759b912e197cc16acda3cb68d1271d77e0f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_som_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: som_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Somali sentences\ + \ to English \nSomali: {{sentence_som_Latn}}\nEnglish: " +include: flores +task: flores_som_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_sot_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_sot_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d5d4e24709a334418b7a23a5d0852f7e5ea665b8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_sot_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: sot_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Southern Sotho\ + \ sentences to English \nSouthern Sotho: {{sentence_sot_Latn}}\nEnglish: " +include: flores +task: flores_sot_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_swh_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_swh_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..06dd9fcc0d384c4926a681e64f1c185c1111fe94 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_swh_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: swh_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Swahili sentences\ + \ to English \nSwahili: {{sentence_swh_Latn}}\nEnglish: " +include: flores +task: flores_swh_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_taq_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_taq_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5380298e28c3be0c4c9ba536dfdcb685dd7356f2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_taq_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: taq_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Tamasheq sentences\ + \ to English \nTamasheq: {{sentence_taq_Latn}}\nEnglish: " +include: flores +task: flores_taq_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_taq_Tfng-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_taq_Tfng-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7cfb54197cdaffd83c578da681c1b5d36c9f4265 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_taq_Tfng-eng_Latn.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: taq_Tfng-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Tamasheq (Tifinagh\ + \ script) sentences to English \nTamasheq (Tifinagh script): {{sentence_taq_Tfng}}\n\ + English: " +include: flores +task: flores_taq_Tfng-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_tsn_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_tsn_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b8d04febf4a6ebf564df918c236ede2ccc016b34 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_tsn_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: tsn_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Setswana sentences\ + \ to English \nSetswana: {{sentence_tsn_Latn}}\nEnglish: " +include: flores +task: flores_tsn_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_tso_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_tso_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3c357e9df91da2e9b05faf883128ad9b81028331 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_tso_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: tso_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Tsonga sentences\ + \ to English \nTsonga: {{sentence_tso_Latn}}\nEnglish: " +include: flores +task: flores_tso_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_wol_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_wol_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0f1210fec591a63d43bd3afebd770b844ffd28a8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_wol_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: wol_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Wolof sentences\ + \ to English \nWolof: {{sentence_wol_Latn}}\nEnglish: " +include: flores +task: flores_wol_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_xho_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_xho_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f28e1bb3eed67659b3ac23ef9f97e6cd9c5ba7d8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_xho_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: xho_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Xhosa sentences\ + \ to English \nXhosa: {{sentence_xho_Latn}}\nEnglish: " +include: flores +task: flores_xho_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores new file mode 100644 index 0000000000000000000000000000000000000000..ab71d6563002c5deed46fb73f2b61bd585b7b9ae --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores @@ -0,0 +1,27 @@ +tag: +- african_flores_tasks +- flores_eng-afr +- flores_eng-afr_prompt_2 +- afrobench_MT_tasks +dataset_path: facebook/flores +dataset_kwargs: {trust_remote_code: True} +output_type: generate_until +validation_split: dev +fewshot_split: dev +test_split: devtest +metric_list: + - metric: bleu + aggregation: bleu + higher_is_better: true + - metric: chrf + aggregation: chrf + higher_is_better: true +generation_kwargs: + until: + - "**" + - + do_sample: false + temperature: 0.0 +repeats: 1 +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-afr_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-afr_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e2769adf1ec72d54aab8dd1911b91f14c6c56db7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-afr_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-afr_Latn +doc_to_target: sentence_afr_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Afrikaans \nEnglish: {{sentence_eng_Latn}} \nAfrikaans: " +include: flores +task: flores_eng_Latn-afr_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-amh_Ethi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-amh_Ethi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0a53e8c2f24cba707d059a83dfa18d3f83791021 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-amh_Ethi.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-amh_Ethi +doc_to_target: sentence_amh_Ethi +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Amharic \nEnglish: {{sentence_eng_Latn}} \nAmharic: " +include: flores +task: flores_eng_Latn-amh_Ethi_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-arz_Arab.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-arz_Arab.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0362666cb7d58d02ed5af9e9429f8b56e1a12d47 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-arz_Arab.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-arz_Arab +doc_to_target: sentence_arz_Arab +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Egyptian Arabic \nEnglish: {{sentence_eng_Latn}} \nEgyptian Arabic: " +include: flores +task: flores_eng_Latn-arz_Arab_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-bam_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-bam_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b38459211670064398a117bdb4b5a63c342471d8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-bam_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-bam_Latn +doc_to_target: sentence_bam_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Bambara \nEnglish: {{sentence_eng_Latn}} \nBambara: " +include: flores +task: flores_eng_Latn-bam_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-ban_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-ban_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cff3c15bd4f226d62932f443416cc5e824dae612 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-ban_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-ban_Latn +doc_to_target: sentence_ban_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Balinese \nEnglish: {{sentence_eng_Latn}} \nBalinese: " +include: flores +task: flores_eng_Latn-ban_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-bem_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-bem_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ef6552a2b4fc29aa64cb6c3e4b3f1304260c9d76 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-bem_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-bem_Latn +doc_to_target: sentence_bem_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Bemba \nEnglish: {{sentence_eng_Latn}} \nBemba: " +include: flores +task: flores_eng_Latn-bem_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-dik_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-dik_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bfcf7180903bace41e71d175a45c65ff68167344 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-dik_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-dik_Latn +doc_to_target: sentence_dik_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Southwestern Dinka \nEnglish: {{sentence_eng_Latn}} \nSouthwestern Dinka: " +include: flores +task: flores_eng_Latn-dik_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-ewe_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-ewe_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5ecc34e50ab717b0fa3d8d6608cb952692446f89 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-ewe_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-ewe_Latn +doc_to_target: sentence_ewe_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Ewe \nEnglish: {{sentence_eng_Latn}} \nEwe: " +include: flores +task: flores_eng_Latn-ewe_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-fon_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-fon_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ed029237af79c6aaabe9942cb911a556718c014c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-fon_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-fon_Latn +doc_to_target: sentence_fon_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Fon \nEnglish: {{sentence_eng_Latn}} \nFon: " +include: flores +task: flores_eng_Latn-fon_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-fra_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-fra_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9d54e66c20d87b05dc59ee76a468f89fa5aca761 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-fra_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-fra_Latn +doc_to_target: sentence_fra_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to French \nEnglish: {{sentence_eng_Latn}} \nFrench: " +include: flores +task: flores_eng_Latn-fra_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-fuv_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-fuv_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a821f58fa428af19d22b819428db35a52f4a6725 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-fuv_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-fuv_Latn +doc_to_target: sentence_fuv_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Nigerian Fulfulde \nEnglish: {{sentence_eng_Latn}} \nNigerian Fulfulde: " +include: flores +task: flores_eng_Latn-fuv_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-kea_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-kea_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4a67cb9fef15715713918aaafe29f1147f40acda --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-kea_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-kea_Latn +doc_to_target: sentence_kea_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Kabuverdianu \nEnglish: {{sentence_eng_Latn}} \nKabuverdianu: " +include: flores +task: flores_eng_Latn-kea_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-som_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-som_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f87466dc875c1b2402ccd195f164949c94aa3e5e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-som_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-som_Latn +doc_to_target: sentence_som_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Somali \nEnglish: {{sentence_eng_Latn}} \nSomali: " +include: flores +task: flores_eng_Latn-som_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-sot_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-sot_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..674d162b64e72d5c3d58521643a8dae6042b9cf5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-sot_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-sot_Latn +doc_to_target: sentence_sot_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Southern Sotho \nEnglish: {{sentence_eng_Latn}} \nSouthern Sotho: " +include: flores +task: flores_eng_Latn-sot_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-swh_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-swh_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1558af98e0011ddf66f5ec63bcde425414a539a2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-swh_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-swh_Latn +doc_to_target: sentence_swh_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Swahili \nEnglish: {{sentence_eng_Latn}} \nSwahili: " +include: flores +task: flores_eng_Latn-swh_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-taq_Tfng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-taq_Tfng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b69f1dbd4dd96f81e055b758bfc103a81f7c116c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-taq_Tfng.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: eng_Latn-taq_Tfng +doc_to_target: sentence_taq_Tfng +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Tamasheq (Tifinagh script) \nEnglish: {{sentence_eng_Latn}} \nTamasheq (Tifinagh\ + \ script): " +include: flores +task: flores_eng_Latn-taq_Tfng_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-tir_Ethi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-tir_Ethi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4340591d8e397caa4525832e1225b364574c4664 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-tir_Ethi.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-tir_Ethi +doc_to_target: sentence_tir_Ethi +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Tigrinya \nEnglish: {{sentence_eng_Latn}} \nTigrinya: " +include: flores +task: flores_eng_Latn-tir_Ethi_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-tso_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-tso_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0d027a2aa2fc08aaa9fa792391eb69d46ddee802 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-tso_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-tso_Latn +doc_to_target: sentence_tso_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Tsonga \nEnglish: {{sentence_eng_Latn}} \nTsonga: " +include: flores +task: flores_eng_Latn-tso_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-tum_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-tum_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1accaeaf4bd9cbbfb5c46c6341a3bb81663767be --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-tum_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-tum_Latn +doc_to_target: sentence_tum_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Tumbuka \nEnglish: {{sentence_eng_Latn}} \nTumbuka: " +include: flores +task: flores_eng_Latn-tum_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-twi_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-twi_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4a45df82e6c60141396deafa19cd7882b2edb689 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-twi_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-twi_Latn +doc_to_target: sentence_twi_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Twi \nEnglish: {{sentence_eng_Latn}} \nTwi: " +include: flores +task: flores_eng_Latn-twi_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-tzm_Tfng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-tzm_Tfng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6a3faa15d24df79d839a226cc374ace410133a12 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-tzm_Tfng.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-tzm_Tfng +doc_to_target: sentence_tzm_Tfng +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Central Atlas Tamazight \nEnglish: {{sentence_eng_Latn}} \nCentral Atlas Tamazight: " +include: flores +task: flores_eng_Latn-tzm_Tfng_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-umb_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-umb_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3f21c6fe1939f288d56b6bd1229ce6755babb807 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-umb_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-umb_Latn +doc_to_target: sentence_umb_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Umbundu \nEnglish: {{sentence_eng_Latn}} \nUmbundu: " +include: flores +task: flores_eng_Latn-umb_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-wol_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-wol_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..263ded277f0e1a596eac0bf1af5ab1858cf6cd42 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-wol_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-wol_Latn +doc_to_target: sentence_wol_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Wolof \nEnglish: {{sentence_eng_Latn}} \nWolof: " +include: flores +task: flores_eng_Latn-wol_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-xho_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-xho_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a92e46f996b3dfacb06bd5d8d589d43763800e1a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-xho_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-xho_Latn +doc_to_target: sentence_xho_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Xhosa \nEnglish: {{sentence_eng_Latn}} \nXhosa: " +include: flores +task: flores_eng_Latn-xho_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-yor_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-yor_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..80ec895c70fc2f09261ec6df48f2e6bc9755f479 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-yor_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-yor_Latn +doc_to_target: sentence_yor_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Yoruba \nEnglish: {{sentence_eng_Latn}} \nYoruba: " +include: flores +task: flores_eng_Latn-yor_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-zul_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-zul_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..593cdfe3c7c6878bef06e9af476476fdcbbfdfd6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-zul_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-zul_Latn +doc_to_target: sentence_zul_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Zulu \nEnglish: {{sentence_eng_Latn}} \nZulu: " +include: flores +task: flores_eng_Latn-zul_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/flores b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/flores new file mode 100644 index 0000000000000000000000000000000000000000..74f9f33eb22662bec79709bd64d8d31f3fb8eae0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/flores @@ -0,0 +1,24 @@ +tag: +- flores_tasks +- flores_afr-eng +dataset_path: facebook/flores +dataset_kwargs: {trust_remote_code: True} +output_type: generate_until +validation_split: dev +fewshot_split: dev +test_split: devtest +metric_list: + - metric: bleu + aggregation: bleu + higher_is_better: true + - metric: chrf + aggregation: chrf + higher_is_better: true +generation_kwargs: + until: + - "\n" + do_sample: false + temperature: 0.0 +repeats: 1 +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_ewe_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_ewe_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..77133ad60cc57992882213908c1abe2300fae291 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_ewe_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: ewe_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Ewe and English linguist, translate the following Ewe sentences\ + \ to English \nEwe: {{sentence_ewe_Latn}}\nEnglish: " +include: flores +task: flores_ewe_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_fon_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_fon_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..436bf4ac3e5319e8cf2da5791607f8d4f7564eca --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_fon_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: fon_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Fon and English linguist, translate the following Fon sentences\ + \ to English \nFon: {{sentence_fon_Latn}}\nEnglish: " +include: flores +task: flores_fon_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_ibo_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_ibo_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7705911a67b2584a9d1afc3cd5c4294a37a22ece --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_ibo_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: ibo_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Igbo and English linguist, translate the following Igbo sentences\ + \ to English \nIgbo: {{sentence_ibo_Latn}}\nEnglish: " +include: flores +task: flores_ibo_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_kam_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_kam_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ed27b6d79c71b5c1f4690cd409a86aabf9901124 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_kam_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: kam_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Kamba and English linguist, translate the following Kamba sentences\ + \ to English \nKamba: {{sentence_kam_Latn}}\nEnglish: " +include: flores +task: flores_kam_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_kbp_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_kbp_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5c1a0961e08908cf1e846a87e7ddce3641e80ead --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_kbp_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: kbp_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Kabiyè and English linguist, translate the following Kabiyè sentences\ + \ to English \nKabiyè: {{sentence_kbp_Latn}}\nEnglish: " +include: flores +task: flores_kbp_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_kea_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_kea_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..67dd9e73fa327338fecec80728ac645f78996b92 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_kea_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: kea_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Kabuverdianu and English linguist, translate the following Kabuverdianu\ + \ sentences to English \nKabuverdianu: {{sentence_kea_Latn}}\nEnglish: " +include: flores +task: flores_kea_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_kik_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_kik_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..14a6be5dfd5c86c44c8f3acb23af0f36d5445ead --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_kik_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: kik_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Kikuyu and English linguist, translate the following Kikuyu sentences\ + \ to English \nKikuyu: {{sentence_kik_Latn}}\nEnglish: " +include: flores +task: flores_kik_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_knc_Arab-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_knc_Arab-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c8c7f8095e11d407d75360ee5e4794e45fc17eeb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_knc_Arab-eng_Latn.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: knc_Arab-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Central Kanuri (Arabic script) and English linguist, translate\ + \ the following Central Kanuri (Arabic script) sentences to English \nCentral Kanuri\ + \ (Arabic script): {{sentence_knc_Arab}}\nEnglish: " +include: flores +task: flores_knc_Arab-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_lin_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_lin_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ea7e736d949726bf2296091e4309a12d493dbe4f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_lin_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: lin_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Lingala and English linguist, translate the following Lingala sentences\ + \ to English \nLingala: {{sentence_lin_Latn}}\nEnglish: " +include: flores +task: flores_lin_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_lua_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_lua_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..327f014489f7502ac062537c72d4846a7099b64d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_lua_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: lua_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Luba-Kasai and English linguist, translate the following Luba-Kasai\ + \ sentences to English \nLuba-Kasai: {{sentence_lua_Latn}}\nEnglish: " +include: flores +task: flores_lua_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_lug_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_lug_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9bfa92fa280f98278f7735634231cb99d44bc71f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_lug_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: lug_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Luganda and English linguist, translate the following Luganda sentences\ + \ to English \nLuganda: {{sentence_lug_Latn}}\nEnglish: " +include: flores +task: flores_lug_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_luo_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_luo_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a66fded383a914454aed5e904278e60bf85d1e62 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_luo_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: luo_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Luo and English linguist, translate the following Luo sentences\ + \ to English \nLuo: {{sentence_luo_Latn}}\nEnglish: " +include: flores +task: flores_luo_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_mos_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_mos_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e428853bf22859e37bbd10dcd4e113188b7519ef --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_mos_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: mos_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Mossi and English linguist, translate the following Mossi sentences\ + \ to English \nMossi: {{sentence_mos_Latn}}\nEnglish: " +include: flores +task: flores_mos_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_nso_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_nso_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..054aa409b729cc70d70b1d53a49945f92096f4b9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_nso_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: nso_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Northern Sotho and English linguist, translate the following Northern\ + \ Sotho sentences to English \nNorthern Sotho: {{sentence_nso_Latn}}\nEnglish: " +include: flores +task: flores_nso_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_nya_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_nya_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e23c57c6807991e1b509c0a03a7c93a23eac3015 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_nya_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: nya_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Nyanja and English linguist, translate the following Nyanja sentences\ + \ to English \nNyanja: {{sentence_nya_Latn}}\nEnglish: " +include: flores +task: flores_nya_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_plt_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_plt_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3ddfd864c3830d4921e1fcda79155c733809b305 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_plt_Latn-eng_Latn.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: plt_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Plateau Malagasy and English linguist, translate the following\ + \ Plateau Malagasy sentences to English \nPlateau Malagasy: {{sentence_plt_Latn}}\n\ + English: " +include: flores +task: flores_plt_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_run_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_run_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..64a82f716b71950711e6b055a6e30f45356a082c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_run_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: run_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Rundi and English linguist, translate the following Rundi sentences\ + \ to English \nRundi: {{sentence_run_Latn}}\nEnglish: " +include: flores +task: flores_run_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_sag_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_sag_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..48408f94054fee78fe7c3be6460de563e9e60f0a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_sag_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: sag_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Sango and English linguist, translate the following Sango sentences\ + \ to English \nSango: {{sentence_sag_Latn}}\nEnglish: " +include: flores +task: flores_sag_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_sna_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_sna_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ff1626419b69a8e94349bfd63093ad33914bbcec --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_sna_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: sna_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Shona and English linguist, translate the following Shona sentences\ + \ to English \nShona: {{sentence_sna_Latn}}\nEnglish: " +include: flores +task: flores_sna_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_som_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_som_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7e27e2a5b3d1754f4c47755a1b92c7ac95938e67 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_som_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: som_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Somali and English linguist, translate the following Somali sentences\ + \ to English \nSomali: {{sentence_som_Latn}}\nEnglish: " +include: flores +task: flores_som_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_sot_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_sot_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cc70b6f62b317e27cd8c00d95d89e103945028dc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_sot_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: sot_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Southern Sotho and English linguist, translate the following Southern\ + \ Sotho sentences to English \nSouthern Sotho: {{sentence_sot_Latn}}\nEnglish: " +include: flores +task: flores_sot_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_ssw_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_ssw_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0cd61ae8e8ceb6b9a0f985457f26d25961272036 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_ssw_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: ssw_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Swati and English linguist, translate the following Swati sentences\ + \ to English \nSwati: {{sentence_ssw_Latn}}\nEnglish: " +include: flores +task: flores_ssw_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_sun_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_sun_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..000108f77ea1a0b578c1ca980ed0d74490f24fdf --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_sun_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: sun_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Sundanese and English linguist, translate the following Sundanese\ + \ sentences to English \nSundanese: {{sentence_sun_Latn}}\nEnglish: " +include: flores +task: flores_sun_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_swh_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_swh_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1c81805c1f22b9e6c6bd55ba925fa7dfb80f0cf1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_swh_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: swh_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Swahili and English linguist, translate the following Swahili sentences\ + \ to English \nSwahili: {{sentence_swh_Latn}}\nEnglish: " +include: flores +task: flores_swh_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_taq_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_taq_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6febb3004bc3f39f94287ac41a9883d95f056fe1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_taq_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: taq_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Tamasheq and English linguist, translate the following Tamasheq\ + \ sentences to English \nTamasheq: {{sentence_taq_Latn}}\nEnglish: " +include: flores +task: flores_taq_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_taq_Tfng-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_taq_Tfng-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6290ab94d3be2e52750f9af5900d7c27f27cb5af --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_taq_Tfng-eng_Latn.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: taq_Tfng-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Tamasheq (Tifinagh script) and English linguist, translate the\ + \ following Tamasheq (Tifinagh script) sentences to English \nTamasheq (Tifinagh\ + \ script): {{sentence_taq_Tfng}}\nEnglish: " +include: flores +task: flores_taq_Tfng-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_tir_Ethi-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_tir_Ethi-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..60133a3b735a0b1d91901bdd7f1ef2122b0f0f03 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_tir_Ethi-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: tir_Ethi-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Tigrinya and English linguist, translate the following Tigrinya\ + \ sentences to English \nTigrinya: {{sentence_tir_Ethi}}\nEnglish: " +include: flores +task: flores_tir_Ethi-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_tsn_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_tsn_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..40417bde77b4a0ede2e9c87c56668be101579f3b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_tsn_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: tsn_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Setswana and English linguist, translate the following Setswana\ + \ sentences to English \nSetswana: {{sentence_tsn_Latn}}\nEnglish: " +include: flores +task: flores_tsn_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_tso_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_tso_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..56d4632500b86964d0d665f1827cd129bc508d63 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_tso_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: tso_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Tsonga and English linguist, translate the following Tsonga sentences\ + \ to English \nTsonga: {{sentence_tso_Latn}}\nEnglish: " +include: flores +task: flores_tso_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_tum_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_tum_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cc4bb541f7a5692a067ad75f5e3a86490487cc70 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_tum_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: tum_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Tumbuka and English linguist, translate the following Tumbuka sentences\ + \ to English \nTumbuka: {{sentence_tum_Latn}}\nEnglish: " +include: flores +task: flores_tum_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_umb_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_umb_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e7df76cf07cb4bf8f772721136fd4d92280b3820 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_umb_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: umb_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Umbundu and English linguist, translate the following Umbundu sentences\ + \ to English \nUmbundu: {{sentence_umb_Latn}}\nEnglish: " +include: flores +task: flores_umb_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_wol_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_wol_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..22275ca15cd1829db481c0c77363a10649be101f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_wol_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: wol_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Wolof and English linguist, translate the following Wolof sentences\ + \ to English \nWolof: {{sentence_wol_Latn}}\nEnglish: " +include: flores +task: flores_wol_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_xho_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_xho_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..85ae368b6efa8dab3a8a2b110e438b043ef3c74f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_xho_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: xho_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Xhosa and English linguist, translate the following Xhosa sentences\ + \ to English \nXhosa: {{sentence_xho_Latn}}\nEnglish: " +include: flores +task: flores_xho_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_zul_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_zul_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ea2c2edb8439fffb452187b86ed1690501b20b3d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_zul_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: zul_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Zulu and English linguist, translate the following Zulu sentences\ + \ to English \nZulu: {{sentence_zul_Latn}}\nEnglish: " +include: flores +task: flores_zul_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores new file mode 100644 index 0000000000000000000000000000000000000000..ac7dc1651e4729ae0357c6d958745400ddc35ea1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores @@ -0,0 +1,27 @@ +tag: +- african_flores_tasks +- flores_eng-afr +- flores_eng-afr_prompt_3 +- afrobench_MT_tasks +dataset_path: facebook/flores +dataset_kwargs: {trust_remote_code: True} +output_type: generate_until +validation_split: dev +fewshot_split: dev +test_split: devtest +metric_list: + - metric: bleu + aggregation: bleu + higher_is_better: true + - metric: chrf + aggregation: chrf + higher_is_better: true +generation_kwargs: + until: + - "**" + - + do_sample: false + temperature: 0.0 +repeats: 1 +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-ace_Arab.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-ace_Arab.yaml new file mode 100644 index 0000000000000000000000000000000000000000..53cf711fa19132b8668d1c4a6024e1b96f54751b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-ace_Arab.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: eng_Latn-ace_Arab +doc_to_target: sentence_ace_Arab +doc_to_text: "As a Acehnese (Arabic script) and English linguist, translate the following\ + \ English sentences to Acehnese (Arabic script) \nEnglish: {{sentence_eng_Latn}}\ + \ \nAcehnese (Arabic script): " +include: flores +task: flores_eng_Latn-ace_Arab_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-acq_Arab.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-acq_Arab.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e809c866eb602e76defc6c3fca983e02bc213a52 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-acq_Arab.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: eng_Latn-acq_Arab +doc_to_target: sentence_acq_Arab +doc_to_text: "As a Ta’izzi-Adeni Arabic and English linguist, translate the following\ + \ English sentences to Ta’izzi-Adeni Arabic \nEnglish: {{sentence_eng_Latn}} \n\ + Ta’izzi-Adeni Arabic: " +include: flores +task: flores_eng_Latn-acq_Arab_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-aeb_Arab.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-aeb_Arab.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9e8263fe6af0b65e5c935c7d56146b6940c6850b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-aeb_Arab.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-aeb_Arab +doc_to_target: sentence_aeb_Arab +doc_to_text: "As a Tunisian Arabic and English linguist, translate the following English\ + \ sentences to Tunisian Arabic \nEnglish: {{sentence_eng_Latn}} \nTunisian Arabic: " +include: flores +task: flores_eng_Latn-aeb_Arab_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-afr_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-afr_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..86421c268959192fae2dcbc19a1a4b935d6bff29 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-afr_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-afr_Latn +doc_to_target: sentence_afr_Latn +doc_to_text: "As a Afrikaans and English linguist, translate the following English\ + \ sentences to Afrikaans \nEnglish: {{sentence_eng_Latn}} \nAfrikaans: " +include: flores +task: flores_eng_Latn-afr_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-aka_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-aka_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3373390566a317a7438e216cea67b926d5dd20fa --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-aka_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-aka_Latn +doc_to_target: sentence_aka_Latn +doc_to_text: "As a Akan and English linguist, translate the following English sentences\ + \ to Akan \nEnglish: {{sentence_eng_Latn}} \nAkan: " +include: flores +task: flores_eng_Latn-aka_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-amh_Ethi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-amh_Ethi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ba3e0116586dfb106bc57103dc685ddd8856570d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-amh_Ethi.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-amh_Ethi +doc_to_target: sentence_amh_Ethi +doc_to_text: "As a Amharic and English linguist, translate the following English sentences\ + \ to Amharic \nEnglish: {{sentence_eng_Latn}} \nAmharic: " +include: flores +task: flores_eng_Latn-amh_Ethi_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-ary_Arab.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-ary_Arab.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c732756a2ea06e33114c117c630f9b3fccab32fe --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-ary_Arab.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-ary_Arab +doc_to_target: sentence_ary_Arab +doc_to_text: "As a Moroccan Arabic and English linguist, translate the following English\ + \ sentences to Moroccan Arabic \nEnglish: {{sentence_eng_Latn}} \nMoroccan Arabic: " +include: flores +task: flores_eng_Latn-ary_Arab_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-arz_Arab.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-arz_Arab.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f11bc38a2dc1c1cc565979a47b51b5aad9bc830e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-arz_Arab.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-arz_Arab +doc_to_target: sentence_arz_Arab +doc_to_text: "As a Egyptian Arabic and English linguist, translate the following English\ + \ sentences to Egyptian Arabic \nEnglish: {{sentence_eng_Latn}} \nEgyptian Arabic: " +include: flores +task: flores_eng_Latn-arz_Arab_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-bam_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-bam_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c762962832885fe21c75986f7ce006789217dbd4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-bam_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-bam_Latn +doc_to_target: sentence_bam_Latn +doc_to_text: "As a Bambara and English linguist, translate the following English sentences\ + \ to Bambara \nEnglish: {{sentence_eng_Latn}} \nBambara: " +include: flores +task: flores_eng_Latn-bam_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-ban_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-ban_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..601aecf5cebebdb6572fadf8f82d2963b9b87d5c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-ban_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-ban_Latn +doc_to_target: sentence_ban_Latn +doc_to_text: "As a Balinese and English linguist, translate the following English\ + \ sentences to Balinese \nEnglish: {{sentence_eng_Latn}} \nBalinese: " +include: flores +task: flores_eng_Latn-ban_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-bem_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-bem_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fadabdb9356f28fda88a68e69646a0fd60141e9c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-bem_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-bem_Latn +doc_to_target: sentence_bem_Latn +doc_to_text: "As a Bemba and English linguist, translate the following English sentences\ + \ to Bemba \nEnglish: {{sentence_eng_Latn}} \nBemba: " +include: flores +task: flores_eng_Latn-bem_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-cjk_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-cjk_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c522831373d25e05103918ba43f36183106cc509 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-cjk_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-cjk_Latn +doc_to_target: sentence_cjk_Latn +doc_to_text: "As a Chokwe and English linguist, translate the following English sentences\ + \ to Chokwe \nEnglish: {{sentence_eng_Latn}} \nChokwe: " +include: flores +task: flores_eng_Latn-cjk_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-dik_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-dik_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..acfeb83ad758573c632ca1b3e9e08f190b86fa30 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-dik_Latn.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: eng_Latn-dik_Latn +doc_to_target: sentence_dik_Latn +doc_to_text: "As a Southwestern Dinka and English linguist, translate the following\ + \ English sentences to Southwestern Dinka \nEnglish: {{sentence_eng_Latn}} \nSouthwestern\ + \ Dinka: " +include: flores +task: flores_eng_Latn-dik_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-dyu_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-dyu_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..796dc6d2f633c5baba22d2cce8592f0f01e3fe42 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-dyu_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-dyu_Latn +doc_to_target: sentence_dyu_Latn +doc_to_text: "As a Dyula and English linguist, translate the following English sentences\ + \ to Dyula \nEnglish: {{sentence_eng_Latn}} \nDyula: " +include: flores +task: flores_eng_Latn-dyu_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-ewe_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-ewe_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..31a07891820793360f26b2d093e98b5982816ea6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-ewe_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-ewe_Latn +doc_to_target: sentence_ewe_Latn +doc_to_text: "As a Ewe and English linguist, translate the following English sentences\ + \ to Ewe \nEnglish: {{sentence_eng_Latn}} \nEwe: " +include: flores +task: flores_eng_Latn-ewe_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-fon_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-fon_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6cdc7308d63891ed0fc65e779e394b71eafb3bb1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-fon_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-fon_Latn +doc_to_target: sentence_fon_Latn +doc_to_text: "As a Fon and English linguist, translate the following English sentences\ + \ to Fon \nEnglish: {{sentence_eng_Latn}} \nFon: " +include: flores +task: flores_eng_Latn-fon_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-fra_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-fra_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3896879db152bc1e583fd24f5823321d0f6eda4d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-fra_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-fra_Latn +doc_to_target: sentence_fra_Latn +doc_to_text: "As a French and English linguist, translate the following English sentences\ + \ to French \nEnglish: {{sentence_eng_Latn}} \nFrench: " +include: flores +task: flores_eng_Latn-fra_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-fuv_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-fuv_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6b63249be8c1e81e837e9a024dd19ecd822f748b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-fuv_Latn.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: eng_Latn-fuv_Latn +doc_to_target: sentence_fuv_Latn +doc_to_text: "As a Nigerian Fulfulde and English linguist, translate the following\ + \ English sentences to Nigerian Fulfulde \nEnglish: {{sentence_eng_Latn}} \nNigerian\ + \ Fulfulde: " +include: flores +task: flores_eng_Latn-fuv_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-gaz_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-gaz_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..95cde87c38c66448967d595f60709c2f908af5f0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-gaz_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-gaz_Latn +doc_to_target: sentence_gaz_Latn +doc_to_text: "As a Oromo and English linguist, translate the following English sentences\ + \ to Oromo \nEnglish: {{sentence_eng_Latn}} \nOromo: " +include: flores +task: flores_eng_Latn-gaz_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-hau_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-hau_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..eec82e34503bb64fcab1c90cf507499760f95a15 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-hau_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-hau_Latn +doc_to_target: sentence_hau_Latn +doc_to_text: "As a Hausa and English linguist, translate the following English sentences\ + \ to Hausa \nEnglish: {{sentence_eng_Latn}} \nHausa: " +include: flores +task: flores_eng_Latn-hau_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-ibo_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-ibo_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..838990b364097652e9ba4ed68726147e4424d05e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-ibo_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-ibo_Latn +doc_to_target: sentence_ibo_Latn +doc_to_text: "As a Igbo and English linguist, translate the following English sentences\ + \ to Igbo \nEnglish: {{sentence_eng_Latn}} \nIgbo: " +include: flores +task: flores_eng_Latn-ibo_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-kab_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-kab_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..16888ad8f28b64a7bb9715fdf8f193e18ce06072 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-kab_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-kab_Latn +doc_to_target: sentence_kab_Latn +doc_to_text: "As a Kabyle and English linguist, translate the following English sentences\ + \ to Kabyle \nEnglish: {{sentence_eng_Latn}} \nKabyle: " +include: flores +task: flores_eng_Latn-kab_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-kam_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-kam_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d48c52d16017b2e1241afd154539148d3f0d0ae4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-kam_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-kam_Latn +doc_to_target: sentence_kam_Latn +doc_to_text: "As a Kamba and English linguist, translate the following English sentences\ + \ to Kamba \nEnglish: {{sentence_eng_Latn}} \nKamba: " +include: flores +task: flores_eng_Latn-kam_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-kbp_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-kbp_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c992a28f7e168e3753e378631fa6ce716e7ee69e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-kbp_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-kbp_Latn +doc_to_target: sentence_kbp_Latn +doc_to_text: "As a Kabiyè and English linguist, translate the following English sentences\ + \ to Kabiyè \nEnglish: {{sentence_eng_Latn}} \nKabiyè: " +include: flores +task: flores_eng_Latn-kbp_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-kea_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-kea_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d8ce1b502edea3c9f00fe89dbf9dc382010e4bff --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-kea_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-kea_Latn +doc_to_target: sentence_kea_Latn +doc_to_text: "As a Kabuverdianu and English linguist, translate the following English\ + \ sentences to Kabuverdianu \nEnglish: {{sentence_eng_Latn}} \nKabuverdianu: " +include: flores +task: flores_eng_Latn-kea_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-kik_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-kik_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fc7975c2b23bb486ead2962f28064a5fcab6102f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-kik_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-kik_Latn +doc_to_target: sentence_kik_Latn +doc_to_text: "As a Kikuyu and English linguist, translate the following English sentences\ + \ to Kikuyu \nEnglish: {{sentence_eng_Latn}} \nKikuyu: " +include: flores +task: flores_eng_Latn-kik_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-kin_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-kin_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1e2b91d461378cb7c8ff098d237037eefdcacc03 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-kin_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-kin_Latn +doc_to_target: sentence_kin_Latn +doc_to_text: "As a Kinyarwanda and English linguist, translate the following English\ + \ sentences to Kinyarwanda \nEnglish: {{sentence_eng_Latn}} \nKinyarwanda: " +include: flores +task: flores_eng_Latn-kin_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-kmb_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-kmb_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..270f29b629e6f1f06da31ba154d977b0281fd63b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-kmb_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-kmb_Latn +doc_to_target: sentence_kmb_Latn +doc_to_text: "As a Kimbundu and English linguist, translate the following English\ + \ sentences to Kimbundu \nEnglish: {{sentence_eng_Latn}} \nKimbundu: " +include: flores +task: flores_eng_Latn-kmb_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-knc_Arab.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-knc_Arab.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bd2994d36152fc1dcb4a7a2561cc41982dd6fed1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-knc_Arab.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: eng_Latn-knc_Arab +doc_to_target: sentence_knc_Arab +doc_to_text: "As a Central Kanuri (Arabic script) and English linguist, translate\ + \ the following English sentences to Central Kanuri (Arabic script) \nEnglish: {{sentence_eng_Latn}}\ + \ \nCentral Kanuri (Arabic script): " +include: flores +task: flores_eng_Latn-knc_Arab_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-kon_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-kon_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ae9e1201808061f32c0e9d9260b8d2900f7bd7d4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-kon_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-kon_Latn +doc_to_target: sentence_kon_Latn +doc_to_text: "As a Kikongo and English linguist, translate the following English sentences\ + \ to Kikongo \nEnglish: {{sentence_eng_Latn}} \nKikongo: " +include: flores +task: flores_eng_Latn-kon_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-lin_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-lin_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0945c697c27b39ed91cff296dd162735f0629f4a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-lin_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-lin_Latn +doc_to_target: sentence_lin_Latn +doc_to_text: "As a Lingala and English linguist, translate the following English sentences\ + \ to Lingala \nEnglish: {{sentence_eng_Latn}} \nLingala: " +include: flores +task: flores_eng_Latn-lin_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-lua_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-lua_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ff92a2cf381a3a4c94a5543901be39a647d24eb7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-lua_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-lua_Latn +doc_to_target: sentence_lua_Latn +doc_to_text: "As a Luba-Kasai and English linguist, translate the following English\ + \ sentences to Luba-Kasai \nEnglish: {{sentence_eng_Latn}} \nLuba-Kasai: " +include: flores +task: flores_eng_Latn-lua_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-luo_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-luo_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..803ed75d8b732c81f859285f974cd216afb86784 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-luo_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-luo_Latn +doc_to_target: sentence_luo_Latn +doc_to_text: "As a Luo and English linguist, translate the following English sentences\ + \ to Luo \nEnglish: {{sentence_eng_Latn}} \nLuo: " +include: flores +task: flores_eng_Latn-luo_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-mos_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-mos_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0e959db1653eff6ca0054ec5032144a96c2c5713 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-mos_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-mos_Latn +doc_to_target: sentence_mos_Latn +doc_to_text: "As a Mossi and English linguist, translate the following English sentences\ + \ to Mossi \nEnglish: {{sentence_eng_Latn}} \nMossi: " +include: flores +task: flores_eng_Latn-mos_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-nya_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-nya_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9311e264e1f7e617a22c52e7ac969b1001f7c5e7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-nya_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-nya_Latn +doc_to_target: sentence_nya_Latn +doc_to_text: "As a Nyanja and English linguist, translate the following English sentences\ + \ to Nyanja \nEnglish: {{sentence_eng_Latn}} \nNyanja: " +include: flores +task: flores_eng_Latn-nya_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-sot_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-sot_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a45cf383057f37f504f799d7cb241ec61274fd83 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-sot_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-sot_Latn +doc_to_target: sentence_sot_Latn +doc_to_text: "As a Southern Sotho and English linguist, translate the following English\ + \ sentences to Southern Sotho \nEnglish: {{sentence_eng_Latn}} \nSouthern Sotho: " +include: flores +task: flores_eng_Latn-sot_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/README.md b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/README.md new file mode 100644 index 0000000000000000000000000000000000000000..641877cb7c01a5b19791b20c95a246753ddee75a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/README.md @@ -0,0 +1,23 @@ +# + +## Paper +Title: `INJONGO: A Multicultural Intent Detection and Slot-filling Dataset for 16 African Languages` + +Paper Link: https://arxiv.org/abs/2502.09814 + +## Abstract +>Slot-filling and intent detection are well-established tasks in Conversational AI. However, current large-scale benchmarks for these tasks often exclude evaluations of low-resource languages and rely on translations from English benchmarks, thereby predominantly reflecting Western-centric concepts. In this paper, we introduce Injongo -- a multicultural, open-source benchmark dataset for 16 African languages with utterances generated by native speakers across diverse domains, including banking, travel, home, and dining. Through extensive experiments, we benchmark the fine-tuning multilingual transformer models and the prompting large language models (LLMs), and show the advantage of leveraging African-cultural utterances over Western-centric utterances for improving cross-lingual transfer from the English language. Experimental results reveal that current LLMs struggle with the slot-filling task, with GPT-4o achieving an average performance of 26 F1-score. In contrast, intent detection performance is notably better, with an average accuracy of 70.6%, though it still falls behind the fine-tuning baselines. Compared to the English language, GPT-4o and fine-tuning baselines perform similarly on intent detection, achieving an accuracy of approximately 81%. Our findings suggest that the performance of LLMs is still behind for many low-resource African languages, and more work is needed to further improve their downstream performance. + +### Citation + +``` +@misc{yu2025injongomulticulturalintentdetection, + title={INJONGO: A Multicultural Intent Detection and Slot-filling Dataset for 16 African Languages}, + author={Hao Yu and Jesujoba O. Alabi and Andiswa Bukula and Jian Yun Zhuang and En-Shiun Annie Lee and Tadesse Kebede Guge and Israel Abebe Azime and Happy Buzaaba and Blessing Kudzaishe Sibanda and Godson K. Kalipe and Jonathan Mukiibi and Salomon Kabongo Kabenamualu and Mmasibidi Setaka and Lolwethu Ndolela and Nkiruka Odu and Rooweither Mabuya and Shamsuddeen Hassan Muhammad and Salomey Osei and Sokhar Samb and Juliet W. Murage and Dietrich Klakow and David Ifeoluwa Adelani}, + year={2025}, + eprint={2502.09814}, + archivePrefix={arXiv}, + primaryClass={cs.CL}, + url={https://arxiv.org/abs/2502.09814}, +} +``` diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/gen_utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/gen_utils.py new file mode 100644 index 0000000000000000000000000000000000000000..112041999df20d26a31becc30633720a16457b18 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/gen_utils.py @@ -0,0 +1,159 @@ +import argparse +import os + +import yaml + + +class FunctionTag: + def __init__(self, value): + self.value = value + + +def prompt_func(mode, lang, intent): + prompt_map = { + "prompt_1": "Given the text: '{{text}}', determine the correct intent from the following list: " + f"[{', '.join(intent)}]. Only output one intent from the list.", + "prompt_2": "Analyze the text: '{{text}}'. Choose the most appropriate intent from these options: " + f"[{', '.join(intent)}]. Respond with only the selected intent.", + "prompt_3": "You are a linguistic analyst trained to understand user intent. Based on the text: '{{text}}', " + f"choose the intent that best matches from this list: [{', '.join(intent)}]. Return only the intent.", + "prompt_4": f"You are a {lang} linguistic analyst trained to understand {lang} user intent. Based on the {lang}" + "text: '{{text}}', choose the intent that best matches from this list: " + f"[{', '.join(intent)}]. Return only the intent.", + "prompt_5": f"The following text is in {lang}: '{{{{text}}}}'. Given the list of intents: [{', '.join(intent)}], " + "identify the intent expressed in the text. Return only the identified intent.", + } + return prompt_map[mode] + + +def gen_lang_yamls(output_dir: str, overwrite: bool, mode: str) -> None: + """ + Generate a yaml file for each language. + + :param output_dir: The directory to output the files to. + :param overwrite: Whether to overwrite files if they already exist. + """ + err = [] + languages = { + "amh": "Amharic", + "ewe": "Ewe", + "hau": "Hausa", + "ibo": "Igbo", + "kin": "Kinyarwanda", + "lin": "Lingala", + "lug": "Luganda", + "orm": "Oromo", + "sna": "Shona", + "sot": "Sotho", + "swa": "Swahili", + "twi": "Twi", + "wol": "Wolof", + "xho": "Xhosa", + "yor": "Yoruba", + "zul": "Zulu", + "eng": "English", + } + + intents = [ + "alarm", + "balance", + "bill_balance", + "book_flight", + "book_hotel", + "calendar_update", + "cancel_reservation", + "car_rental", + "confirm_reservation", + "cook_time", + "exchange_rate", + "food_last", + "freeze_account", + "ingredients_list", + "interest_rate", + "international_visa", + "make_call", + "meal_suggestion", + "min_payment", + "pay_bill", + "pin_change", + "play_music", + "plug_type", + "recipe", + "restaurant_reservation", + "restaurant_reviews", + "restaurant_suggestion", + "share_location", + "shopping_list_update", + "spending_history", + "text", + "time", + "timezone", + "transactions", + "transfer", + "translate", + "travel_notification", + "travel_suggestion", + "update_playlist", + "weather", + ] + + for lang in languages.keys(): + try: + file_name = f"injongointent_{lang}.yaml" + task_name = f"injongointent_{lang}_{mode}" + yaml_template = "injongointent" + yaml_details = { + "include": yaml_template, + "task": task_name, + "dataset_name": lang, + "doc_to_text": prompt_func(mode, languages[lang], intents), + } + os.makedirs(f"{output_dir}/{mode}", exist_ok=True) + with open( + f"{output_dir}/{mode}/{file_name}", + "w" if overwrite else "x", + encoding="utf8", + ) as f: + f.write("# Generated by utils.py\n") + yaml.dump( + yaml_details, + f, + allow_unicode=True, + ) + except FileExistsError: + err.append(file_name) + + if len(err) > 0: + raise FileExistsError( + "Files were not created because they already exist (use --overwrite flag):" + f" {', '.join(err)}" + ) + + +def main() -> None: + """Parse CLI args and generate language-specific yaml files.""" + parser = argparse.ArgumentParser() + parser.add_argument( + "--overwrite", + default=True, + action="store_true", + help="Overwrite files if they already exist", + ) + parser.add_argument( + "--output-dir", + default="./", + help="Directory to write yaml files to", + ) + parser.add_argument( + "--mode", + default="prompt_3", + choices=["prompt_1", "prompt_2", "prompt_3", "prompt_4", "prompt_5"], + help="Prompt number", + ) + args = parser.parse_args() + + gen_lang_yamls(output_dir=args.output_dir, overwrite=args.overwrite, mode=args.mode) + + +if __name__ == "__main__": + main() diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/injongointent.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/injongointent.yaml new file mode 100644 index 0000000000000000000000000000000000000000..220f4c514f0afb8ec9105d54c90f834b4fd57780 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/injongointent.yaml @@ -0,0 +1,13 @@ +group: injongointent +task: + - injongointent_prompt_1 + - injongointent_prompt_2 + - injongointent_prompt_3 + - injongointent_prompt_4 + - injongointent_prompt_5 +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0b3a3ee270683d5cc57a6e6ce81a3fe971f6c04e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_amh.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: amh +doc_to_text: 'Given the text: ''{{text}}'', determine the correct intent from the + following list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Only + output one intent from the list.' +include: injongointent +task: injongointent_amh_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_eng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..240c37d5f1cd4197314c51532ab25ab2e915e2ad --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_eng.yaml @@ -0,0 +1,16 @@ +# Generated by utils.py +dataset_name: eng +validation_split: train +test_split: test +fewshot_split: train +doc_to_text: 'Given the text: ''{{text}}'', determine the correct intent from the + following list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Only + output one intent from the list.' +include: injongointent +task: injongointent_eng_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_ewe.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_ewe.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c08d8bb0c151a812b1cd6d5131e4a0d1664f8725 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_ewe.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: ewe +doc_to_text: 'Given the text: ''{{text}}'', determine the correct intent from the + following list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Only + output one intent from the list.' +include: injongointent +task: injongointent_ewe_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9e1338c72cda04bbbbca713f83dd6aa715f19014 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_hau.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: hau +doc_to_text: 'Given the text: ''{{text}}'', determine the correct intent from the + following list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Only + output one intent from the list.' +include: injongointent +task: injongointent_hau_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e4a956d23c414332d353f9c7d04ac3ce876831ed --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_ibo.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: ibo +doc_to_text: 'Given the text: ''{{text}}'', determine the correct intent from the + following list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Only + output one intent from the list.' +include: injongointent +task: injongointent_ibo_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f55d787a3f4952afcbb3f70d9432bfc7dbf0a84e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_kin.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: kin +doc_to_text: 'Given the text: ''{{text}}'', determine the correct intent from the + following list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Only + output one intent from the list.' +include: injongointent +task: injongointent_kin_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_orm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_orm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2b95a4e7b9cb3afe8c745cbd62a94c3fad6a5314 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_orm.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: orm +doc_to_text: 'Given the text: ''{{text}}'', determine the correct intent from the + following list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Only + output one intent from the list.' +include: injongointent +task: injongointent_orm_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_sna.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_sna.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6cbf0105abe8b3157df4c6898e3873fb25beba28 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_sna.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: sna +doc_to_text: 'Given the text: ''{{text}}'', determine the correct intent from the + following list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Only + output one intent from the list.' +include: injongointent +task: injongointent_sna_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_sot.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_sot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ad3b4497f8dbbb678048dc9dfa8aa8894dd241fe --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_sot.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: sot +doc_to_text: 'Given the text: ''{{text}}'', determine the correct intent from the + following list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Only + output one intent from the list.' +include: injongointent +task: injongointent_sot_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..73fc61c7e6ab11fbf71d8842818f020f147b5443 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_twi.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: twi +doc_to_text: 'Given the text: ''{{text}}'', determine the correct intent from the + following list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Only + output one intent from the list.' +include: injongointent +task: injongointent_twi_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..682e01c12972c8b9a98e53711b307ebbf62676fa --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_1/injongointent_yor.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: yor +doc_to_text: 'Given the text: ''{{text}}'', determine the correct intent from the + following list: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Only + output one intent from the list.' +include: injongointent +task: injongointent_yor_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_eng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cc03705d5868b6c67b9973937f4bba59f08bd8c7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/injongointent/prompt_2/injongointent_eng.yaml @@ -0,0 +1,16 @@ +# Generated by utils.py +dataset_name: eng +validation_split: train +test_split: test +fewshot_split: train +doc_to_text: 'Analyze the text: ''{{text}}''. Choose the most appropriate intent from + these options: [alarm, balance, bill_balance, book_flight, book_hotel, calendar_update, + cancel_reservation, car_rental, confirm_reservation, cook_time, exchange_rate, food_last, + freeze_account, ingredients_list, interest_rate, international_visa, make_call, + meal_suggestion, min_payment, pay_bill, pin_change, play_music, plug_type, recipe, + restaurant_reservation, restaurant_reviews, restaurant_suggestion, share_location, + shopping_list_update, spending_history, text, time, timezone, transactions, transfer, + translate, travel_notification, travel_suggestion, update_playlist, weather]. Respond + with only the selected intent.' +include: injongointent +task: injongointent_eng_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_pcm-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_pcm-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7ec0c66a0a5c7fadbc7b9ec715fe53596d4c2b51 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/african-english/mafand_pcm-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-pcm +include: mafand +task: mafand_pcm-en_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand new file mode 100644 index 0000000000000000000000000000000000000000..35548392e118f0625d9adae32849389d5239cd3e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand @@ -0,0 +1,28 @@ +tag: +- mafand_tasks +- mafand_eng-afr +- mafand_eng-afr_prompt_2 +- afrobench_MT_tasks +dataset_path: masakhane/mafand +dataset_kwargs: {trust_remote_code: True} +output_type: generate_until +validation_split: validation +fewshot_split: validation +test_split: test +doc_to_target: !function utils.get_target_reverse +doc_to_text: !function utils.create_reverse_prompt_2 +metric_list: + - metric: bleu + aggregation: bleu + higher_is_better: true + - metric: chrf + aggregation: chrf + higher_is_better: true +generation_kwargs: + until: + - "\n" + do_sample: false + temperature: 0.0 +repeats: 1 +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..568a845e403663c3d24e30383cab8eebe0d37151 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-ibo.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-ibo +include: mafand +task: mafand_en-ibo_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-nya.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-nya.yaml new file mode 100644 index 0000000000000000000000000000000000000000..90a728107edf37bddd1d4eb80bcc6ddfaa49572e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-nya.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-nya +include: mafand +task: mafand_en-nya_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-pcm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-pcm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..73229c4f18ac24014cf15f161454910a921e1d02 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-pcm.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-pcm +include: mafand +task: mafand_en-pcm_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..21d9fc0e71589cb2e1c832a6b82e6d8fb5288b89 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-swa.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-swa +include: mafand +task: mafand_en-swa_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-tsn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-tsn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b3dd43626d00774a8e9bd6e432322e4d608887ba --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-tsn.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-tsn +include: mafand +task: mafand_en-tsn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..89c070c7c5e9f6d53fde695f156408f37038242d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_en-yor.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-yor +include: mafand +task: mafand_en-yor_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_fr-bam.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_fr-bam.yaml new file mode 100644 index 0000000000000000000000000000000000000000..15c6e981d1cb878e26a08755f5bf1fd7629f2525 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_fr-bam.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fr-bam +include: mafand +task: mafand_fr-bam_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_fr-bbj.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_fr-bbj.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4f5101a752865bf641a150be2aff3b449239d3f8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_fr-bbj.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fr-bbj +include: mafand +task: mafand_fr-bbj_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_fr-ewe.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_fr-ewe.yaml new file mode 100644 index 0000000000000000000000000000000000000000..29d4214cf433e56e8f2d479299cf413e2d211d34 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_fr-ewe.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fr-ewe +include: mafand +task: mafand_fr-ewe_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_fr-wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_fr-wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3740ca9b73557e472f867b7b5c33131a441e18fa --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/mafand_fr-wol.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fr-wol +include: mafand +task: mafand_fr-wol_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..0df3a329824d44fa94eb830ae943fa30dd32bab7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_2/english-african/utils.py @@ -0,0 +1,121 @@ +languages = { + "amh": "Amharic", + "bam": "Bambara", + "bbj": "Gbomala", + "ewe": "Ewe", + "fon": "Fon", + "hau": "Hausa", + "ibo": "Igbo", + "kin": "Kinyarwanda", + "lug": "Luganda", + "luo": "Luo", + "mos": "Mossi", + "nya": "Chichewa", + "pcm": "Nigerian Pidgin", + "sna": "Shona", + "swa": "Swahili", + "tsn": "Setswana", + "twi": "Twi", + "wol": "Wolof", + "xho": "Xhosa", + "yor": "Yoruba", + "zul": "Zulu", +} + + +def get_target(doc): + target = ( + doc["translation"]["en"] + if "en" in doc["translation"].keys() + else doc["translation"]["fr"] + ) + return target + + +def get_target_reverse(doc): + target_key = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + target = doc["translation"][target_key] + return target + + +def create_text_prompt_1(doc): + source_key = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + source_sentence = doc["translation"][source_key] + source_lang = "English" if "en" in doc["translation"].keys() else "French" + prompt = ( + "You are an advanced Translator, a specialized assistant designed to translate documents from " + f"{languages[source_key]} into {source_lang}. \nYour main goal is to ensure translations are grammatically " + f"correct and human-oriented. \n{languages[source_key]}: {source_sentence} \n{source_lang}: " + ) + return prompt + + +def create_reverse_prompt_1(doc): + target_lang = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + source_key = "en" if "en" in doc["translation"].keys() else "fr" + source_lang = "English" if source_key == "en" else "French" + source_sentence = doc["translation"][source_key] + prompt = ( + "You are an advanced Translator, a specialized assistant designed to translate documents from " + f"{source_lang} into {languages[target_lang]}. \nYour main goal is to ensure translations are " + f"grammatically correct and human-oriented. \n{source_lang}: {source_sentence} \n{languages[target_lang]}: " + ) + return prompt + + +def create_text_prompt_2(doc): + source_key = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + source_sentence = doc["translation"][source_key] + source_lang = "English" if "en" in doc["translation"].keys() else "French" + prompt = ( + f"{languages[source_key]} sentence: {source_sentence} \n{source_lang} sentence: ", + ) + return prompt + + +def create_reverse_prompt_2(doc): + target_lang = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + source_key = "en" if "en" in doc["translation"].keys() else "fr" + source_lang = "English" if source_key == "en" else "French" + source_sentence = doc["translation"][source_key] + prompt = ( + f"{source_lang} sentence: {source_sentence} \n{languages[target_lang]} sentence: \n", + ) + return prompt + + +def create_text_prompt_3(doc): + source_key = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + source_sentence = doc["translation"][source_key] + source_lang = "English" if "en" in doc["translation"].keys() else "French" + prompt = ( + f"You are a translation expert. Translate the following {languages[source_key]} sentences " + f"to {source_lang}. \n{languages[source_key]} sentence: {source_sentence}\n{source_lang} sentence: " + ) + return prompt + + +def create_reverse_prompt_3(doc): + target_lang = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + source_key = "en" if "en" in doc["translation"].keys() else "fr" + source_lang = "English" if source_key == "en" else "French" + source_sentence = doc["translation"][source_key] + prompt = ( + f"You are a translation expert. Translate the following {source_lang} sentence into {languages[target_lang]}\n" + f"{source_lang} sentence: {source_sentence}\n{languages[target_lang]} sentence: " + ) + return prompt diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_amh-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_amh-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..856318b5ad1b19fde25dd12ee3a2fc712b053b1a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_amh-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-amh +include: mafand +task: mafand_amh-en_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_bam-fr.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_bam-fr.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bed4252375b82e34d27c96adf368217957e33063 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_bam-fr.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fr-bam +include: mafand +task: mafand_bam-fr_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_ewe-fr.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_ewe-fr.yaml new file mode 100644 index 0000000000000000000000000000000000000000..39a345cb6b6b86684e67978bd52c0e3071302a04 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_ewe-fr.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fr-ewe +include: mafand +task: mafand_ewe-fr_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_fon-fr.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_fon-fr.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1b464bb913f4888d825a2ab2aa2d566ef5d422d4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_fon-fr.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fr-fon +include: mafand +task: mafand_fon-fr_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_hau-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_hau-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9c0b0f15fe011d52ff4bbd167e23453a621e2928 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_hau-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-hau +include: mafand +task: mafand_hau-en_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_ibo-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_ibo-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2f78f55a3b27a4461efd85e741503ead06e4bcce --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_ibo-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-ibo +include: mafand +task: mafand_ibo-en_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_lug-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_lug-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ad19b3c85e6aae28a01c7e3476f8492e804c6d83 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_lug-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-lug +include: mafand +task: mafand_lug-en_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_luo-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_luo-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a3a367493d5da3804d2eda16eba676886765939e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_luo-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-luo +include: mafand +task: mafand_luo-en_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_mos-fr.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_mos-fr.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4ea419312925c06760ddc6f6efbf343594f2b932 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_mos-fr.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fr-mos +include: mafand +task: mafand_mos-fr_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_pcm-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_pcm-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..95ad3380334ced0554dc21ab4e6a07f23352f51d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_pcm-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-pcm +include: mafand +task: mafand_pcm-en_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_sna-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_sna-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2d86ccc3ad64220d354d4e9e230ee585e556b32c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_sna-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-sna +include: mafand +task: mafand_sna-en_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_swa-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_swa-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3c70f2e3e77f4ee94edd757311e8b37816911104 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_swa-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-swa +include: mafand +task: mafand_swa-en_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_twi-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_twi-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a37d2395a4721c6553a878f6693221cbac6a22a6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_twi-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-twi +include: mafand +task: mafand_twi-en_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_wol-fr.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_wol-fr.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ed778cbe6690aafa7abd33fad50ef41fc25dbaea --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_wol-fr.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fr-wol +include: mafand +task: mafand_wol-fr_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_xho-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_xho-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..93e9e2fee5b39b755b9e060697e509a5f63d58e0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_xho-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-xho +include: mafand +task: mafand_xho-en_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_yor-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_yor-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..78301f7e658cd489eb6433f3ac5a10a0f0cde49b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_yor-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-yor +include: mafand +task: mafand_yor-en_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_zul-en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_zul-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..06177d14ec829fe74a12030d88e06a5fee7bc9a0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/mafand_zul-en.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-zul +include: mafand +task: mafand_zul-en_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..0df3a329824d44fa94eb830ae943fa30dd32bab7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/african-english/utils.py @@ -0,0 +1,121 @@ +languages = { + "amh": "Amharic", + "bam": "Bambara", + "bbj": "Gbomala", + "ewe": "Ewe", + "fon": "Fon", + "hau": "Hausa", + "ibo": "Igbo", + "kin": "Kinyarwanda", + "lug": "Luganda", + "luo": "Luo", + "mos": "Mossi", + "nya": "Chichewa", + "pcm": "Nigerian Pidgin", + "sna": "Shona", + "swa": "Swahili", + "tsn": "Setswana", + "twi": "Twi", + "wol": "Wolof", + "xho": "Xhosa", + "yor": "Yoruba", + "zul": "Zulu", +} + + +def get_target(doc): + target = ( + doc["translation"]["en"] + if "en" in doc["translation"].keys() + else doc["translation"]["fr"] + ) + return target + + +def get_target_reverse(doc): + target_key = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + target = doc["translation"][target_key] + return target + + +def create_text_prompt_1(doc): + source_key = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + source_sentence = doc["translation"][source_key] + source_lang = "English" if "en" in doc["translation"].keys() else "French" + prompt = ( + "You are an advanced Translator, a specialized assistant designed to translate documents from " + f"{languages[source_key]} into {source_lang}. \nYour main goal is to ensure translations are grammatically " + f"correct and human-oriented. \n{languages[source_key]}: {source_sentence} \n{source_lang}: " + ) + return prompt + + +def create_reverse_prompt_1(doc): + target_lang = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + source_key = "en" if "en" in doc["translation"].keys() else "fr" + source_lang = "English" if source_key == "en" else "French" + source_sentence = doc["translation"][source_key] + prompt = ( + "You are an advanced Translator, a specialized assistant designed to translate documents from " + f"{source_lang} into {languages[target_lang]}. \nYour main goal is to ensure translations are " + f"grammatically correct and human-oriented. \n{source_lang}: {source_sentence} \n{languages[target_lang]}: " + ) + return prompt + + +def create_text_prompt_2(doc): + source_key = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + source_sentence = doc["translation"][source_key] + source_lang = "English" if "en" in doc["translation"].keys() else "French" + prompt = ( + f"{languages[source_key]} sentence: {source_sentence} \n{source_lang} sentence: ", + ) + return prompt + + +def create_reverse_prompt_2(doc): + target_lang = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + source_key = "en" if "en" in doc["translation"].keys() else "fr" + source_lang = "English" if source_key == "en" else "French" + source_sentence = doc["translation"][source_key] + prompt = ( + f"{source_lang} sentence: {source_sentence} \n{languages[target_lang]} sentence: \n", + ) + return prompt + + +def create_text_prompt_3(doc): + source_key = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + source_sentence = doc["translation"][source_key] + source_lang = "English" if "en" in doc["translation"].keys() else "French" + prompt = ( + f"You are a translation expert. Translate the following {languages[source_key]} sentences " + f"to {source_lang}. \n{languages[source_key]} sentence: {source_sentence}\n{source_lang} sentence: " + ) + return prompt + + +def create_reverse_prompt_3(doc): + target_lang = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + source_key = "en" if "en" in doc["translation"].keys() else "fr" + source_lang = "English" if source_key == "en" else "French" + source_sentence = doc["translation"][source_key] + prompt = ( + f"You are a translation expert. Translate the following {source_lang} sentence into {languages[target_lang]}\n" + f"{source_lang} sentence: {source_sentence}\n{languages[target_lang]} sentence: " + ) + return prompt diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand new file mode 100644 index 0000000000000000000000000000000000000000..9a59654e4feba09f57277b539633c2b0efda291e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand @@ -0,0 +1,28 @@ +tag: +- mafand_tasks +- mafand_eng-afr +- mafand_eng-afr_prompt_3 +- afrobench_MT_tasks +dataset_path: masakhane/mafand +dataset_kwargs: {trust_remote_code: True} +output_type: generate_until +validation_split: validation +fewshot_split: validation +test_split: test +doc_to_target: !function utils.get_target_reverse +doc_to_text: !function utils.create_reverse_prompt_3 +metric_list: + - metric: bleu + aggregation: bleu + higher_is_better: true + - metric: chrf + aggregation: chrf + higher_is_better: true +generation_kwargs: + until: + - "\n" + do_sample: false + temperature: 0.0 +repeats: 1 +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..10872430688882c53e5f2dba5aea70d1194c6020 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-amh.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-amh +include: mafand +task: mafand_en-amh_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..72df05e551444a9e82de940171c646797f1c18d6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-ibo.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-ibo +include: mafand +task: mafand_en-ibo_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..44c48678e89c3dd7e72d310fd050b5a1d3bc6092 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-kin.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-kin +include: mafand +task: mafand_en-kin_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2beae91b569db6fc361da97e0879e854af005e4d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-lug.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-lug +include: mafand +task: mafand_en-lug_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-luo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-luo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b4c1aa8becf693cb16d4d1be2820707e521a1052 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-luo.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-luo +include: mafand +task: mafand_en-luo_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-nya.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-nya.yaml new file mode 100644 index 0000000000000000000000000000000000000000..eee7af0ce8ed105fa08769793ad816c2f4d17318 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-nya.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-nya +include: mafand +task: mafand_en-nya_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-pcm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-pcm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6e60642562403f3b86d2e13fb3ea48368dc84883 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-pcm.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-pcm +include: mafand +task: mafand_en-pcm_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-sna.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-sna.yaml new file mode 100644 index 0000000000000000000000000000000000000000..82abd862535426ee078cd45f057c792643062981 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-sna.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-sna +include: mafand +task: mafand_en-sna_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8a7135ff6921556928338e52e3dbcc8afa731025 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-swa.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-swa +include: mafand +task: mafand_en-swa_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-tsn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-tsn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b976b5fd24007928854afe3b533ffd8b66ea0851 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-tsn.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-tsn +include: mafand +task: mafand_en-tsn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..53345a2668eccb93b11568c22f6d218598f20ba2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-twi.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-twi +include: mafand +task: mafand_en-twi_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4eba7f6994b577b4769b6b74a2848ecc0b3b0fb0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-xho.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-xho +include: mafand +task: mafand_en-xho_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9b20e9f920deb657630ef0aa5aa6a443934f0519 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-yor.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-yor +include: mafand +task: mafand_en-yor_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cb5280b995b0754ae6a4f6cd33bfc82350d0e8cb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_en-zul.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: en-zul +include: mafand +task: mafand_en-zul_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_fr-bam.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_fr-bam.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3e94be00dea4c013238a543cc3ceeb2982b92ce4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_fr-bam.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fr-bam +include: mafand +task: mafand_fr-bam_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_fr-ewe.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_fr-ewe.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7139c81fd0c991bf9ee6a24013a34a8b0b700efc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_fr-ewe.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fr-ewe +include: mafand +task: mafand_fr-ewe_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_fr-fon.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_fr-fon.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b42292ce56e6abc7127e9988e5619e0eee3d56ab --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_fr-fon.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fr-fon +include: mafand +task: mafand_fr-fon_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_fr-mos.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_fr-mos.yaml new file mode 100644 index 0000000000000000000000000000000000000000..044047c346abcb945443b2b500eef7bd32f2caad --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_fr-mos.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fr-mos +include: mafand +task: mafand_fr-mos_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_fr-wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_fr-wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9fc1bca3b94f64489d26632c70c022558e7793b3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/mafand_fr-wol.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fr-wol +include: mafand +task: mafand_fr-wol_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..0df3a329824d44fa94eb830ae943fa30dd32bab7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/mafand/prompt_3/english-african/utils.py @@ -0,0 +1,121 @@ +languages = { + "amh": "Amharic", + "bam": "Bambara", + "bbj": "Gbomala", + "ewe": "Ewe", + "fon": "Fon", + "hau": "Hausa", + "ibo": "Igbo", + "kin": "Kinyarwanda", + "lug": "Luganda", + "luo": "Luo", + "mos": "Mossi", + "nya": "Chichewa", + "pcm": "Nigerian Pidgin", + "sna": "Shona", + "swa": "Swahili", + "tsn": "Setswana", + "twi": "Twi", + "wol": "Wolof", + "xho": "Xhosa", + "yor": "Yoruba", + "zul": "Zulu", +} + + +def get_target(doc): + target = ( + doc["translation"]["en"] + if "en" in doc["translation"].keys() + else doc["translation"]["fr"] + ) + return target + + +def get_target_reverse(doc): + target_key = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + target = doc["translation"][target_key] + return target + + +def create_text_prompt_1(doc): + source_key = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + source_sentence = doc["translation"][source_key] + source_lang = "English" if "en" in doc["translation"].keys() else "French" + prompt = ( + "You are an advanced Translator, a specialized assistant designed to translate documents from " + f"{languages[source_key]} into {source_lang}. \nYour main goal is to ensure translations are grammatically " + f"correct and human-oriented. \n{languages[source_key]}: {source_sentence} \n{source_lang}: " + ) + return prompt + + +def create_reverse_prompt_1(doc): + target_lang = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + source_key = "en" if "en" in doc["translation"].keys() else "fr" + source_lang = "English" if source_key == "en" else "French" + source_sentence = doc["translation"][source_key] + prompt = ( + "You are an advanced Translator, a specialized assistant designed to translate documents from " + f"{source_lang} into {languages[target_lang]}. \nYour main goal is to ensure translations are " + f"grammatically correct and human-oriented. \n{source_lang}: {source_sentence} \n{languages[target_lang]}: " + ) + return prompt + + +def create_text_prompt_2(doc): + source_key = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + source_sentence = doc["translation"][source_key] + source_lang = "English" if "en" in doc["translation"].keys() else "French" + prompt = ( + f"{languages[source_key]} sentence: {source_sentence} \n{source_lang} sentence: ", + ) + return prompt + + +def create_reverse_prompt_2(doc): + target_lang = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + source_key = "en" if "en" in doc["translation"].keys() else "fr" + source_lang = "English" if source_key == "en" else "French" + source_sentence = doc["translation"][source_key] + prompt = ( + f"{source_lang} sentence: {source_sentence} \n{languages[target_lang]} sentence: \n", + ) + return prompt + + +def create_text_prompt_3(doc): + source_key = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + source_sentence = doc["translation"][source_key] + source_lang = "English" if "en" in doc["translation"].keys() else "French" + prompt = ( + f"You are a translation expert. Translate the following {languages[source_key]} sentences " + f"to {source_lang}. \n{languages[source_key]} sentence: {source_sentence}\n{source_lang} sentence: " + ) + return prompt + + +def create_reverse_prompt_3(doc): + target_lang = [key for key in doc["translation"].keys() if key not in ["en", "fr"]][ + 0 + ] + source_key = "en" if "en" in doc["translation"].keys() else "fr" + source_lang = "English" if source_key == "en" else "French" + source_sentence = doc["translation"][source_key] + prompt = ( + f"You are a translation expert. Translate the following {source_lang} sentence into {languages[target_lang]}\n" + f"{source_lang} sentence: {source_sentence}\n{languages[target_lang]} sentence: " + ) + return prompt diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/README.md b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/README.md new file mode 100644 index 0000000000000000000000000000000000000000..ca96648e07bb2c54fbf0c79d968b2ef4cb6aba75 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/README.md @@ -0,0 +1,76 @@ +# + +## Paper +Title: `MasakhaNER 2.0: Africa-centric Transfer Learning for Named Entity Recognition` + +Paper Link: https://aclanthology.org/2022.emnlp-main.298/ + +## Abstract +>African languages are spoken by over a billion people, but they are under-represented in NLP research and development. Multiple challenges exist, including the limited availability of annotated training and evaluation datasets as well as the lack of understanding of which settings, languages, and recently proposed methods like cross-lingual transfer will be effective. In this paper, we aim to move towards solutions for these challenges, focusing on the task of named entity recognition (NER). We present the creation of the largest to-date human-annotated NER dataset for 20 African languages. We study the behaviour of state-of-the-art cross-lingual transfer methods in an Africa-centric setting, empirically demonstrating that the choice of source transfer language significantly affects performance. While much previous work defaults to using English as the source language, our results show that choosing the best transfer language improves zero-shot F1 scores by an average of 14% over 20 languages as compared to using English. + +HomePage: https://github.com/masakhane-io/masakhane-ner + +### Citation + +``` +@inproceedings{adelani-etal-2022-masakhaner, + title = "{M}asakha{NER} 2.0: {A}frica-centric Transfer Learning for Named Entity Recognition", + author = "Adelani, David Ifeoluwa and + Neubig, Graham and + Ruder, Sebastian and + Rijhwani, Shruti and + Beukman, Michael and + Palen-Michel, Chester and + Lignos, Constantine and + Alabi, Jesujoba O. and + Muhammad, Shamsuddeen H. and + Nabende, Peter and + Dione, Cheikh M. Bamba and + Bukula, Andiswa and + Mabuya, Rooweither and + Dossou, Bonaventure F. P. and + Sibanda, Blessing and + Buzaaba, Happy and + Mukiibi, Jonathan and + Kalipe, Godson and + Mbaye, Derguene and + Taylor, Amelia and + Kabore, Fatoumata and + Emezue, Chris Chinenye and + Aremu, Anuoluwapo and + Ogayo, Perez and + Gitau, Catherine and + Munkoh-Buabeng, Edwin and + Memdjokam Koagne, Victoire and + Tapo, Allahsera Auguste and + Macucwa, Tebogo and + Marivate, Vukosi and + Mboning, Elvis and + Gwadabe, Tajuddeen and + Adewumi, Tosin and + Ahia, Orevaoghene and + Nakatumba-Nabende, Joyce and + Mokono, Neo L. and + Ezeani, Ignatius and + Chukwuneke, Chiamaka and + Adeyemi, Mofetoluwa and + Hacheme, Gilles Q. and + Abdulmumim, Idris and + Ogundepo, Odunayo and + Yousuf, Oreen and + Moteu Ngoli, Tatiana and + Klakow, Dietrich", + editor = "Goldberg, Yoav and + Kozareva, Zornitsa and + Zhang, Yue", + booktitle = "Proceedings of the 2022 Conference on Empirical Methods in Natural Language Processing", + month = dec, + year = "2022", + address = "Abu Dhabi, United Arab Emirates", + publisher = "Association for Computational Linguistics", + url = "https://aclanthology.org/2022.emnlp-main.298/", + doi = "10.18653/v1/2022.emnlp-main.298", + pages = "4488--4508", + abstract = "African languages are spoken by over a billion people, but they are under-represented in NLP research and development. Multiple challenges exist, including the limited availability of annotated training and evaluation datasets as well as the lack of understanding of which settings, languages, and recently proposed methods like cross-lingual transfer will be effective. In this paper, we aim to move towards solutions for these challenges, focusing on the task of named entity recognition (NER). We present the creation of the largest to-date human-annotated NER dataset for 20 African languages. We study the behaviour of state-of-the-art cross-lingual transfer methods in an Africa-centric setting, empirically demonstrating that the choice of source transfer language significantly affects performance. While much previous work defaults to using English as the source language, our results show that choosing the best transfer language improves zero-shot F1 scores by an average of 14{\%} over 20 languages as compared to using English." +} +``` diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/gen_utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/gen_utils.py new file mode 100644 index 0000000000000000000000000000000000000000..4d1012021f567ab02ccdd6259788e00ea1f759e9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/gen_utils.py @@ -0,0 +1,138 @@ +import argparse +import os + +import yaml + + +class FunctionTag: + def __init__(self, value): + self.value = value + + +def prompt_func(mode, lang): + prompt_map = { + "prompt_1": "Named entities refers to names of location, organisation and personal name. \n For example, " + "'David is an employee of Amazon and he is visiting New York next week to see Esther' will be \n" + "PERSON: David $ ORGANIZATION: Amazon $ LOCATION: New York $ PERSON: Esther \n\n" + "Ensure the output strictly follows the format: label: entity $ label: entity, with each unique " + "entity on a separate label line, avoiding grouped entities (e.g., avoid LOC: entity, entity) or " + "irrelevant entries like none. \n\nText: {{text}} \n" + "Return only the output", + "prompt_2": "You are working as a named entity recognition expert and your task is to label a given text " + "with named entity labels. Your task is to identify and label any named entities present in the " + "text. The named entity labels that you will be using are PER (person), LOC (location), " + "ORG (organization) and DATE (date). Label multi-word entities as a single named entity. " + "For words which are not part of any named entity, do not return any value for it. \n" + "Ensure the output strictly follows the format: label: entity $$ label: entity, with each unique " + "entity on a separate label line, avoiding grouped entities (e.g., avoid LOC: entity, entity) or " + "irrelevant entries like none. Return only the output \n\nText: {{text}}", + "prompt_3": f"You are a Named Entity Recognition expert in {lang} language. \nExtract all named entities from " + f"the following {lang} text and categorize them into PERSON, LOCATION, ORGANIZATION, or DATE. " + f"Ensure the output strictly follows the format: label: entity $$ label: entity, with each unique " + "entity on a separate label line, avoiding grouped entities (e.g., avoid LOC: entity, entity) or " + "irrelevant entries like none. Return only the output \n\nText: {{text}}", + "prompt_4": f"As a {lang} linguist, label all named entities in the {lang} text below with the categories: " + "PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the output strictly follows the format: label: " + "entity $$ label: entity, with each unique entity on a separate label line, avoiding grouped " + "entities (e.g., avoid LOC: entity, entity) or irrelevant entries like none. Return only the " + "output. \n\nText: {{text}}", + "prompt_5": "Provide a concise list of named entities in the text below. Use the following labels: " + "PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the output strictly follows the format: label: " + "entity $$ label: entity, with each unique entity on a separate label line, avoiding grouped " + "entities (e.g., avoid LOC: entity, entity) or irrelevant entries like none. Return only the " + "output. \n\nText: {{text}}", + } + return prompt_map[mode] + + +def gen_lang_yamls(output_dir: str, overwrite: bool, mode: str) -> None: + """ + Generate a yaml file for each language. + + :param output_dir: The directory to output the files to. + :param overwrite: Whether to overwrite files if they already exist. + """ + err = [] + languages = { + "am": "Amharic", + "bm": "Bambara", + "bbj": "Ghomala", + "ee": "Ewe", + "ha": "Hausa", + "ig": "Igbo", + "rw": "Kinyarwanda", + "lg": "Luganda", + "luo": "Luo", + "mos": "Mossi", + "ny": "Chichewa", + "pcm": "Nigerian Pidgin", + "sn": "chiShona", + "sw": "Kiswahili", + "tn": "Setswana", + "tw": "Twi", + "wo": "Wolof", + "xh": "isiXhosa", + "yo": "Yoruba", + "zu": "isiZulu", + } + + for lang in languages.keys(): + try: + file_name = f"masakhaner_{lang}.yaml" + task_name = f"masakhaner_{lang}_{mode}" + yaml_template = "masakhaner" + yaml_details = { + "include": yaml_template, + "task": task_name, + "dataset_name": lang, + "doc_to_text": prompt_func(mode, languages[lang]), + } + os.makedirs(f"{output_dir}/{mode}", exist_ok=True) + with open( + f"{output_dir}/{mode}/{file_name}", + "w" if overwrite else "x", + encoding="utf8", + ) as f: + f.write("# Generated by utils.py\n") + yaml.dump( + yaml_details, + f, + allow_unicode=True, + ) + except FileExistsError: + err.append(file_name) + + if len(err) > 0: + raise FileExistsError( + "Files were not created because they already exist (use --overwrite flag):" + f" {', '.join(err)}" + ) + + +def main() -> None: + """Parse CLI args and generate language-specific yaml files.""" + parser = argparse.ArgumentParser() + parser.add_argument( + "--overwrite", + default=True, + action="store_true", + help="Overwrite files if they already exist", + ) + parser.add_argument( + "--output-dir", + default="./", + help="Directory to write yaml files to", + ) + parser.add_argument( + "--mode", + default="prompt_1", + choices=["prompt_1", "prompt_2", "prompt_3", "prompt_4", "prompt_5"], + help="Prompt number", + ) + args = parser.parse_args() + + gen_lang_yamls(output_dir=args.output_dir, overwrite=args.overwrite, mode=args.mode) + + +if __name__ == "__main__": + main() diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/masakhaner.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/masakhaner.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b0d374e80c43cc8831d167887e386f3500773b48 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/masakhaner.yaml @@ -0,0 +1,13 @@ +group: masakhaner +task: + - masakhaner_prompt_1 + - masakhaner_prompt_2 + - masakhaner_prompt_3 + - masakhaner_prompt_4 + - masakhaner_prompt_5 +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner new file mode 100644 index 0000000000000000000000000000000000000000..706eb36644524b2aaa10b686ce120048e6322390 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner @@ -0,0 +1,26 @@ +tag: +- masakhaner_tasks +- masakhaner_prompt_1 +dataset_path: masakhane/masakhaner-x +dataset_name: null +dataset_kwargs: {trust_remote_code: True} +output_type: generate_until +generation_kwargs: + do_sample: false + until: + - + - <|im_end|> +validation_split: validation +test_split: test +fewshot_split: train +doc_to_target: target +filter_list: + - name: flexible-extract + filter: + - function: format_span +metric_list: + - metric: f1 + aggregation: !function utils.span_f1_agg + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_am.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_am.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b2128752f754eb8c46f7608c89dbefd7a3800480 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_am.yaml @@ -0,0 +1,11 @@ +# Generated by utils.py +dataset_name: am +doc_to_text: "Named entities refers to names of location, organisation and personal\ + \ name. \n For example, 'David is an employee of Amazon and he is visiting New York\ + \ next week to see Esther' will be \nPERSON: David $ ORGANIZATION: Amazon $ LOCATION:\ + \ New York $ PERSON: Esther \n\nEnsure the output strictly follows the format: label:\ + \ entity $ label: entity, with each unique entity on a separate label line, avoiding\ + \ grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries like\ + \ none. \n\nText: {{text}} \nReturn only the output" +include: masakhaner +task: masakhaner_am_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_bbj.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_bbj.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3f3a72bdc0257aedee8787af75a70c14560bfc53 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_bbj.yaml @@ -0,0 +1,11 @@ +# Generated by utils.py +dataset_name: bbj +doc_to_text: "Named entities refers to names of location, organisation and personal\ + \ name. \n For example, 'David is an employee of Amazon and he is visiting New York\ + \ next week to see Esther' will be \nPERSON: David $ ORGANIZATION: Amazon $ LOCATION:\ + \ New York $ PERSON: Esther \n\nEnsure the output strictly follows the format: label:\ + \ entity $ label: entity, with each unique entity on a separate label line, avoiding\ + \ grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries like\ + \ none. \n\nText: {{text}} \nReturn only the output" +include: masakhaner +task: masakhaner_bbj_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_bm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_bm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c38bdee947c2d34a4a7797eae1251fc476be0f53 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_bm.yaml @@ -0,0 +1,11 @@ +# Generated by utils.py +dataset_name: bm +doc_to_text: "Named entities refers to names of location, organisation and personal\ + \ name. \n For example, 'David is an employee of Amazon and he is visiting New York\ + \ next week to see Esther' will be \nPERSON: David $ ORGANIZATION: Amazon $ LOCATION:\ + \ New York $ PERSON: Esther \n\nEnsure the output strictly follows the format: label:\ + \ entity $ label: entity, with each unique entity on a separate label line, avoiding\ + \ grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries like\ + \ none. \n\nText: {{text}} \nReturn only the output" +include: masakhaner +task: masakhaner_bm_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_ee.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_ee.yaml new file mode 100644 index 0000000000000000000000000000000000000000..97903908e30cfe376626f7c668d5d9862593d73e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_ee.yaml @@ -0,0 +1,11 @@ +# Generated by utils.py +dataset_name: ee +doc_to_text: "Named entities refers to names of location, organisation and personal\ + \ name. \n For example, 'David is an employee of Amazon and he is visiting New York\ + \ next week to see Esther' will be \nPERSON: David $ ORGANIZATION: Amazon $ LOCATION:\ + \ New York $ PERSON: Esther \n\nEnsure the output strictly follows the format: label:\ + \ entity $ label: entity, with each unique entity on a separate label line, avoiding\ + \ grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries like\ + \ none. \n\nText: {{text}} \nReturn only the output" +include: masakhaner +task: masakhaner_ee_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_ha.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_ha.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ad11710407bda79b9561cd05725c21eb945ca292 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_ha.yaml @@ -0,0 +1,11 @@ +# Generated by utils.py +dataset_name: ha +doc_to_text: "Named entities refers to names of location, organisation and personal\ + \ name. \n For example, 'David is an employee of Amazon and he is visiting New York\ + \ next week to see Esther' will be \nPERSON: David $ ORGANIZATION: Amazon $ LOCATION:\ + \ New York $ PERSON: Esther \n\nEnsure the output strictly follows the format: label:\ + \ entity $ label: entity, with each unique entity on a separate label line, avoiding\ + \ grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries like\ + \ none. \n\nText: {{text}} \nReturn only the output" +include: masakhaner +task: masakhaner_ha_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_ig.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_ig.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0f06c0655595ae81b985266e4e629b080d49130c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_ig.yaml @@ -0,0 +1,11 @@ +# Generated by utils.py +dataset_name: ig +doc_to_text: "Named entities refers to names of location, organisation and personal\ + \ name. \n For example, 'David is an employee of Amazon and he is visiting New York\ + \ next week to see Esther' will be \nPERSON: David $ ORGANIZATION: Amazon $ LOCATION:\ + \ New York $ PERSON: Esther \n\nEnsure the output strictly follows the format: label:\ + \ entity $ label: entity, with each unique entity on a separate label line, avoiding\ + \ grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries like\ + \ none. \n\nText: {{text}} \nReturn only the output" +include: masakhaner +task: masakhaner_ig_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_lg.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_lg.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1823b20f63e1e4d2bcf5edc1427be758c3d16a62 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_lg.yaml @@ -0,0 +1,11 @@ +# Generated by utils.py +dataset_name: lg +doc_to_text: "Named entities refers to names of location, organisation and personal\ + \ name. \n For example, 'David is an employee of Amazon and he is visiting New York\ + \ next week to see Esther' will be \nPERSON: David $ ORGANIZATION: Amazon $ LOCATION:\ + \ New York $ PERSON: Esther \n\nEnsure the output strictly follows the format: label:\ + \ entity $ label: entity, with each unique entity on a separate label line, avoiding\ + \ grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries like\ + \ none. \n\nText: {{text}} \nReturn only the output" +include: masakhaner +task: masakhaner_lg_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_luo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_luo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..55b6d82968ed9a09ef43a12bd5cba91ea4ab5c87 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_luo.yaml @@ -0,0 +1,11 @@ +# Generated by utils.py +dataset_name: luo +doc_to_text: "Named entities refers to names of location, organisation and personal\ + \ name. \n For example, 'David is an employee of Amazon and he is visiting New York\ + \ next week to see Esther' will be \nPERSON: David $ ORGANIZATION: Amazon $ LOCATION:\ + \ New York $ PERSON: Esther \n\nEnsure the output strictly follows the format: label:\ + \ entity $ label: entity, with each unique entity on a separate label line, avoiding\ + \ grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries like\ + \ none. \n\nText: {{text}} \nReturn only the output" +include: masakhaner +task: masakhaner_luo_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_mos.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_mos.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ac5ddf43cd0eefd5fbe85785c7b4687135924938 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_mos.yaml @@ -0,0 +1,11 @@ +# Generated by utils.py +dataset_name: mos +doc_to_text: "Named entities refers to names of location, organisation and personal\ + \ name. \n For example, 'David is an employee of Amazon and he is visiting New York\ + \ next week to see Esther' will be \nPERSON: David $ ORGANIZATION: Amazon $ LOCATION:\ + \ New York $ PERSON: Esther \n\nEnsure the output strictly follows the format: label:\ + \ entity $ label: entity, with each unique entity on a separate label line, avoiding\ + \ grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries like\ + \ none. \n\nText: {{text}} \nReturn only the output" +include: masakhaner +task: masakhaner_mos_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_ny.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_ny.yaml new file mode 100644 index 0000000000000000000000000000000000000000..36d12ad2c00c0101aa2405af5824b6cbf310d132 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_ny.yaml @@ -0,0 +1,11 @@ +# Generated by utils.py +dataset_name: ny +doc_to_text: "Named entities refers to names of location, organisation and personal\ + \ name. \n For example, 'David is an employee of Amazon and he is visiting New York\ + \ next week to see Esther' will be \nPERSON: David $ ORGANIZATION: Amazon $ LOCATION:\ + \ New York $ PERSON: Esther \n\nEnsure the output strictly follows the format: label:\ + \ entity $ label: entity, with each unique entity on a separate label line, avoiding\ + \ grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries like\ + \ none. \n\nText: {{text}} \nReturn only the output" +include: masakhaner +task: masakhaner_ny_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_pcm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_pcm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8c09bf44c682758e53deb9262e4aef393d8bbc8e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_pcm.yaml @@ -0,0 +1,11 @@ +# Generated by utils.py +dataset_name: pcm +doc_to_text: "Named entities refers to names of location, organisation and personal\ + \ name. \n For example, 'David is an employee of Amazon and he is visiting New York\ + \ next week to see Esther' will be \nPERSON: David $ ORGANIZATION: Amazon $ LOCATION:\ + \ New York $ PERSON: Esther \n\nEnsure the output strictly follows the format: label:\ + \ entity $ label: entity, with each unique entity on a separate label line, avoiding\ + \ grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries like\ + \ none. \n\nText: {{text}} \nReturn only the output" +include: masakhaner +task: masakhaner_pcm_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_rw.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_rw.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7398e5fbe7b55f837b900158b7fdc4a3b7e2ac92 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_rw.yaml @@ -0,0 +1,11 @@ +# Generated by utils.py +dataset_name: rw +doc_to_text: "Named entities refers to names of location, organisation and personal\ + \ name. \n For example, 'David is an employee of Amazon and he is visiting New York\ + \ next week to see Esther' will be \nPERSON: David $ ORGANIZATION: Amazon $ LOCATION:\ + \ New York $ PERSON: Esther \n\nEnsure the output strictly follows the format: label:\ + \ entity $ label: entity, with each unique entity on a separate label line, avoiding\ + \ grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries like\ + \ none. \n\nText: {{text}} \nReturn only the output" +include: masakhaner +task: masakhaner_rw_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_sn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_sn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ecdd3260fcb8b0fd147dfba666aa0a42e4687323 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_sn.yaml @@ -0,0 +1,11 @@ +# Generated by utils.py +dataset_name: sn +doc_to_text: "Named entities refers to names of location, organisation and personal\ + \ name. \n For example, 'David is an employee of Amazon and he is visiting New York\ + \ next week to see Esther' will be \nPERSON: David $ ORGANIZATION: Amazon $ LOCATION:\ + \ New York $ PERSON: Esther \n\nEnsure the output strictly follows the format: label:\ + \ entity $ label: entity, with each unique entity on a separate label line, avoiding\ + \ grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries like\ + \ none. \n\nText: {{text}} \nReturn only the output" +include: masakhaner +task: masakhaner_sn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_sw.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_sw.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f2bd3379c3b3508067ae500c5cad947ba7006b74 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_sw.yaml @@ -0,0 +1,11 @@ +# Generated by utils.py +dataset_name: sw +doc_to_text: "Named entities refers to names of location, organisation and personal\ + \ name. \n For example, 'David is an employee of Amazon and he is visiting New York\ + \ next week to see Esther' will be \nPERSON: David $ ORGANIZATION: Amazon $ LOCATION:\ + \ New York $ PERSON: Esther \n\nEnsure the output strictly follows the format: label:\ + \ entity $ label: entity, with each unique entity on a separate label line, avoiding\ + \ grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries like\ + \ none. \n\nText: {{text}} \nReturn only the output" +include: masakhaner +task: masakhaner_sw_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_tn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_tn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..50d80dcb79ab8f2bb0f423d42159dd1d56fd262f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_tn.yaml @@ -0,0 +1,11 @@ +# Generated by utils.py +dataset_name: tn +doc_to_text: "Named entities refers to names of location, organisation and personal\ + \ name. \n For example, 'David is an employee of Amazon and he is visiting New York\ + \ next week to see Esther' will be \nPERSON: David $ ORGANIZATION: Amazon $ LOCATION:\ + \ New York $ PERSON: Esther \n\nEnsure the output strictly follows the format: label:\ + \ entity $ label: entity, with each unique entity on a separate label line, avoiding\ + \ grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries like\ + \ none. \n\nText: {{text}} \nReturn only the output" +include: masakhaner +task: masakhaner_tn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_tw.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_tw.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6c8a8d40575c4867cc0ec33a96343f1eb9c29f7b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_tw.yaml @@ -0,0 +1,11 @@ +# Generated by utils.py +dataset_name: tw +doc_to_text: "Named entities refers to names of location, organisation and personal\ + \ name. \n For example, 'David is an employee of Amazon and he is visiting New York\ + \ next week to see Esther' will be \nPERSON: David $ ORGANIZATION: Amazon $ LOCATION:\ + \ New York $ PERSON: Esther \n\nEnsure the output strictly follows the format: label:\ + \ entity $ label: entity, with each unique entity on a separate label line, avoiding\ + \ grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries like\ + \ none. \n\nText: {{text}} \nReturn only the output" +include: masakhaner +task: masakhaner_tw_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_wo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_wo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5e5f6eeaecb9ad56ee2cf035cf1390f489d8ba98 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_wo.yaml @@ -0,0 +1,11 @@ +# Generated by utils.py +dataset_name: wo +doc_to_text: "Named entities refers to names of location, organisation and personal\ + \ name. \n For example, 'David is an employee of Amazon and he is visiting New York\ + \ next week to see Esther' will be \nPERSON: David $ ORGANIZATION: Amazon $ LOCATION:\ + \ New York $ PERSON: Esther \n\nEnsure the output strictly follows the format: label:\ + \ entity $ label: entity, with each unique entity on a separate label line, avoiding\ + \ grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries like\ + \ none. \n\nText: {{text}} \nReturn only the output" +include: masakhaner +task: masakhaner_wo_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_xh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_xh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8b27051f5df77d39561c0fcee81033a357a9220d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_xh.yaml @@ -0,0 +1,11 @@ +# Generated by utils.py +dataset_name: xh +doc_to_text: "Named entities refers to names of location, organisation and personal\ + \ name. \n For example, 'David is an employee of Amazon and he is visiting New York\ + \ next week to see Esther' will be \nPERSON: David $ ORGANIZATION: Amazon $ LOCATION:\ + \ New York $ PERSON: Esther \n\nEnsure the output strictly follows the format: label:\ + \ entity $ label: entity, with each unique entity on a separate label line, avoiding\ + \ grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries like\ + \ none. \n\nText: {{text}} \nReturn only the output" +include: masakhaner +task: masakhaner_xh_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_yo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_yo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2fdb71aa53d4eacdb5cfcdada220423909f9515d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_yo.yaml @@ -0,0 +1,11 @@ +# Generated by utils.py +dataset_name: yo +doc_to_text: "Named entities refers to names of location, organisation and personal\ + \ name. \n For example, 'David is an employee of Amazon and he is visiting New York\ + \ next week to see Esther' will be \nPERSON: David $ ORGANIZATION: Amazon $ LOCATION:\ + \ New York $ PERSON: Esther \n\nEnsure the output strictly follows the format: label:\ + \ entity $ label: entity, with each unique entity on a separate label line, avoiding\ + \ grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries like\ + \ none. \n\nText: {{text}} \nReturn only the output" +include: masakhaner +task: masakhaner_yo_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_zu.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_zu.yaml new file mode 100644 index 0000000000000000000000000000000000000000..83b9d4b0fd5ef3ab2d9f62639d49a9cea1e3a1a1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/masakhaner_zu.yaml @@ -0,0 +1,11 @@ +# Generated by utils.py +dataset_name: zu +doc_to_text: "Named entities refers to names of location, organisation and personal\ + \ name. \n For example, 'David is an employee of Amazon and he is visiting New York\ + \ next week to see Esther' will be \nPERSON: David $ ORGANIZATION: Amazon $ LOCATION:\ + \ New York $ PERSON: Esther \n\nEnsure the output strictly follows the format: label:\ + \ entity $ label: entity, with each unique entity on a separate label line, avoiding\ + \ grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries like\ + \ none. \n\nText: {{text}} \nReturn only the output" +include: masakhaner +task: masakhaner_zu_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..76909044e7f35948156f8bb506ce2fce563ec689 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_1/utils.py @@ -0,0 +1,146 @@ +import collections +import re + +from lm_eval.utils import weighted_f1_score + + +def doc_to_target(doc): + return transform_text(doc["ner_tags"]) + + +def transform_text(text): + entities = [] + current_entity = "" + current_tag = "" + + for pair in text.split("\n"): + if pair: # Check if the line is not empty + word, tag = pair.strip().split(": ") + tag = tag.upper() + word = word.lower() + word = word.strip(",.").strip() + + if tag.startswith("B-"): + if current_entity: + entities.append(f"{current_tag}: {current_entity}") + current_tag = tag.split("-")[1] + current_entity = word + elif tag.startswith("I-") and tag.split("-")[1] == current_tag: + current_entity += word + else: + if current_entity: + entities.append(f"{current_tag}: {current_entity}") + current_entity = "" + current_tag = "" + if current_entity: + entities.append(f"{current_tag}: {current_entity}") + + # Join all the transformed output lines with $$ as separator + return " $$ ".join(entities) + + +def span_f1_agg(items): + """Computes Span based F1 score. + + This function is copied from + https://github.com/google-research/multilingual-t5/blob/master/multilingual_t5/evaluation/metrics.py + + Args: + targets: list of strings or list of list of strings if multiple references + are present. + predictions: list of strings + + Returns: + span f1 across all targets and predictions (Based on CoNLL script) + """ + unzipped_list = list(zip(*items)) + targets = unzipped_list[0] + predictions = unzipped_list[1] + + true_positives = collections.defaultdict(int) + false_positives = collections.defaultdict(int) + false_negatives = collections.defaultdict(int) + + def normalize_text(strings): + def get_blank_spaces_pattern(): + return re.compile(r"\s{3,}|\t") + + def remove_blank_spaces(text): + text = re.sub(pattern=get_blank_spaces_pattern(), repl="", string=text) + text = re.sub("\s+", " ", text) + return text + + def remove_punctuation(text): + my_punctuation = '!"$%&\'()*+,-./:;<=>?[\\]^_`{|}~•@.""-,`' + text = re.sub( + "[" + my_punctuation + "]+", " ", str(text) + ) # strip punctuation + return text + + def remove_articles(text): + regex = re.compile(r"\b(a|an|the)\b", re.UNICODE) + return re.sub(regex, " ", text) + + def lowercase(text): + text = text.lower() + return text + + strings = remove_punctuation(strings) + strings = remove_articles(strings) + strings = remove_blank_spaces(strings) + strings = lowercase(strings) + + return strings + + def tags_to_spans(tag_sequence, delimiter="$$"): + """Extract spans from IOB1 or BIO tags.""" + if isinstance(tag_sequence, list): + tag_sequence = " ".join(i.strip() for i in tag_sequence) + tag_sequence_split = [ + item.strip() + for sub in tag_sequence.strip().split(delimiter) + for item in sub.split("$") + if item + ] + tag_sequence_split = [ + item.strip() + for value in tag_sequence_split + for sub in value.split(". ") + for item in sub.split(", ") + ] + tags_entities = [] + for tag_entity in tag_sequence_split: + tag_entity_split = tag_entity.split(": ") + if len(tag_entity_split) != 2: + continue + tag = normalize_text(tag_entity_split[0].strip()) + entity = normalize_text(tag_entity_split[1].rstrip().lstrip()) + tags_entities.append((tag, entity)) + return tags_entities + + def compute_f1_metrics(true_positive, false_positive, false_negative): + precision = float(true_positive) / float(true_positive + false_positive + 1e-13) + recall = float(true_positive) / float(true_positive + false_negative + 1e-13) + f1_measures = 2.0 * ((precision * recall) / (precision + recall + 1e-13)) + return precision, recall, f1_measures + + for target, pred in zip(targets, predictions): + gold_spans = tags_to_spans(target) + predicted_spans = tags_to_spans(pred) + + for span in predicted_spans: + if span in gold_spans: + true_positives[span[0]] += 1 + gold_spans.remove(span) + else: + false_positives[span[0]] += 1 + # These spans weren't predicted. + for span in gold_spans: + false_negatives[span[0]] += 1 + + _, _, f1_measure = compute_f1_metrics( + sum(true_positives.values()), + sum(false_positives.values()), + sum(false_negatives.values()), + ) + return f1_measure diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner new file mode 100644 index 0000000000000000000000000000000000000000..2fd5eb829ce60a5a16d970dbf6b0078e42135cf4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner @@ -0,0 +1,26 @@ +tag: +- masakhaner_tasks +- masakhaner_prompt_2 +dataset_path: masakhane/masakhaner-x +dataset_name: null +dataset_kwargs: {trust_remote_code: True} +output_type: generate_until +generation_kwargs: + do_sample: false + until: + - + - <|im_end|> +validation_split: validation +test_split: test +fewshot_split: train +doc_to_target: target +filter_list: + - name: flexible-extract + filter: + - function: format_span +metric_list: + - metric: f1 + aggregation: !function utils.span_f1_agg + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_am.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_am.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bd1bd33551e152f6f29ff9d248d76ec236be75d4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_am.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: am +doc_to_text: "You are working as a named entity recognition expert and your task is\ + \ to label a given text with named entity labels. Your task is to identify and label\ + \ any named entities present in the text. The named entity labels that you will\ + \ be using are PER (person), LOC (location), ORG (organization) and DATE (date).\ + \ Label multi-word entities as a single named entity. For words which are not part\ + \ of any named entity, do not return any value for it. \nEnsure the output strictly\ + \ follows the format: label: entity $$ label: entity, with each unique entity on\ + \ a separate label line, avoiding grouped entities (e.g., avoid LOC: entity, entity)\ + \ or irrelevant entries like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_am_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_bbj.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_bbj.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5d817ecbe3596b40fad80fe4b95af7ace7bd9e35 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_bbj.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: bbj +doc_to_text: "You are working as a named entity recognition expert and your task is\ + \ to label a given text with named entity labels. Your task is to identify and label\ + \ any named entities present in the text. The named entity labels that you will\ + \ be using are PER (person), LOC (location), ORG (organization) and DATE (date).\ + \ Label multi-word entities as a single named entity. For words which are not part\ + \ of any named entity, do not return any value for it. \nEnsure the output strictly\ + \ follows the format: label: entity $$ label: entity, with each unique entity on\ + \ a separate label line, avoiding grouped entities (e.g., avoid LOC: entity, entity)\ + \ or irrelevant entries like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_bbj_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_bm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_bm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f99a03c7486f8f7170d19b1935e223210915c137 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_bm.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: bm +doc_to_text: "You are working as a named entity recognition expert and your task is\ + \ to label a given text with named entity labels. Your task is to identify and label\ + \ any named entities present in the text. The named entity labels that you will\ + \ be using are PER (person), LOC (location), ORG (organization) and DATE (date).\ + \ Label multi-word entities as a single named entity. For words which are not part\ + \ of any named entity, do not return any value for it. \nEnsure the output strictly\ + \ follows the format: label: entity $$ label: entity, with each unique entity on\ + \ a separate label line, avoiding grouped entities (e.g., avoid LOC: entity, entity)\ + \ or irrelevant entries like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_bm_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_ee.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_ee.yaml new file mode 100644 index 0000000000000000000000000000000000000000..da31685e7d9fb0fa462403e0347347600c705d41 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_ee.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: ee +doc_to_text: "You are working as a named entity recognition expert and your task is\ + \ to label a given text with named entity labels. Your task is to identify and label\ + \ any named entities present in the text. The named entity labels that you will\ + \ be using are PER (person), LOC (location), ORG (organization) and DATE (date).\ + \ Label multi-word entities as a single named entity. For words which are not part\ + \ of any named entity, do not return any value for it. \nEnsure the output strictly\ + \ follows the format: label: entity $$ label: entity, with each unique entity on\ + \ a separate label line, avoiding grouped entities (e.g., avoid LOC: entity, entity)\ + \ or irrelevant entries like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_ee_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_ha.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_ha.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8075046a92f4c88ce8a75db0416732b4f0e96f45 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_ha.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: ha +doc_to_text: "You are working as a named entity recognition expert and your task is\ + \ to label a given text with named entity labels. Your task is to identify and label\ + \ any named entities present in the text. The named entity labels that you will\ + \ be using are PER (person), LOC (location), ORG (organization) and DATE (date).\ + \ Label multi-word entities as a single named entity. For words which are not part\ + \ of any named entity, do not return any value for it. \nEnsure the output strictly\ + \ follows the format: label: entity $$ label: entity, with each unique entity on\ + \ a separate label line, avoiding grouped entities (e.g., avoid LOC: entity, entity)\ + \ or irrelevant entries like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_ha_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_ig.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_ig.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c8771f510a5de2a7ff56acd593b4de3b2309dd07 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_ig.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: ig +doc_to_text: "You are working as a named entity recognition expert and your task is\ + \ to label a given text with named entity labels. Your task is to identify and label\ + \ any named entities present in the text. The named entity labels that you will\ + \ be using are PER (person), LOC (location), ORG (organization) and DATE (date).\ + \ Label multi-word entities as a single named entity. For words which are not part\ + \ of any named entity, do not return any value for it. \nEnsure the output strictly\ + \ follows the format: label: entity $$ label: entity, with each unique entity on\ + \ a separate label line, avoiding grouped entities (e.g., avoid LOC: entity, entity)\ + \ or irrelevant entries like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_ig_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_lg.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_lg.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5c6729e368b3cf53900035c5730d9b729a9e2baa --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_lg.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: lg +doc_to_text: "You are working as a named entity recognition expert and your task is\ + \ to label a given text with named entity labels. Your task is to identify and label\ + \ any named entities present in the text. The named entity labels that you will\ + \ be using are PER (person), LOC (location), ORG (organization) and DATE (date).\ + \ Label multi-word entities as a single named entity. For words which are not part\ + \ of any named entity, do not return any value for it. \nEnsure the output strictly\ + \ follows the format: label: entity $$ label: entity, with each unique entity on\ + \ a separate label line, avoiding grouped entities (e.g., avoid LOC: entity, entity)\ + \ or irrelevant entries like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_lg_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_luo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_luo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a458235f101ae430c77f9da1124a0b8d9fdcda38 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_luo.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: luo +doc_to_text: "You are working as a named entity recognition expert and your task is\ + \ to label a given text with named entity labels. Your task is to identify and label\ + \ any named entities present in the text. The named entity labels that you will\ + \ be using are PER (person), LOC (location), ORG (organization) and DATE (date).\ + \ Label multi-word entities as a single named entity. For words which are not part\ + \ of any named entity, do not return any value for it. \nEnsure the output strictly\ + \ follows the format: label: entity $$ label: entity, with each unique entity on\ + \ a separate label line, avoiding grouped entities (e.g., avoid LOC: entity, entity)\ + \ or irrelevant entries like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_luo_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_mos.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_mos.yaml new file mode 100644 index 0000000000000000000000000000000000000000..816b9bdedc4578c6d2ade8cb05258a4ebc7280de --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_mos.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: mos +doc_to_text: "You are working as a named entity recognition expert and your task is\ + \ to label a given text with named entity labels. Your task is to identify and label\ + \ any named entities present in the text. The named entity labels that you will\ + \ be using are PER (person), LOC (location), ORG (organization) and DATE (date).\ + \ Label multi-word entities as a single named entity. For words which are not part\ + \ of any named entity, do not return any value for it. \nEnsure the output strictly\ + \ follows the format: label: entity $$ label: entity, with each unique entity on\ + \ a separate label line, avoiding grouped entities (e.g., avoid LOC: entity, entity)\ + \ or irrelevant entries like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_mos_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_ny.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_ny.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2f8c4c13c89495b4da3b07ad432b3969310037f8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_ny.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: ny +doc_to_text: "You are working as a named entity recognition expert and your task is\ + \ to label a given text with named entity labels. Your task is to identify and label\ + \ any named entities present in the text. The named entity labels that you will\ + \ be using are PER (person), LOC (location), ORG (organization) and DATE (date).\ + \ Label multi-word entities as a single named entity. For words which are not part\ + \ of any named entity, do not return any value for it. \nEnsure the output strictly\ + \ follows the format: label: entity $$ label: entity, with each unique entity on\ + \ a separate label line, avoiding grouped entities (e.g., avoid LOC: entity, entity)\ + \ or irrelevant entries like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_ny_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_pcm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_pcm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..75dc6ec048cab2dc550d8174dbcc44d966a6ff8e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_pcm.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: pcm +doc_to_text: "You are working as a named entity recognition expert and your task is\ + \ to label a given text with named entity labels. Your task is to identify and label\ + \ any named entities present in the text. The named entity labels that you will\ + \ be using are PER (person), LOC (location), ORG (organization) and DATE (date).\ + \ Label multi-word entities as a single named entity. For words which are not part\ + \ of any named entity, do not return any value for it. \nEnsure the output strictly\ + \ follows the format: label: entity $$ label: entity, with each unique entity on\ + \ a separate label line, avoiding grouped entities (e.g., avoid LOC: entity, entity)\ + \ or irrelevant entries like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_pcm_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_rw.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_rw.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fb93e2d4b24c68bc46b0f141d4c43a51b39ea41e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_rw.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: rw +doc_to_text: "You are working as a named entity recognition expert and your task is\ + \ to label a given text with named entity labels. Your task is to identify and label\ + \ any named entities present in the text. The named entity labels that you will\ + \ be using are PER (person), LOC (location), ORG (organization) and DATE (date).\ + \ Label multi-word entities as a single named entity. For words which are not part\ + \ of any named entity, do not return any value for it. \nEnsure the output strictly\ + \ follows the format: label: entity $$ label: entity, with each unique entity on\ + \ a separate label line, avoiding grouped entities (e.g., avoid LOC: entity, entity)\ + \ or irrelevant entries like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_rw_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_sn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_sn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..60380a512424253dc84f826922fc1fd2f9e75d72 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_sn.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: sn +doc_to_text: "You are working as a named entity recognition expert and your task is\ + \ to label a given text with named entity labels. Your task is to identify and label\ + \ any named entities present in the text. The named entity labels that you will\ + \ be using are PER (person), LOC (location), ORG (organization) and DATE (date).\ + \ Label multi-word entities as a single named entity. For words which are not part\ + \ of any named entity, do not return any value for it. \nEnsure the output strictly\ + \ follows the format: label: entity $$ label: entity, with each unique entity on\ + \ a separate label line, avoiding grouped entities (e.g., avoid LOC: entity, entity)\ + \ or irrelevant entries like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_sn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_sw.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_sw.yaml new file mode 100644 index 0000000000000000000000000000000000000000..82cf74ae26a0cec7d9d659713c7d0203ab829a1f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_sw.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: sw +doc_to_text: "You are working as a named entity recognition expert and your task is\ + \ to label a given text with named entity labels. Your task is to identify and label\ + \ any named entities present in the text. The named entity labels that you will\ + \ be using are PER (person), LOC (location), ORG (organization) and DATE (date).\ + \ Label multi-word entities as a single named entity. For words which are not part\ + \ of any named entity, do not return any value for it. \nEnsure the output strictly\ + \ follows the format: label: entity $$ label: entity, with each unique entity on\ + \ a separate label line, avoiding grouped entities (e.g., avoid LOC: entity, entity)\ + \ or irrelevant entries like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_sw_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_tn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_tn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1852ebe9ae79b51e586fed461b88e7b03fc8557c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_tn.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: tn +doc_to_text: "You are working as a named entity recognition expert and your task is\ + \ to label a given text with named entity labels. Your task is to identify and label\ + \ any named entities present in the text. The named entity labels that you will\ + \ be using are PER (person), LOC (location), ORG (organization) and DATE (date).\ + \ Label multi-word entities as a single named entity. For words which are not part\ + \ of any named entity, do not return any value for it. \nEnsure the output strictly\ + \ follows the format: label: entity $$ label: entity, with each unique entity on\ + \ a separate label line, avoiding grouped entities (e.g., avoid LOC: entity, entity)\ + \ or irrelevant entries like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_tn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_tw.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_tw.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ea354958bcf1590a41fc4e48426d403cae4f9454 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_tw.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: tw +doc_to_text: "You are working as a named entity recognition expert and your task is\ + \ to label a given text with named entity labels. Your task is to identify and label\ + \ any named entities present in the text. The named entity labels that you will\ + \ be using are PER (person), LOC (location), ORG (organization) and DATE (date).\ + \ Label multi-word entities as a single named entity. For words which are not part\ + \ of any named entity, do not return any value for it. \nEnsure the output strictly\ + \ follows the format: label: entity $$ label: entity, with each unique entity on\ + \ a separate label line, avoiding grouped entities (e.g., avoid LOC: entity, entity)\ + \ or irrelevant entries like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_tw_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_wo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_wo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e7cd0d754be2d6fc09054f50d18d29fa07a8551a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_wo.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: wo +doc_to_text: "You are working as a named entity recognition expert and your task is\ + \ to label a given text with named entity labels. Your task is to identify and label\ + \ any named entities present in the text. The named entity labels that you will\ + \ be using are PER (person), LOC (location), ORG (organization) and DATE (date).\ + \ Label multi-word entities as a single named entity. For words which are not part\ + \ of any named entity, do not return any value for it. \nEnsure the output strictly\ + \ follows the format: label: entity $$ label: entity, with each unique entity on\ + \ a separate label line, avoiding grouped entities (e.g., avoid LOC: entity, entity)\ + \ or irrelevant entries like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_wo_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_xh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_xh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9451f0edd121337f8d4dd316284c05a8ea73ff6b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_xh.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: xh +doc_to_text: "You are working as a named entity recognition expert and your task is\ + \ to label a given text with named entity labels. Your task is to identify and label\ + \ any named entities present in the text. The named entity labels that you will\ + \ be using are PER (person), LOC (location), ORG (organization) and DATE (date).\ + \ Label multi-word entities as a single named entity. For words which are not part\ + \ of any named entity, do not return any value for it. \nEnsure the output strictly\ + \ follows the format: label: entity $$ label: entity, with each unique entity on\ + \ a separate label line, avoiding grouped entities (e.g., avoid LOC: entity, entity)\ + \ or irrelevant entries like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_xh_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_yo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_yo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fc0d92c50ed1cf7182b55ed192b2b440d58317e9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_yo.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: yo +doc_to_text: "You are working as a named entity recognition expert and your task is\ + \ to label a given text with named entity labels. Your task is to identify and label\ + \ any named entities present in the text. The named entity labels that you will\ + \ be using are PER (person), LOC (location), ORG (organization) and DATE (date).\ + \ Label multi-word entities as a single named entity. For words which are not part\ + \ of any named entity, do not return any value for it. \nEnsure the output strictly\ + \ follows the format: label: entity $$ label: entity, with each unique entity on\ + \ a separate label line, avoiding grouped entities (e.g., avoid LOC: entity, entity)\ + \ or irrelevant entries like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_yo_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_zu.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_zu.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e06bf3cef883c5ae3f9bf12d7abf5c27618d37e8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/masakhaner_zu.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: zu +doc_to_text: "You are working as a named entity recognition expert and your task is\ + \ to label a given text with named entity labels. Your task is to identify and label\ + \ any named entities present in the text. The named entity labels that you will\ + \ be using are PER (person), LOC (location), ORG (organization) and DATE (date).\ + \ Label multi-word entities as a single named entity. For words which are not part\ + \ of any named entity, do not return any value for it. \nEnsure the output strictly\ + \ follows the format: label: entity $$ label: entity, with each unique entity on\ + \ a separate label line, avoiding grouped entities (e.g., avoid LOC: entity, entity)\ + \ or irrelevant entries like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_zu_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..76909044e7f35948156f8bb506ce2fce563ec689 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_2/utils.py @@ -0,0 +1,146 @@ +import collections +import re + +from lm_eval.utils import weighted_f1_score + + +def doc_to_target(doc): + return transform_text(doc["ner_tags"]) + + +def transform_text(text): + entities = [] + current_entity = "" + current_tag = "" + + for pair in text.split("\n"): + if pair: # Check if the line is not empty + word, tag = pair.strip().split(": ") + tag = tag.upper() + word = word.lower() + word = word.strip(",.").strip() + + if tag.startswith("B-"): + if current_entity: + entities.append(f"{current_tag}: {current_entity}") + current_tag = tag.split("-")[1] + current_entity = word + elif tag.startswith("I-") and tag.split("-")[1] == current_tag: + current_entity += word + else: + if current_entity: + entities.append(f"{current_tag}: {current_entity}") + current_entity = "" + current_tag = "" + if current_entity: + entities.append(f"{current_tag}: {current_entity}") + + # Join all the transformed output lines with $$ as separator + return " $$ ".join(entities) + + +def span_f1_agg(items): + """Computes Span based F1 score. + + This function is copied from + https://github.com/google-research/multilingual-t5/blob/master/multilingual_t5/evaluation/metrics.py + + Args: + targets: list of strings or list of list of strings if multiple references + are present. + predictions: list of strings + + Returns: + span f1 across all targets and predictions (Based on CoNLL script) + """ + unzipped_list = list(zip(*items)) + targets = unzipped_list[0] + predictions = unzipped_list[1] + + true_positives = collections.defaultdict(int) + false_positives = collections.defaultdict(int) + false_negatives = collections.defaultdict(int) + + def normalize_text(strings): + def get_blank_spaces_pattern(): + return re.compile(r"\s{3,}|\t") + + def remove_blank_spaces(text): + text = re.sub(pattern=get_blank_spaces_pattern(), repl="", string=text) + text = re.sub("\s+", " ", text) + return text + + def remove_punctuation(text): + my_punctuation = '!"$%&\'()*+,-./:;<=>?[\\]^_`{|}~•@.""-,`' + text = re.sub( + "[" + my_punctuation + "]+", " ", str(text) + ) # strip punctuation + return text + + def remove_articles(text): + regex = re.compile(r"\b(a|an|the)\b", re.UNICODE) + return re.sub(regex, " ", text) + + def lowercase(text): + text = text.lower() + return text + + strings = remove_punctuation(strings) + strings = remove_articles(strings) + strings = remove_blank_spaces(strings) + strings = lowercase(strings) + + return strings + + def tags_to_spans(tag_sequence, delimiter="$$"): + """Extract spans from IOB1 or BIO tags.""" + if isinstance(tag_sequence, list): + tag_sequence = " ".join(i.strip() for i in tag_sequence) + tag_sequence_split = [ + item.strip() + for sub in tag_sequence.strip().split(delimiter) + for item in sub.split("$") + if item + ] + tag_sequence_split = [ + item.strip() + for value in tag_sequence_split + for sub in value.split(". ") + for item in sub.split(", ") + ] + tags_entities = [] + for tag_entity in tag_sequence_split: + tag_entity_split = tag_entity.split(": ") + if len(tag_entity_split) != 2: + continue + tag = normalize_text(tag_entity_split[0].strip()) + entity = normalize_text(tag_entity_split[1].rstrip().lstrip()) + tags_entities.append((tag, entity)) + return tags_entities + + def compute_f1_metrics(true_positive, false_positive, false_negative): + precision = float(true_positive) / float(true_positive + false_positive + 1e-13) + recall = float(true_positive) / float(true_positive + false_negative + 1e-13) + f1_measures = 2.0 * ((precision * recall) / (precision + recall + 1e-13)) + return precision, recall, f1_measures + + for target, pred in zip(targets, predictions): + gold_spans = tags_to_spans(target) + predicted_spans = tags_to_spans(pred) + + for span in predicted_spans: + if span in gold_spans: + true_positives[span[0]] += 1 + gold_spans.remove(span) + else: + false_positives[span[0]] += 1 + # These spans weren't predicted. + for span in gold_spans: + false_negatives[span[0]] += 1 + + _, _, f1_measure = compute_f1_metrics( + sum(true_positives.values()), + sum(false_positives.values()), + sum(false_negatives.values()), + ) + return f1_measure diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner new file mode 100644 index 0000000000000000000000000000000000000000..7f32f86b1e194826a7ffe7d4edb0935eac80c491 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner @@ -0,0 +1,26 @@ +tag: +- masakhaner_tasks +- masakhaner_prompt_3 +dataset_path: masakhane/masakhaner-x +dataset_name: null +dataset_kwargs: {trust_remote_code: True} +output_type: generate_until +generation_kwargs: + do_sample: false + until: + - + - <|im_end|> +validation_split: validation +test_split: test +fewshot_split: train +doc_to_target: target +filter_list: + - name: flexible-extract + filter: + - function: format_span +metric_list: + - metric: f1 + aggregation: !function utils.span_f1_agg + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_am.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_am.yaml new file mode 100644 index 0000000000000000000000000000000000000000..54ad8b54111743ecf392d944c6201bfc56e5362c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_am.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: am +doc_to_text: "You are a Named Entity Recognition expert in Amharic language. \nExtract\ + \ all named entities from the following Amharic text and categorize them into PERSON,\ + \ LOCATION, ORGANIZATION, or DATE. Ensure the output strictly follows the format:\ + \ label: entity $$ label: entity, with each unique entity on a separate label line,\ + \ avoiding grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries\ + \ like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_am_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_bbj.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_bbj.yaml new file mode 100644 index 0000000000000000000000000000000000000000..23e724f424a862308d1947b176cc24b5eb040d47 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_bbj.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: bbj +doc_to_text: "You are a Named Entity Recognition expert in Ghomala language. \nExtract\ + \ all named entities from the following Ghomala text and categorize them into PERSON,\ + \ LOCATION, ORGANIZATION, or DATE. Ensure the output strictly follows the format:\ + \ label: entity $$ label: entity, with each unique entity on a separate label line,\ + \ avoiding grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries\ + \ like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_bbj_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_bm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_bm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..62b5b80e7c26335f304f7ed9673dd9f1c94ef970 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_bm.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: bm +doc_to_text: "You are a Named Entity Recognition expert in Bambara language. \nExtract\ + \ all named entities from the following Bambara text and categorize them into PERSON,\ + \ LOCATION, ORGANIZATION, or DATE. Ensure the output strictly follows the format:\ + \ label: entity $$ label: entity, with each unique entity on a separate label line,\ + \ avoiding grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries\ + \ like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_bm_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_ee.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_ee.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2cdadd27559ab964cacfd396be6fb29a3ae392e8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_ee.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: ee +doc_to_text: "You are a Named Entity Recognition expert in Ewe language. \nExtract\ + \ all named entities from the following Ewe text and categorize them into PERSON,\ + \ LOCATION, ORGANIZATION, or DATE. Ensure the output strictly follows the format:\ + \ label: entity $$ label: entity, with each unique entity on a separate label line,\ + \ avoiding grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries\ + \ like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_ee_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_ha.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_ha.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9d19d26f67447f8916f74b5e26d1bccc0e65bc57 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_ha.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: ha +doc_to_text: "You are a Named Entity Recognition expert in Hausa language. \nExtract\ + \ all named entities from the following Hausa text and categorize them into PERSON,\ + \ LOCATION, ORGANIZATION, or DATE. Ensure the output strictly follows the format:\ + \ label: entity $$ label: entity, with each unique entity on a separate label line,\ + \ avoiding grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries\ + \ like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_ha_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_ig.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_ig.yaml new file mode 100644 index 0000000000000000000000000000000000000000..edf6119689b51760de8a77b70c7dd4461f41b99d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_ig.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: ig +doc_to_text: "You are a Named Entity Recognition expert in Igbo language. \nExtract\ + \ all named entities from the following Igbo text and categorize them into PERSON,\ + \ LOCATION, ORGANIZATION, or DATE. Ensure the output strictly follows the format:\ + \ label: entity $$ label: entity, with each unique entity on a separate label line,\ + \ avoiding grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries\ + \ like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_ig_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_lg.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_lg.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9318a78207c0f3f9e6d2de954e6707100143258e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_lg.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: lg +doc_to_text: "You are a Named Entity Recognition expert in Luganda language. \nExtract\ + \ all named entities from the following Luganda text and categorize them into PERSON,\ + \ LOCATION, ORGANIZATION, or DATE. Ensure the output strictly follows the format:\ + \ label: entity $$ label: entity, with each unique entity on a separate label line,\ + \ avoiding grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries\ + \ like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_lg_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_luo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_luo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..61254fc358a2bd8f2ed2a5dd6aca3d09f88ae232 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_luo.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: luo +doc_to_text: "You are a Named Entity Recognition expert in Luo language. \nExtract\ + \ all named entities from the following Luo text and categorize them into PERSON,\ + \ LOCATION, ORGANIZATION, or DATE. Ensure the output strictly follows the format:\ + \ label: entity $$ label: entity, with each unique entity on a separate label line,\ + \ avoiding grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries\ + \ like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_luo_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_mos.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_mos.yaml new file mode 100644 index 0000000000000000000000000000000000000000..84ff6b24aaffaf14564de139bde92b1c850bcddd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_mos.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: mos +doc_to_text: "You are a Named Entity Recognition expert in Mossi language. \nExtract\ + \ all named entities from the following Mossi text and categorize them into PERSON,\ + \ LOCATION, ORGANIZATION, or DATE. Ensure the output strictly follows the format:\ + \ label: entity $$ label: entity, with each unique entity on a separate label line,\ + \ avoiding grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries\ + \ like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_mos_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_ny.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_ny.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bd592c5b93e39a488d5c5d909137e1caada54840 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_ny.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: ny +doc_to_text: "You are a Named Entity Recognition expert in Chichewa language. \nExtract\ + \ all named entities from the following Chichewa text and categorize them into PERSON,\ + \ LOCATION, ORGANIZATION, or DATE. Ensure the output strictly follows the format:\ + \ label: entity $$ label: entity, with each unique entity on a separate label line,\ + \ avoiding grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries\ + \ like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_ny_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_pcm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_pcm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b448b244b8d6580c0ebc53817060de633dd39efb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_pcm.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: pcm +doc_to_text: "You are a Named Entity Recognition expert in Nigerian Pidgin language.\ + \ \nExtract all named entities from the following Nigerian Pidgin text and categorize\ + \ them into PERSON, LOCATION, ORGANIZATION, or DATE. Ensure the output strictly\ + \ follows the format: label: entity $$ label: entity, with each unique entity on\ + \ a separate label line, avoiding grouped entities (e.g., avoid LOC: entity, entity)\ + \ or irrelevant entries like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_pcm_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_rw.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_rw.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5356ce8b011c53b77ab18302512e30b52c727e37 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_rw.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: rw +doc_to_text: "You are a Named Entity Recognition expert in Kinyarwanda language. \n\ + Extract all named entities from the following Kinyarwanda text and categorize them\ + \ into PERSON, LOCATION, ORGANIZATION, or DATE. Ensure the output strictly follows\ + \ the format: label: entity $$ label: entity, with each unique entity on a separate\ + \ label line, avoiding grouped entities (e.g., avoid LOC: entity, entity) or irrelevant\ + \ entries like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_rw_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_sn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_sn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ab356ae80061b8c49348513b8f6fa05b2dea9473 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_sn.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: sn +doc_to_text: "You are a Named Entity Recognition expert in chiShona language. \nExtract\ + \ all named entities from the following chiShona text and categorize them into PERSON,\ + \ LOCATION, ORGANIZATION, or DATE. Ensure the output strictly follows the format:\ + \ label: entity $$ label: entity, with each unique entity on a separate label line,\ + \ avoiding grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries\ + \ like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_sn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_sw.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_sw.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bb3d69595796755c03f4f02f201d36e121e4b6bd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_sw.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: sw +doc_to_text: "You are a Named Entity Recognition expert in Kiswahili language. \n\ + Extract all named entities from the following Kiswahili text and categorize them\ + \ into PERSON, LOCATION, ORGANIZATION, or DATE. Ensure the output strictly follows\ + \ the format: label: entity $$ label: entity, with each unique entity on a separate\ + \ label line, avoiding grouped entities (e.g., avoid LOC: entity, entity) or irrelevant\ + \ entries like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_sw_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_tn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_tn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d42d164ad81853e92855fd51e73bd0e276d4c761 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_tn.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: tn +doc_to_text: "You are a Named Entity Recognition expert in Setswana language. \nExtract\ + \ all named entities from the following Setswana text and categorize them into PERSON,\ + \ LOCATION, ORGANIZATION, or DATE. Ensure the output strictly follows the format:\ + \ label: entity $$ label: entity, with each unique entity on a separate label line,\ + \ avoiding grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries\ + \ like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_tn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_tw.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_tw.yaml new file mode 100644 index 0000000000000000000000000000000000000000..62b4e2af7c4916e851c834a27c8357a91140d45b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_tw.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: tw +doc_to_text: "You are a Named Entity Recognition expert in Twi language. \nExtract\ + \ all named entities from the following Twi text and categorize them into PERSON,\ + \ LOCATION, ORGANIZATION, or DATE. Ensure the output strictly follows the format:\ + \ label: entity $$ label: entity, with each unique entity on a separate label line,\ + \ avoiding grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries\ + \ like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_tw_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_wo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_wo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6db45e2bccb590f2586494d3fb586b3e5966a17b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_wo.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: wo +doc_to_text: "You are a Named Entity Recognition expert in Wolof language. \nExtract\ + \ all named entities from the following Wolof text and categorize them into PERSON,\ + \ LOCATION, ORGANIZATION, or DATE. Ensure the output strictly follows the format:\ + \ label: entity $$ label: entity, with each unique entity on a separate label line,\ + \ avoiding grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries\ + \ like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_wo_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_xh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_xh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6a697b274e71c54293edab7d6dbefabd719843ea --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_xh.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: xh +doc_to_text: "You are a Named Entity Recognition expert in isiXhosa language. \nExtract\ + \ all named entities from the following isiXhosa text and categorize them into PERSON,\ + \ LOCATION, ORGANIZATION, or DATE. Ensure the output strictly follows the format:\ + \ label: entity $$ label: entity, with each unique entity on a separate label line,\ + \ avoiding grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries\ + \ like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_xh_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_yo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_yo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..589cd5b35a6b27c8a9075ee22b9f8fd98a11860d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_yo.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: yo +doc_to_text: "You are a Named Entity Recognition expert in Yoruba language. \nExtract\ + \ all named entities from the following Yoruba text and categorize them into PERSON,\ + \ LOCATION, ORGANIZATION, or DATE. Ensure the output strictly follows the format:\ + \ label: entity $$ label: entity, with each unique entity on a separate label line,\ + \ avoiding grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries\ + \ like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_yo_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_zu.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_zu.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c25d5a0c89da79ea982a35f8d032c3c489a16a80 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/masakhaner_zu.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: zu +doc_to_text: "You are a Named Entity Recognition expert in isiZulu language. \nExtract\ + \ all named entities from the following isiZulu text and categorize them into PERSON,\ + \ LOCATION, ORGANIZATION, or DATE. Ensure the output strictly follows the format:\ + \ label: entity $$ label: entity, with each unique entity on a separate label line,\ + \ avoiding grouped entities (e.g., avoid LOC: entity, entity) or irrelevant entries\ + \ like none. Return only the output \n\nText: {{text}}" +include: masakhaner +task: masakhaner_zu_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..76909044e7f35948156f8bb506ce2fce563ec689 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_3/utils.py @@ -0,0 +1,146 @@ +import collections +import re + +from lm_eval.utils import weighted_f1_score + + +def doc_to_target(doc): + return transform_text(doc["ner_tags"]) + + +def transform_text(text): + entities = [] + current_entity = "" + current_tag = "" + + for pair in text.split("\n"): + if pair: # Check if the line is not empty + word, tag = pair.strip().split(": ") + tag = tag.upper() + word = word.lower() + word = word.strip(",.").strip() + + if tag.startswith("B-"): + if current_entity: + entities.append(f"{current_tag}: {current_entity}") + current_tag = tag.split("-")[1] + current_entity = word + elif tag.startswith("I-") and tag.split("-")[1] == current_tag: + current_entity += word + else: + if current_entity: + entities.append(f"{current_tag}: {current_entity}") + current_entity = "" + current_tag = "" + if current_entity: + entities.append(f"{current_tag}: {current_entity}") + + # Join all the transformed output lines with $$ as separator + return " $$ ".join(entities) + + +def span_f1_agg(items): + """Computes Span based F1 score. + + This function is copied from + https://github.com/google-research/multilingual-t5/blob/master/multilingual_t5/evaluation/metrics.py + + Args: + targets: list of strings or list of list of strings if multiple references + are present. + predictions: list of strings + + Returns: + span f1 across all targets and predictions (Based on CoNLL script) + """ + unzipped_list = list(zip(*items)) + targets = unzipped_list[0] + predictions = unzipped_list[1] + + true_positives = collections.defaultdict(int) + false_positives = collections.defaultdict(int) + false_negatives = collections.defaultdict(int) + + def normalize_text(strings): + def get_blank_spaces_pattern(): + return re.compile(r"\s{3,}|\t") + + def remove_blank_spaces(text): + text = re.sub(pattern=get_blank_spaces_pattern(), repl="", string=text) + text = re.sub("\s+", " ", text) + return text + + def remove_punctuation(text): + my_punctuation = '!"$%&\'()*+,-./:;<=>?[\\]^_`{|}~•@.""-,`' + text = re.sub( + "[" + my_punctuation + "]+", " ", str(text) + ) # strip punctuation + return text + + def remove_articles(text): + regex = re.compile(r"\b(a|an|the)\b", re.UNICODE) + return re.sub(regex, " ", text) + + def lowercase(text): + text = text.lower() + return text + + strings = remove_punctuation(strings) + strings = remove_articles(strings) + strings = remove_blank_spaces(strings) + strings = lowercase(strings) + + return strings + + def tags_to_spans(tag_sequence, delimiter="$$"): + """Extract spans from IOB1 or BIO tags.""" + if isinstance(tag_sequence, list): + tag_sequence = " ".join(i.strip() for i in tag_sequence) + tag_sequence_split = [ + item.strip() + for sub in tag_sequence.strip().split(delimiter) + for item in sub.split("$") + if item + ] + tag_sequence_split = [ + item.strip() + for value in tag_sequence_split + for sub in value.split(". ") + for item in sub.split(", ") + ] + tags_entities = [] + for tag_entity in tag_sequence_split: + tag_entity_split = tag_entity.split(": ") + if len(tag_entity_split) != 2: + continue + tag = normalize_text(tag_entity_split[0].strip()) + entity = normalize_text(tag_entity_split[1].rstrip().lstrip()) + tags_entities.append((tag, entity)) + return tags_entities + + def compute_f1_metrics(true_positive, false_positive, false_negative): + precision = float(true_positive) / float(true_positive + false_positive + 1e-13) + recall = float(true_positive) / float(true_positive + false_negative + 1e-13) + f1_measures = 2.0 * ((precision * recall) / (precision + recall + 1e-13)) + return precision, recall, f1_measures + + for target, pred in zip(targets, predictions): + gold_spans = tags_to_spans(target) + predicted_spans = tags_to_spans(pred) + + for span in predicted_spans: + if span in gold_spans: + true_positives[span[0]] += 1 + gold_spans.remove(span) + else: + false_positives[span[0]] += 1 + # These spans weren't predicted. + for span in gold_spans: + false_negatives[span[0]] += 1 + + _, _, f1_measure = compute_f1_metrics( + sum(true_positives.values()), + sum(false_positives.values()), + sum(false_negatives.values()), + ) + return f1_measure diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner new file mode 100644 index 0000000000000000000000000000000000000000..5c0ae52e62da27b202b794918ac568747195de34 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner @@ -0,0 +1,26 @@ +tag: +- masakhaner_tasks +- masakhaner_prompt_4 +dataset_path: masakhane/masakhaner-x +dataset_name: null +dataset_kwargs: {trust_remote_code: True} +output_type: generate_until +generation_kwargs: + do_sample: false + until: + - + - <|im_end|> +validation_split: validation +test_split: test +fewshot_split: train +doc_to_target: target +filter_list: + - name: flexible-extract + filter: + - function: format_span +metric_list: + - metric: f1 + aggregation: !function utils.span_f1_agg + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_am.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_am.yaml new file mode 100644 index 0000000000000000000000000000000000000000..19b06221f2366dbeb7418961ecbf00f6b1146f1d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_am.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: am +doc_to_text: "As a Amharic linguist, label all named entities in the Amharic text\ + \ below with the categories: PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the\ + \ output strictly follows the format: label: entity $$ label: entity, with each\ + \ unique entity on a separate label line, avoiding grouped entities (e.g., avoid\ + \ LOC: entity, entity) or irrelevant entries like none. Return only the output.\ + \ \n\nText: {{text}}" +include: masakhaner +task: masakhaner_am_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_bbj.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_bbj.yaml new file mode 100644 index 0000000000000000000000000000000000000000..03ed5210a03e4ce5bc9afde45166519152a153c7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_bbj.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: bbj +doc_to_text: "As a Ghomala linguist, label all named entities in the Ghomala text\ + \ below with the categories: PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the\ + \ output strictly follows the format: label: entity $$ label: entity, with each\ + \ unique entity on a separate label line, avoiding grouped entities (e.g., avoid\ + \ LOC: entity, entity) or irrelevant entries like none. Return only the output.\ + \ \n\nText: {{text}}" +include: masakhaner +task: masakhaner_bbj_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_bm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_bm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e719db9ac9f7ffb2011e7af1f12bf6319cb1b9cc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_bm.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: bm +doc_to_text: "As a Bambara linguist, label all named entities in the Bambara text\ + \ below with the categories: PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the\ + \ output strictly follows the format: label: entity $$ label: entity, with each\ + \ unique entity on a separate label line, avoiding grouped entities (e.g., avoid\ + \ LOC: entity, entity) or irrelevant entries like none. Return only the output.\ + \ \n\nText: {{text}}" +include: masakhaner +task: masakhaner_bm_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_ee.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_ee.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fe5fc75d28eef016bac579673cf1db875ee0a9f3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_ee.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: ee +doc_to_text: "As a Ewe linguist, label all named entities in the Ewe text below with\ + \ the categories: PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the output strictly\ + \ follows the format: label: entity $$ label: entity, with each unique entity on\ + \ a separate label line, avoiding grouped entities (e.g., avoid LOC: entity, entity)\ + \ or irrelevant entries like none. Return only the output. \n\nText: {{text}}" +include: masakhaner +task: masakhaner_ee_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_ha.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_ha.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9f88b9d19d4545c1131573a2303c6014421ed54b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_ha.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: ha +doc_to_text: "As a Hausa linguist, label all named entities in the Hausa text below\ + \ with the categories: PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the output\ + \ strictly follows the format: label: entity $$ label: entity, with each unique\ + \ entity on a separate label line, avoiding grouped entities (e.g., avoid LOC: entity,\ + \ entity) or irrelevant entries like none. Return only the output. \n\nText: {{text}}" +include: masakhaner +task: masakhaner_ha_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_ig.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_ig.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d4712d7e8edd36a87c59c9f0bc759f7b884ec830 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_ig.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: ig +doc_to_text: "As a Igbo linguist, label all named entities in the Igbo text below\ + \ with the categories: PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the output\ + \ strictly follows the format: label: entity $$ label: entity, with each unique\ + \ entity on a separate label line, avoiding grouped entities (e.g., avoid LOC: entity,\ + \ entity) or irrelevant entries like none. Return only the output. \n\nText: {{text}}" +include: masakhaner +task: masakhaner_ig_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_lg.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_lg.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cd7bde4a6f098bc8e946d83b556a00e38de43a5b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_lg.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: lg +doc_to_text: "As a Luganda linguist, label all named entities in the Luganda text\ + \ below with the categories: PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the\ + \ output strictly follows the format: label: entity $$ label: entity, with each\ + \ unique entity on a separate label line, avoiding grouped entities (e.g., avoid\ + \ LOC: entity, entity) or irrelevant entries like none. Return only the output.\ + \ \n\nText: {{text}}" +include: masakhaner +task: masakhaner_lg_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_luo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_luo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..92c0ddfa2c58c90a9da84e3dd3e002f9eb8c1098 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_luo.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: luo +doc_to_text: "As a Luo linguist, label all named entities in the Luo text below with\ + \ the categories: PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the output strictly\ + \ follows the format: label: entity $$ label: entity, with each unique entity on\ + \ a separate label line, avoiding grouped entities (e.g., avoid LOC: entity, entity)\ + \ or irrelevant entries like none. Return only the output. \n\nText: {{text}}" +include: masakhaner +task: masakhaner_luo_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_mos.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_mos.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2eb75d8e554dba222cf9f92fc6c8f013e7d232b2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_mos.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: mos +doc_to_text: "As a Mossi linguist, label all named entities in the Mossi text below\ + \ with the categories: PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the output\ + \ strictly follows the format: label: entity $$ label: entity, with each unique\ + \ entity on a separate label line, avoiding grouped entities (e.g., avoid LOC: entity,\ + \ entity) or irrelevant entries like none. Return only the output. \n\nText: {{text}}" +include: masakhaner +task: masakhaner_mos_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_ny.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_ny.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e8cb8218aff23a34f48455b2cbb897112a407f06 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_ny.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: ny +doc_to_text: "As a Chichewa linguist, label all named entities in the Chichewa text\ + \ below with the categories: PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the\ + \ output strictly follows the format: label: entity $$ label: entity, with each\ + \ unique entity on a separate label line, avoiding grouped entities (e.g., avoid\ + \ LOC: entity, entity) or irrelevant entries like none. Return only the output.\ + \ \n\nText: {{text}}" +include: masakhaner +task: masakhaner_ny_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_pcm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_pcm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..93f8ae3adb27f74df18c13a0ed886d176b62eead --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_pcm.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: pcm +doc_to_text: "As a Nigerian Pidgin linguist, label all named entities in the Nigerian\ + \ Pidgin text below with the categories: PERSON, LOCATION, ORGANIZATION, and DATE.\ + \ Ensure the output strictly follows the format: label: entity $$ label: entity,\ + \ with each unique entity on a separate label line, avoiding grouped entities (e.g.,\ + \ avoid LOC: entity, entity) or irrelevant entries like none. Return only the output.\ + \ \n\nText: {{text}}" +include: masakhaner +task: masakhaner_pcm_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_rw.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_rw.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d64d49925bf1668e0f2b5ebdb6516f4ba0668f88 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_rw.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: rw +doc_to_text: "As a Kinyarwanda linguist, label all named entities in the Kinyarwanda\ + \ text below with the categories: PERSON, LOCATION, ORGANIZATION, and DATE. Ensure\ + \ the output strictly follows the format: label: entity $$ label: entity, with each\ + \ unique entity on a separate label line, avoiding grouped entities (e.g., avoid\ + \ LOC: entity, entity) or irrelevant entries like none. Return only the output.\ + \ \n\nText: {{text}}" +include: masakhaner +task: masakhaner_rw_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_sn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_sn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..40230fb1cebb4f1aa9d9001360389ef1cdfda64e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_sn.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: sn +doc_to_text: "As a chiShona linguist, label all named entities in the chiShona text\ + \ below with the categories: PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the\ + \ output strictly follows the format: label: entity $$ label: entity, with each\ + \ unique entity on a separate label line, avoiding grouped entities (e.g., avoid\ + \ LOC: entity, entity) or irrelevant entries like none. Return only the output.\ + \ \n\nText: {{text}}" +include: masakhaner +task: masakhaner_sn_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_sw.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_sw.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7b27554ddd1faa7a796b5249c340b4204fdfaa5a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_sw.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: sw +doc_to_text: "As a Kiswahili linguist, label all named entities in the Kiswahili text\ + \ below with the categories: PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the\ + \ output strictly follows the format: label: entity $$ label: entity, with each\ + \ unique entity on a separate label line, avoiding grouped entities (e.g., avoid\ + \ LOC: entity, entity) or irrelevant entries like none. Return only the output.\ + \ \n\nText: {{text}}" +include: masakhaner +task: masakhaner_sw_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_tn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_tn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..88080456ba554e20a63f3c68638908b31d5294cd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_tn.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: tn +doc_to_text: "As a Setswana linguist, label all named entities in the Setswana text\ + \ below with the categories: PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the\ + \ output strictly follows the format: label: entity $$ label: entity, with each\ + \ unique entity on a separate label line, avoiding grouped entities (e.g., avoid\ + \ LOC: entity, entity) or irrelevant entries like none. Return only the output.\ + \ \n\nText: {{text}}" +include: masakhaner +task: masakhaner_tn_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_tw.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_tw.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8d2eec6befd2442e314abba2e2655dc8aa0baf4a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_tw.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: tw +doc_to_text: "As a Twi linguist, label all named entities in the Twi text below with\ + \ the categories: PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the output strictly\ + \ follows the format: label: entity $$ label: entity, with each unique entity on\ + \ a separate label line, avoiding grouped entities (e.g., avoid LOC: entity, entity)\ + \ or irrelevant entries like none. Return only the output. \n\nText: {{text}}" +include: masakhaner +task: masakhaner_tw_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_wo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_wo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..41501cb385fe50b38d5d30afe592da4705585881 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_wo.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: wo +doc_to_text: "As a Wolof linguist, label all named entities in the Wolof text below\ + \ with the categories: PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the output\ + \ strictly follows the format: label: entity $$ label: entity, with each unique\ + \ entity on a separate label line, avoiding grouped entities (e.g., avoid LOC: entity,\ + \ entity) or irrelevant entries like none. Return only the output. \n\nText: {{text}}" +include: masakhaner +task: masakhaner_wo_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_xh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_xh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4b29fda3f444dd526ee7cc94f6579e74b6f63b97 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_xh.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: xh +doc_to_text: "As a isiXhosa linguist, label all named entities in the isiXhosa text\ + \ below with the categories: PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the\ + \ output strictly follows the format: label: entity $$ label: entity, with each\ + \ unique entity on a separate label line, avoiding grouped entities (e.g., avoid\ + \ LOC: entity, entity) or irrelevant entries like none. Return only the output.\ + \ \n\nText: {{text}}" +include: masakhaner +task: masakhaner_xh_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_yo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_yo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a0c327bd53c90651e9c7e6b699d3fb9fb52748a1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_yo.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: yo +doc_to_text: "As a Yoruba linguist, label all named entities in the Yoruba text below\ + \ with the categories: PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the output\ + \ strictly follows the format: label: entity $$ label: entity, with each unique\ + \ entity on a separate label line, avoiding grouped entities (e.g., avoid LOC: entity,\ + \ entity) or irrelevant entries like none. Return only the output. \n\nText: {{text}}" +include: masakhaner +task: masakhaner_yo_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_zu.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_zu.yaml new file mode 100644 index 0000000000000000000000000000000000000000..24961ec759862a77b2ce608f4a5954ec62f139fe --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/masakhaner_zu.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: zu +doc_to_text: "As a isiZulu linguist, label all named entities in the isiZulu text\ + \ below with the categories: PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the\ + \ output strictly follows the format: label: entity $$ label: entity, with each\ + \ unique entity on a separate label line, avoiding grouped entities (e.g., avoid\ + \ LOC: entity, entity) or irrelevant entries like none. Return only the output.\ + \ \n\nText: {{text}}" +include: masakhaner +task: masakhaner_zu_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..76909044e7f35948156f8bb506ce2fce563ec689 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_4/utils.py @@ -0,0 +1,146 @@ +import collections +import re + +from lm_eval.utils import weighted_f1_score + + +def doc_to_target(doc): + return transform_text(doc["ner_tags"]) + + +def transform_text(text): + entities = [] + current_entity = "" + current_tag = "" + + for pair in text.split("\n"): + if pair: # Check if the line is not empty + word, tag = pair.strip().split(": ") + tag = tag.upper() + word = word.lower() + word = word.strip(",.").strip() + + if tag.startswith("B-"): + if current_entity: + entities.append(f"{current_tag}: {current_entity}") + current_tag = tag.split("-")[1] + current_entity = word + elif tag.startswith("I-") and tag.split("-")[1] == current_tag: + current_entity += word + else: + if current_entity: + entities.append(f"{current_tag}: {current_entity}") + current_entity = "" + current_tag = "" + if current_entity: + entities.append(f"{current_tag}: {current_entity}") + + # Join all the transformed output lines with $$ as separator + return " $$ ".join(entities) + + +def span_f1_agg(items): + """Computes Span based F1 score. + + This function is copied from + https://github.com/google-research/multilingual-t5/blob/master/multilingual_t5/evaluation/metrics.py + + Args: + targets: list of strings or list of list of strings if multiple references + are present. + predictions: list of strings + + Returns: + span f1 across all targets and predictions (Based on CoNLL script) + """ + unzipped_list = list(zip(*items)) + targets = unzipped_list[0] + predictions = unzipped_list[1] + + true_positives = collections.defaultdict(int) + false_positives = collections.defaultdict(int) + false_negatives = collections.defaultdict(int) + + def normalize_text(strings): + def get_blank_spaces_pattern(): + return re.compile(r"\s{3,}|\t") + + def remove_blank_spaces(text): + text = re.sub(pattern=get_blank_spaces_pattern(), repl="", string=text) + text = re.sub("\s+", " ", text) + return text + + def remove_punctuation(text): + my_punctuation = '!"$%&\'()*+,-./:;<=>?[\\]^_`{|}~•@.""-,`' + text = re.sub( + "[" + my_punctuation + "]+", " ", str(text) + ) # strip punctuation + return text + + def remove_articles(text): + regex = re.compile(r"\b(a|an|the)\b", re.UNICODE) + return re.sub(regex, " ", text) + + def lowercase(text): + text = text.lower() + return text + + strings = remove_punctuation(strings) + strings = remove_articles(strings) + strings = remove_blank_spaces(strings) + strings = lowercase(strings) + + return strings + + def tags_to_spans(tag_sequence, delimiter="$$"): + """Extract spans from IOB1 or BIO tags.""" + if isinstance(tag_sequence, list): + tag_sequence = " ".join(i.strip() for i in tag_sequence) + tag_sequence_split = [ + item.strip() + for sub in tag_sequence.strip().split(delimiter) + for item in sub.split("$") + if item + ] + tag_sequence_split = [ + item.strip() + for value in tag_sequence_split + for sub in value.split(". ") + for item in sub.split(", ") + ] + tags_entities = [] + for tag_entity in tag_sequence_split: + tag_entity_split = tag_entity.split(": ") + if len(tag_entity_split) != 2: + continue + tag = normalize_text(tag_entity_split[0].strip()) + entity = normalize_text(tag_entity_split[1].rstrip().lstrip()) + tags_entities.append((tag, entity)) + return tags_entities + + def compute_f1_metrics(true_positive, false_positive, false_negative): + precision = float(true_positive) / float(true_positive + false_positive + 1e-13) + recall = float(true_positive) / float(true_positive + false_negative + 1e-13) + f1_measures = 2.0 * ((precision * recall) / (precision + recall + 1e-13)) + return precision, recall, f1_measures + + for target, pred in zip(targets, predictions): + gold_spans = tags_to_spans(target) + predicted_spans = tags_to_spans(pred) + + for span in predicted_spans: + if span in gold_spans: + true_positives[span[0]] += 1 + gold_spans.remove(span) + else: + false_positives[span[0]] += 1 + # These spans weren't predicted. + for span in gold_spans: + false_negatives[span[0]] += 1 + + _, _, f1_measure = compute_f1_metrics( + sum(true_positives.values()), + sum(false_positives.values()), + sum(false_negatives.values()), + ) + return f1_measure diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner new file mode 100644 index 0000000000000000000000000000000000000000..09cd77e13106cca8862dcfa31b86c7742b97985a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner @@ -0,0 +1,26 @@ +tag: +- masakhaner_tasks +- masakhaner_prompt_5 +dataset_path: masakhane/masakhaner-x +dataset_name: null +dataset_kwargs: {trust_remote_code: True} +output_type: generate_until +generation_kwargs: + do_sample: false + until: + - + - <|im_end|> +validation_split: validation +test_split: test +fewshot_split: train +doc_to_target: target +filter_list: + - name: flexible-extract + filter: + - function: format_span +metric_list: + - metric: f1 + aggregation: !function utils.span_f1_agg + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_am.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_am.yaml new file mode 100644 index 0000000000000000000000000000000000000000..90c485745377aa9532eb0f6e7b35b15cd31d5414 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_am.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: am +doc_to_text: "Provide a concise list of named entities in the text below. Use the\ + \ following labels: PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the output\ + \ strictly follows the format: label: entity $$ label: entity, with each unique\ + \ entity on a separate label line, avoiding grouped entities (e.g., avoid LOC: entity,\ + \ entity) or irrelevant entries like none. Return only the output. \n\nText: {{text}}" +include: masakhaner +task: masakhaner_am_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_bbj.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_bbj.yaml new file mode 100644 index 0000000000000000000000000000000000000000..74726694ef67fd5f572c1a58b0b637cf410a9997 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_bbj.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: bbj +doc_to_text: "Provide a concise list of named entities in the text below. Use the\ + \ following labels: PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the output\ + \ strictly follows the format: label: entity $$ label: entity, with each unique\ + \ entity on a separate label line, avoiding grouped entities (e.g., avoid LOC: entity,\ + \ entity) or irrelevant entries like none. Return only the output. \n\nText: {{text}}" +include: masakhaner +task: masakhaner_bbj_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_bm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_bm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c97e0c22a609ceb20a40b5185a58baad850e4e81 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_bm.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: bm +doc_to_text: "Provide a concise list of named entities in the text below. Use the\ + \ following labels: PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the output\ + \ strictly follows the format: label: entity $$ label: entity, with each unique\ + \ entity on a separate label line, avoiding grouped entities (e.g., avoid LOC: entity,\ + \ entity) or irrelevant entries like none. Return only the output. \n\nText: {{text}}" +include: masakhaner +task: masakhaner_bm_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_ee.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_ee.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6371649d2db361bfda7fabe8ed353ca529324375 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_ee.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: ee +doc_to_text: "Provide a concise list of named entities in the text below. Use the\ + \ following labels: PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the output\ + \ strictly follows the format: label: entity $$ label: entity, with each unique\ + \ entity on a separate label line, avoiding grouped entities (e.g., avoid LOC: entity,\ + \ entity) or irrelevant entries like none. Return only the output. \n\nText: {{text}}" +include: masakhaner +task: masakhaner_ee_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_ha.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_ha.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1d68c7eed339d51ba9d72863624a2ee9842be64e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_ha.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: ha +doc_to_text: "Provide a concise list of named entities in the text below. Use the\ + \ following labels: PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the output\ + \ strictly follows the format: label: entity $$ label: entity, with each unique\ + \ entity on a separate label line, avoiding grouped entities (e.g., avoid LOC: entity,\ + \ entity) or irrelevant entries like none. Return only the output. \n\nText: {{text}}" +include: masakhaner +task: masakhaner_ha_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_ig.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_ig.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3b8a429593210e2db50535011174cdbff26ad9c2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_ig.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: ig +doc_to_text: "Provide a concise list of named entities in the text below. Use the\ + \ following labels: PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the output\ + \ strictly follows the format: label: entity $$ label: entity, with each unique\ + \ entity on a separate label line, avoiding grouped entities (e.g., avoid LOC: entity,\ + \ entity) or irrelevant entries like none. Return only the output. \n\nText: {{text}}" +include: masakhaner +task: masakhaner_ig_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_lg.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_lg.yaml new file mode 100644 index 0000000000000000000000000000000000000000..84bdc8b9af8e76d09da546db020390531450ed85 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_lg.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: lg +doc_to_text: "Provide a concise list of named entities in the text below. Use the\ + \ following labels: PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the output\ + \ strictly follows the format: label: entity $$ label: entity, with each unique\ + \ entity on a separate label line, avoiding grouped entities (e.g., avoid LOC: entity,\ + \ entity) or irrelevant entries like none. Return only the output. \n\nText: {{text}}" +include: masakhaner +task: masakhaner_lg_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_luo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_luo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..55a0b5744cdaa4cf894ac882642978f66c7dbe5f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_luo.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: luo +doc_to_text: "Provide a concise list of named entities in the text below. Use the\ + \ following labels: PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the output\ + \ strictly follows the format: label: entity $$ label: entity, with each unique\ + \ entity on a separate label line, avoiding grouped entities (e.g., avoid LOC: entity,\ + \ entity) or irrelevant entries like none. Return only the output. \n\nText: {{text}}" +include: masakhaner +task: masakhaner_luo_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_mos.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_mos.yaml new file mode 100644 index 0000000000000000000000000000000000000000..06bcc4467d43d8559b8e9b2cd4b0a89fa6b3fa40 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_mos.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: mos +doc_to_text: "Provide a concise list of named entities in the text below. Use the\ + \ following labels: PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the output\ + \ strictly follows the format: label: entity $$ label: entity, with each unique\ + \ entity on a separate label line, avoiding grouped entities (e.g., avoid LOC: entity,\ + \ entity) or irrelevant entries like none. Return only the output. \n\nText: {{text}}" +include: masakhaner +task: masakhaner_mos_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_ny.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_ny.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e400f10e186f0e4ae6119fa622065a23da73c680 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_ny.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: ny +doc_to_text: "Provide a concise list of named entities in the text below. Use the\ + \ following labels: PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the output\ + \ strictly follows the format: label: entity $$ label: entity, with each unique\ + \ entity on a separate label line, avoiding grouped entities (e.g., avoid LOC: entity,\ + \ entity) or irrelevant entries like none. Return only the output. \n\nText: {{text}}" +include: masakhaner +task: masakhaner_ny_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_pcm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_pcm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b9d897bc01a217c01d9fbffdf496d060ec54b434 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_pcm.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: pcm +doc_to_text: "Provide a concise list of named entities in the text below. Use the\ + \ following labels: PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the output\ + \ strictly follows the format: label: entity $$ label: entity, with each unique\ + \ entity on a separate label line, avoiding grouped entities (e.g., avoid LOC: entity,\ + \ entity) or irrelevant entries like none. Return only the output. \n\nText: {{text}}" +include: masakhaner +task: masakhaner_pcm_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_rw.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_rw.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0742bc4dd567005a51861bab9b6208d64c66166f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_rw.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: rw +doc_to_text: "Provide a concise list of named entities in the text below. Use the\ + \ following labels: PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the output\ + \ strictly follows the format: label: entity $$ label: entity, with each unique\ + \ entity on a separate label line, avoiding grouped entities (e.g., avoid LOC: entity,\ + \ entity) or irrelevant entries like none. Return only the output. \n\nText: {{text}}" +include: masakhaner +task: masakhaner_rw_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_sn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_sn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..56711335c8c11879338fbc3b245bc49d0702733e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_sn.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: sn +doc_to_text: "Provide a concise list of named entities in the text below. Use the\ + \ following labels: PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the output\ + \ strictly follows the format: label: entity $$ label: entity, with each unique\ + \ entity on a separate label line, avoiding grouped entities (e.g., avoid LOC: entity,\ + \ entity) or irrelevant entries like none. Return only the output. \n\nText: {{text}}" +include: masakhaner +task: masakhaner_sn_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_sw.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_sw.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c418beb45612e31ff1738909d5d1c181bfbab079 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_sw.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: sw +doc_to_text: "Provide a concise list of named entities in the text below. Use the\ + \ following labels: PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the output\ + \ strictly follows the format: label: entity $$ label: entity, with each unique\ + \ entity on a separate label line, avoiding grouped entities (e.g., avoid LOC: entity,\ + \ entity) or irrelevant entries like none. Return only the output. \n\nText: {{text}}" +include: masakhaner +task: masakhaner_sw_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_tn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_tn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bf94a1081352827a7fe09eb913bfabd9e5f0c576 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_tn.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: tn +doc_to_text: "Provide a concise list of named entities in the text below. Use the\ + \ following labels: PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the output\ + \ strictly follows the format: label: entity $$ label: entity, with each unique\ + \ entity on a separate label line, avoiding grouped entities (e.g., avoid LOC: entity,\ + \ entity) or irrelevant entries like none. Return only the output. \n\nText: {{text}}" +include: masakhaner +task: masakhaner_tn_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_tw.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_tw.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cad2e2e3e64819dc6ab3151929a6ec76bb868821 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_tw.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: tw +doc_to_text: "Provide a concise list of named entities in the text below. Use the\ + \ following labels: PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the output\ + \ strictly follows the format: label: entity $$ label: entity, with each unique\ + \ entity on a separate label line, avoiding grouped entities (e.g., avoid LOC: entity,\ + \ entity) or irrelevant entries like none. Return only the output. \n\nText: {{text}}" +include: masakhaner +task: masakhaner_tw_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_wo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_wo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ec7af039234047cda3500be81b361bae294bace2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_wo.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: wo +doc_to_text: "Provide a concise list of named entities in the text below. Use the\ + \ following labels: PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the output\ + \ strictly follows the format: label: entity $$ label: entity, with each unique\ + \ entity on a separate label line, avoiding grouped entities (e.g., avoid LOC: entity,\ + \ entity) or irrelevant entries like none. Return only the output. \n\nText: {{text}}" +include: masakhaner +task: masakhaner_wo_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_xh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_xh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..debb164aef161a178ca53046e9b674a677d5fc08 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_xh.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: xh +doc_to_text: "Provide a concise list of named entities in the text below. Use the\ + \ following labels: PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the output\ + \ strictly follows the format: label: entity $$ label: entity, with each unique\ + \ entity on a separate label line, avoiding grouped entities (e.g., avoid LOC: entity,\ + \ entity) or irrelevant entries like none. Return only the output. \n\nText: {{text}}" +include: masakhaner +task: masakhaner_xh_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_yo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_yo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9abe1acbcb473c1bd091277c8a1913c792fdd0a3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_yo.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: yo +doc_to_text: "Provide a concise list of named entities in the text below. Use the\ + \ following labels: PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the output\ + \ strictly follows the format: label: entity $$ label: entity, with each unique\ + \ entity on a separate label line, avoiding grouped entities (e.g., avoid LOC: entity,\ + \ entity) or irrelevant entries like none. Return only the output. \n\nText: {{text}}" +include: masakhaner +task: masakhaner_yo_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_zu.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_zu.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c5af591aa464ef88ff6a4d4b62e55c504cb777c4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/masakhaner_zu.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: zu +doc_to_text: "Provide a concise list of named entities in the text below. Use the\ + \ following labels: PERSON, LOCATION, ORGANIZATION, and DATE. Ensure the output\ + \ strictly follows the format: label: entity $$ label: entity, with each unique\ + \ entity on a separate label line, avoiding grouped entities (e.g., avoid LOC: entity,\ + \ entity) or irrelevant entries like none. Return only the output. \n\nText: {{text}}" +include: masakhaner +task: masakhaner_zu_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..76909044e7f35948156f8bb506ce2fce563ec689 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhaner/prompt_5/utils.py @@ -0,0 +1,146 @@ +import collections +import re + +from lm_eval.utils import weighted_f1_score + + +def doc_to_target(doc): + return transform_text(doc["ner_tags"]) + + +def transform_text(text): + entities = [] + current_entity = "" + current_tag = "" + + for pair in text.split("\n"): + if pair: # Check if the line is not empty + word, tag = pair.strip().split(": ") + tag = tag.upper() + word = word.lower() + word = word.strip(",.").strip() + + if tag.startswith("B-"): + if current_entity: + entities.append(f"{current_tag}: {current_entity}") + current_tag = tag.split("-")[1] + current_entity = word + elif tag.startswith("I-") and tag.split("-")[1] == current_tag: + current_entity += word + else: + if current_entity: + entities.append(f"{current_tag}: {current_entity}") + current_entity = "" + current_tag = "" + if current_entity: + entities.append(f"{current_tag}: {current_entity}") + + # Join all the transformed output lines with $$ as separator + return " $$ ".join(entities) + + +def span_f1_agg(items): + """Computes Span based F1 score. + + This function is copied from + https://github.com/google-research/multilingual-t5/blob/master/multilingual_t5/evaluation/metrics.py + + Args: + targets: list of strings or list of list of strings if multiple references + are present. + predictions: list of strings + + Returns: + span f1 across all targets and predictions (Based on CoNLL script) + """ + unzipped_list = list(zip(*items)) + targets = unzipped_list[0] + predictions = unzipped_list[1] + + true_positives = collections.defaultdict(int) + false_positives = collections.defaultdict(int) + false_negatives = collections.defaultdict(int) + + def normalize_text(strings): + def get_blank_spaces_pattern(): + return re.compile(r"\s{3,}|\t") + + def remove_blank_spaces(text): + text = re.sub(pattern=get_blank_spaces_pattern(), repl="", string=text) + text = re.sub("\s+", " ", text) + return text + + def remove_punctuation(text): + my_punctuation = '!"$%&\'()*+,-./:;<=>?[\\]^_`{|}~•@.""-,`' + text = re.sub( + "[" + my_punctuation + "]+", " ", str(text) + ) # strip punctuation + return text + + def remove_articles(text): + regex = re.compile(r"\b(a|an|the)\b", re.UNICODE) + return re.sub(regex, " ", text) + + def lowercase(text): + text = text.lower() + return text + + strings = remove_punctuation(strings) + strings = remove_articles(strings) + strings = remove_blank_spaces(strings) + strings = lowercase(strings) + + return strings + + def tags_to_spans(tag_sequence, delimiter="$$"): + """Extract spans from IOB1 or BIO tags.""" + if isinstance(tag_sequence, list): + tag_sequence = " ".join(i.strip() for i in tag_sequence) + tag_sequence_split = [ + item.strip() + for sub in tag_sequence.strip().split(delimiter) + for item in sub.split("$") + if item + ] + tag_sequence_split = [ + item.strip() + for value in tag_sequence_split + for sub in value.split(". ") + for item in sub.split(", ") + ] + tags_entities = [] + for tag_entity in tag_sequence_split: + tag_entity_split = tag_entity.split(": ") + if len(tag_entity_split) != 2: + continue + tag = normalize_text(tag_entity_split[0].strip()) + entity = normalize_text(tag_entity_split[1].rstrip().lstrip()) + tags_entities.append((tag, entity)) + return tags_entities + + def compute_f1_metrics(true_positive, false_positive, false_negative): + precision = float(true_positive) / float(true_positive + false_positive + 1e-13) + recall = float(true_positive) / float(true_positive + false_negative + 1e-13) + f1_measures = 2.0 * ((precision * recall) / (precision + recall + 1e-13)) + return precision, recall, f1_measures + + for target, pred in zip(targets, predictions): + gold_spans = tags_to_spans(target) + predicted_spans = tags_to_spans(pred) + + for span in predicted_spans: + if span in gold_spans: + true_positives[span[0]] += 1 + gold_spans.remove(span) + else: + false_positives[span[0]] += 1 + # These spans weren't predicted. + for span in gold_spans: + false_negatives[span[0]] += 1 + + _, _, f1_measure = compute_f1_metrics( + sum(true_positives.values()), + sum(false_positives.values()), + sum(false_negatives.values()), + ) + return f1_measure diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/README.md b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/README.md new file mode 100644 index 0000000000000000000000000000000000000000..16df2df1d62f2d83d6d34e22373d6680a246eaa8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/README.md @@ -0,0 +1,99 @@ +# + +## Paper +Title: `MasakhaNEWS: News Topic Classification for African languages` + +Paper Link: https://aclanthology.org/2023.ijcnlp-main.10/ + +## Abstract +>African languages are severely under-represented in NLP research due to lack of datasets covering several NLP tasks. While there are individual language specific datasets that are being expanded to different tasks, only a handful of NLP tasks (e.g. named entity recognition and machine translation) have standardized benchmark datasets covering several geographical and typologically-diverse African languages. In this paper, we develop MasakhaNEWS -- a new benchmark dataset for news topic classification covering 16 languages widely spoken in Africa. We provide an evaluation of baseline models by training classical machine learning models and fine-tuning several language models. Furthermore, we explore several alternatives to full fine-tuning of language models that are better suited for zero-shot and few-shot learning such as cross-lingual parameter-efficient fine-tuning (like MAD-X), pattern exploiting training (PET), prompting language models (like ChatGPT), and prompt-free sentence transformer fine-tuning (SetFit and Cohere Embedding API). Our evaluation in zero-shot setting shows the potential of prompting ChatGPT for news topic classification in low-resource African languages, achieving an average performance of 70 F1 points without leveraging additional supervision like MAD-X. In few-shot setting, we show that with as little as 10 examples per label, we achieved more than 90% (i.e. 86.0 F1 points) of the performance of full supervised training (92.6 F1 points) leveraging the PET approach. + +HomePage: https://github.com/masakhane-io/masakhane-news + +### Citation + +``` +@inproceedings{adelani-etal-2023-masakhanews, + title = "{M}asakha{NEWS}: News Topic Classification for {A}frican languages", + author = "Adelani, David Ifeoluwa and + Masiak, Marek and + Azime, Israel Abebe and + Alabi, Jesujoba and + Tonja, Atnafu Lambebo and + Mwase, Christine and + Ogundepo, Odunayo and + Dossou, Bonaventure F. P. and + Oladipo, Akintunde and + Nixdorf, Doreen and + Emezue, Chris Chinenye and + Al-azzawi, Sana and + Sibanda, Blessing and + David, Davis and + Ndolela, Lolwethu and + Mukiibi, Jonathan and + Ajayi, Tunde and + Moteu, Tatiana and + Odhiambo, Brian and + Owodunni, Abraham and + Obiefuna, Nnaemeka and + Mohamed, Muhidin and + Muhammad, Shamsuddeen Hassan and + Ababu, Teshome Mulugeta and + Salahudeen, Saheed Abdullahi and + Yigezu, Mesay Gemeda and + Gwadabe, Tajuddeen and + Abdulmumin, Idris and + Taye, Mahlet and + Awoyomi, Oluwabusayo and + Shode, Iyanuoluwa and + Adelani, Tolulope and + Abdulganiyu, Habiba and + Omotayo, Abdul-Hakeem and + Adeeko, Adetola and + Afolabi, Abeeb and + Aremu, Anuoluwapo and + Samuel, Olanrewaju and + Siro, Clemencia and + Kimotho, Wangari and + Ogbu, Onyekachi and + Mbonu, Chinedu and + Chukwuneke, Chiamaka and + Fanijo, Samuel and + Ojo, Jessica and + Awosan, Oyinkansola and + Kebede, Tadesse and + Sakayo, Toadoum Sari and + Nyatsine, Pamela and + Sidume, Freedmore and + Yousuf, Oreen and + Oduwole, Mardiyyah and + Tshinu, Kanda and + Kimanuka, Ussen and + Diko, Thina and + Nxakama, Siyanda and + Nigusse, Sinodos and + Johar, Abdulmejid and + Mohamed, Shafie and + Hassan, Fuad Mire and + Mehamed, Moges Ahmed and + Ngabire, Evrard and + Jules, Jules and + Ssenkungu, Ivan and + Stenetorp, Pontus", + editor = "Park, Jong C. and + Arase, Yuki and + Hu, Baotian and + Lu, Wei and + Wijaya, Derry and + Purwarianti, Ayu and + Krisnadhi, Adila Alfa", + booktitle = "Proceedings of the 13th International Joint Conference on Natural Language Processing and the 3rd Conference of the Asia-Pacific Chapter of the Association for Computational Linguistics (Volume 1: Long Papers)", + month = nov, + year = "2023", + address = "Nusa Dua, Bali", + publisher = "Association for Computational Linguistics", + url = "https://aclanthology.org/2023.ijcnlp-main.10/", + doi = "10.18653/v1/2023.ijcnlp-main.10", + pages = "144--159" +} +``` diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/masakhanews.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/masakhanews.yaml new file mode 100644 index 0000000000000000000000000000000000000000..93b6f29d8cdc05cf0904e4a8fe9afdd35d111c88 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/masakhanews.yaml @@ -0,0 +1,13 @@ +group: masakhanews +task: + - masakhanews_prompt_1 + - masakhanews_prompt_2 + - masakhanews_prompt_3 + - masakhanews_prompt_4 + - masakhanews_prompt_5 +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/masakhanews b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/masakhanews new file mode 100644 index 0000000000000000000000000000000000000000..282a38422e526b9f8ce8731f950ae04e2a6cbf08 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/masakhanews @@ -0,0 +1,43 @@ +tag: +- masakhanews_tasks +- masakhanews_prompt_1 +- afrobench_TC_tasks +dataset_path: masakhane/masakhanews +dataset_name: null +output_type: multiple_choice +validation_split: validation +test_split: test +fewshot_split: validation +doc_to_target: label +doc_to_choice: + - "business" + - "entertainment" + - "health" + - "politics" + - "religion" + - "sports" + - "technology" +should_decontaminate: true +doc_to_decontamination_query: headline_text +metric_list: + - metric: f1 + aggregation: !function utils.weighted_f1_score + # aggregation: mean + average: weighted + hf_evaluate: true + higher_is_better: True + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" + - metric: acc + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/masakhanews_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/masakhanews_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d45b784facda2e60335d1980b9a1038c5cb91ec0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/masakhanews_amh.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: amh +doc_to_text: "Given the categories technology, business, politics, sports, health,\ + \ entertainment, or religion; what category does the text: '{{headline_text}}' belong\ + \ to: \n\n" +include: masakhanews +task: masakhanews_amh_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/masakhanews_eng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/masakhanews_eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..40685c17d8616dafb3797ab84153f77242e6a364 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/masakhanews_eng.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng +doc_to_text: "Given the categories technology, business, politics, sports, health,\ + \ entertainment, or religion; what category does the text: '{{headline_text}}' belong\ + \ to: \n\n" +include: masakhanews +task: masakhanews_eng_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/masakhanews_fra.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/masakhanews_fra.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2371172156b0cfb8532d93366518f2faf8793aed --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/masakhanews_fra.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: fra +doc_to_text: "Given the categories technology, business, politics, sports, health,\ + \ entertainment, or religion; what category does the text: '{{headline_text}}' belong\ + \ to: \n\n" +include: masakhanews +task: masakhanews_fra_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/masakhanews_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/masakhanews_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c7288982d35eaaad09407c59b6b6d02af3fb637a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/masakhanews_hau.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: hau +doc_to_text: "Given the categories technology, business, politics, sports, health,\ + \ entertainment, or religion; what category does the text: '{{headline_text}}' belong\ + \ to: \n\n" +include: masakhanews +task: masakhanews_hau_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/masakhanews_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/masakhanews_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4bf65cca861c670be75aa9b63a914590d4d996c4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/masakhanews_ibo.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: ibo +doc_to_text: "Given the categories technology, business, politics, sports, health,\ + \ entertainment, or religion; what category does the text: '{{headline_text}}' belong\ + \ to: \n\n" +include: masakhanews +task: masakhanews_ibo_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/masakhanews_lin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/masakhanews_lin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c6cdbe8de5ef1593bd2f98d75ec4ecca3fc33084 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/masakhanews_lin.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: lin +doc_to_text: "Given the categories technology, business, politics, sports, health,\ + \ entertainment, or religion; what category does the text: '{{headline_text}}' belong\ + \ to: \n\n" +include: masakhanews +task: masakhanews_lin_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/masakhanews_lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/masakhanews_lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e2f0ec1bada04bf7c72649621330e828bd449951 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/masakhanews_lug.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: lug +doc_to_text: "Given the categories technology, business, politics, sports, health,\ + \ entertainment, or religion; what category does the text: '{{headline_text}}' belong\ + \ to: \n\n" +include: masakhanews +task: masakhanews_lug_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/masakhanews_orm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/masakhanews_orm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a9bff1ac5113f06c5d035e46c0e139fbeb0a8d28 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/masakhanews_orm.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: orm +doc_to_text: "Given the categories technology, business, politics, sports, health,\ + \ entertainment, or religion; what category does the text: '{{headline_text}}' belong\ + \ to: \n\n" +include: masakhanews +task: masakhanews_orm_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/masakhanews_pcm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/masakhanews_pcm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..119b01bb158d15d307949226dc71a055d169e4fa --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/masakhanews_pcm.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: pcm +doc_to_text: "Given the categories technology, business, politics, sports, health,\ + \ entertainment, or religion; what category does the text: '{{headline_text}}' belong\ + \ to: \n\n" +include: masakhanews +task: masakhanews_pcm_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/masakhanews_run.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/masakhanews_run.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d8bc2923fa069f71d5bb00d4f272c235d6e7f0c6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/masakhanews_run.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: run +doc_to_text: "Given the categories technology, business, politics, sports, health,\ + \ entertainment, or religion; what category does the text: '{{headline_text}}' belong\ + \ to: \n\n" +include: masakhanews +task: masakhanews_run_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/masakhanews_sna.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/masakhanews_sna.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ee4fabdc96a680e0482040f00daca794eaf87dfe --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/masakhanews_sna.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: sna +doc_to_text: "Given the categories technology, business, politics, sports, health,\ + \ entertainment, or religion; what category does the text: '{{headline_text}}' belong\ + \ to: \n\n" +include: masakhanews +task: masakhanews_sna_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/masakhanews_som.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/masakhanews_som.yaml new file mode 100644 index 0000000000000000000000000000000000000000..88d7774c1b4c1c08367d8ebc79bc1ca7214cab9f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/masakhanews_som.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: som +doc_to_text: "Given the categories technology, business, politics, sports, health,\ + \ entertainment, or religion; what category does the text: '{{headline_text}}' belong\ + \ to: \n\n" +include: masakhanews +task: masakhanews_som_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/masakhanews_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/masakhanews_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c4e02aae282bdba08b8f057c71ab580a7ee9c031 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/masakhanews_swa.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: swa +doc_to_text: "Given the categories technology, business, politics, sports, health,\ + \ entertainment, or religion; what category does the text: '{{headline_text}}' belong\ + \ to: \n\n" +include: masakhanews +task: masakhanews_swa_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/masakhanews_tir.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/masakhanews_tir.yaml new file mode 100644 index 0000000000000000000000000000000000000000..72fa30ae7012379011b04a1fb3255fcad2a9a4e7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/masakhanews_tir.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: tir +doc_to_text: "Given the categories technology, business, politics, sports, health,\ + \ entertainment, or religion; what category does the text: '{{headline_text}}' belong\ + \ to: \n\n" +include: masakhanews +task: masakhanews_tir_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/masakhanews_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/masakhanews_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1d98b3b681de265b99207db0d3d2347d952d05a0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/masakhanews_xho.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: xho +doc_to_text: "Given the categories technology, business, politics, sports, health,\ + \ entertainment, or religion; what category does the text: '{{headline_text}}' belong\ + \ to: \n\n" +include: masakhanews +task: masakhanews_xho_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/masakhanews_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/masakhanews_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3ef4eec0e46425b9b02771ee6a360f3498cc0a1a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/masakhanews_yor.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: yor +doc_to_text: "Given the categories technology, business, politics, sports, health,\ + \ entertainment, or religion; what category does the text: '{{headline_text}}' belong\ + \ to: \n\n" +include: masakhanews +task: masakhanews_yor_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..3e735e2deb1f9c53152c072615aebe8ba3acb90b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_1/utils.py @@ -0,0 +1 @@ +from lm_eval.utils import weighted_f1_score diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_2/masakhanews b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_2/masakhanews new file mode 100644 index 0000000000000000000000000000000000000000..c174d2c7ff991b749881260b1ccb93d63a5e9f26 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_2/masakhanews @@ -0,0 +1,43 @@ +tag: +- masakhanews_tasks +- masakhanews_prompt_2 +- afrobench_TC_tasks +dataset_path: masakhane/masakhanews +dataset_name: null +output_type: multiple_choice +validation_split: validation +test_split: test +fewshot_split: validation +doc_to_target: label +doc_to_choice: + - "business" + - "entertainment" + - "health" + - "politics" + - "religion" + - "sports" + - "technology" +should_decontaminate: true +doc_to_decontamination_query: headline_text +metric_list: + - metric: f1 + aggregation: !function utils.weighted_f1_score + # aggregation: mean + average: weighted + hf_evaluate: true + higher_is_better: True + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" + - metric: acc + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_2/masakhanews_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_2/masakhanews_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cee7619cfb4c215a9e05551b668d2ea4e9e517ca --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_2/masakhanews_amh.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: amh +doc_to_text: 'Does this Amharic topic; ''{{headline_text}}'' belong to one of the + following categories: technology, business, politics, sports, health, entertainment, + or religion? category only + + + ' +include: masakhanews +task: masakhanews_amh_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_2/masakhanews_eng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_2/masakhanews_eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d3d6dd16c461eb7df5b730aa3283f6695f130503 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_2/masakhanews_eng.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: eng +doc_to_text: 'Does this English topic; ''{{headline_text}}'' belong to one of the + following categories: technology, business, politics, sports, health, entertainment, + or religion? category only + + + ' +include: masakhanews +task: masakhanews_eng_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_2/masakhanews_fra.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_2/masakhanews_fra.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c35a6a1d34753893185c2d0fcd5dd82e73853e35 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_2/masakhanews_fra.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: fra +doc_to_text: 'Does this French topic; ''{{headline_text}}'' belong to one of the following + categories: technology, business, politics, sports, health, entertainment, or religion? + category only + + + ' +include: masakhanews +task: masakhanews_fra_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_2/masakhanews_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_2/masakhanews_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..93e9cc5a7f3e53fd8fc97ee1ecf8bfbb85911939 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_2/masakhanews_hau.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: hau +doc_to_text: 'Does this Hausa topic; ''{{headline_text}}'' belong to one of the following + categories: technology, business, politics, sports, health, entertainment, or religion? + category only + + + ' +include: masakhanews +task: masakhanews_hau_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_2/masakhanews_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_2/masakhanews_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1638e435c76f6fdd671a6bf165d81debc4be3b3a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_2/masakhanews_ibo.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: ibo +doc_to_text: 'Does this Igbo topic; ''{{headline_text}}'' belong to one of the following + categories: technology, business, politics, sports, health, entertainment, or religion? + category only + + + ' +include: masakhanews +task: masakhanews_ibo_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_2/masakhanews_lin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_2/masakhanews_lin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0010d0e1ad36356f8c5b9bccccffc62f46730d93 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_2/masakhanews_lin.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: lin +doc_to_text: 'Does this Lingala topic; ''{{headline_text}}'' belong to one of the + following categories: technology, business, politics, sports, health, entertainment, + or religion? category only + + + ' +include: masakhanews +task: masakhanews_lin_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_2/masakhanews_lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_2/masakhanews_lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d526067289d1145ed494418408b0de55d51a74ca --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_2/masakhanews_lug.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: lug +doc_to_text: 'Does this Luganda topic; ''{{headline_text}}'' belong to one of the + following categories: technology, business, politics, sports, health, entertainment, + or religion? category only + + + ' +include: masakhanews +task: masakhanews_lug_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_2/masakhanews_orm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_2/masakhanews_orm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cd04c845d4eb1d46dd48052f0a582d7557e610a8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_2/masakhanews_orm.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: orm +doc_to_text: 'Does this Afaan Oromoo topic; ''{{headline_text}}'' belong to one of + the following categories: technology, business, politics, sports, health, entertainment, + or religion? category only + + + ' +include: masakhanews +task: masakhanews_orm_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_2/masakhanews_pcm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_2/masakhanews_pcm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..de685e3ac8a62a2e5ca5b3d0b16119a77caa1994 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_2/masakhanews_pcm.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: pcm +doc_to_text: 'Does this Nigerian Pidgin topic; ''{{headline_text}}'' belong to one + of the following categories: technology, business, politics, sports, health, entertainment, + or religion? category only + + + ' +include: masakhanews +task: masakhanews_pcm_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_2/masakhanews_run.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_2/masakhanews_run.yaml new file mode 100644 index 0000000000000000000000000000000000000000..62236d590bfc009043dd8dca6ab3c43edcaff995 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_2/masakhanews_run.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: run +doc_to_text: 'Does this Kirundi topic; ''{{headline_text}}'' belong to one of the + following categories: technology, business, politics, sports, health, entertainment, + or religion? category only + + + ' +include: masakhanews +task: masakhanews_run_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_2/masakhanews_sna.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_2/masakhanews_sna.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2a97e176b865335804af5efe9e911894acbbcc78 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_2/masakhanews_sna.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: sna +doc_to_text: 'Does this Shona topic; ''{{headline_text}}'' belong to one of the following + categories: technology, business, politics, sports, health, entertainment, or religion? + category only + + + ' +include: masakhanews +task: masakhanews_sna_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_2/masakhanews_som.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_2/masakhanews_som.yaml new file mode 100644 index 0000000000000000000000000000000000000000..318b9b87beabeef3837c826cbf128b3b8d1b4e8d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_2/masakhanews_som.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: som +doc_to_text: 'Does this Somali topic; ''{{headline_text}}'' belong to one of the following + categories: technology, business, politics, sports, health, entertainment, or religion? + category only + + + ' +include: masakhanews +task: masakhanews_som_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_2/masakhanews_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_2/masakhanews_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..75b9345f3229c6e105f4cd99e65a970c1061b7d4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_2/masakhanews_swa.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: swa +doc_to_text: 'Does this Swahili topic; ''{{headline_text}}'' belong to one of the + following categories: technology, business, politics, sports, health, entertainment, + or religion? category only + + + ' +include: masakhanews +task: masakhanews_swa_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_2/masakhanews_tir.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_2/masakhanews_tir.yaml new file mode 100644 index 0000000000000000000000000000000000000000..258a2bd3d7431083088e09f74044ce518cbaa7b1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_2/masakhanews_tir.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: tir +doc_to_text: 'Does this Tigrinya topic; ''{{headline_text}}'' belong to one of the + following categories: technology, business, politics, sports, health, entertainment, + or religion? category only + + + ' +include: masakhanews +task: masakhanews_tir_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_2/masakhanews_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_2/masakhanews_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..30c4c3ac3abdb3b239406a2d36efc9331b1597d7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_2/masakhanews_xho.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: xho +doc_to_text: 'Does this Xhosa topic; ''{{headline_text}}'' belong to one of the following + categories: technology, business, politics, sports, health, entertainment, or religion? + category only + + + ' +include: masakhanews +task: masakhanews_xho_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_2/masakhanews_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_2/masakhanews_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..067cf10632de430b650053d5537a733052e34b09 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_2/masakhanews_yor.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: yor +doc_to_text: 'Does this Yoruba topic; ''{{headline_text}}'' belong to one of the following + categories: technology, business, politics, sports, health, entertainment, or religion? + category only + + + ' +include: masakhanews +task: masakhanews_yor_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_2/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_2/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..3e735e2deb1f9c53152c072615aebe8ba3acb90b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_2/utils.py @@ -0,0 +1 @@ +from lm_eval.utils import weighted_f1_score diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_3/masakhanews b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_3/masakhanews new file mode 100644 index 0000000000000000000000000000000000000000..ecc2108967078bb24a1efd15acdd8387d47e173c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_3/masakhanews @@ -0,0 +1,43 @@ +tag: +- masakhanews_tasks +- masakhanews_prompt_3 +- afrobench_TC_tasks +dataset_path: masakhane/masakhanews +dataset_name: null +output_type: multiple_choice +validation_split: validation +test_split: test +fewshot_split: validation +doc_to_target: label +doc_to_choice: + - "business" + - "entertainment" + - "health" + - "politics" + - "religion" + - "sports" + - "technology" +should_decontaminate: true +doc_to_decontamination_query: headline_text +metric_list: + - metric: f1 + aggregation: !function utils.weighted_f1_score + # aggregation: mean + average: weighted + hf_evaluate: true + higher_is_better: True + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" + - metric: acc + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_3/masakhanews_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_3/masakhanews_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dec10d2963dd00fce7e8dbd7f24f8a61a178e0a7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_3/masakhanews_amh.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: amh +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories technology, religion, politics, sports, health, entertainment, or business;\ + \ what is the topic of the Amharic statement below? Return only the category. \n\ + \ntext: {{headline_text}} \\category:\n\n" +include: masakhanews +task: masakhanews_amh_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_3/masakhanews_eng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_3/masakhanews_eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a8b7159e215dd6bc5a766d51b06f77289e4ce1a7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_3/masakhanews_eng.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: eng +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories technology, religion, politics, sports, health, entertainment, or business;\ + \ what is the topic of the English statement below? Return only the category. \n\ + \ntext: {{headline_text}} \\category:\n\n" +include: masakhanews +task: masakhanews_eng_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_3/masakhanews_fra.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_3/masakhanews_fra.yaml new file mode 100644 index 0000000000000000000000000000000000000000..328316a8361d29a4db6ab882b46944fc65b2ff9b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_3/masakhanews_fra.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: fra +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories technology, religion, politics, sports, health, entertainment, or business;\ + \ what is the topic of the French statement below? Return only the category. \n\n\ + text: {{headline_text}} \\category:\n\n" +include: masakhanews +task: masakhanews_fra_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_3/masakhanews_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_3/masakhanews_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c18ff2779cc9a9d149afe1eb7c438e3d18e8af2c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_3/masakhanews_hau.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: hau +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories technology, religion, politics, sports, health, entertainment, or business;\ + \ what is the topic of the Hausa statement below? Return only the category. \n\n\ + text: {{headline_text}} \\category:\n\n" +include: masakhanews +task: masakhanews_hau_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_3/masakhanews_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_3/masakhanews_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3a91db840f2b72f041cffb827ea87520e28434cb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_3/masakhanews_ibo.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: ibo +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories technology, religion, politics, sports, health, entertainment, or business;\ + \ what is the topic of the Igbo statement below? Return only the category. \n\n\ + text: {{headline_text}} \\category:\n\n" +include: masakhanews +task: masakhanews_ibo_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_3/masakhanews_lin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_3/masakhanews_lin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..19c4cca2e5f65851b6c44a1baa6dd2842ce3bd5f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_3/masakhanews_lin.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: lin +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories technology, religion, politics, sports, health, entertainment, or business;\ + \ what is the topic of the Lingala statement below? Return only the category. \n\ + \ntext: {{headline_text}} \\category:\n\n" +include: masakhanews +task: masakhanews_lin_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_3/masakhanews_lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_3/masakhanews_lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9e3d4319fc82762f63292da9f916166132e53a42 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_3/masakhanews_lug.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: lug +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories technology, religion, politics, sports, health, entertainment, or business;\ + \ what is the topic of the Luganda statement below? Return only the category. \n\ + \ntext: {{headline_text}} \\category:\n\n" +include: masakhanews +task: masakhanews_lug_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_3/masakhanews_orm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_3/masakhanews_orm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9bacf0420b81df236351a3698e37cf3eca8983e7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_3/masakhanews_orm.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: orm +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories technology, religion, politics, sports, health, entertainment, or business;\ + \ what is the topic of the Afaan Oromoo statement below? Return only the category.\ + \ \n\ntext: {{headline_text}} \\category:\n\n" +include: masakhanews +task: masakhanews_orm_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_3/masakhanews_pcm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_3/masakhanews_pcm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e873becd56c21cd0d92841994a4fb6bed5119b51 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_3/masakhanews_pcm.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: pcm +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories technology, religion, politics, sports, health, entertainment, or business;\ + \ what is the topic of the Nigerian Pidgin statement below? Return only the category.\ + \ \n\ntext: {{headline_text}} \\category:\n\n" +include: masakhanews +task: masakhanews_pcm_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_3/masakhanews_run.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_3/masakhanews_run.yaml new file mode 100644 index 0000000000000000000000000000000000000000..307e13710dde72132a0df4011500aca4ccfd9e22 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_3/masakhanews_run.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: run +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories technology, religion, politics, sports, health, entertainment, or business;\ + \ what is the topic of the Kirundi statement below? Return only the category. \n\ + \ntext: {{headline_text}} \\category:\n\n" +include: masakhanews +task: masakhanews_run_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_3/masakhanews_sna.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_3/masakhanews_sna.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ee69be3de11efe89bac3dd355f4a9ffe99206c37 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_3/masakhanews_sna.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: sna +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories technology, religion, politics, sports, health, entertainment, or business;\ + \ what is the topic of the Shona statement below? Return only the category. \n\n\ + text: {{headline_text}} \\category:\n\n" +include: masakhanews +task: masakhanews_sna_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_3/masakhanews_som.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_3/masakhanews_som.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8c181fddb819a9f42f18a31141bc71f837f759cd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_3/masakhanews_som.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: som +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories technology, religion, politics, sports, health, entertainment, or business;\ + \ what is the topic of the Somali statement below? Return only the category. \n\n\ + text: {{headline_text}} \\category:\n\n" +include: masakhanews +task: masakhanews_som_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_3/masakhanews_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_3/masakhanews_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fbe1c4200f9953d603eae6262067facfc09fe694 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_3/masakhanews_swa.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: swa +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories technology, religion, politics, sports, health, entertainment, or business;\ + \ what is the topic of the Swahili statement below? Return only the category. \n\ + \ntext: {{headline_text}} \\category:\n\n" +include: masakhanews +task: masakhanews_swa_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_3/masakhanews_tir.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_3/masakhanews_tir.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b6055da2859d8932b3d6c40130d895846069f285 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_3/masakhanews_tir.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: tir +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories technology, religion, politics, sports, health, entertainment, or business;\ + \ what is the topic of the Tigrinya statement below? Return only the category. \n\ + \ntext: {{headline_text}} \\category:\n\n" +include: masakhanews +task: masakhanews_tir_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_3/masakhanews_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_3/masakhanews_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..110fc08778130c4880b904a682211df4800a2cd5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_3/masakhanews_xho.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: xho +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories technology, religion, politics, sports, health, entertainment, or business;\ + \ what is the topic of the Xhosa statement below? Return only the category. \n\n\ + text: {{headline_text}} \\category:\n\n" +include: masakhanews +task: masakhanews_xho_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_3/masakhanews_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_3/masakhanews_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d31e9b23fdfbdc5bbc85be20457c80c9b90c4a31 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_3/masakhanews_yor.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: yor +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories technology, religion, politics, sports, health, entertainment, or business;\ + \ what is the topic of the Yoruba statement below? Return only the category. \n\n\ + text: {{headline_text}} \\category:\n\n" +include: masakhanews +task: masakhanews_yor_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_3/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_3/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..3e735e2deb1f9c53152c072615aebe8ba3acb90b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_3/utils.py @@ -0,0 +1 @@ +from lm_eval.utils import weighted_f1_score diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_4/masakhanews b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_4/masakhanews new file mode 100644 index 0000000000000000000000000000000000000000..a1801f4e00b1d90a885ed9d73a14c2745cd73f01 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_4/masakhanews @@ -0,0 +1,43 @@ +tag: +- masakhanews_tasks +- masakhanews_prompt_4 +- afrobench_TC_tasks +dataset_path: masakhane/masakhanews +dataset_name: null +output_type: multiple_choice +validation_split: validation +test_split: test +fewshot_split: validation +doc_to_target: label +doc_to_choice: + - "business" + - "entertainment" + - "health" + - "politics" + - "religion" + - "sports" + - "technology" +should_decontaminate: true +doc_to_decontamination_query: headline_text +metric_list: + - metric: f1 + aggregation: !function utils.weighted_f1_score + # aggregation: mean + average: weighted + hf_evaluate: true + higher_is_better: True + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" + - metric: acc + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_4/masakhanews_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_4/masakhanews_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a76305859c1e27a1b00b3be6492a76b207d313da --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_4/masakhanews_amh.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: amh +doc_to_text: "Label the following text as technology, religion, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{headline_text}} \\category: \n\n" +include: masakhanews +task: masakhanews_amh_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_4/masakhanews_eng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_4/masakhanews_eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8567113756fde2e8e98c5f6f2f68073a5d14550b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_4/masakhanews_eng.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng +doc_to_text: "Label the following text as technology, religion, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{headline_text}} \\category: \n\n" +include: masakhanews +task: masakhanews_eng_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_4/masakhanews_fra.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_4/masakhanews_fra.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3f86635f6d5edf28590264ae45a0f3546d868feb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_4/masakhanews_fra.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: fra +doc_to_text: "Label the following text as technology, religion, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{headline_text}} \\category: \n\n" +include: masakhanews +task: masakhanews_fra_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_4/masakhanews_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_4/masakhanews_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c1b7ce562b51486f41ce75c6716eda24d56caf1f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_4/masakhanews_hau.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: hau +doc_to_text: "Label the following text as technology, religion, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{headline_text}} \\category: \n\n" +include: masakhanews +task: masakhanews_hau_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_4/masakhanews_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_4/masakhanews_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d76a905d62d2e93f09608684592dff02c60f131c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_4/masakhanews_ibo.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: ibo +doc_to_text: "Label the following text as technology, religion, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{headline_text}} \\category: \n\n" +include: masakhanews +task: masakhanews_ibo_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_4/masakhanews_lin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_4/masakhanews_lin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e0247529b5c4496cce3f52651f6969e894484bb1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_4/masakhanews_lin.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: lin +doc_to_text: "Label the following text as technology, religion, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{headline_text}} \\category: \n\n" +include: masakhanews +task: masakhanews_lin_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_4/masakhanews_lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_4/masakhanews_lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ca02c0a5fcbd248c82e945646d932614c8e515e8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_4/masakhanews_lug.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: lug +doc_to_text: "Label the following text as technology, religion, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{headline_text}} \\category: \n\n" +include: masakhanews +task: masakhanews_lug_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_4/masakhanews_orm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_4/masakhanews_orm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..781eb4cc977bd0ed74698737c56c9f97190b9623 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_4/masakhanews_orm.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: orm +doc_to_text: "Label the following text as technology, religion, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{headline_text}} \\category: \n\n" +include: masakhanews +task: masakhanews_orm_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_4/masakhanews_pcm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_4/masakhanews_pcm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..93ad9f482b3539efd0b4e7ab89b64ab75ce91147 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_4/masakhanews_pcm.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: pcm +doc_to_text: "Label the following text as technology, religion, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{headline_text}} \\category: \n\n" +include: masakhanews +task: masakhanews_pcm_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_4/masakhanews_run.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_4/masakhanews_run.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f5d985481f1b4443cadd4c6d1ef12424825e02cb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_4/masakhanews_run.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: run +doc_to_text: "Label the following text as technology, religion, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{headline_text}} \\category: \n\n" +include: masakhanews +task: masakhanews_run_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_4/masakhanews_sna.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_4/masakhanews_sna.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2676db850ff3b7ec98dfdd2b596e8d3011b30915 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_4/masakhanews_sna.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: sna +doc_to_text: "Label the following text as technology, religion, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{headline_text}} \\category: \n\n" +include: masakhanews +task: masakhanews_sna_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_4/masakhanews_som.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_4/masakhanews_som.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6562da417b3a50c0d712038db88bd4f205c13df8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_4/masakhanews_som.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: som +doc_to_text: "Label the following text as technology, religion, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{headline_text}} \\category: \n\n" +include: masakhanews +task: masakhanews_som_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_4/masakhanews_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_4/masakhanews_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3bb9764ad0dc0c69ba98fc85fcb1a51cda37c3b8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_4/masakhanews_swa.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: swa +doc_to_text: "Label the following text as technology, religion, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{headline_text}} \\category: \n\n" +include: masakhanews +task: masakhanews_swa_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_4/masakhanews_tir.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_4/masakhanews_tir.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3dfb1d4e7de9510175386192bcdf7f4524181308 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_4/masakhanews_tir.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: tir +doc_to_text: "Label the following text as technology, religion, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{headline_text}} \\category: \n\n" +include: masakhanews +task: masakhanews_tir_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_4/masakhanews_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_4/masakhanews_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9c1b51c20386a2c1d5196d4da47223603b5637ab --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_4/masakhanews_xho.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: xho +doc_to_text: "Label the following text as technology, religion, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{headline_text}} \\category: \n\n" +include: masakhanews +task: masakhanews_xho_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_4/masakhanews_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_4/masakhanews_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9d22d1c7f59e8e2103d605b3e5c9c4dd08811bfb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_4/masakhanews_yor.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: yor +doc_to_text: "Label the following text as technology, religion, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{headline_text}} \\category: \n\n" +include: masakhanews +task: masakhanews_yor_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_4/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_4/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..3e735e2deb1f9c53152c072615aebe8ba3acb90b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_4/utils.py @@ -0,0 +1 @@ +from lm_eval.utils import weighted_f1_score diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_5/masakhanews b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_5/masakhanews new file mode 100644 index 0000000000000000000000000000000000000000..8d76af03ab044d68314853c0a0005a05141c1dca --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_5/masakhanews @@ -0,0 +1,43 @@ +tag: +- masakhanews_tasks +- masakhanews_prompt_5 +- afrobench_TC_tasks +dataset_path: masakhane/masakhanews +dataset_name: null +output_type: multiple_choice +validation_split: validation +test_split: test +fewshot_split: validation +doc_to_target: label +doc_to_choice: + - "business" + - "entertainment" + - "health" + - "politics" + - "religion" + - "sports" + - "technology" +should_decontaminate: true +doc_to_decontamination_query: headline_text +metric_list: + - metric: f1 + aggregation: !function utils.weighted_f1_score + # aggregation: mean + average: weighted + hf_evaluate: true + higher_is_better: True + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" + - metric: acc + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_5/masakhanews_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_5/masakhanews_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..759ce913fe968c78eed1f302719b61dd0d62aa2a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_5/masakhanews_amh.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: amh +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Amharic text. For each input, classify the topic as technology, business, politics,\ + \ sports, health, entertainment, or religion. Use the following guidelines: \n\n\ + \ technology: The text discusses scientific discoveries, technological advancements,\ + \ or related topics. \npolitics: The text covers political events, policies, or\ + \ related topics. \nsports: The text talks about sports events, athletes, or related\ + \ topics. \nhealth: The text addresses health issues, medical advancements, or related\ + \ topics. \nentertainment: The text pertains to movies, music, celebrities, or related\ + \ topics. \nreligion: The text talks about relgions, religious institutions and\ + \ beliefs or related topics. \n\nbusiness: The text covers economy, business, or\ + \ related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{headline_text}} \\category: \n\n" +include: masakhanews +task: masakhanews_amh_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_5/masakhanews_eng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_5/masakhanews_eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2c03032b48ecf521e9563277d2b149c703171321 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_5/masakhanews_eng.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: eng +doc_to_text: "You are tasked with performing topic classification on the following\ + \ English text. For each input, classify the topic as technology, business, politics,\ + \ sports, health, entertainment, or religion. Use the following guidelines: \n\n\ + \ technology: The text discusses scientific discoveries, technological advancements,\ + \ or related topics. \npolitics: The text covers political events, policies, or\ + \ related topics. \nsports: The text talks about sports events, athletes, or related\ + \ topics. \nhealth: The text addresses health issues, medical advancements, or related\ + \ topics. \nentertainment: The text pertains to movies, music, celebrities, or related\ + \ topics. \nreligion: The text talks about relgions, religious institutions and\ + \ beliefs or related topics. \n\nbusiness: The text covers economy, business, or\ + \ related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{headline_text}} \\category: \n\n" +include: masakhanews +task: masakhanews_eng_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_5/masakhanews_fra.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_5/masakhanews_fra.yaml new file mode 100644 index 0000000000000000000000000000000000000000..603d149d733336355b4874a5bbffe61786a9edd7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_5/masakhanews_fra.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: fra +doc_to_text: "You are tasked with performing topic classification on the following\ + \ French text. For each input, classify the topic as technology, business, politics,\ + \ sports, health, entertainment, or religion. Use the following guidelines: \n\n\ + \ technology: The text discusses scientific discoveries, technological advancements,\ + \ or related topics. \npolitics: The text covers political events, policies, or\ + \ related topics. \nsports: The text talks about sports events, athletes, or related\ + \ topics. \nhealth: The text addresses health issues, medical advancements, or related\ + \ topics. \nentertainment: The text pertains to movies, music, celebrities, or related\ + \ topics. \nreligion: The text talks about relgions, religious institutions and\ + \ beliefs or related topics. \n\nbusiness: The text covers economy, business, or\ + \ related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{headline_text}} \\category: \n\n" +include: masakhanews +task: masakhanews_fra_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_5/masakhanews_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_5/masakhanews_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..04a478cf6a5b63269c1ef2ef061d50fd08f95c11 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_5/masakhanews_hau.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: hau +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Hausa text. For each input, classify the topic as technology, business, politics,\ + \ sports, health, entertainment, or religion. Use the following guidelines: \n\n\ + \ technology: The text discusses scientific discoveries, technological advancements,\ + \ or related topics. \npolitics: The text covers political events, policies, or\ + \ related topics. \nsports: The text talks about sports events, athletes, or related\ + \ topics. \nhealth: The text addresses health issues, medical advancements, or related\ + \ topics. \nentertainment: The text pertains to movies, music, celebrities, or related\ + \ topics. \nreligion: The text talks about relgions, religious institutions and\ + \ beliefs or related topics. \n\nbusiness: The text covers economy, business, or\ + \ related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{headline_text}} \\category: \n\n" +include: masakhanews +task: masakhanews_hau_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_5/masakhanews_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_5/masakhanews_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ce3cc15b942e3cdbf02fa8b885aa2b796b409544 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_5/masakhanews_ibo.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: ibo +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Igbo text. For each input, classify the topic as technology, business, politics,\ + \ sports, health, entertainment, or religion. Use the following guidelines: \n\n\ + \ technology: The text discusses scientific discoveries, technological advancements,\ + \ or related topics. \npolitics: The text covers political events, policies, or\ + \ related topics. \nsports: The text talks about sports events, athletes, or related\ + \ topics. \nhealth: The text addresses health issues, medical advancements, or related\ + \ topics. \nentertainment: The text pertains to movies, music, celebrities, or related\ + \ topics. \nreligion: The text talks about relgions, religious institutions and\ + \ beliefs or related topics. \n\nbusiness: The text covers economy, business, or\ + \ related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{headline_text}} \\category: \n\n" +include: masakhanews +task: masakhanews_ibo_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_5/masakhanews_lin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_5/masakhanews_lin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e22303fe79bca34cef9784b7ecea4fe1d1a39ab7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_5/masakhanews_lin.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: lin +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Lingala text. For each input, classify the topic as technology, business, politics,\ + \ sports, health, entertainment, or religion. Use the following guidelines: \n\n\ + \ technology: The text discusses scientific discoveries, technological advancements,\ + \ or related topics. \npolitics: The text covers political events, policies, or\ + \ related topics. \nsports: The text talks about sports events, athletes, or related\ + \ topics. \nhealth: The text addresses health issues, medical advancements, or related\ + \ topics. \nentertainment: The text pertains to movies, music, celebrities, or related\ + \ topics. \nreligion: The text talks about relgions, religious institutions and\ + \ beliefs or related topics. \n\nbusiness: The text covers economy, business, or\ + \ related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{headline_text}} \\category: \n\n" +include: masakhanews +task: masakhanews_lin_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_5/masakhanews_lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_5/masakhanews_lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fe949b6f3c7c60a48d72ecf58d47aac7d36cb130 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_5/masakhanews_lug.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: lug +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Luganda text. For each input, classify the topic as technology, business, politics,\ + \ sports, health, entertainment, or religion. Use the following guidelines: \n\n\ + \ technology: The text discusses scientific discoveries, technological advancements,\ + \ or related topics. \npolitics: The text covers political events, policies, or\ + \ related topics. \nsports: The text talks about sports events, athletes, or related\ + \ topics. \nhealth: The text addresses health issues, medical advancements, or related\ + \ topics. \nentertainment: The text pertains to movies, music, celebrities, or related\ + \ topics. \nreligion: The text talks about relgions, religious institutions and\ + \ beliefs or related topics. \n\nbusiness: The text covers economy, business, or\ + \ related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{headline_text}} \\category: \n\n" +include: masakhanews +task: masakhanews_lug_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_5/masakhanews_orm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_5/masakhanews_orm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..413e88125dc1e69cb4aac285ed8f8fc59b887bfe --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_5/masakhanews_orm.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: orm +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Afaan Oromoo text. For each input, classify the topic as technology, business,\ + \ politics, sports, health, entertainment, or religion. Use the following guidelines:\ + \ \n\n technology: The text discusses scientific discoveries, technological advancements,\ + \ or related topics. \npolitics: The text covers political events, policies, or\ + \ related topics. \nsports: The text talks about sports events, athletes, or related\ + \ topics. \nhealth: The text addresses health issues, medical advancements, or related\ + \ topics. \nentertainment: The text pertains to movies, music, celebrities, or related\ + \ topics. \nreligion: The text talks about relgions, religious institutions and\ + \ beliefs or related topics. \n\nbusiness: The text covers economy, business, or\ + \ related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{headline_text}} \\category: \n\n" +include: masakhanews +task: masakhanews_orm_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_5/masakhanews_pcm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_5/masakhanews_pcm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b9322857eaaa71c77ed2399e23470b8085ebd4f5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_5/masakhanews_pcm.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: pcm +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Nigerian Pidgin text. For each input, classify the topic as technology, business,\ + \ politics, sports, health, entertainment, or religion. Use the following guidelines:\ + \ \n\n technology: The text discusses scientific discoveries, technological advancements,\ + \ or related topics. \npolitics: The text covers political events, policies, or\ + \ related topics. \nsports: The text talks about sports events, athletes, or related\ + \ topics. \nhealth: The text addresses health issues, medical advancements, or related\ + \ topics. \nentertainment: The text pertains to movies, music, celebrities, or related\ + \ topics. \nreligion: The text talks about relgions, religious institutions and\ + \ beliefs or related topics. \n\nbusiness: The text covers economy, business, or\ + \ related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{headline_text}} \\category: \n\n" +include: masakhanews +task: masakhanews_pcm_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_5/masakhanews_run.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_5/masakhanews_run.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6f207fb703debacf6655975485fe188b13f4313d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_5/masakhanews_run.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: run +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Kirundi text. For each input, classify the topic as technology, business, politics,\ + \ sports, health, entertainment, or religion. Use the following guidelines: \n\n\ + \ technology: The text discusses scientific discoveries, technological advancements,\ + \ or related topics. \npolitics: The text covers political events, policies, or\ + \ related topics. \nsports: The text talks about sports events, athletes, or related\ + \ topics. \nhealth: The text addresses health issues, medical advancements, or related\ + \ topics. \nentertainment: The text pertains to movies, music, celebrities, or related\ + \ topics. \nreligion: The text talks about relgions, religious institutions and\ + \ beliefs or related topics. \n\nbusiness: The text covers economy, business, or\ + \ related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{headline_text}} \\category: \n\n" +include: masakhanews +task: masakhanews_run_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_5/masakhanews_sna.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_5/masakhanews_sna.yaml new file mode 100644 index 0000000000000000000000000000000000000000..737d335e6df283c7bf9f81c186c6e90f0cd81991 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_5/masakhanews_sna.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: sna +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Shona text. For each input, classify the topic as technology, business, politics,\ + \ sports, health, entertainment, or religion. Use the following guidelines: \n\n\ + \ technology: The text discusses scientific discoveries, technological advancements,\ + \ or related topics. \npolitics: The text covers political events, policies, or\ + \ related topics. \nsports: The text talks about sports events, athletes, or related\ + \ topics. \nhealth: The text addresses health issues, medical advancements, or related\ + \ topics. \nentertainment: The text pertains to movies, music, celebrities, or related\ + \ topics. \nreligion: The text talks about relgions, religious institutions and\ + \ beliefs or related topics. \n\nbusiness: The text covers economy, business, or\ + \ related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{headline_text}} \\category: \n\n" +include: masakhanews +task: masakhanews_sna_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_5/masakhanews_som.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_5/masakhanews_som.yaml new file mode 100644 index 0000000000000000000000000000000000000000..39bb80c47bd7f16e34ec4ebefbdea0a08e6a6bef --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_5/masakhanews_som.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: som +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Somali text. For each input, classify the topic as technology, business, politics,\ + \ sports, health, entertainment, or religion. Use the following guidelines: \n\n\ + \ technology: The text discusses scientific discoveries, technological advancements,\ + \ or related topics. \npolitics: The text covers political events, policies, or\ + \ related topics. \nsports: The text talks about sports events, athletes, or related\ + \ topics. \nhealth: The text addresses health issues, medical advancements, or related\ + \ topics. \nentertainment: The text pertains to movies, music, celebrities, or related\ + \ topics. \nreligion: The text talks about relgions, religious institutions and\ + \ beliefs or related topics. \n\nbusiness: The text covers economy, business, or\ + \ related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{headline_text}} \\category: \n\n" +include: masakhanews +task: masakhanews_som_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_5/masakhanews_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_5/masakhanews_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c59e359c21af54c0f9e78950925fb16e2e0e5b29 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_5/masakhanews_swa.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: swa +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Swahili text. For each input, classify the topic as technology, business, politics,\ + \ sports, health, entertainment, or religion. Use the following guidelines: \n\n\ + \ technology: The text discusses scientific discoveries, technological advancements,\ + \ or related topics. \npolitics: The text covers political events, policies, or\ + \ related topics. \nsports: The text talks about sports events, athletes, or related\ + \ topics. \nhealth: The text addresses health issues, medical advancements, or related\ + \ topics. \nentertainment: The text pertains to movies, music, celebrities, or related\ + \ topics. \nreligion: The text talks about relgions, religious institutions and\ + \ beliefs or related topics. \n\nbusiness: The text covers economy, business, or\ + \ related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{headline_text}} \\category: \n\n" +include: masakhanews +task: masakhanews_swa_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_5/masakhanews_tir.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_5/masakhanews_tir.yaml new file mode 100644 index 0000000000000000000000000000000000000000..959de7a803f556ecde803744c6dc1451ca4493d0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_5/masakhanews_tir.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: tir +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Tigrinya text. For each input, classify the topic as technology, business, politics,\ + \ sports, health, entertainment, or religion. Use the following guidelines: \n\n\ + \ technology: The text discusses scientific discoveries, technological advancements,\ + \ or related topics. \npolitics: The text covers political events, policies, or\ + \ related topics. \nsports: The text talks about sports events, athletes, or related\ + \ topics. \nhealth: The text addresses health issues, medical advancements, or related\ + \ topics. \nentertainment: The text pertains to movies, music, celebrities, or related\ + \ topics. \nreligion: The text talks about relgions, religious institutions and\ + \ beliefs or related topics. \n\nbusiness: The text covers economy, business, or\ + \ related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{headline_text}} \\category: \n\n" +include: masakhanews +task: masakhanews_tir_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_5/masakhanews_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_5/masakhanews_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..35cad7295a830661882c30930153700936817082 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_5/masakhanews_xho.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: xho +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Xhosa text. For each input, classify the topic as technology, business, politics,\ + \ sports, health, entertainment, or religion. Use the following guidelines: \n\n\ + \ technology: The text discusses scientific discoveries, technological advancements,\ + \ or related topics. \npolitics: The text covers political events, policies, or\ + \ related topics. \nsports: The text talks about sports events, athletes, or related\ + \ topics. \nhealth: The text addresses health issues, medical advancements, or related\ + \ topics. \nentertainment: The text pertains to movies, music, celebrities, or related\ + \ topics. \nreligion: The text talks about relgions, religious institutions and\ + \ beliefs or related topics. \n\nbusiness: The text covers economy, business, or\ + \ related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{headline_text}} \\category: \n\n" +include: masakhanews +task: masakhanews_xho_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_5/masakhanews_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_5/masakhanews_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e83c70454d5cd330383f49a7fa3bbaf0f1226790 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_5/masakhanews_yor.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: yor +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Yoruba text. For each input, classify the topic as technology, business, politics,\ + \ sports, health, entertainment, or religion. Use the following guidelines: \n\n\ + \ technology: The text discusses scientific discoveries, technological advancements,\ + \ or related topics. \npolitics: The text covers political events, policies, or\ + \ related topics. \nsports: The text talks about sports events, athletes, or related\ + \ topics. \nhealth: The text addresses health issues, medical advancements, or related\ + \ topics. \nentertainment: The text pertains to movies, music, celebrities, or related\ + \ topics. \nreligion: The text talks about relgions, religious institutions and\ + \ beliefs or related topics. \n\nbusiness: The text covers economy, business, or\ + \ related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{headline_text}} \\category: \n\n" +include: masakhanews +task: masakhanews_yor_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_5/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_5/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..3e735e2deb1f9c53152c072615aebe8ba3acb90b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/prompt_5/utils.py @@ -0,0 +1 @@ +from lm_eval.utils import weighted_f1_score diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..310a7aeb5af2b998d57c6a793f27b00c8ab04029 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhanews/utils.py @@ -0,0 +1,127 @@ +import argparse +import os + +import yaml + + +def prompt_func(mode, lang): + prompt_map = { + "prompt_1": "Given the categories technology, business, politics, sports, health, entertainment, or religion; what category does the text: '{{headline}}' belong to: \n\n", + "prompt_2": f"Does this {lang} topic; " + "'{{headline}}' belong to one of the following categories: technology, business, politics, sports, health, entertainment, or religion? category only\n\n", + "prompt_3": f"You are an assistant able to classify topics in texts. \n\n" + f"Given the categories technology, religion, politics, sports, health, entertainment, or business; what is " + f"the topic of the {lang} statement below? Return only the category. " + "\n\ntext: {{headline}} \category:\n\n", + "prompt_4": "Label the following text as technology, religion, politics, sports, health, entertainment, or geography. Provide only the category as your " + "response. \n\ntext: {{headline}} \category: \n\n", + "prompt_5": f"You are tasked with performing topic classification on the following {lang} text. " + f"For each input, classify the topic as technology, business, politics, sports, health, entertainment, or religion. " + f"Use the following guidelines: \n\n " + f"technology: The text discusses scientific discoveries, technological advancements, or related topics. \n" + f"politics: The text covers political events, policies, or related topics. \n" + f"sports: The text talks about sports events, athletes, or related topics. \n" + f"health: The text addresses health issues, medical advancements, or related topics. \n" + f"entertainment: The text pertains to movies, music, celebrities, or related topics. \n" + f"religion: The text talks about relgions, religious institutions and beliefs or related topics. \n\n" + f"business: The text covers economy, business, or related topics. \n\n" + f"If the text contains multiple topics, choose the dominant topic. " + f"For ambiguous or unclear topics, select the category that best reflects the overall content. " + "Please provide a single classification for each input.\n\ntext: {{headline}} \category: \n\n", + } + return prompt_map[mode] + + +def gen_lang_yamls(output_dir: str, overwrite: bool, mode: str) -> None: + """ + Generate a yaml file for each language. + + :param output_dir: The directory to output the files to. + :param overwrite: Whether to overwrite files if they already exist. + """ + err = [] + languages = { + "amh": "Amharic", + "eng": "English", + "fra": "French", + "hau": "Hausa", + "ibo": "Igbo", + "lin": "Lingala", + "lug": "Luganda", + "orm": "Afaan Oromoo", + "pcm": "Nigerian Pidgin", + "run": "Kirundi", + "sna": "Shona", + "som": "Somali", + "swa": "Swahili", + "tir": "Tigrinya", + "xho": "Xhosa", + "yor": "Yoruba", + } + + for lang in languages.keys(): + try: + file_name = f"masakhanews_{lang}.yaml" + task_name = f"masakhanews_{lang}_{mode}" + yaml_template = "masakhanews" + yaml_details = { + "include": yaml_template, + "task": task_name, + "dataset_name": lang, + "doc_to_text": prompt_func(mode, languages[lang]), + } + file_path = os.path.join(output_dir, mode) + os.makedirs(file_path, exist_ok=True) + + with open( + f"{output_dir}/{mode}/{file_name}", + "w" if overwrite else "x", + encoding="utf8", + ) as f: + f.write("# Generated by utils.py\n") + yaml.dump( + yaml_details, + f, + allow_unicode=True, + ) + except FileExistsError: + err.append(file_name) + + if len(err) > 0: + raise FileExistsError( + "Files were not created because they already exist (use --overwrite flag):" + f" {', '.join(err)}" + ) + + +def main() -> None: + """Parse CLI args and generate language-specific yaml files.""" + parser = argparse.ArgumentParser() + parser.add_argument( + "--overwrite", + default=True, + action="store_true", + help="Overwrite files if they already exist", + ) + parser.add_argument( + "--output-dir", + default="./", + help="Directory to write yaml files to", + ) + + PROMPT_CHOICES = ["prompt_1", "prompt_2", "prompt_3", "prompt_4", "prompt_5"] + parser.add_argument( + "--mode", + nargs="*", + default=PROMPT_CHOICES, + choices=PROMPT_CHOICES, + help="Prompt number(s)", + ) + args = parser.parse_args() + + for mode in args.mode: + gen_lang_yamls(output_dir=args.output_dir, overwrite=args.overwrite, mode=mode) + + +if __name__ == "__main__": + main() diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/README.md b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/README.md new file mode 100644 index 0000000000000000000000000000000000000000..1fcf11c780e88864fef93b46ef536cc11f33e60b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/README.md @@ -0,0 +1,75 @@ +# + +## Paper +Title: `MasakhaPOS: Part-of-Speech Tagging for Typologically Diverse African languages` + +Paper Link: https://aclanthology.org/2023.acl-long.609/ + +## Abstract +>In this paper, we present AfricaPOS, the largest part-of-speech (POS) dataset for 20 typologically diverse African languages. We discuss the challenges in annotating POS for these languages using the universal dependencies (UD) guidelines. We conducted extensive POS baseline experiments using both conditional random field and several multilingual pre-trained language models. We applied various cross-lingual transfer models trained with data available in the UD. Evaluating on the AfricaPOS dataset, we show that choosing the best transfer language(s) in both single-source and multi-source setups greatly improves the POS tagging performance of the target languages, in particular when combined with parameter-fine-tuning methods. Crucially, transferring knowledge from a language that matches the language family and morphosyntactic properties seems to be more effective for POS tagging in unseen languages. + +HomePage: https://github.com/masakhane-io/masakhane-pos + +### Citation + +``` +@inproceedings{dione-etal-2023-masakhapos, + title = "{M}asakha{POS}: Part-of-Speech Tagging for Typologically Diverse {A}frican languages", + author = "Dione, Cheikh M. Bamba and + Adelani, David Ifeoluwa and + Nabende, Peter and + Alabi, Jesujoba and + Sindane, Thapelo and + Buzaaba, Happy and + Muhammad, Shamsuddeen Hassan and + Emezue, Chris Chinenye and + Ogayo, Perez and + Aremu, Anuoluwapo and + Gitau, Catherine and + Mbaye, Derguene and + Mukiibi, Jonathan and + Sibanda, Blessing and + Dossou, Bonaventure F. P. and + Bukula, Andiswa and + Mabuya, Rooweither and + Tapo, Allahsera Auguste and + Munkoh-Buabeng, Edwin and + Memdjokam Koagne, Victoire and + Ouoba Kabore, Fatoumata and + Taylor, Amelia and + Kalipe, Godson and + Macucwa, Tebogo and + Marivate, Vukosi and + Gwadabe, Tajuddeen and + Elvis, Mboning Tchiaze and + Onyenwe, Ikechukwu and + Atindogbe, Gratien and + Adelani, Tolulope and + Akinade, Idris and + Samuel, Olanrewaju and + Nahimana, Marien and + Musabeyezu, Th{\'e}og{\`e}ne and + Niyomutabazi, Emile and + Chimhenga, Ester and + Gotosa, Kudzai and + Mizha, Patrick and + Agbolo, Apelete and + Traore, Seydou and + Uchechukwu, Chinedu and + Yusuf, Aliyu and + Abdullahi, Muhammad and + Klakow, Dietrich", + editor = "Rogers, Anna and + Boyd-Graber, Jordan and + Okazaki, Naoaki", + booktitle = "Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)", + month = jul, + year = "2023", + address = "Toronto, Canada", + publisher = "Association for Computational Linguistics", + url = "https://aclanthology.org/2023.acl-long.609/", + doi = "10.18653/v1/2023.acl-long.609", + pages = "10883--10900", + abstract = "In this paper, we present AfricaPOS, the largest part-of-speech (POS) dataset for 20 typologically diverse African languages. We discuss the challenges in annotating POS for these languages using the universal dependencies (UD) guidelines. We conducted extensive POS baseline experiments using both conditional random field and several multilingual pre-trained language models. We applied various cross-lingual transfer models trained with data available in the UD. Evaluating on the AfricaPOS dataset, we show that choosing the best transfer language(s) in both single-source and multi-source setups greatly improves the POS tagging performance of the target languages, in particular when combined with parameter-fine-tuning methods. Crucially, transferring knowledge from a language that matches the language family and morphosyntactic properties seems to be more effective for POS tagging in unseen languages." +} +``` diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/gen_utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/gen_utils.py new file mode 100644 index 0000000000000000000000000000000000000000..52b9dafb435cf5f24664d7fb9c8ba73a687a7d4c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/gen_utils.py @@ -0,0 +1,151 @@ +import argparse +import os + +import yaml + + +class FunctionTag: + def __init__(self, value): + self.value = value + + +def prompt_func(mode, lang): + prompt_map = { + "prompt_1": "Please provide the POS tags for each word in the input sentence. The input will be a list of " + "words in the sentence. The output format should be a list of tuples, where each tuple consists of " + "a word from the input text and its corresponding POS tag label from the tag label set: ['ADJ', " + "'ADP', 'ADV', 'AUX', 'CCONJ, 'DET', 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', " + "'SCONJ', 'SYM', 'VERB', 'X']. \nYour response should include only a list of tuples, in the order " + "that the words appear in the input sentence, including punctuations, with each tuple containing the corresponding POS tag " + "label for a word. \n\nSentence: {{tokens}} \nOutput: ", + "prompt_2": f"You are an expert in tagging words and sentences in {lang} with the right POS tag. " + f"\n\nPlease provide the POS tags for each word in the {lang} sentence. The input is a list of words in" + " the sentence. POS tag label set: ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET', 'INTJ', 'NOUN', " + "'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM', 'VERB', 'X']. The output format should " + "be a list of tuples, where each tuple consists of a word from the input text and its corresponding" + " POS tag label from the POS tag label set provided\nYour response should include only a list of " + "tuples, in the order that the words appear in the input sentence, including punctuations, with each tuple containing the " + "corresponding POS tag label for a word. \n\nSentence: {{tokens}} \nOutput: ", + "prompt_3": f"Acting as a {lang} linguist and without making any corrections or changes to the text, perform a part of " + "speech (POS) analysis of the sentences using the following POS tag label annotation ['ADJ', " + "'ADP', 'ADV', 'AUX', 'CCONJ, 'DET', 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', " + "'SCONJ', 'SYM', 'VERB', 'X']. The input will be a list of words in the sentence. The output format should " + "be a list of tuples, where each tuple consists of a word from the input text and its corresponding" + " POS tag label from the POS tag label set provided\nYour response should include only a list of " + "tuples, in the order that the words appear in the input sentence, including punctuations, with each tuple containing the " + "corresponding POS tag label for a word. \n\nSentence: {{tokens}} \nOutput: ", + "prompt_4": "Annotate each word in the provided sentence with the appropriate POS tag. The annotation " + "list is given as: ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET', 'INTJ', 'NOUN', 'NUM', 'PART', " + "'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM', 'VERB', 'X']. The input sentence will be a list of words" + " in the sentence. The output format should " + "be a list of tuples, where each tuple consists of a word from the input text and its corresponding" + " POS tag label from the POS tag label set provided\nYour response should include only a list of " + "tuples, in the order that the words appear in the input sentence, including punctuations, with each tuple containing the " + "corresponding POS tag label for a word. \n\nSentence: {{tokens}} \nOutput: ", + "prompt_5": "Given the following sentence, identify the part of speech (POS) for each word. Use the following " + "POS tag set: \nNOUN: Noun (person, place, thing), \nVERB: Verb (action, state), " + "\nADJ: Adjective (describes a noun), \nADV: Adverb (modifies a verb, adjective, or adverb), " + "\nPRON: Pronoun (replaces a noun), \nDET: Determiner (introduces a noun), " + "\nADP: Adposition (preposition or postposition), \nCCONJ: Conjunction (connects words, phrases, clauses)" + "\nPUNCT: Punctuation, \nPROPN: Proper Noun, \nAUX: Auxiliary verb (helper verb), " + "\nSCONJ: Subordinating conjunction \nPART: Particle, \nSYM: Symbol, \nINTJ: Interjection, " + "\nNUM: Numeral, \nX: others. The output format should " + "be a list of tuples, where each tuple consists of a word from the input text and its corresponding" + " POS tag label key only from the POS tag set provided\nYour response should include only a list of " + "tuples, in the order that the words appear in the input sentence, including punctuations, with each tuple containing the " + "corresponding POS tag label for a word. \n\nSentence: {{tokens}} \nOutput: ", + } + return prompt_map[mode] + + +def gen_lang_yamls(output_dir: str, overwrite: bool, mode: str) -> None: + """ + Generate a yaml file for each language. + + :param output_dir: The directory to output the files to. + :param overwrite: Whether to overwrite files if they already exist. + """ + err = [] + languages = { + "bam": "Bambara", + "bbj": "Ghomala", + "ewe": "Ewe", + "fon": "Fon", + "hau": "Hausa", + "ibo": "Igbo", + "kin": "Kinyarwanda", + "lug": "Luganda", + "luo": "Dholuo", + "mos": "Mossi", + "nya": "Chichewa", + "pcm": "Nigerian Pidgin", + "sna": "chiShona", + "swa": "Kiswahili", + "tsn": "Setswana", + "twi": "Twi", + "wol": "Wolof", + "xho": "isiXhosa", + "yor": "Yoruba", + "zul": "isiZulu", + } + + for lang in languages.keys(): + try: + file_name = f"masakhapos_{lang}.yaml" + task_name = f"masakhapos_{lang}_{mode}" + yaml_template = "masakhapos_yaml" + yaml_details = { + "include": yaml_template, + "task": task_name, + "dataset_name": lang, + "doc_to_text": prompt_func(mode, languages[lang]), + } + os.makedirs(f"{output_dir}/{mode}", exist_ok=True) + with open( + f"{output_dir}/{mode}/{file_name}", + "w" if overwrite else "x", + encoding="utf8", + ) as f: + f.write("# Generated by utils.py\n") + yaml.dump( + yaml_details, + f, + allow_unicode=True, + ) + except FileExistsError: + err.append(file_name) + + if len(err) > 0: + raise FileExistsError( + "Files were not created because they already exist (use --overwrite flag):" + f" {', '.join(err)}" + ) + + +def main() -> None: + """Parse CLI args and generate language-specific yaml files.""" + parser = argparse.ArgumentParser() + parser.add_argument( + "--overwrite", + default=True, + action="store_true", + help="Overwrite files if they already exist", + ) + parser.add_argument( + "--output-dir", + default="./", + help="Directory to write yaml files to", + ) + parser.add_argument( + "--mode", + default="prompt_1", + choices=["prompt_1", "prompt_2", "prompt_3", "prompt_4", "prompt_5"], + help="Prompt number", + ) + args = parser.parse_args() + + gen_lang_yamls(output_dir=args.output_dir, overwrite=args.overwrite, mode=args.mode) + + +if __name__ == "__main__": + main() diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/masakhapos.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/masakhapos.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3fb1574eb32a0203198a4d210c788765cf476f34 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/masakhapos.yaml @@ -0,0 +1,13 @@ +group: masakhapos +task: + - masakhapos_prompt_1 + - masakhapos_prompt_2 + - masakhapos_prompt_3 + - masakhapos_prompt_4 + - masakhapos_prompt_5 +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_bam.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_bam.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b1c64e387ae638c83e30b1172f458c3976d20728 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_bam.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: bam +doc_to_text: "Please provide the POS tags for each word in the input sentence. The\ + \ input will be a list of words in the sentence. The output format should be a list\ + \ of tuples, where each tuple consists of a word from the input text and its corresponding\ + \ POS tag label from the tag label set: ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET',\ + \ 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM', 'VERB',\ + \ 'X']. \nYour response should include only a list of tuples, in the order that\ + \ the words appear in the input sentence, including punctuations, with each tuple\ + \ containing the corresponding POS tag label for a word. \n\nSentence: {{tokens}}\ + \ \nOutput: " +include: masakhapos_yaml +task: masakhapos_bam_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_bbj.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_bbj.yaml new file mode 100644 index 0000000000000000000000000000000000000000..418c8e0ca6c411620056f280d51696e730107c2c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_bbj.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: bbj +doc_to_text: "Please provide the POS tags for each word in the input sentence. The\ + \ input will be a list of words in the sentence. The output format should be a list\ + \ of tuples, where each tuple consists of a word from the input text and its corresponding\ + \ POS tag label from the tag label set: ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET',\ + \ 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM', 'VERB',\ + \ 'X']. \nYour response should include only a list of tuples, in the order that\ + \ the words appear in the input sentence, including punctuations, with each tuple\ + \ containing the corresponding POS tag label for a word. \n\nSentence: {{tokens}}\ + \ \nOutput: " +include: masakhapos_yaml +task: masakhapos_bbj_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_ewe.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_ewe.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1eeb249744fc6f75bb7a08896fa0caaacdc1e84d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_ewe.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: ewe +doc_to_text: "Please provide the POS tags for each word in the input sentence. The\ + \ input will be a list of words in the sentence. The output format should be a list\ + \ of tuples, where each tuple consists of a word from the input text and its corresponding\ + \ POS tag label from the tag label set: ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET',\ + \ 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM', 'VERB',\ + \ 'X']. \nYour response should include only a list of tuples, in the order that\ + \ the words appear in the input sentence, including punctuations, with each tuple\ + \ containing the corresponding POS tag label for a word. \n\nSentence: {{tokens}}\ + \ \nOutput: " +include: masakhapos_yaml +task: masakhapos_ewe_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..431ed8f1656568111d4206a7a33c954b51cfa743 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_hau.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: hau +doc_to_text: "Please provide the POS tags for each word in the input sentence. The\ + \ input will be a list of words in the sentence. The output format should be a list\ + \ of tuples, where each tuple consists of a word from the input text and its corresponding\ + \ POS tag label from the tag label set: ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET',\ + \ 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM', 'VERB',\ + \ 'X']. \nYour response should include only a list of tuples, in the order that\ + \ the words appear in the input sentence, including punctuations, with each tuple\ + \ containing the corresponding POS tag label for a word. \n\nSentence: {{tokens}}\ + \ \nOutput: " +include: masakhapos_yaml +task: masakhapos_hau_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0cb171fe3c93c8be6d6bee9b41e6596d769b5deb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_ibo.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: ibo +doc_to_text: "Please provide the POS tags for each word in the input sentence. The\ + \ input will be a list of words in the sentence. The output format should be a list\ + \ of tuples, where each tuple consists of a word from the input text and its corresponding\ + \ POS tag label from the tag label set: ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET',\ + \ 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM', 'VERB',\ + \ 'X']. \nYour response should include only a list of tuples, in the order that\ + \ the words appear in the input sentence, including punctuations, with each tuple\ + \ containing the corresponding POS tag label for a word. \n\nSentence: {{tokens}}\ + \ \nOutput: " +include: masakhapos_yaml +task: masakhapos_ibo_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dced04f22e7424c3d0c4f3a39f4cf58c331f759b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_kin.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: kin +doc_to_text: "Please provide the POS tags for each word in the input sentence. The\ + \ input will be a list of words in the sentence. The output format should be a list\ + \ of tuples, where each tuple consists of a word from the input text and its corresponding\ + \ POS tag label from the tag label set: ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET',\ + \ 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM', 'VERB',\ + \ 'X']. \nYour response should include only a list of tuples, in the order that\ + \ the words appear in the input sentence, including punctuations, with each tuple\ + \ containing the corresponding POS tag label for a word. \n\nSentence: {{tokens}}\ + \ \nOutput: " +include: masakhapos_yaml +task: masakhapos_kin_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e773f6430c0f38d842c63ff8752a78d8a54dd87d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_lug.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: lug +doc_to_text: "Please provide the POS tags for each word in the input sentence. The\ + \ input will be a list of words in the sentence. The output format should be a list\ + \ of tuples, where each tuple consists of a word from the input text and its corresponding\ + \ POS tag label from the tag label set: ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET',\ + \ 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM', 'VERB',\ + \ 'X']. \nYour response should include only a list of tuples, in the order that\ + \ the words appear in the input sentence, including punctuations, with each tuple\ + \ containing the corresponding POS tag label for a word. \n\nSentence: {{tokens}}\ + \ \nOutput: " +include: masakhapos_yaml +task: masakhapos_lug_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_luo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_luo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4544e2b1bce03c8f8fc8d0e82c1c6fbeab6f3570 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_luo.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: luo +doc_to_text: "Please provide the POS tags for each word in the input sentence. The\ + \ input will be a list of words in the sentence. The output format should be a list\ + \ of tuples, where each tuple consists of a word from the input text and its corresponding\ + \ POS tag label from the tag label set: ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET',\ + \ 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM', 'VERB',\ + \ 'X']. \nYour response should include only a list of tuples, in the order that\ + \ the words appear in the input sentence, including punctuations, with each tuple\ + \ containing the corresponding POS tag label for a word. \n\nSentence: {{tokens}}\ + \ \nOutput: " +include: masakhapos_yaml +task: masakhapos_luo_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_mos.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_mos.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b0c7d3f6a3cd272812926812588744bd737dbb51 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_mos.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: mos +doc_to_text: "Please provide the POS tags for each word in the input sentence. The\ + \ input will be a list of words in the sentence. The output format should be a list\ + \ of tuples, where each tuple consists of a word from the input text and its corresponding\ + \ POS tag label from the tag label set: ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET',\ + \ 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM', 'VERB',\ + \ 'X']. \nYour response should include only a list of tuples, in the order that\ + \ the words appear in the input sentence, including punctuations, with each tuple\ + \ containing the corresponding POS tag label for a word. \n\nSentence: {{tokens}}\ + \ \nOutput: " +include: masakhapos_yaml +task: masakhapos_mos_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_nya.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_nya.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f8d4fcbf23feecfaa7ef927dc0a2d9c090370469 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_nya.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: nya +doc_to_text: "Please provide the POS tags for each word in the input sentence. The\ + \ input will be a list of words in the sentence. The output format should be a list\ + \ of tuples, where each tuple consists of a word from the input text and its corresponding\ + \ POS tag label from the tag label set: ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET',\ + \ 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM', 'VERB',\ + \ 'X']. \nYour response should include only a list of tuples, in the order that\ + \ the words appear in the input sentence, including punctuations, with each tuple\ + \ containing the corresponding POS tag label for a word. \n\nSentence: {{tokens}}\ + \ \nOutput: " +include: masakhapos_yaml +task: masakhapos_nya_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_pcm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_pcm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2d05924ee2ba0c702fbb17de84cce0ed03e536bb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_pcm.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: pcm +doc_to_text: "Please provide the POS tags for each word in the input sentence. The\ + \ input will be a list of words in the sentence. The output format should be a list\ + \ of tuples, where each tuple consists of a word from the input text and its corresponding\ + \ POS tag label from the tag label set: ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET',\ + \ 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM', 'VERB',\ + \ 'X']. \nYour response should include only a list of tuples, in the order that\ + \ the words appear in the input sentence, including punctuations, with each tuple\ + \ containing the corresponding POS tag label for a word. \n\nSentence: {{tokens}}\ + \ \nOutput: " +include: masakhapos_yaml +task: masakhapos_pcm_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_sna.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_sna.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7afa02f4f8b72801d5d782165a68694ef41cdc5a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_sna.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: sna +doc_to_text: "Please provide the POS tags for each word in the input sentence. The\ + \ input will be a list of words in the sentence. The output format should be a list\ + \ of tuples, where each tuple consists of a word from the input text and its corresponding\ + \ POS tag label from the tag label set: ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET',\ + \ 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM', 'VERB',\ + \ 'X']. \nYour response should include only a list of tuples, in the order that\ + \ the words appear in the input sentence, including punctuations, with each tuple\ + \ containing the corresponding POS tag label for a word. \n\nSentence: {{tokens}}\ + \ \nOutput: " +include: masakhapos_yaml +task: masakhapos_sna_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ab2f123e1a42759c4f600bb90c4f8450cbc84edf --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_swa.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: swa +doc_to_text: "Please provide the POS tags for each word in the input sentence. The\ + \ input will be a list of words in the sentence. The output format should be a list\ + \ of tuples, where each tuple consists of a word from the input text and its corresponding\ + \ POS tag label from the tag label set: ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET',\ + \ 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM', 'VERB',\ + \ 'X']. \nYour response should include only a list of tuples, in the order that\ + \ the words appear in the input sentence, including punctuations, with each tuple\ + \ containing the corresponding POS tag label for a word. \n\nSentence: {{tokens}}\ + \ \nOutput: " +include: masakhapos_yaml +task: masakhapos_swa_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_tsn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_tsn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ca02f064a837e69871250046a44d5ed63253ec1f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_tsn.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: tsn +doc_to_text: "Please provide the POS tags for each word in the input sentence. The\ + \ input will be a list of words in the sentence. The output format should be a list\ + \ of tuples, where each tuple consists of a word from the input text and its corresponding\ + \ POS tag label from the tag label set: ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET',\ + \ 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM', 'VERB',\ + \ 'X']. \nYour response should include only a list of tuples, in the order that\ + \ the words appear in the input sentence, including punctuations, with each tuple\ + \ containing the corresponding POS tag label for a word. \n\nSentence: {{tokens}}\ + \ \nOutput: " +include: masakhapos_yaml +task: masakhapos_tsn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f22c093639eefc5747c288506d2cb28f90cd6ca6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_twi.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: twi +doc_to_text: "Please provide the POS tags for each word in the input sentence. The\ + \ input will be a list of words in the sentence. The output format should be a list\ + \ of tuples, where each tuple consists of a word from the input text and its corresponding\ + \ POS tag label from the tag label set: ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET',\ + \ 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM', 'VERB',\ + \ 'X']. \nYour response should include only a list of tuples, in the order that\ + \ the words appear in the input sentence, including punctuations, with each tuple\ + \ containing the corresponding POS tag label for a word. \n\nSentence: {{tokens}}\ + \ \nOutput: " +include: masakhapos_yaml +task: masakhapos_twi_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e0bdd23a8a2203243fa657388b7eae8a2be1a28b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_wol.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: wol +doc_to_text: "Please provide the POS tags for each word in the input sentence. The\ + \ input will be a list of words in the sentence. The output format should be a list\ + \ of tuples, where each tuple consists of a word from the input text and its corresponding\ + \ POS tag label from the tag label set: ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET',\ + \ 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM', 'VERB',\ + \ 'X']. \nYour response should include only a list of tuples, in the order that\ + \ the words appear in the input sentence, including punctuations, with each tuple\ + \ containing the corresponding POS tag label for a word. \n\nSentence: {{tokens}}\ + \ \nOutput: " +include: masakhapos_yaml +task: masakhapos_wol_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f712a874546298594bff74f47a85aa67bc5ae23b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_xho.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: xho +doc_to_text: "Please provide the POS tags for each word in the input sentence. The\ + \ input will be a list of words in the sentence. The output format should be a list\ + \ of tuples, where each tuple consists of a word from the input text and its corresponding\ + \ POS tag label from the tag label set: ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET',\ + \ 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM', 'VERB',\ + \ 'X']. \nYour response should include only a list of tuples, in the order that\ + \ the words appear in the input sentence, including punctuations, with each tuple\ + \ containing the corresponding POS tag label for a word. \n\nSentence: {{tokens}}\ + \ \nOutput: " +include: masakhapos_yaml +task: masakhapos_xho_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_yaml new file mode 100644 index 0000000000000000000000000000000000000000..bdca7a85d905f3e177b496b139ed9705f1a3e620 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_yaml @@ -0,0 +1,32 @@ +tag: +- masakhapos_tasks +- masakhapos_prompt_1 +dataset_path: masakhane/masakhapos +dataset_name: null +dataset_kwargs: {trust_remote_code: True} +output_type: generate_until +generation_kwargs: + do_sample: false + until: + - + - <|im_end|> +validation_split: validation +test_split: test +fewshot_split: train +doc_to_target: !function utils.doc_to_target +should_decontaminate: true +doc_to_decontamination_query: "Sentence: {{token}}\nOutput:" +filter_list: + - filter: + - function: regex_pos + name: flexible-extract +metric_list: + - metric: acc + aggregation: !function utils.acc_score + higher_is_better: true + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..efa8750a6200a2be388806f4f8da57f52f781b3c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_yor.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: yor +doc_to_text: "Please provide the POS tags for each word in the input sentence. The\ + \ input will be a list of words in the sentence. The output format should be a list\ + \ of tuples, where each tuple consists of a word from the input text and its corresponding\ + \ POS tag label from the tag label set: ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET',\ + \ 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM', 'VERB',\ + \ 'X']. \nYour response should include only a list of tuples, in the order that\ + \ the words appear in the input sentence, including punctuations, with each tuple\ + \ containing the corresponding POS tag label for a word. \n\nSentence: {{tokens}}\ + \ \nOutput: " +include: masakhapos_yaml +task: masakhapos_yor_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..362c9934b856664dc1ca336d8420b170c5532813 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_zul.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: zul +doc_to_text: "Please provide the POS tags for each word in the input sentence. The\ + \ input will be a list of words in the sentence. The output format should be a list\ + \ of tuples, where each tuple consists of a word from the input text and its corresponding\ + \ POS tag label from the tag label set: ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET',\ + \ 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM', 'VERB',\ + \ 'X']. \nYour response should include only a list of tuples, in the order that\ + \ the words appear in the input sentence, including punctuations, with each tuple\ + \ containing the corresponding POS tag label for a word. \n\nSentence: {{tokens}}\ + \ \nOutput: " +include: masakhapos_yaml +task: masakhapos_zul_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/masakhapos_bam.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/masakhapos_bam.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bde25d7e5c36fa84add36210bf728999f9dafcb2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/masakhapos_bam.yaml @@ -0,0 +1,14 @@ +# Generated by utils.py +dataset_name: bam +doc_to_text: "You are an expert in tagging words and sentences in Bambara with the\ + \ right POS tag. \n\nPlease provide the POS tags for each word in the Bambara sentence.\ + \ The input is a list of words in the sentence. POS tag label set: ['ADJ', 'ADP',\ + \ 'ADV', 'AUX', 'CCONJ, 'DET', 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT',\ + \ 'SCONJ', 'SYM', 'VERB', 'X']. The output format should be a list of tuples, where\ + \ each tuple consists of a word from the input text and its corresponding POS tag\ + \ label from the POS tag label set provided\nYour response should include only a\ + \ list of tuples, in the order that the words appear in the input sentence, including\ + \ punctuations, with each tuple containing the corresponding POS tag label for a\ + \ word. \n\nSentence: {{tokens}} \nOutput: " +include: masakhapos_yaml +task: masakhapos_bam_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/masakhapos_bbj.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/masakhapos_bbj.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8439e6b03f209094e973cf0f9faddfd1a32495b0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/masakhapos_bbj.yaml @@ -0,0 +1,14 @@ +# Generated by utils.py +dataset_name: bbj +doc_to_text: "You are an expert in tagging words and sentences in Ghomala with the\ + \ right POS tag. \n\nPlease provide the POS tags for each word in the Ghomala sentence.\ + \ The input is a list of words in the sentence. POS tag label set: ['ADJ', 'ADP',\ + \ 'ADV', 'AUX', 'CCONJ, 'DET', 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT',\ + \ 'SCONJ', 'SYM', 'VERB', 'X']. The output format should be a list of tuples, where\ + \ each tuple consists of a word from the input text and its corresponding POS tag\ + \ label from the POS tag label set provided\nYour response should include only a\ + \ list of tuples, in the order that the words appear in the input sentence, including\ + \ punctuations, with each tuple containing the corresponding POS tag label for a\ + \ word. \n\nSentence: {{tokens}} \nOutput: " +include: masakhapos_yaml +task: masakhapos_bbj_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/masakhapos_ewe.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/masakhapos_ewe.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5ffa2ba95963fbe4cac38e5a419df3e98b140750 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/masakhapos_ewe.yaml @@ -0,0 +1,14 @@ +# Generated by utils.py +dataset_name: ewe +doc_to_text: "You are an expert in tagging words and sentences in Ewe with the right\ + \ POS tag. \n\nPlease provide the POS tags for each word in the Ewe sentence. The\ + \ input is a list of words in the sentence. POS tag label set: ['ADJ', 'ADP', 'ADV',\ + \ 'AUX', 'CCONJ, 'DET', 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT',\ + \ 'SCONJ', 'SYM', 'VERB', 'X']. The output format should be a list of tuples, where\ + \ each tuple consists of a word from the input text and its corresponding POS tag\ + \ label from the POS tag label set provided\nYour response should include only a\ + \ list of tuples, in the order that the words appear in the input sentence, including\ + \ punctuations, with each tuple containing the corresponding POS tag label for a\ + \ word. \n\nSentence: {{tokens}} \nOutput: " +include: masakhapos_yaml +task: masakhapos_ewe_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/masakhapos_fon.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/masakhapos_fon.yaml new file mode 100644 index 0000000000000000000000000000000000000000..548f2de48255080669b96408c1975eff7958770b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/masakhapos_fon.yaml @@ -0,0 +1,14 @@ +# Generated by utils.py +dataset_name: fon +doc_to_text: "You are an expert in tagging words and sentences in Fon with the right\ + \ POS tag. \n\nPlease provide the POS tags for each word in the Fon sentence. The\ + \ input is a list of words in the sentence. POS tag label set: ['ADJ', 'ADP', 'ADV',\ + \ 'AUX', 'CCONJ, 'DET', 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT',\ + \ 'SCONJ', 'SYM', 'VERB', 'X']. The output format should be a list of tuples, where\ + \ each tuple consists of a word from the input text and its corresponding POS tag\ + \ label from the POS tag label set provided\nYour response should include only a\ + \ list of tuples, in the order that the words appear in the input sentence, including\ + \ punctuations, with each tuple containing the corresponding POS tag label for a\ + \ word. \n\nSentence: {{tokens}} \nOutput: " +include: masakhapos_yaml +task: masakhapos_fon_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/masakhapos_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/masakhapos_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4bc034571803b9fee3f6af8db6e567d64f2a2e61 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/masakhapos_hau.yaml @@ -0,0 +1,14 @@ +# Generated by utils.py +dataset_name: hau +doc_to_text: "You are an expert in tagging words and sentences in Hausa with the right\ + \ POS tag. \n\nPlease provide the POS tags for each word in the Hausa sentence.\ + \ The input is a list of words in the sentence. POS tag label set: ['ADJ', 'ADP',\ + \ 'ADV', 'AUX', 'CCONJ, 'DET', 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT',\ + \ 'SCONJ', 'SYM', 'VERB', 'X']. The output format should be a list of tuples, where\ + \ each tuple consists of a word from the input text and its corresponding POS tag\ + \ label from the POS tag label set provided\nYour response should include only a\ + \ list of tuples, in the order that the words appear in the input sentence, including\ + \ punctuations, with each tuple containing the corresponding POS tag label for a\ + \ word. \n\nSentence: {{tokens}} \nOutput: " +include: masakhapos_yaml +task: masakhapos_hau_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/masakhapos_luo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/masakhapos_luo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..42ccb34fec23488a68562f06ebe2e05811f4e057 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/masakhapos_luo.yaml @@ -0,0 +1,14 @@ +# Generated by utils.py +dataset_name: luo +doc_to_text: "You are an expert in tagging words and sentences in Dholuo with the\ + \ right POS tag. \n\nPlease provide the POS tags for each word in the Dholuo sentence.\ + \ The input is a list of words in the sentence. POS tag label set: ['ADJ', 'ADP',\ + \ 'ADV', 'AUX', 'CCONJ, 'DET', 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT',\ + \ 'SCONJ', 'SYM', 'VERB', 'X']. The output format should be a list of tuples, where\ + \ each tuple consists of a word from the input text and its corresponding POS tag\ + \ label from the POS tag label set provided\nYour response should include only a\ + \ list of tuples, in the order that the words appear in the input sentence, including\ + \ punctuations, with each tuple containing the corresponding POS tag label for a\ + \ word. \n\nSentence: {{tokens}} \nOutput: " +include: masakhapos_yaml +task: masakhapos_luo_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/masakhapos_mos.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/masakhapos_mos.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cfa74aefef204c134d692d17913371137a696a1b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/masakhapos_mos.yaml @@ -0,0 +1,14 @@ +# Generated by utils.py +dataset_name: mos +doc_to_text: "You are an expert in tagging words and sentences in Mossi with the right\ + \ POS tag. \n\nPlease provide the POS tags for each word in the Mossi sentence.\ + \ The input is a list of words in the sentence. POS tag label set: ['ADJ', 'ADP',\ + \ 'ADV', 'AUX', 'CCONJ, 'DET', 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT',\ + \ 'SCONJ', 'SYM', 'VERB', 'X']. The output format should be a list of tuples, where\ + \ each tuple consists of a word from the input text and its corresponding POS tag\ + \ label from the POS tag label set provided\nYour response should include only a\ + \ list of tuples, in the order that the words appear in the input sentence, including\ + \ punctuations, with each tuple containing the corresponding POS tag label for a\ + \ word. \n\nSentence: {{tokens}} \nOutput: " +include: masakhapos_yaml +task: masakhapos_mos_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/masakhapos_nya.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/masakhapos_nya.yaml new file mode 100644 index 0000000000000000000000000000000000000000..27de8386357d493a950920afb895edd9eb689adf --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/masakhapos_nya.yaml @@ -0,0 +1,14 @@ +# Generated by utils.py +dataset_name: nya +doc_to_text: "You are an expert in tagging words and sentences in Chichewa with the\ + \ right POS tag. \n\nPlease provide the POS tags for each word in the Chichewa sentence.\ + \ The input is a list of words in the sentence. POS tag label set: ['ADJ', 'ADP',\ + \ 'ADV', 'AUX', 'CCONJ, 'DET', 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT',\ + \ 'SCONJ', 'SYM', 'VERB', 'X']. The output format should be a list of tuples, where\ + \ each tuple consists of a word from the input text and its corresponding POS tag\ + \ label from the POS tag label set provided\nYour response should include only a\ + \ list of tuples, in the order that the words appear in the input sentence, including\ + \ punctuations, with each tuple containing the corresponding POS tag label for a\ + \ word. \n\nSentence: {{tokens}} \nOutput: " +include: masakhapos_yaml +task: masakhapos_nya_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/masakhapos_pcm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/masakhapos_pcm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0c532569d338696c50b8746c4b1ac9ded2b20d22 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/masakhapos_pcm.yaml @@ -0,0 +1,14 @@ +# Generated by utils.py +dataset_name: pcm +doc_to_text: "You are an expert in tagging words and sentences in Nigerian Pidgin\ + \ with the right POS tag. \n\nPlease provide the POS tags for each word in the Nigerian\ + \ Pidgin sentence. The input is a list of words in the sentence. POS tag label set:\ + \ ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET', 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON',\ + \ 'PROPN', 'PUNCT', 'SCONJ', 'SYM', 'VERB', 'X']. The output format should be a\ + \ list of tuples, where each tuple consists of a word from the input text and its\ + \ corresponding POS tag label from the POS tag label set provided\nYour response\ + \ should include only a list of tuples, in the order that the words appear in the\ + \ input sentence, including punctuations, with each tuple containing the corresponding\ + \ POS tag label for a word. \n\nSentence: {{tokens}} \nOutput: " +include: masakhapos_yaml +task: masakhapos_pcm_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/masakhapos_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/masakhapos_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b1ca8780834ded2c13edc50203f610c1b8147693 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/masakhapos_swa.yaml @@ -0,0 +1,14 @@ +# Generated by utils.py +dataset_name: swa +doc_to_text: "You are an expert in tagging words and sentences in Kiswahili with the\ + \ right POS tag. \n\nPlease provide the POS tags for each word in the Kiswahili\ + \ sentence. The input is a list of words in the sentence. POS tag label set: ['ADJ',\ + \ 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET', 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN',\ + \ 'PUNCT', 'SCONJ', 'SYM', 'VERB', 'X']. The output format should be a list of tuples,\ + \ where each tuple consists of a word from the input text and its corresponding\ + \ POS tag label from the POS tag label set provided\nYour response should include\ + \ only a list of tuples, in the order that the words appear in the input sentence,\ + \ including punctuations, with each tuple containing the corresponding POS tag label\ + \ for a word. \n\nSentence: {{tokens}} \nOutput: " +include: masakhapos_yaml +task: masakhapos_swa_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/masakhapos_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/masakhapos_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..22a6f414cdbd3485cb822a95f8b2a41012174907 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/masakhapos_twi.yaml @@ -0,0 +1,14 @@ +# Generated by utils.py +dataset_name: twi +doc_to_text: "You are an expert in tagging words and sentences in Twi with the right\ + \ POS tag. \n\nPlease provide the POS tags for each word in the Twi sentence. The\ + \ input is a list of words in the sentence. POS tag label set: ['ADJ', 'ADP', 'ADV',\ + \ 'AUX', 'CCONJ, 'DET', 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT',\ + \ 'SCONJ', 'SYM', 'VERB', 'X']. The output format should be a list of tuples, where\ + \ each tuple consists of a word from the input text and its corresponding POS tag\ + \ label from the POS tag label set provided\nYour response should include only a\ + \ list of tuples, in the order that the words appear in the input sentence, including\ + \ punctuations, with each tuple containing the corresponding POS tag label for a\ + \ word. \n\nSentence: {{tokens}} \nOutput: " +include: masakhapos_yaml +task: masakhapos_twi_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/masakhapos_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/masakhapos_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b0d8d8deda904adfe211b7a1b138742ba90c57a6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/masakhapos_xho.yaml @@ -0,0 +1,14 @@ +# Generated by utils.py +dataset_name: xho +doc_to_text: "You are an expert in tagging words and sentences in isiXhosa with the\ + \ right POS tag. \n\nPlease provide the POS tags for each word in the isiXhosa sentence.\ + \ The input is a list of words in the sentence. POS tag label set: ['ADJ', 'ADP',\ + \ 'ADV', 'AUX', 'CCONJ, 'DET', 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT',\ + \ 'SCONJ', 'SYM', 'VERB', 'X']. The output format should be a list of tuples, where\ + \ each tuple consists of a word from the input text and its corresponding POS tag\ + \ label from the POS tag label set provided\nYour response should include only a\ + \ list of tuples, in the order that the words appear in the input sentence, including\ + \ punctuations, with each tuple containing the corresponding POS tag label for a\ + \ word. \n\nSentence: {{tokens}} \nOutput: " +include: masakhapos_yaml +task: masakhapos_xho_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/masakhapos_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/masakhapos_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1a9d1b78326ba004acfd95ba7f1c1682f240cb6e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/masakhapos_yor.yaml @@ -0,0 +1,14 @@ +# Generated by utils.py +dataset_name: yor +doc_to_text: "You are an expert in tagging words and sentences in Yoruba with the\ + \ right POS tag. \n\nPlease provide the POS tags for each word in the Yoruba sentence.\ + \ The input is a list of words in the sentence. POS tag label set: ['ADJ', 'ADP',\ + \ 'ADV', 'AUX', 'CCONJ, 'DET', 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT',\ + \ 'SCONJ', 'SYM', 'VERB', 'X']. The output format should be a list of tuples, where\ + \ each tuple consists of a word from the input text and its corresponding POS tag\ + \ label from the POS tag label set provided\nYour response should include only a\ + \ list of tuples, in the order that the words appear in the input sentence, including\ + \ punctuations, with each tuple containing the corresponding POS tag label for a\ + \ word. \n\nSentence: {{tokens}} \nOutput: " +include: masakhapos_yaml +task: masakhapos_yor_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_3/masakhapos_bam.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_3/masakhapos_bam.yaml new file mode 100644 index 0000000000000000000000000000000000000000..64bf664f58c9c3ebf4a5192c9f84909cfd7e97c1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_3/masakhapos_bam.yaml @@ -0,0 +1,14 @@ +# Generated by utils.py +dataset_name: bam +doc_to_text: "Acting as a Bambara linguist and without making any corrections or changes\ + \ to the text, perform a part of speech (POS) analysis of the sentences using the\ + \ following POS tag label annotation ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET',\ + \ 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM', 'VERB',\ + \ 'X']. The input will be a list of words in the sentence. The output format should\ + \ be a list of tuples, where each tuple consists of a word from the input text and\ + \ its corresponding POS tag label from the POS tag label set provided\nYour response\ + \ should include only a list of tuples, in the order that the words appear in the\ + \ input sentence, including punctuations, with each tuple containing the corresponding\ + \ POS tag label for a word. \n\nSentence: {{tokens}} \nOutput: " +include: masakhapos_yaml +task: masakhapos_bam_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_3/masakhapos_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_3/masakhapos_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..613384cf036ccb0232274c55521c30e27ee039b6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_3/masakhapos_hau.yaml @@ -0,0 +1,14 @@ +# Generated by utils.py +dataset_name: hau +doc_to_text: "Acting as a Hausa linguist and without making any corrections or changes\ + \ to the text, perform a part of speech (POS) analysis of the sentences using the\ + \ following POS tag label annotation ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET',\ + \ 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM', 'VERB',\ + \ 'X']. The input will be a list of words in the sentence. The output format should\ + \ be a list of tuples, where each tuple consists of a word from the input text and\ + \ its corresponding POS tag label from the POS tag label set provided\nYour response\ + \ should include only a list of tuples, in the order that the words appear in the\ + \ input sentence, including punctuations, with each tuple containing the corresponding\ + \ POS tag label for a word. \n\nSentence: {{tokens}} \nOutput: " +include: masakhapos_yaml +task: masakhapos_hau_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_3/masakhapos_nya.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_3/masakhapos_nya.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fecb644d99aa1621c9ac5a6f34bcc87de7f0d377 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_3/masakhapos_nya.yaml @@ -0,0 +1,14 @@ +# Generated by utils.py +dataset_name: nya +doc_to_text: "Acting as a Chichewa linguist and without making any corrections or\ + \ changes to the text, perform a part of speech (POS) analysis of the sentences\ + \ using the following POS tag label annotation ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ,\ + \ 'DET', 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM',\ + \ 'VERB', 'X']. The input will be a list of words in the sentence. The output format\ + \ should be a list of tuples, where each tuple consists of a word from the input\ + \ text and its corresponding POS tag label from the POS tag label set provided\n\ + Your response should include only a list of tuples, in the order that the words\ + \ appear in the input sentence, including punctuations, with each tuple containing\ + \ the corresponding POS tag label for a word. \n\nSentence: {{tokens}} \nOutput: " +include: masakhapos_yaml +task: masakhapos_nya_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_3/masakhapos_tsn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_3/masakhapos_tsn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a37aa2e611c87e94ca1e4444b7e583244c4598b3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_3/masakhapos_tsn.yaml @@ -0,0 +1,14 @@ +# Generated by utils.py +dataset_name: tsn +doc_to_text: "Acting as a Setswana linguist and without making any corrections or\ + \ changes to the text, perform a part of speech (POS) analysis of the sentences\ + \ using the following POS tag label annotation ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ,\ + \ 'DET', 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM',\ + \ 'VERB', 'X']. The input will be a list of words in the sentence. The output format\ + \ should be a list of tuples, where each tuple consists of a word from the input\ + \ text and its corresponding POS tag label from the POS tag label set provided\n\ + Your response should include only a list of tuples, in the order that the words\ + \ appear in the input sentence, including punctuations, with each tuple containing\ + \ the corresponding POS tag label for a word. \n\nSentence: {{tokens}} \nOutput: " +include: masakhapos_yaml +task: masakhapos_tsn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_4/masakhapos_bbj.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_4/masakhapos_bbj.yaml new file mode 100644 index 0000000000000000000000000000000000000000..24680e2dbfb841086a49469a56b25d32e8efa1ef --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_4/masakhapos_bbj.yaml @@ -0,0 +1,13 @@ +# Generated by utils.py +dataset_name: bbj +doc_to_text: "Annotate each word in the provided sentence with the appropriate POS\ + \ tag. The annotation list is given as: ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET',\ + \ 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM', 'VERB',\ + \ 'X']. The input sentence will be a list of words in the sentence. The output format\ + \ should be a list of tuples, where each tuple consists of a word from the input\ + \ text and its corresponding POS tag label from the POS tag label set provided\n\ + Your response should include only a list of tuples, in the order that the words\ + \ appear in the input sentence, including punctuations, with each tuple containing\ + \ the corresponding POS tag label for a word. \n\nSentence: {{tokens}} \nOutput: " +include: masakhapos_yaml +task: masakhapos_bbj_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_5/masakhapos_pcm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_5/masakhapos_pcm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cd5ea9278b841721283b09b5920f8d395674b81f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_5/masakhapos_pcm.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: pcm +doc_to_text: "Given the following sentence, identify the part of speech (POS) for\ + \ each word. Use the following POS tag set: \nNOUN: Noun (person, place, thing),\ + \ \nVERB: Verb (action, state), \nADJ: Adjective (describes a noun), \nADV: Adverb\ + \ (modifies a verb, adjective, or adverb), \nPRON: Pronoun (replaces a noun), \n\ + DET: Determiner (introduces a noun), \nADP: Adposition (preposition or postposition),\ + \ \nCCONJ: Conjunction (connects words, phrases, clauses)\nPUNCT: Punctuation, \n\ + PROPN: Proper Noun, \nAUX: Auxiliary verb (helper verb), \nSCONJ: Subordinating\ + \ conjunction \nPART: Particle, \nSYM: Symbol, \nINTJ: Interjection, \nNUM: Numeral,\ + \ \nX: others. The output format should be a list of tuples, where each tuple consists\ + \ of a word from the input text and its corresponding POS tag label key only from\ + \ the POS tag set provided\nYour response should include only a list of tuples,\ + \ in the order that the words appear in the input sentence, including punctuations,\ + \ with each tuple containing the corresponding POS tag label for a word. \n\nSentence:\ + \ {{tokens}} \nOutput: " +include: masakhapos_yaml +task: masakhapos_pcm_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_5/masakhapos_zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_5/masakhapos_zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4e400bfe74d1505f9335dcd6baf3ff21c949b8b3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_5/masakhapos_zul.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: zul +doc_to_text: "Given the following sentence, identify the part of speech (POS) for\ + \ each word. Use the following POS tag set: \nNOUN: Noun (person, place, thing),\ + \ \nVERB: Verb (action, state), \nADJ: Adjective (describes a noun), \nADV: Adverb\ + \ (modifies a verb, adjective, or adverb), \nPRON: Pronoun (replaces a noun), \n\ + DET: Determiner (introduces a noun), \nADP: Adposition (preposition or postposition),\ + \ \nCCONJ: Conjunction (connects words, phrases, clauses)\nPUNCT: Punctuation, \n\ + PROPN: Proper Noun, \nAUX: Auxiliary verb (helper verb), \nSCONJ: Subordinating\ + \ conjunction \nPART: Particle, \nSYM: Symbol, \nINTJ: Interjection, \nNUM: Numeral,\ + \ \nX: others. The output format should be a list of tuples, where each tuple consists\ + \ of a word from the input text and its corresponding POS tag label key only from\ + \ the POS tag set provided\nYour response should include only a list of tuples,\ + \ in the order that the words appear in the input sentence, including punctuations,\ + \ with each tuple containing the corresponding POS tag label for a word. \n\nSentence:\ + \ {{tokens}} \nOutput: " +include: masakhapos_yaml +task: masakhapos_zul_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..d7976f846c42a3b8d347553cacc97779dea15671 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/utils.py @@ -0,0 +1,40 @@ +from lm_eval.utils import weighted_f1_score + + +def doc_to_text(doc): + output = """Please provide the POS tags for each word in the input sentence. The input will be a list of words in + the sentence. The output format should be a list of tuples, where each tuple consists of a word from the input text + and its corresponding POS tag label from the tag label set: ["ADJ", "ADP", "ADV", "AUX", "CCONJ, "DET", "INTJ", + "NOUN", "NUM", "PART", "PRON", "PROPN", "PUNCT" "SCONJ", "SYM", "VERB", "X"]. \nYour response should include only a + list of tuples, in the order that the words appear in the input sentence, with each tuple containing the + corresponding POS tag label for a word. + + Input: {tokens} + Output: """ + + text = output.format(subject=doc["tokens"]) + return text + + +def doc_to_target(doc): + pos_tag_map = { + 0: "NOUN", + 1: "PUNCT", + 2: "ADP", + 3: "NUM", + 4: "SYM", + 5: "SCONJ", + 6: "ADJ", + 7: "PART", + 8: "DET", + 9: "CCONJ", + 10: "PROPN", + 11: "PRON", + 12: "X", + 13: "_", + 14: "ADV", + 15: "INTJ", + 16: "VERB", + 17: "AUX", + } + return [pos_tag_map[tag] for tag in doc["upos"]] diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/naijarc/prompt_2/naijarc_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/naijarc/prompt_2/naijarc_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8384fad18343389dd8a22a1b7d2ae21e1de0e22e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/naijarc/prompt_2/naijarc_ibo.yaml @@ -0,0 +1,17 @@ +# Generated by utils.py +dataset_name: ibo +doc_to_text: 'Passage: {{story}} + + Question: {{question.strip()}} + + 1: {{options_A}} + + 2: {{options_B}} + + 3: {{options_C}} + + 4: {{options_D}} + + Please select the correct answer from the given choices:' +include: naijarc +task: naijarc_ibo_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/naijarc/prompt_4/naijarc_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/naijarc/prompt_4/naijarc_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..926d7a8f1615a83902e98ff65633e0fd19838d8d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/naijarc/prompt_4/naijarc_ibo.yaml @@ -0,0 +1,21 @@ +# Generated by utils.py +dataset_name: ibo +doc_to_text: '{{story}} + + Based on the above passage, answer the following question: + + {{question.strip()}} + + Choices: + + A) {{options_A}} + + B) {{options_B}} + + C) {{options_C}} + + D) {{options_D}} + + Please provide the correct answer from the choices given:' +include: naijarc +task: naijarc_ibo_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/naijarc/prompt_5/naijarc_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/naijarc/prompt_5/naijarc_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0959e3277d10fb768565622143eee4e9728fd3c1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/naijarc/prompt_5/naijarc_yor.yaml @@ -0,0 +1,19 @@ +# Generated by utils.py +dataset_name: yor +doc_to_text: 'Read the passage: {{story}} + + Then answer the question: {{question.strip()}} + + Options: + + A. {{options_A}} + + B. {{options_B}} + + C. {{options_C}} + + D. {{options_D}} + + Please choose the correct option from the above list:' +include: naijarc +task: naijarc_yor_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/README.md b/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/README.md new file mode 100644 index 0000000000000000000000000000000000000000..fa2413190b57192fe7a4a4250bf9fb41eb5950a3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/README.md @@ -0,0 +1,35 @@ +# + +## Paper +Title: `NollySenti: Leveraging Transfer Learning and Machine Translation for Nigerian Movie Sentiment Classification` + +Paper Link: https://aclanthology.org/2023.acl-short.85/ + +## Abstract +>Africa has over 2000 indigenous languages but they are under-represented in NLP research due to lack of datasets. In recent years, there have been progress in developing labelled corpora for African languages. However, they are often available in a single domain and may not generalize to other domains. In this paper, we focus on the task of sentiment classification for cross-domain adaptation. We create a new dataset, Nollywood movie reviews for five languages widely spoken in Nigeria (English, Hausa, Igbo, Nigerian Pidgin, and Yoruba). We provide an extensive empirical evaluation using classical machine learning methods and pre-trained language models. By leveraging transfer learning, we compare the performance of cross-domain adaptation from Twitter domain, and cross-lingual adaptation from English language. Our evaluation shows that transfer from English in the same target domain leads to more than 5% improvement in accuracy compared to transfer from Twitter in the same language. To further mitigate the domain difference, we leverage machine translation from English to other Nigerian languages, which leads to a further improvement of 7% over cross-lingual evaluation. While machine translation to low-resource languages are often of low quality, our analysis shows that sentiment related words are often preserved. + +HomePage: https://github.com/IyanuSh/NollySenti + +### Citation + +``` +@inproceedings{shode-etal-2023-nollysenti, + title = "{N}olly{S}enti: Leveraging Transfer Learning and Machine Translation for {N}igerian Movie Sentiment Classification", + author = "Shode, Iyanuoluwa and + Adelani, David Ifeoluwa and + Peng, JIng and + Feldman, Anna", + editor = "Rogers, Anna and + Boyd-Graber, Jordan and + Okazaki, Naoaki", + booktitle = "Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics (Volume 2: Short Papers)", + month = jul, + year = "2023", + address = "Toronto, Canada", + publisher = "Association for Computational Linguistics", + url = "https://aclanthology.org/2023.acl-short.85/", + doi = "10.18653/v1/2023.acl-short.85", + pages = "986--998", + abstract = "Africa has over 2000 indigenous languages but they are under-represented in NLP research due to lack of datasets. In recent years, there have been progress in developing labelled corpora for African languages. However, they are often available in a single domain and may not generalize to other domains. In this paper, we focus on the task of sentiment classification for cross-domain adaptation. We create a new dataset, Nollywood movie reviews for five languages widely spoken in Nigeria (English, Hausa, Igbo, Nigerian Pidgin, and Yoruba). We provide an extensive empirical evaluation using classical machine learning methods and pre-trained language models. By leveraging transfer learning, we compare the performance of cross-domain adaptation from Twitter domain, and cross-lingual adaptation from English language. Our evaluation shows that transfer from English in the same target domain leads to more than 5{\%} improvement in accuracy compared to transfer from Twitter in the same language. To further mitigate the domain difference, we leverage machine translation from English to other Nigerian languages, which leads to a further improvement of 7{\%} over cross-lingual evaluation. While machine translation to low-resource languages are often of low quality, our analysis shows that sentiment related words are often preserved." +} +``` diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_1/nollysenti_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_1/nollysenti_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6dc1cfabadc7019a92b7d023982641ac60a0b9c2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_1/nollysenti_yor.yaml @@ -0,0 +1,3 @@ +dataset_name: yo +include: nollysenti +task: nollysenti_yor_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_2/nollysenti_eng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_2/nollysenti_eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ac3bb04d137a207aad2ac307bd2eefc7e5effc2d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_2/nollysenti_eng.yaml @@ -0,0 +1,4 @@ +dataset_name: en +include: nollysenti +doc_to_text: 'Does this English movie description; "{{review}}" have a Positive or Negative sentiment? Labels only\n' +task: nollysenti_eng_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_2/nollysenti_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_2/nollysenti_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f87bce673c68bacdcf3e516bb58c116ada8209e4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_2/nollysenti_hau.yaml @@ -0,0 +1,4 @@ +dataset_name: ha +include: nollysenti +doc_to_text: 'Does this Hausa movie description; "{{review}}" have a Positive or Negative sentiment? Labels only\n' +task: nollysenti_hau_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_2/nollysenti_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_2/nollysenti_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2f7ae185dff1e0108d5d4b6d0bd5fa318c3c182b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_2/nollysenti_ibo.yaml @@ -0,0 +1,4 @@ +dataset_name: ig +include: nollysenti +doc_to_text: 'Does this Igbo movie description; "{{review}}" have a Positive or Negative sentiment? Labels only\n' +task: nollysenti_ibo_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_2/nollysenti_pcm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_2/nollysenti_pcm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b0305c7673fc5f2a527f96205a2b6730efff4db3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_2/nollysenti_pcm.yaml @@ -0,0 +1,4 @@ +dataset_name: pcm +include: nollysenti +doc_to_text: 'Does this Naija Pidgin movie description; "{{review}}" have a Positive or Negative sentiment? Labels only\n' +task: nollysenti_pcm_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_2/nollysenti_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_2/nollysenti_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..03c89d8bd05dec45bfc07f5af8c2dc8ed76388ae --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_2/nollysenti_yor.yaml @@ -0,0 +1,4 @@ +dataset_name: yo +include: nollysenti +doc_to_text: 'Does this Yoruba movie description; "{{review}}" have a Positive or Negative sentiment? Labels only\n' +task: nollysenti_yor_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_2/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_2/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..3e735e2deb1f9c53152c072615aebe8ba3acb90b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_2/utils.py @@ -0,0 +1 @@ +from lm_eval.utils import weighted_f1_score diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_3/nollysenti b/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_3/nollysenti new file mode 100644 index 0000000000000000000000000000000000000000..472928acdc7b964d60fbd0eb992af298319afcc4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_3/nollysenti @@ -0,0 +1,37 @@ +tag: + - afrobench_sentiment_tasks + - nollysenti_prompt_3 +dataset_path: Davlan/nollysenti +dataset_kwargs: {trust_remote_code: True} +output_type: multiple_choice +validation_split: validation +test_split: test +fewshot_split: validation +doc_to_target: label +doc_to_choice: + - "positive" + - "negative" +should_decontaminate: true +doc_to_decontamination_query: review +metric_list: + - metric: f1 + aggregation: !function utils.weighted_f1_score + # aggregation: mean + average: weighted + hf_evaluate: true + higher_is_better: True + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" + - metric: acc + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_3/nollysenti_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_3/nollysenti_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2f6bb7b29581858b860b5919afbab5e5b22ebc28 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_3/nollysenti_ibo.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: ig +doc_to_text: "You are an assistant able to detect sentiment in movie reviews. \n\nGiven\ + \ the sentiment labels Positive or Negative; what is the sentiment of the\ + \ Igbo statement below? Return only the labels\n\nReview: {{review}}\n" +include: nollysenti +task: nollysenti_ibo_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_3/nollysenti_pcm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_3/nollysenti_pcm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f98519f3ed329da73ab2272fd33305670d8f2ec1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_3/nollysenti_pcm.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: pcm +doc_to_text: "You are an assistant able to detect sentiment in movie reviews. \n\nGiven\ + \ the sentiment labels Positive or Negative; what is the sentiment of the\ + \ Naija Pidgin statement below? Return only the labels\n\nReview: {{review}}\n" +include: nollysenti +task: nollysenti_pcm_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_3/nollysenti_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_3/nollysenti_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fd64d1eda4fa7048690527046e71c6af21eb0d51 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_3/nollysenti_yor.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: yo +doc_to_text: "You are an assistant able to detect sentiment in movie reviews. \n\nGiven\ + \ the sentiment labels Positive or Negative; what is the sentiment of the\ + \ Yoruba statement below? Return only the labels\n\nReview: {{review}}\n" +include: nollysenti +task: nollysenti_yor_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_3/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_3/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..3e735e2deb1f9c53152c072615aebe8ba3acb90b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_3/utils.py @@ -0,0 +1 @@ +from lm_eval.utils import weighted_f1_score diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_4/nollysenti b/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_4/nollysenti new file mode 100644 index 0000000000000000000000000000000000000000..de1bb486dc1c84ea828d1cb99deb16af6e3f1644 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_4/nollysenti @@ -0,0 +1,37 @@ +tag: + - afrobench_sentiment_tasks + - nollysenti_prompt_4 +dataset_path: Davlan/nollysenti +dataset_kwargs: {trust_remote_code: True} +output_type: multiple_choice +validation_split: validation +test_split: test +fewshot_split: validation +doc_to_target: label +doc_to_choice: + - "positive" + - "negative" +should_decontaminate: true +doc_to_decontamination_query: review +metric_list: + - metric: f1 + aggregation: !function utils.weighted_f1_score + # aggregation: mean + average: weighted + hf_evaluate: true + higher_is_better: True + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" + - metric: acc + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_4/nollysenti_eng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_4/nollysenti_eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d8e01ab6efb4450b392b7d6278088c7f74114f61 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_4/nollysenti_eng.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: en +doc_to_text: "Label the following text as Positive, or Negative. Provide\ + \ only the label as your response. \n\ntext: {{review}} \nlabel: \n" +include: nollysenti +task: nollysenti_eng_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_4/nollysenti_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_4/nollysenti_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..abc9570484fbd79acebb9ba2b7be840bb9391c4d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_4/nollysenti_hau.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: ha +doc_to_text: "Label the following text as Positive, or Negative. Provide\ + \ only the label as your response. \n\ntext: {{review}} \nlabel: \n" +include: nollysenti +task: nollysenti_hau_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_4/nollysenti_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_4/nollysenti_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8962cf729075203d9c853470791aa15f7eb97023 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_4/nollysenti_ibo.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: ig +doc_to_text: "Label the following text as Positive, or Negative. Provide\ + \ only the label as your response. \n\ntext: {{review}} \nlabel: \n" +include: nollysenti +task: nollysenti_ibo_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_4/nollysenti_pcm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_4/nollysenti_pcm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..36d43b795461972411b56413b1bc11386cc34d78 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_4/nollysenti_pcm.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: pcm +doc_to_text: "Label the following text as Positive, or Negative. Provide\ + \ only the label as your response. \n\ntext: {{review}} \nlabel: \n" +include: nollysenti +task: nollysenti_pcm_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_4/nollysenti_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_4/nollysenti_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2c100c4dd367e2d610e8881d0d7d932c3473f38c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_4/nollysenti_yor.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: yo +doc_to_text: "Label the following text as Positive, or Negative. Provide\ + \ only the label as your response. \n\ntext: {{review}} \nlabel: \n" +include: nollysenti +task: nollysenti_yor_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_4/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_4/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..3e735e2deb1f9c53152c072615aebe8ba3acb90b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_4/utils.py @@ -0,0 +1 @@ +from lm_eval.utils import weighted_f1_score diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_5/nollysenti b/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_5/nollysenti new file mode 100644 index 0000000000000000000000000000000000000000..2e25f2f088edcb81f754f3b7fd7f9a5e92e18b12 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_5/nollysenti @@ -0,0 +1,37 @@ +tag: + - afrobench_sentiment_tasks + - nollysenti_prompt_5 +dataset_path: Davlan/nollysenti +dataset_kwargs: {trust_remote_code: True} +output_type: multiple_choice +validation_split: validation +test_split: test +fewshot_split: validation +doc_to_target: label +doc_to_choice: + - "positive" + - "negative" +should_decontaminate: true +doc_to_decontamination_query: review +metric_list: + - metric: f1 + aggregation: !function utils.weighted_f1_score + # aggregation: mean + average: weighted + hf_evaluate: true + higher_is_better: True + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" + - metric: acc + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_5/nollysenti_eng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_5/nollysenti_eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d485ffe154c61f91924a5c0015e5defeb8ea83a2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_5/nollysenti_eng.yaml @@ -0,0 +1,12 @@ +# Generated by utils.py +dataset_name: en +doc_to_text: "You are tasked with performing sentiment classification on the following\ + \ English text. For each input, classify the sentiment as positive, negative.\ + \ Use the following guidelines: \n\n Positive: The text expresses happiness,\ + \ satisfaction, or optimism. \nNegative: The text conveys disappointment, dissatisfaction,\ + \ or pessimism. \n\nIf the text contains both positive and negative sentiments, choose\ + \ the dominant sentiment. For ambiguous or unclear sentiments, select the label\ + \ that best reflects the overall tone. Please provide a single classification for\ + \ each input.\n\ntext: {{review}} \nlabel: \n" +include: nollysenti +task: nollysenti_eng_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_5/nollysenti_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_5/nollysenti_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7ed16af77a33c39aa1569a38047ef92091837152 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_5/nollysenti_hau.yaml @@ -0,0 +1,12 @@ +# Generated by utils.py +dataset_name: ha +doc_to_text: "You are tasked with performing sentiment classification on the following\ + \ Hausa text. For each input, classify the sentiment as positive, negative.\ + \ Use the following guidelines: \n\n Positive: The text expresses happiness,\ + \ satisfaction, or optimism. \nNegative: The text conveys disappointment, dissatisfaction,\ + \ or pessimism. \n\nIf the text contains both positive and negative sentiments, choose\ + \ the dominant sentiment. For ambiguous or unclear sentiments, select the label\ + \ that best reflects the overall tone. Please provide a single classification for\ + \ each input.\n\ntext: {{review}} \nlabel: \n" +include: nollysenti +task: nollysenti_hau_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_5/nollysenti_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_5/nollysenti_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c75f26900298951c5934b17964ca0cd744d86726 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_5/nollysenti_ibo.yaml @@ -0,0 +1,12 @@ +# Generated by utils.py +dataset_name: ig +doc_to_text: "You are tasked with performing sentiment classification on the following\ + \ Igbo text. For each input, classify the sentiment as positive, negative.\ + \ Use the following guidelines: \n\n Positive: The text expresses happiness,\ + \ satisfaction, or optimism. \nNegative: The text conveys disappointment, dissatisfaction,\ + \ or pessimism. \n\nIf the text contains both positive and negative sentiments, choose\ + \ the dominant sentiment. For ambiguous or unclear sentiments, select the label\ + \ that best reflects the overall tone. Please provide a single classification for\ + \ each input. \n\ntext: {{review}} \nlabel: \n" +include: nollysenti +task: nollysenti_ibo_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_5/nollysenti_pcm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_5/nollysenti_pcm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..29b5cda0b66b083a2cbcdf8d6750d447e7890519 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_5/nollysenti_pcm.yaml @@ -0,0 +1,12 @@ +# Generated by utils.py +dataset_name: pcm +doc_to_text: "You are tasked with performing sentiment classification on the following\ + \ Naija Pidgin text. For each input, classify the sentiment as positive, negative.\ + \ Use the following guidelines: \n\n Positive: The text expresses happiness,\ + \ satisfaction, or optimism. \nNegative: The text conveys disappointment, dissatisfaction,\ + \ or pessimism. \n\nIf the text contains both positive and negative sentiments, choose\ + \ the dominant sentiment. For ambiguous or unclear sentiments, select the label\ + \ that best reflects the overall tone. Please provide a single classification for\ + \ each input. \n\ntext: {{review}} \nlabel: \n" +include: nollysenti +task: nollysenti_pcm_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_5/nollysenti_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_5/nollysenti_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f1aea0284e191356e15db16036a4d1abfbc1c5aa --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_5/nollysenti_yor.yaml @@ -0,0 +1,12 @@ +# Generated by utils.py +dataset_name: yo +doc_to_text: "You are tasked with performing sentiment classification on the following\ + \ Yoruba text. For each input, classify the sentiment as positive, negative.\ + \ Use the following guidelines: \n\n Positive: The text expresses happiness,\ + \ satisfaction, or optimism. \nNegative: The text conveys disappointment, dissatisfaction,\ + \ or pessimism. \n\nIf the text contains both positive and negative sentiments, choose\ + \ the dominant sentiment. For ambiguous or unclear sentiments, select the label\ + \ that best reflects the overall tone. Please provide a single classification for\ + \ each input. \n\ntext: {{review}} \nlabel: \n" +include: nollysenti +task: nollysenti_yor_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_5/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_5/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..3e735e2deb1f9c53152c072615aebe8ba3acb90b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/nollysenti/prompt_5/utils.py @@ -0,0 +1 @@ +from lm_eval.utils import weighted_f1_score diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/README.md b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/README.md new file mode 100644 index 0000000000000000000000000000000000000000..d68cf8c99cb4d7cb8c68eb7d015e6cb26daca3cb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/README.md @@ -0,0 +1,38 @@ +# + +## Paper +Title: `NTREX-128 – News Test References for MT Evaluation of 128 Languages` + +Paper Link: https://aclanthology.org/2022.sumeval-1.4/ + +## Abstract +>We release NTREX-128, a data set for machine translation (MT) evaluation from English into a total of 128 target languages. The paper describes the data creation process and proposes a quality filtering method based on human evaluation. We show experimental results which confirm that the directionality of test sets translation indeed plays an important role wrt. the usefulness of the corresponding metrics’ scores. Thus, we recommend that the NTREX-128 data set should be used for evaluation of Englishsourced translation models but not in reverse direction. The test set release introduces another benchmark for the evaluation of massively multilingual machine translation research. + +HomePage: https://github.com/MicrosoftTranslator/NTREX + +### Citation + +``` +@inproceedings{federmann-etal-2022-ntrex, + title = "{NTREX}-128 {--} News Test References for {MT} Evaluation of 128 Languages", + author = "Federmann, Christian and + Kocmi, Tom and + Xin, Ying", + editor = "Ahuja, Kabir and + Anastasopoulos, Antonios and + Patra, Barun and + Neubig, Graham and + Choudhury, Monojit and + Dandapat, Sandipan and + Sitaram, Sunayana and + Chaudhary, Vishrav", + booktitle = "Proceedings of the First Workshop on Scaling Up Multilingual Evaluation", + month = nov, + year = "2022", + address = "Online", + publisher = "Association for Computational Linguistics", + url = "https://aclanthology.org/2022.sumeval-1.4/", + doi = "10.18653/v1/2022.sumeval-1.4", + pages = "21--24" +} +``` diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/gen_utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/gen_utils.py new file mode 100644 index 0000000000000000000000000000000000000000..ba549de25b69b0892f6e80c923c44f7ca001cd79 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/gen_utils.py @@ -0,0 +1,171 @@ +import argparse +import os + +import yaml + + +class FunctionTag: + def __init__(self, value): + self.value = value + + +def prompt_func(mode, lang, lang_dict): + language_column_name = f"sentence_{lang}" + prompt_map = { + "prompt_1": f"{lang_dict[lang]}: {{{{{language_column_name}}}}} \nEnglish: ", + "prompt_1_reverse": f"English: {{{{sentence_eng_Latn}}}} \n{lang_dict[lang]}: ", + "prompt_2": f"You are a translation expert. Translate the following {lang_dict[lang]} sentences to English \n" + f"{lang_dict[lang]}: {{{{{language_column_name}}}}}\nEnglish: ", + "prompt_2_reverse": f"You are a translation expert. Translate the following English sentences to " + f"{lang_dict[lang]} " + "\nEnglish: {{sentence_eng_Latn}} " + f"\n{lang_dict[lang]}: ", + "prompt_3": f"As a {lang_dict[lang]} and English linguist, translate the following {lang_dict[lang]} sentences " + f"to English \n{lang_dict[lang]}: {{{{{language_column_name}}}}}\nEnglish: ", + "prompt_3_reverse": f"As a {lang_dict[lang]} and English linguist, translate the following English sentences to " + f"{lang_dict[lang]} " + "\nEnglish: {{sentence_eng_Latn}} " + f"\n{lang_dict[lang]}: ", + } + return prompt_map[mode] + + +def gen_lang_yamls(output_dir: str, overwrite: bool, mode: str, reverse: bool) -> None: + """ + Generate a yaml file for each language. + + :param output_dir: The directory to output the files to. + :param overwrite: Whether to overwrite files if they already exist. + """ + err = [] + languages = { + "afr_Latn": "Afrikaans", + "amh_Ethi": "Amharic", + "arb_Arab": "Arabic", + "bem_Latn": "Bemba", + "ewe_Latn": "Ewe", + "fra_Latn": "French", + "hau_Latn": "Hausa", + "ibo_Latn": "Igbo", + "kin_Latn": "Kinyarwanda", + "mey_Arab": "Hassaniya Arabic", + "mlg_Latn": "Malagasy", + "msa_Latn": "Malay", + "nde_Latn": "North Ndebele", + "nso_Latn": "Northern Sotho", + "nya_Latn": "Chichewa", + "orm_Ethi": "Oromo", + "shi_Arab": "Tachelhit", + "sna_Latn": "Shona (Latin)", + "som_Latn": "Somali", + "ssw_Latn": "Swati", + "swa_Latn": "Swahili", + "tam_Taml": "Tamil", + "tel_Telu": "Telugu", + "tir_Ethi": "Tigrinya", + "ton_Latn": "Tongan", + "tsn_Latn": "Tswana", + "urd_Arab": "Urdu", + "ven_Latn": "Venda", + "wol_Latn": "Wolof", + "xho_Latn": "Xhosa", + "yor_Latn": "Yoruba", + "zul_Latn": "Zulu", + } + + for lang in languages.keys(): + try: + if not reverse: + file_name = f"ntrex_{lang}-eng_Latn.yaml" + task_name = f"ntrex_{lang}-eng_Latn_{mode}" + yaml_template = "ntrex" + yaml_details = { + "include": yaml_template, + "dataset_name": f"{lang}", + "task": task_name, + "doc_to_target": "sentence_eng_Latn", + "doc_to_text": prompt_func(mode, lang, languages), + } + os.makedirs(f"{output_dir}/{mode}/african-english", exist_ok=True) + with open( + f"{output_dir}/{mode}/african-english/{file_name}", + "w" if overwrite else "x", + encoding="utf8", + ) as f: + f.write("# Generated by utils.py\n") + yaml.dump( + yaml_details, + f, + allow_unicode=True, + ) + else: + file_name = f"ntrex_eng_Latn-{lang}.yaml" + task_name = f"ntrex_eng_Latn-{lang}_{mode}" + yaml_template = "ntrex" + yaml_details = { + "include": yaml_template, + "dataset_name": f"{lang}", + "task": task_name, + "doc_to_target": f"sentence_{lang}", + "doc_to_text": prompt_func(f"{mode}_reverse", lang, languages), + } + os.makedirs(f"{output_dir}/{mode}/english-african", exist_ok=True) + with open( + f"{output_dir}/{mode}/english-african/{file_name}", + "w" if overwrite else "x", + encoding="utf8", + ) as f: + f.write("# Generated by utils.py\n") + yaml.dump( + yaml_details, + f, + allow_unicode=True, + ) + except FileExistsError: + err.append(file_name) + + if len(err) > 0: + raise FileExistsError( + "Files were not created because they already exist (use --overwrite flag):" + f" {', '.join(err)}" + ) + + +def main() -> None: + """Parse CLI args and generate language-specific yaml files.""" + parser = argparse.ArgumentParser() + parser.add_argument( + "--overwrite", + default=True, + action="store_true", + help="Overwrite files if they already exist", + ) + parser.add_argument( + "--output-dir", + default="./", + help="Directory to write yaml files to", + ) + parser.add_argument( + "--mode", + default="prompt_1", + choices=["prompt_1", "prompt_2", "prompt_3"], + help="Prompt number", + ) + parser.add_argument( + "--reverse", + default=False, + choices=[True, False], + help="Reverse the translation direction", + ) + args = parser.parse_args() + + gen_lang_yamls( + output_dir=args.output_dir, + overwrite=args.overwrite, + mode=args.mode, + reverse=args.reverse, + ) + + +if __name__ == "__main__": + main() diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/ntrex.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/ntrex.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c30b08cea2ffdbf775cfeeb8957c47e9e807518a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/ntrex.yaml @@ -0,0 +1,14 @@ +group: african_ntrex +task: + - ntrex_eng-afr_prompt_1 + - ntrex_eng-afr_prompt_2 + - ntrex_eng-afr_prompt_3 + - ntrex_afr-eng_prompt_1 + - ntrex_afr-eng_prompt_2 + - ntrex_afr-eng_prompt_3 +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex new file mode 100644 index 0000000000000000000000000000000000000000..3c2659d752c9f14412d23f3c1e553fbb03a16b03 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex @@ -0,0 +1,26 @@ +tag: +- ntrex_tasks +- ntrex_afr-eng +- ntrex_afr-eng_prompt_1 +- afrobench_MT_tasks +dataset_path: masakhane/ntrex_african +dataset_kwargs: {trust_remote_code: True} +output_type: generate_until +validation_split: test +fewshot_split: test +test_split: test +metric_list: + - metric: bleu + aggregation: bleu + higher_is_better: true + - metric: chrf + aggregation: chrf + higher_is_better: true +generation_kwargs: + until: + - "\n" + do_sample: false + temperature: 0.0 +repeats: 1 +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_afr_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_afr_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..eb11904366801d649186548e124027489497a4cb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_afr_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: afr_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Afrikaans: {{sentence_afr_Latn}} \nEnglish: " +include: ntrex +task: ntrex_afr_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_amh_Ethi-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_amh_Ethi-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0114a212b89bee62243b3adedad49066998d1785 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_amh_Ethi-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: amh_Ethi +doc_to_target: sentence_eng_Latn +doc_to_text: "Amharic: {{sentence_amh_Ethi}} \nEnglish: " +include: ntrex +task: ntrex_amh_Ethi-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_arb_Arab-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_arb_Arab-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4ddc8c4bbd403a3b83c15172d119ae183247c522 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_arb_Arab-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: arb_Arab +doc_to_target: sentence_eng_Latn +doc_to_text: "Arabic: {{sentence_arb_Arab}} \nEnglish: " +include: ntrex +task: ntrex_arb_Arab-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_bem_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_bem_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c33ab35a18175300ffbf938b2431652ecf86017e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_bem_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: bem_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Bemba: {{sentence_bem_Latn}} \nEnglish: " +include: ntrex +task: ntrex_bem_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_ewe_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_ewe_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c5f69c0051ac2292ef1282ac6c8844ee61bc5148 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_ewe_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: ewe_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Ewe: {{sentence_ewe_Latn}} \nEnglish: " +include: ntrex +task: ntrex_ewe_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_fra_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_fra_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fa3fad61684684f7155bf40704397cff7d5bcbc8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_fra_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: fra_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "French: {{sentence_fra_Latn}} \nEnglish: " +include: ntrex +task: ntrex_fra_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_hau_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_hau_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8b6d0f28b84d4c89d96f3db9de8478201265fade --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_hau_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: hau_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Hausa: {{sentence_hau_Latn}} \nEnglish: " +include: ntrex +task: ntrex_hau_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_ibo_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_ibo_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..992598614c1d9fb0929ca024260a31b953a1204e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_ibo_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: ibo_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Igbo: {{sentence_ibo_Latn}} \nEnglish: " +include: ntrex +task: ntrex_ibo_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_kin_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_kin_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..eee96a62b961371c1fd1f069e97cd94ebef5b4d5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_kin_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: kin_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Kinyarwanda: {{sentence_kin_Latn}} \nEnglish: " +include: ntrex +task: ntrex_kin_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_mey_Arab-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_mey_Arab-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6650e644ad9b84df3c93bb6622543f8984bc4f8b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_mey_Arab-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: mey_Arab +doc_to_target: sentence_eng_Latn +doc_to_text: "Hassaniya Arabic: {{sentence_mey_Arab}} \nEnglish: " +include: ntrex +task: ntrex_mey_Arab-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_mlg_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_mlg_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..375522c5c8560747a2775ec380b4964296dec7e3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_mlg_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: mlg_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Malagasy: {{sentence_mlg_Latn}} \nEnglish: " +include: ntrex +task: ntrex_mlg_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_msa_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_msa_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..65aaaa8014abf84963112a1b7f0239f4129c20bb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_msa_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: msa_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Malay: {{sentence_msa_Latn}} \nEnglish: " +include: ntrex +task: ntrex_msa_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_nde_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_nde_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d63548fb439470b4d46fb7225fa521f31becc77f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_nde_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: nde_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "North Ndebele: {{sentence_nde_Latn}} \nEnglish: " +include: ntrex +task: ntrex_nde_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_nso_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_nso_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4cf1cccf8a2562b0c958457561c7c4c9a5ae6776 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_nso_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: nso_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Northern Sotho: {{sentence_nso_Latn}} \nEnglish: " +include: ntrex +task: ntrex_nso_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_nya_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_nya_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ee4ac6d73f198367a96c684921e6e65e9a0adea7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_nya_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: nya_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Chichewa: {{sentence_nya_Latn}} \nEnglish: " +include: ntrex +task: ntrex_nya_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_orm_Ethi-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_orm_Ethi-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..446873065b536f58bfa12e5886f49edd1b7ea5ee --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_orm_Ethi-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: orm_Ethi +doc_to_target: sentence_eng_Latn +doc_to_text: "Oromo: {{sentence_orm_Ethi}} \nEnglish: " +include: ntrex +task: ntrex_orm_Ethi-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_shi_Arab-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_shi_Arab-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..10972893f3f453f91d12845d9fea3e43558c1fc4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_shi_Arab-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: shi_Arab +doc_to_target: sentence_eng_Latn +doc_to_text: "Tachelhit: {{sentence_shi_Arab}} \nEnglish: " +include: ntrex +task: ntrex_shi_Arab-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_sna_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_sna_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..63d83528835e8ae79f82d09007a4494ccaf1229c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_sna_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: sna_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Shona (Latin): {{sentence_sna_Latn}} \nEnglish: " +include: ntrex +task: ntrex_sna_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_som_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_som_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d6eb91e0310fcebd6483a3d43aca793e3a6934b9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_som_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: som_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Somali: {{sentence_som_Latn}} \nEnglish: " +include: ntrex +task: ntrex_som_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_swa_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_swa_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..863222f7325fab67ff5afe3a13bef0cc0f4df035 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_swa_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: swa_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Swahili: {{sentence_swa_Latn}} \nEnglish: " +include: ntrex +task: ntrex_swa_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_tam_Taml-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_tam_Taml-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..993b480f20e34eab5f1c4cdfb644e09e0e978264 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_tam_Taml-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: tam_Taml +doc_to_target: sentence_eng_Latn +doc_to_text: "Tamil: {{sentence_tam_Taml}} \nEnglish: " +include: ntrex +task: ntrex_tam_Taml-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_tel_Telu-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_tel_Telu-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d91e9a1f762a013ed992d04a2c9e9f0049d8f7eb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_tel_Telu-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: tel_Telu +doc_to_target: sentence_eng_Latn +doc_to_text: "Telugu: {{sentence_tel_Telu}} \nEnglish: " +include: ntrex +task: ntrex_tel_Telu-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_ton_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_ton_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5676a1a99997aca3d0bfc4120003ccb4edef3099 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_ton_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: ton_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Tongan: {{sentence_ton_Latn}} \nEnglish: " +include: ntrex +task: ntrex_ton_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_urd_Arab-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_urd_Arab-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4e07e6787868ec0a56e7b79b2246fcd2211c19d1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_urd_Arab-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: urd_Arab +doc_to_target: sentence_eng_Latn +doc_to_text: "Urdu: {{sentence_urd_Arab}} \nEnglish: " +include: ntrex +task: ntrex_urd_Arab-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_ven_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_ven_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7ba8ceaf4921b087cf38dce53a8c9bb49c359389 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_ven_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: ven_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Venda: {{sentence_ven_Latn}} \nEnglish: " +include: ntrex +task: ntrex_ven_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_wol_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_wol_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8dcacb69de3f8fd83c5714494665cfb7f8cc7be1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_wol_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: wol_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Wolof: {{sentence_wol_Latn}} \nEnglish: " +include: ntrex +task: ntrex_wol_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_xho_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_xho_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1b6abc9dcbf53879148418592fd155f95026bba8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_xho_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: xho_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Xhosa: {{sentence_xho_Latn}} \nEnglish: " +include: ntrex +task: ntrex_xho_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_yor_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_yor_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e98aecd5b188aabf46c2c00b9a126616fee55f6f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_yor_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: yor_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Yoruba: {{sentence_yor_Latn}} \nEnglish: " +include: ntrex +task: ntrex_yor_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_zul_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_zul_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a38abee1148ad1b77a5395afa48621070ad3c239 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/african-english/ntrex_zul_Latn-eng_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: zul_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "Zulu: {{sentence_zul_Latn}} \nEnglish: " +include: ntrex +task: ntrex_zul_Latn-eng_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex new file mode 100644 index 0000000000000000000000000000000000000000..2b5aa84f990e10804a9cdc8ca69901bfb55e5d71 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex @@ -0,0 +1,26 @@ +tag: +- ntrex_tasks +- ntrex_eng-afr +- ntrex_eng-afr_prompt_1 +- afrobench_MT_tasks +dataset_path: masakhane/ntrex_african +dataset_kwargs: {trust_remote_code: True} +output_type: generate_until +validation_split: test +fewshot_split: test +test_split: test +metric_list: + - metric: bleu + aggregation: bleu + higher_is_better: true + - metric: chrf + aggregation: chrf + higher_is_better: true +generation_kwargs: + until: + - "\n" + do_sample: false + temperature: 0.0 +repeats: 1 +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-afr_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-afr_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..40471f80151bacf355f8bf8ff617027f9da68ef7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-afr_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: afr_Latn +doc_to_target: sentence_afr_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nAfrikaans: " +include: ntrex +task: ntrex_eng_Latn-afr_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-amh_Ethi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-amh_Ethi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6e4dfba5dc799649532e9e6b28c862b25afb9566 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-amh_Ethi.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: amh_Ethi +doc_to_target: sentence_amh_Ethi +doc_to_text: "English: {{sentence_eng_Latn}} \nAmharic: " +include: ntrex +task: ntrex_eng_Latn-amh_Ethi_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-arb_Arab.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-arb_Arab.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1a248a9ac6da1668ce1fab555fb7ad586cf0acaa --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-arb_Arab.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: arb_Arab +doc_to_target: sentence_arb_Arab +doc_to_text: "English: {{sentence_eng_Latn}} \nArabic: " +include: ntrex +task: ntrex_eng_Latn-arb_Arab_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-bem_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-bem_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..035c682256b81ca9cc7dda1aebfc9ac130a75762 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-bem_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: bem_Latn +doc_to_target: sentence_bem_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nBemba: " +include: ntrex +task: ntrex_eng_Latn-bem_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-ewe_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-ewe_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f5deae5c56b3bb203b372298207e7fa8d79cfb58 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-ewe_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: ewe_Latn +doc_to_target: sentence_ewe_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nEwe: " +include: ntrex +task: ntrex_eng_Latn-ewe_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-fra_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-fra_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cf079cf440f75a35edbea04e8afa0703ab0eea7b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-fra_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: fra_Latn +doc_to_target: sentence_fra_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nFrench: " +include: ntrex +task: ntrex_eng_Latn-fra_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-ibo_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-ibo_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1c3a14dfa2200c29eb83825a6efb202905e6e78f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-ibo_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: ibo_Latn +doc_to_target: sentence_ibo_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nIgbo: " +include: ntrex +task: ntrex_eng_Latn-ibo_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-mey_Arab.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-mey_Arab.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fb696cc5ac25f1f43c276c34e26b97b7c82efaee --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-mey_Arab.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: mey_Arab +doc_to_target: sentence_mey_Arab +doc_to_text: "English: {{sentence_eng_Latn}} \nHassaniya Arabic: " +include: ntrex +task: ntrex_eng_Latn-mey_Arab_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-mlg_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-mlg_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..035c98c373ff6738310cb280cd617df60c8b6a2a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-mlg_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: mlg_Latn +doc_to_target: sentence_mlg_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nMalagasy: " +include: ntrex +task: ntrex_eng_Latn-mlg_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-msa_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-msa_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c4c6b7d7f1f904ce5fe6061eb7c4c8caef86a8af --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-msa_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: msa_Latn +doc_to_target: sentence_msa_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nMalay: " +include: ntrex +task: ntrex_eng_Latn-msa_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-nde_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-nde_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c66b44beee186f47ea9f8b4d62776d60e4be3ba9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-nde_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: nde_Latn +doc_to_target: sentence_nde_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nNorth Ndebele: " +include: ntrex +task: ntrex_eng_Latn-nde_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-nya_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-nya_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..74cbd1ffed9675feaff5ead68f147fc2572b4edd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-nya_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: nya_Latn +doc_to_target: sentence_nya_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nChichewa: " +include: ntrex +task: ntrex_eng_Latn-nya_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-orm_Ethi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-orm_Ethi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ad875cab5b7012caecd06b99a8d7047ad50c403c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-orm_Ethi.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: orm_Ethi +doc_to_target: sentence_orm_Ethi +doc_to_text: "English: {{sentence_eng_Latn}} \nOromo: " +include: ntrex +task: ntrex_eng_Latn-orm_Ethi_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-shi_Arab.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-shi_Arab.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5441bbdb6ea535f01c71753b9df5ee3290a7cac3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-shi_Arab.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: shi_Arab +doc_to_target: sentence_shi_Arab +doc_to_text: "English: {{sentence_eng_Latn}} \nTachelhit: " +include: ntrex +task: ntrex_eng_Latn-shi_Arab_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-sna_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-sna_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0bed0f6c195e7945329b7d26b50bb5d2abd62c90 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-sna_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: sna_Latn +doc_to_target: sentence_sna_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nShona (Latin): " +include: ntrex +task: ntrex_eng_Latn-sna_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-som_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-som_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5e4aafdfc79bd2e31747847ec081ae15f3799dc3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-som_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: som_Latn +doc_to_target: sentence_som_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nSomali: " +include: ntrex +task: ntrex_eng_Latn-som_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-ssw_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-ssw_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fa18ebf233e0cdbfd5b7d692356f0eacc1cf669a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-ssw_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: ssw_Latn +doc_to_target: sentence_ssw_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nSwati: " +include: ntrex +task: ntrex_eng_Latn-ssw_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-tam_Taml.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-tam_Taml.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b7e42a36beee8d83d057b6daf7b6cfa488b2d90f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-tam_Taml.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: tam_Taml +doc_to_target: sentence_tam_Taml +doc_to_text: "English: {{sentence_eng_Latn}} \nTamil: " +include: ntrex +task: ntrex_eng_Latn-tam_Taml_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-tel_Telu.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-tel_Telu.yaml new file mode 100644 index 0000000000000000000000000000000000000000..db8eb6b20ef17fb518b1c45a8753e72f205a7e41 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-tel_Telu.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: tel_Telu +doc_to_target: sentence_tel_Telu +doc_to_text: "English: {{sentence_eng_Latn}} \nTelugu: " +include: ntrex +task: ntrex_eng_Latn-tel_Telu_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-tir_Ethi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-tir_Ethi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..45c6ae84c642d58db1ebdbf45feb112c4e872bea --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-tir_Ethi.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: tir_Ethi +doc_to_target: sentence_tir_Ethi +doc_to_text: "English: {{sentence_eng_Latn}} \nTigrinya: " +include: ntrex +task: ntrex_eng_Latn-tir_Ethi_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-tsn_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-tsn_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d5a7a4ca261a1b8bfcdd1614eaa167c81c46c1d0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-tsn_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: tsn_Latn +doc_to_target: sentence_tsn_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nTswana: " +include: ntrex +task: ntrex_eng_Latn-tsn_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-ven_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-ven_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4277ce08a5d44f22996d704e0bfbd7461103a0ab --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-ven_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: ven_Latn +doc_to_target: sentence_ven_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nVenda: " +include: ntrex +task: ntrex_eng_Latn-ven_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-wol_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-wol_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dea533ee5e959705c664d5b6e2ee10244c81d3f1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_1/english-african/ntrex_eng_Latn-wol_Latn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: wol_Latn +doc_to_target: sentence_wol_Latn +doc_to_text: "English: {{sentence_eng_Latn}} \nWolof: " +include: ntrex +task: ntrex_eng_Latn-wol_Latn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_2/african-english/ntrex b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_2/african-english/ntrex new file mode 100644 index 0000000000000000000000000000000000000000..3dc29226bf4677ee34836dbc0c5c206cbb1744bd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_2/african-english/ntrex @@ -0,0 +1,25 @@ +tag: +- ntrex_afr-eng +- ntrex_afr-eng_prompt_2 +- afrobench_MT_tasks +dataset_path: masakhane/ntrex_african +dataset_kwargs: {trust_remote_code: True} +output_type: generate_until +validation_split: test +fewshot_split: test +test_split: test +metric_list: + - metric: bleu + aggregation: bleu + higher_is_better: true + - metric: chrf + aggregation: chrf + higher_is_better: true +generation_kwargs: + until: + - "\n" + do_sample: false + temperature: 0.0 +repeats: 1 +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_2/african-english/ntrex_afr_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_2/african-english/ntrex_afr_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..16cfc7d5d0811aec8fca3bcbc7a436f74391cda5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_2/african-english/ntrex_afr_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: afr_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Afrikaans sentences\ + \ to English \nAfrikaans: {{sentence_afr_Latn}}\nEnglish: " +include: ntrex +task: ntrex_afr_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_2/african-english/ntrex_amh_Ethi-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_2/african-english/ntrex_amh_Ethi-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..20e88c366d9c477928abda6bebd2a73d26d00e36 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_2/african-english/ntrex_amh_Ethi-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: amh_Ethi +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Amharic sentences\ + \ to English \nAmharic: {{sentence_amh_Ethi}}\nEnglish: " +include: ntrex +task: ntrex_amh_Ethi-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_2/african-english/ntrex_arb_Arab-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_2/african-english/ntrex_arb_Arab-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a88a478a12a99d5910360dab8b6fa6fac1b78601 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_2/african-english/ntrex_arb_Arab-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: arb_Arab +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Arabic sentences\ + \ to English \nArabic: {{sentence_arb_Arab}}\nEnglish: " +include: ntrex +task: ntrex_arb_Arab-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_2/african-english/ntrex_bem_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_2/african-english/ntrex_bem_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3e114a3464d6cb98baf2374ccaacbc45c3f91240 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_2/african-english/ntrex_bem_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: bem_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Bemba sentences\ + \ to English \nBemba: {{sentence_bem_Latn}}\nEnglish: " +include: ntrex +task: ntrex_bem_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_2/african-english/ntrex_ewe_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_2/african-english/ntrex_ewe_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5e4facd5106291d0fe52d5315d1f6a88a6f32afe --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_2/african-english/ntrex_ewe_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: ewe_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Ewe sentences\ + \ to English \nEwe: {{sentence_ewe_Latn}}\nEnglish: " +include: ntrex +task: ntrex_ewe_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_2/african-english/ntrex_fra_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_2/african-english/ntrex_fra_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ad46aedf727a431a166cac1b9ec45be707feb9bb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_2/african-english/ntrex_fra_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: fra_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following French sentences\ + \ to English \nFrench: {{sentence_fra_Latn}}\nEnglish: " +include: ntrex +task: ntrex_fra_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_2/african-english/ntrex_kin_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_2/african-english/ntrex_kin_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..45b18a640b749e848a8d7df9c01ac2121afb5c2d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_2/african-english/ntrex_kin_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: kin_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Kinyarwanda sentences\ + \ to English \nKinyarwanda: {{sentence_kin_Latn}}\nEnglish: " +include: ntrex +task: ntrex_kin_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_2/african-english/ntrex_mey_Arab-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_2/african-english/ntrex_mey_Arab-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d155b62c828b30e1505e194d3a93960ed707c1aa --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_2/african-english/ntrex_mey_Arab-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: mey_Arab +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Hassaniya Arabic\ + \ sentences to English \nHassaniya Arabic: {{sentence_mey_Arab}}\nEnglish: " +include: ntrex +task: ntrex_mey_Arab-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_2/african-english/ntrex_mlg_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_2/african-english/ntrex_mlg_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..10a7507bae076af1c5aec92ec0db65da9b94f876 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_2/african-english/ntrex_mlg_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: mlg_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Malagasy sentences\ + \ to English \nMalagasy: {{sentence_mlg_Latn}}\nEnglish: " +include: ntrex +task: ntrex_mlg_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_2/african-english/ntrex_nde_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_2/african-english/ntrex_nde_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c4a39fc2c31bc63eb27fdbfb78edaa8c8c59e0ee --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_2/african-english/ntrex_nde_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: nde_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following North Ndebele\ + \ sentences to English \nNorth Ndebele: {{sentence_nde_Latn}}\nEnglish: " +include: ntrex +task: ntrex_nde_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_2/english-african/ntrex_eng_Latn-xho_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_2/english-african/ntrex_eng_Latn-xho_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e7f51491f88c75f9d2da270209fbe32b56bc529b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_2/english-african/ntrex_eng_Latn-xho_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: xho_Latn +doc_to_target: sentence_xho_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Xhosa \nEnglish: {{sentence_eng_Latn}} \nXhosa: " +include: ntrex +task: ntrex_eng_Latn-xho_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_2/english-african/ntrex_eng_Latn-yor_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_2/english-african/ntrex_eng_Latn-yor_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6f3e4be543796276d04c65001199521701f02ed9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_2/english-african/ntrex_eng_Latn-yor_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: yor_Latn +doc_to_target: sentence_yor_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Yoruba \nEnglish: {{sentence_eng_Latn}} \nYoruba: " +include: ntrex +task: ntrex_eng_Latn-yor_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/african-english/ntrex_orm_Ethi-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/african-english/ntrex_orm_Ethi-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9a38e9312cdb30b6bc62b2d3f23c1e5583f043b6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/african-english/ntrex_orm_Ethi-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: orm_Ethi +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Oromo and English linguist, translate the following Oromo sentences\ + \ to English \nOromo: {{sentence_orm_Ethi}}\nEnglish: " +include: ntrex +task: ntrex_orm_Ethi-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/african-english/ntrex_som_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/african-english/ntrex_som_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..685f38233c655048cb55819812247eefaea19527 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/african-english/ntrex_som_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: som_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Somali and English linguist, translate the following Somali sentences\ + \ to English \nSomali: {{sentence_som_Latn}}\nEnglish: " +include: ntrex +task: ntrex_som_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/african-english/ntrex_tam_Taml-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/african-english/ntrex_tam_Taml-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..834320d846a40fcf6bd53c9445f051c38c38a439 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/african-english/ntrex_tam_Taml-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: tam_Taml +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Tamil and English linguist, translate the following Tamil sentences\ + \ to English \nTamil: {{sentence_tam_Taml}}\nEnglish: " +include: ntrex +task: ntrex_tam_Taml-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/african-english/ntrex_tir_Ethi-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/african-english/ntrex_tir_Ethi-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..60189ee73207fc08911821188f61e23eb12dc62e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/african-english/ntrex_tir_Ethi-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: tir_Ethi +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Tigrinya and English linguist, translate the following Tigrinya\ + \ sentences to English \nTigrinya: {{sentence_tir_Ethi}}\nEnglish: " +include: ntrex +task: ntrex_tir_Ethi-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/african-english/ntrex_ton_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/african-english/ntrex_ton_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ec2b5ba992a535f5f5f4fd6b269653b213f1b39a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/african-english/ntrex_ton_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: ton_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Tongan and English linguist, translate the following Tongan sentences\ + \ to English \nTongan: {{sentence_ton_Latn}}\nEnglish: " +include: ntrex +task: ntrex_ton_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/african-english/ntrex_tsn_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/african-english/ntrex_tsn_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fa63ca4b77edb7b0907e6660ce31df7ce0ea7278 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/african-english/ntrex_tsn_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: tsn_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Tswana and English linguist, translate the following Tswana sentences\ + \ to English \nTswana: {{sentence_tsn_Latn}}\nEnglish: " +include: ntrex +task: ntrex_tsn_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/african-english/ntrex_urd_Arab-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/african-english/ntrex_urd_Arab-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2b520795f2fd8e986b8292b985e36769c76f3553 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/african-english/ntrex_urd_Arab-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: urd_Arab +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Urdu and English linguist, translate the following Urdu sentences\ + \ to English \nUrdu: {{sentence_urd_Arab}}\nEnglish: " +include: ntrex +task: ntrex_urd_Arab-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/african-english/ntrex_ven_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/african-english/ntrex_ven_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..82372de2dd0624f9b068f27ab24f48433267ea28 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/african-english/ntrex_ven_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: ven_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Venda and English linguist, translate the following Venda sentences\ + \ to English \nVenda: {{sentence_ven_Latn}}\nEnglish: " +include: ntrex +task: ntrex_ven_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/african-english/ntrex_xho_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/african-english/ntrex_xho_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7f0528af4efc5cb15035158a7c5789878eaa653b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/african-english/ntrex_xho_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: xho_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Xhosa and English linguist, translate the following Xhosa sentences\ + \ to English \nXhosa: {{sentence_xho_Latn}}\nEnglish: " +include: ntrex +task: ntrex_xho_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/african-english/ntrex_yor_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/african-english/ntrex_yor_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..99d7cf494376be71148044b251c23c7b6f15191d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/african-english/ntrex_yor_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: yor_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Yoruba and English linguist, translate the following Yoruba sentences\ + \ to English \nYoruba: {{sentence_yor_Latn}}\nEnglish: " +include: ntrex +task: ntrex_yor_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/african-english/ntrex_zul_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/african-english/ntrex_zul_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..30f3b307eef0a64f137cd993ff8571b103b2e91e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/african-english/ntrex_zul_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: zul_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Zulu and English linguist, translate the following Zulu sentences\ + \ to English \nZulu: {{sentence_zul_Latn}}\nEnglish: " +include: ntrex +task: ntrex_zul_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-afr_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-afr_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4aaa928ba0d31ca83a7d7eb59462a14715a2abf7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-afr_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: afr_Latn +doc_to_target: sentence_afr_Latn +doc_to_text: "As a Afrikaans and English linguist, translate the following English\ + \ sentences to Afrikaans \nEnglish: {{sentence_eng_Latn}} \nAfrikaans: " +include: ntrex +task: ntrex_eng_Latn-afr_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-arb_Arab.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-arb_Arab.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d0c9e8132374542c605789269c27aabf181dad28 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-arb_Arab.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: arb_Arab +doc_to_target: sentence_arb_Arab +doc_to_text: "As a Arabic and English linguist, translate the following English sentences\ + \ to Arabic \nEnglish: {{sentence_eng_Latn}} \nArabic: " +include: ntrex +task: ntrex_eng_Latn-arb_Arab_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-ewe_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-ewe_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e1c99ad06add81beb54d8b0e3b0d97a987bd2d70 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-ewe_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: ewe_Latn +doc_to_target: sentence_ewe_Latn +doc_to_text: "As a Ewe and English linguist, translate the following English sentences\ + \ to Ewe \nEnglish: {{sentence_eng_Latn}} \nEwe: " +include: ntrex +task: ntrex_eng_Latn-ewe_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-msa_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-msa_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bc9a3245f365cdb7c03e5d67e45a9bb236b6477f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-msa_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: msa_Latn +doc_to_target: sentence_msa_Latn +doc_to_text: "As a Malay and English linguist, translate the following English sentences\ + \ to Malay \nEnglish: {{sentence_eng_Latn}} \nMalay: " +include: ntrex +task: ntrex_eng_Latn-msa_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-nso_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-nso_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d52c1ef1f9c4a1ba82ae0f0722669fbf126569f2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-nso_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: nso_Latn +doc_to_target: sentence_nso_Latn +doc_to_text: "As a Northern Sotho and English linguist, translate the following English\ + \ sentences to Northern Sotho \nEnglish: {{sentence_eng_Latn}} \nNorthern Sotho: " +include: ntrex +task: ntrex_eng_Latn-nso_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-nya_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-nya_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5a3d395516d48af64ed67d177cd0fa8b28fd9a46 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-nya_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: nya_Latn +doc_to_target: sentence_nya_Latn +doc_to_text: "As a Chichewa and English linguist, translate the following English\ + \ sentences to Chichewa \nEnglish: {{sentence_eng_Latn}} \nChichewa: " +include: ntrex +task: ntrex_eng_Latn-nya_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-orm_Ethi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-orm_Ethi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d3de07b02307696d09c97eec6120b69580dffade --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-orm_Ethi.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: orm_Ethi +doc_to_target: sentence_orm_Ethi +doc_to_text: "As a Oromo and English linguist, translate the following English sentences\ + \ to Oromo \nEnglish: {{sentence_eng_Latn}} \nOromo: " +include: ntrex +task: ntrex_eng_Latn-orm_Ethi_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-sna_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-sna_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ce8c50f5cacf69e70aca8f451ab8bc1fa8270158 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-sna_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: sna_Latn +doc_to_target: sentence_sna_Latn +doc_to_text: "As a Shona (Latin) and English linguist, translate the following English\ + \ sentences to Shona (Latin) \nEnglish: {{sentence_eng_Latn}} \nShona (Latin): " +include: ntrex +task: ntrex_eng_Latn-sna_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-som_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-som_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4b7f46323401a4c04b1026507b1163111fa71455 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-som_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: som_Latn +doc_to_target: sentence_som_Latn +doc_to_text: "As a Somali and English linguist, translate the following English sentences\ + \ to Somali \nEnglish: {{sentence_eng_Latn}} \nSomali: " +include: ntrex +task: ntrex_eng_Latn-som_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-ssw_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-ssw_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6f02e88ca3f7d5abb314ea174fe21c35b48af402 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-ssw_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: ssw_Latn +doc_to_target: sentence_ssw_Latn +doc_to_text: "As a Swati and English linguist, translate the following English sentences\ + \ to Swati \nEnglish: {{sentence_eng_Latn}} \nSwati: " +include: ntrex +task: ntrex_eng_Latn-ssw_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-swa_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-swa_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..47090821da435d1b9d4caada3e91221cd1eed3b3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-swa_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: swa_Latn +doc_to_target: sentence_swa_Latn +doc_to_text: "As a Swahili and English linguist, translate the following English sentences\ + \ to Swahili \nEnglish: {{sentence_eng_Latn}} \nSwahili: " +include: ntrex +task: ntrex_eng_Latn-swa_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-tam_Taml.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-tam_Taml.yaml new file mode 100644 index 0000000000000000000000000000000000000000..78d61866bd42b467246479946dfa342a6e7835ff --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-tam_Taml.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: tam_Taml +doc_to_target: sentence_tam_Taml +doc_to_text: "As a Tamil and English linguist, translate the following English sentences\ + \ to Tamil \nEnglish: {{sentence_eng_Latn}} \nTamil: " +include: ntrex +task: ntrex_eng_Latn-tam_Taml_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-tir_Ethi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-tir_Ethi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1f27f4389cb2b2be56b02f2427a8d8df222aed17 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-tir_Ethi.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: tir_Ethi +doc_to_target: sentence_tir_Ethi +doc_to_text: "As a Tigrinya and English linguist, translate the following English\ + \ sentences to Tigrinya \nEnglish: {{sentence_eng_Latn}} \nTigrinya: " +include: ntrex +task: ntrex_eng_Latn-tir_Ethi_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-ton_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-ton_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3ffeb74fbb04205e0bb1b27d0ec855252688f6e5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-ton_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: ton_Latn +doc_to_target: sentence_ton_Latn +doc_to_text: "As a Tongan and English linguist, translate the following English sentences\ + \ to Tongan \nEnglish: {{sentence_eng_Latn}} \nTongan: " +include: ntrex +task: ntrex_eng_Latn-ton_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-tsn_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-tsn_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ed11f2cba88a703b44ce2a077f761b4cd98135c0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-tsn_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: tsn_Latn +doc_to_target: sentence_tsn_Latn +doc_to_text: "As a Tswana and English linguist, translate the following English sentences\ + \ to Tswana \nEnglish: {{sentence_eng_Latn}} \nTswana: " +include: ntrex +task: ntrex_eng_Latn-tsn_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-urd_Arab.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-urd_Arab.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a05e951bef2f38005b9d7fb3133bdf811f69c565 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-urd_Arab.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: urd_Arab +doc_to_target: sentence_urd_Arab +doc_to_text: "As a Urdu and English linguist, translate the following English sentences\ + \ to Urdu \nEnglish: {{sentence_eng_Latn}} \nUrdu: " +include: ntrex +task: ntrex_eng_Latn-urd_Arab_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-ven_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-ven_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4345201694bc0c8a9f9bda487a9ecfb36982c8bf --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-ven_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: ven_Latn +doc_to_target: sentence_ven_Latn +doc_to_text: "As a Venda and English linguist, translate the following English sentences\ + \ to Venda \nEnglish: {{sentence_eng_Latn}} \nVenda: " +include: ntrex +task: ntrex_eng_Latn-ven_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-wol_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-wol_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..48abbb33f870ec3305f8337e62b131bbd38683fb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-wol_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: wol_Latn +doc_to_target: sentence_wol_Latn +doc_to_text: "As a Wolof and English linguist, translate the following English sentences\ + \ to Wolof \nEnglish: {{sentence_eng_Latn}} \nWolof: " +include: ntrex +task: ntrex_eng_Latn-wol_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-xho_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-xho_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b1071a5fb2faad74df4e2f357f412923162b0044 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-xho_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: xho_Latn +doc_to_target: sentence_xho_Latn +doc_to_text: "As a Xhosa and English linguist, translate the following English sentences\ + \ to Xhosa \nEnglish: {{sentence_eng_Latn}} \nXhosa: " +include: ntrex +task: ntrex_eng_Latn-xho_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-yor_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-yor_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..43c1be35ee76adf853e6429e4bb06fea867ce5d6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-yor_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: yor_Latn +doc_to_target: sentence_yor_Latn +doc_to_text: "As a Yoruba and English linguist, translate the following English sentences\ + \ to Yoruba \nEnglish: {{sentence_eng_Latn}} \nYoruba: " +include: ntrex +task: ntrex_eng_Latn-yor_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-zul_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-zul_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..10e890a9b3cbffdbb2205d091d91fa42eae880b1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/ntrex/prompt_3/english-african/ntrex_eng_Latn-zul_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: zul_Latn +doc_to_target: sentence_zul_Latn +doc_to_text: "As a Zulu and English linguist, translate the following English sentences\ + \ to Zulu \nEnglish: {{sentence_eng_Latn}} \nZulu: " +include: ntrex +task: ntrex_eng_Latn-zul_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/README.md b/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/README.md new file mode 100644 index 0000000000000000000000000000000000000000..fe980e87464b07c91d2c766254c760d772d65c36 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/README.md @@ -0,0 +1,25 @@ +# + +## Paper +Title: `Multilingual Massive Multitask Language Understanding (MMMLU)` + +Paper Link: https://arxiv.org/abs/2009.03300 + +## Abstract +>We propose a new test to measure a text model's multitask accuracy. The test covers 57 tasks including elementary mathematics, US history, computer science, law, and more. To attain high accuracy on this test, models must possess extensive world knowledge and problem solving ability. We find that while most recent models have near random-chance accuracy, the very largest GPT-3 model improves over random chance by almost 20 percentage points on average. However, on every one of the 57 tasks, the best models still need substantial improvements before they can reach expert-level accuracy. Models also have lopsided performance and frequently do not know when they are wrong. Worse, they still have near-random accuracy on some socially important subjects such as morality and law. By comprehensively evaluating the breadth and depth of a model's academic and professional understanding, our test can be used to analyze models across many tasks and to identify important shortcomings. + +HomePage: https://huggingface.co/datasets/openai/MMMLU + +### Citation + +``` +@misc{hendrycks2021measuringmassivemultitasklanguage, + title={Measuring Massive Multitask Language Understanding}, + author={Dan Hendrycks and Collin Burns and Steven Basart and Andy Zou and Mantas Mazeika and Dawn Song and Jacob Steinhardt}, + year={2021}, + eprint={2009.03300}, + archivePrefix={arXiv}, + primaryClass={cs.CY}, + url={https://arxiv.org/abs/2009.03300}, +} +``` diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/openai_mmlu.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/openai_mmlu.yaml new file mode 100644 index 0000000000000000000000000000000000000000..541eb43cfdd783b15cad4123437c2dffcf1cc794 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/openai_mmlu.yaml @@ -0,0 +1,13 @@ +group: openai_mmlu +task: + - openai_mmlu_prompt_1 + - openai_mmlu_prompt_2 + - openai_mmlu_prompt_3 + - openai_mmlu_prompt_4 + - openai_mmlu_prompt_5 +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_1/openai_mmlu b/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_1/openai_mmlu new file mode 100644 index 0000000000000000000000000000000000000000..ce4f02eeda277404713974f4699c716b454514f5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_1/openai_mmlu @@ -0,0 +1,22 @@ +tag: + - openai_mmlu_tasks + - openai_mmlu_prompt_1 + - afrobench_mmlu_tasks +dataset_path: openai/MMMLU +output_type: multiple_choice +test_split: test +fewshot_config: + sampler: first_n +doc_to_target: "{{['A', 'B', 'C', 'D'].index(Answer.strip())}}" +should_decontaminate: true +doc_to_decontamination_query: "{{Question}}" +doc_to_choice: ["A", "B", "C", "D"] +metric_list: + - metric: acc + aggregation: mean + weight_by_size: true + - metric: acc_norm + aggregation: mean + weight_by_size: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_1/openai_mmlu_ara.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_1/openai_mmlu_ara.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5c9b86fc1d1c5d8185692c48bc85d991714dbff5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_1/openai_mmlu_ara.yaml @@ -0,0 +1,15 @@ +# Generated by utils.py +dataset_name: AR_XY +doc_to_text: 'Q: {{Question.strip()}} + + A: {{A}} + + B: {{B}} + + C: {{C}} + + D: {{D}} + + Please choose the correct answer from the options above:' +include: openai_mmlu +task: openai_mmlu_ara_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_1/openai_mmlu_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_1/openai_mmlu_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4124252bfc0b549160ac802f18c44004792d3bf2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_1/openai_mmlu_yor.yaml @@ -0,0 +1,15 @@ +# Generated by utils.py +dataset_name: YO_NG +doc_to_text: 'Q: {{Question.strip()}} + + A: {{A}} + + B: {{B}} + + C: {{C}} + + D: {{D}} + + Please choose the correct answer from the options above:' +include: openai_mmlu +task: openai_mmlu_yor_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_2/openai_mmlu b/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_2/openai_mmlu new file mode 100644 index 0000000000000000000000000000000000000000..9f39b0a9d7423b4d5638f23f294b636240570281 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_2/openai_mmlu @@ -0,0 +1,22 @@ +tag: + - openai_mmlu_tasks + - openai_mmlu_prompt_2 + - afrobench_mmlu_tasks +dataset_path: openai/MMMLU +output_type: multiple_choice +test_split: test +fewshot_config: + sampler: first_n +doc_to_target: "{{['A', 'B', 'C', 'D'].index(Answer.strip())}}" +should_decontaminate: true +doc_to_decontamination_query: "{{Question}}" +doc_to_choice: ["A", "B", "C", "D"] +metric_list: + - metric: acc + aggregation: mean + weight_by_size: true + - metric: acc_norm + aggregation: mean + weight_by_size: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_2/openai_mmlu_ara.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_2/openai_mmlu_ara.yaml new file mode 100644 index 0000000000000000000000000000000000000000..550834257a69f7054ae397a403c6dc00d15c8888 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_2/openai_mmlu_ara.yaml @@ -0,0 +1,15 @@ +# Generated by utils.py +dataset_name: AR_XY +doc_to_text: 'Question: {{Question.strip()}} + + 1: {{A}} + + 2: {{B}} + + 3: {{C}} + + 4: {{D}} + + Please select the correct answer from the given choices:' +include: openai_mmlu +task: openai_mmlu_ara_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_2/openai_mmlu_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_2/openai_mmlu_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9b3025fd726ab59f48ee90bb65b294580a6cfc3c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_2/openai_mmlu_swa.yaml @@ -0,0 +1,15 @@ +# Generated by utils.py +dataset_name: SW_KE +doc_to_text: 'Question: {{Question.strip()}} + + 1: {{A}} + + 2: {{B}} + + 3: {{C}} + + 4: {{D}} + + Please select the correct answer from the given choices:' +include: openai_mmlu +task: openai_mmlu_swa_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_2/openai_mmlu_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_2/openai_mmlu_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..145b237ef50234278732605b1e3936bfccb9968a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_2/openai_mmlu_yor.yaml @@ -0,0 +1,15 @@ +# Generated by utils.py +dataset_name: YO_NG +doc_to_text: 'Question: {{Question.strip()}} + + 1: {{A}} + + 2: {{B}} + + 3: {{C}} + + 4: {{D}} + + Please select the correct answer from the given choices:' +include: openai_mmlu +task: openai_mmlu_yor_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_3/openai_mmlu b/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_3/openai_mmlu new file mode 100644 index 0000000000000000000000000000000000000000..95456656739a2490a3e11037e7d9f67f72d60962 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_3/openai_mmlu @@ -0,0 +1,23 @@ +tag: + - openai_mmlu_tasks + - openai_mmlu_prompt_3 + - afrobench_mmlu_tasks +dataset_path: openai/MMMLU +dataset_name: null +output_type: multiple_choice +test_split: test +fewshot_config: + sampler: first_n +doc_to_target: "{{['A', 'B', 'C', 'D'].index(Answer.strip())}}" +should_decontaminate: true +doc_to_decontamination_query: "{{Question}}" +doc_to_choice: ["A", "B", "C", "D"] +metric_list: + - metric: acc + aggregation: mean + weight_by_size: true + - metric: acc_norm + aggregation: mean + weight_by_size: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_3/openai_mmlu_ara.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_3/openai_mmlu_ara.yaml new file mode 100644 index 0000000000000000000000000000000000000000..012192ceee6638f197db8ea8b9210e1529b6b92d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_3/openai_mmlu_ara.yaml @@ -0,0 +1,15 @@ +# Generated by utils.py +dataset_name: AR_XY +doc_to_text: 'Input Question: {{Question.strip()}} + + Option A: {{A}} + + Option B: {{B}} + + Option C: {{C}} + + Option D: {{D}} + + Please indicate the correct option from the list above:' +include: openai_mmlu +task: openai_mmlu_ara_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_3/openai_mmlu_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_3/openai_mmlu_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..431bdb345178bf44b12ea01507cc805cd000113f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_3/openai_mmlu_swa.yaml @@ -0,0 +1,15 @@ +# Generated by utils.py +dataset_name: SW_KE +doc_to_text: 'Input Question: {{Question.strip()}} + + Option A: {{A}} + + Option B: {{B}} + + Option C: {{C}} + + Option D: {{D}} + + Please indicate the correct option from the list above:' +include: openai_mmlu +task: openai_mmlu_swa_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_3/openai_mmlu_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_3/openai_mmlu_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..814fe380267e57da691f727198f2828042aa54c0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_3/openai_mmlu_yor.yaml @@ -0,0 +1,15 @@ +# Generated by utils.py +dataset_name: YO_NG +doc_to_text: 'Input Question: {{Question.strip()}} + + Option A: {{A}} + + Option B: {{B}} + + Option C: {{C}} + + Option D: {{D}} + + Please indicate the correct option from the list above:' +include: openai_mmlu +task: openai_mmlu_yor_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_4/openai_mmlu b/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_4/openai_mmlu new file mode 100644 index 0000000000000000000000000000000000000000..37a5949f93795737f8f61a06fc2824ebb671dbe2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_4/openai_mmlu @@ -0,0 +1,23 @@ +tag: + - openai_mmlu_tasks + - openai_mmlu_prompt_4 + - afrobench_mmlu_tasks +dataset_path: openai/MMMLU +dataset_name: null +output_type: multiple_choice +test_split: test +fewshot_config: + sampler: first_n +doc_to_target: "{{['A', 'B', 'C', 'D'].index(Answer.strip())}}" +should_decontaminate: true +doc_to_decontamination_query: "{{Question}}" +doc_to_choice: ["A", "B", "C", "D"] +metric_list: + - metric: acc + aggregation: mean + weight_by_size: true + - metric: acc_norm + aggregation: mean + weight_by_size: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_4/openai_mmlu_ara.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_4/openai_mmlu_ara.yaml new file mode 100644 index 0000000000000000000000000000000000000000..793eb7441ce36573953525c3c97e60daffb10b02 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_4/openai_mmlu_ara.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: AR_XY +doc_to_text: 'Critically analyze the question and select the most probable answer + from the list: + + {{Question.strip()}} + + Choices: + + A) {{A}} + + B) {{B}} + + C) {{C}} + + D) {{D}}' +include: openai_mmlu +task: openai_mmlu_ara_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_4/openai_mmlu_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_4/openai_mmlu_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..095dd7ff6d04db581bc070eff001d48600014e0e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_4/openai_mmlu_swa.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: SW_KE +doc_to_text: 'Critically analyze the question and select the most probable answer + from the list: + + {{Question.strip()}} + + Choices: + + A) {{A}} + + B) {{B}} + + C) {{C}} + + D) {{D}}' +include: openai_mmlu +task: openai_mmlu_swa_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_4/openai_mmlu_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_4/openai_mmlu_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dd0a9daa1ed5a2ae9882225aefdfe5e653dffcc6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_4/openai_mmlu_yor.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: YO_NG +doc_to_text: 'Critically analyze the question and select the most probable answer + from the list: + + {{Question.strip()}} + + Choices: + + A) {{A}} + + B) {{B}} + + C) {{C}} + + D) {{D}}' +include: openai_mmlu +task: openai_mmlu_yor_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_5/openai_mmlu b/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_5/openai_mmlu new file mode 100644 index 0000000000000000000000000000000000000000..77183eb04c0567b83f87bfd17bbdd18bf003f7dd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_5/openai_mmlu @@ -0,0 +1,23 @@ +tag: + - openai_mmlu_tasks + - openai_mmlu_prompt_5 + - afrobench_mmlu_tasks +dataset_path: openai/MMMLU +dataset_name: null +output_type: multiple_choice +test_split: test +fewshot_config: + sampler: first_n +doc_to_target: "{{['A', 'B', 'C', 'D'].index(Answer.strip())}}" +should_decontaminate: true +doc_to_decontamination_query: "{{Question}}" +doc_to_choice: ["A", "B", "C", "D"] +metric_list: + - metric: acc + aggregation: mean + weight_by_size: true + - metric: acc_norm + aggregation: mean + weight_by_size: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_5/openai_mmlu_ara.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_5/openai_mmlu_ara.yaml new file mode 100644 index 0000000000000000000000000000000000000000..50a6e74ff2198b326716b99a7430102d8aaf0221 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_5/openai_mmlu_ara.yaml @@ -0,0 +1,17 @@ +# Generated by utils.py +dataset_name: AR_XY +doc_to_text: 'Answer the question and pick the correct answer from the options: {{Question.strip()}} + + Options: + + A. {{A}} + + B. {{B}} + + C. {{C}} + + D. {{D}} + + Please choose the correct option from the above list:' +include: openai_mmlu +task: openai_mmlu_ara_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_5/openai_mmlu_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_5/openai_mmlu_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c0cc19860cc5f7bc90d499a1ead811a549170eb6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_5/openai_mmlu_swa.yaml @@ -0,0 +1,17 @@ +# Generated by utils.py +dataset_name: SW_KE +doc_to_text: 'Answer the question and pick the correct answer from the options: {{Question.strip()}} + + Options: + + A. {{A}} + + B. {{B}} + + C. {{C}} + + D. {{D}} + + Please choose the correct option from the above list:' +include: openai_mmlu +task: openai_mmlu_swa_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_5/openai_mmlu_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_5/openai_mmlu_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..691657ef46974107e46c37291fb1efa66364a5b9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/prompt_5/openai_mmlu_yor.yaml @@ -0,0 +1,17 @@ +# Generated by utils.py +dataset_name: YO_NG +doc_to_text: 'Answer the question and pick the correct answer from the options: {{Question.strip()}} + + Options: + + A. {{A}} + + B. {{B}} + + C. {{C}} + + D. {{D}} + + Please choose the correct option from the above list:' +include: openai_mmlu +task: openai_mmlu_yor_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..0fc0fea958c32b2b8d104f586045564c04de8c86 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/openai_mmlu/utils.py @@ -0,0 +1,99 @@ +import argparse +import os + +import yaml + + +def prompt_func(mode, lang): + prompt_map = { + "prompt_1": "Q: {{Question.strip()}}\nA: {{A}}\nB: {{B}}\nC: {{C}}\nD: {{D}}\nPlease choose the correct answer from the options above:", + "prompt_2": "Question: {{Question.strip()}}\n1: {{A}}\n2: {{B}}\n3: {{C}}\n4: {{D}}\nPlease select the correct answer from the given choices:", + "prompt_3": "Input Question: {{Question.strip()}}\nOption A: {{A}}\nOption B: {{B}}\nOption C: {{C}}\nOption D: {{D}}\nPlease indicate the correct option from the list above:", + "prompt_4": "Critically analyze the question and select the most probable answer from the list:\n{{Question.strip()}}\nChoices:\nA) {{A}}\nB) {{B}}\nC) {{C}}\nD) {{D}}", + "prompt_5": "Answer the question and pick the correct answer from the options: {{Question.strip()}}\nOptions:\nA. {{A}}\nB. {{B}}\nC. {{C}}\nD. {{D}}\nPlease choose the correct option from the above list:", + } + return prompt_map[mode] + + +def gen_lang_yamls(output_dir: str, overwrite: bool, mode: str) -> None: + """ + Generate a yaml file for each language. + + :param output_dir: The directory to output the files to. + :param overwrite: Whether to overwrite files if they already exist. + """ + err = [] + languages = { + "ara": "Arabic", + "swa": "Swahili", + "yor": "Yoruba", + } + + lang2_code = { + "ara": "AR_XY", + "swa": "SW_KE", + "yor": "YO_NG", + } + + for lang in languages.keys(): + try: + file_name = f"openai_mmlu_{lang}.yaml" + task_name = f"openai_mmlu_{lang}_{mode}" + yaml_template = "openai_mmlu" + yaml_details = { + "include": yaml_template, + "task": task_name, + "dataset_name": lang2_code[lang], + "doc_to_text": prompt_func(mode, languages[lang]), + } + file_path = os.path.join(output_dir, mode) + os.makedirs(file_path, exist_ok=True) + + with open( + f"{output_dir}/{mode}/{file_name}", + "w" if overwrite else "x", + encoding="utf8", + ) as f: + f.write("# Generated by utils.py\n") + yaml.dump( + yaml_details, + f, + allow_unicode=True, + ) + except FileExistsError: + err.append(file_name) + + if len(err) > 0: + raise FileExistsError( + "Files were not created because they already exist (use --overwrite flag):" + f" {', '.join(err)}" + ) + + +def main() -> None: + """Parse CLI args and generate language-specific yaml files.""" + parser = argparse.ArgumentParser() + parser.add_argument( + "--overwrite", + default=True, + action="store_true", + help="Overwrite files if they already exist", + ) + parser.add_argument( + "--output-dir", + default="./", + help="Directory to write yaml files to", + ) + parser.add_argument( + "--mode", + default="prompt_1", + choices=["prompt_1", "prompt_2", "prompt_3", "prompt_4", "prompt_5"], + help="Prompt number", + ) + args = parser.parse_args() + + gen_lang_yamls(output_dir=args.output_dir, overwrite=args.overwrite, mode=args.mode) + + +if __name__ == "__main__": + main() diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/README.md b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/README.md new file mode 100644 index 0000000000000000000000000000000000000000..3c5239a05e88cbfbadf6670f96d6ed621b0d805c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/README.md @@ -0,0 +1,17 @@ +# + +## Paper +Title: `Sunbird African Language Technology (SALT) dataset` + +Paper Link: https://aclanthology.org/2023.emnlp-main.862/ + +## Abstract +>SALT is a multi-way parallel text and speech corpus of Engish and six languages widely spoken in Uganda and East Africa: Luganda, Lugbara, Acholi, Runyankole, Ateso and Swahili. The core of the dataset is a set of 25,000 sentences covering a range of topics of local relevance, such as agriculture, health and society. Each sentence is translated into all languages, to support machine translation, and speech recordings are made for approximately 5,000 of the sentences both by a variety of speakers in natural settings (suitable for ASR) and by professionals in a studio setting (suitable for text-to-speech). + +HomePage: https://github.com/SunbirdAI/salt + +### Publications + +Multilingual Model and Data Resources for Text-To-Speech in Ugandan Languages. Isaac Owomugisha, Benjamin Akera, Ernest Tonny Mwebaze, John Quinn. 4th Workshop on African Natural Language Processing, 2023. [pdf](https://openreview.net/pdf?id=vaxG0WAPzL) + +Machine Translation For African Languages: Community Creation Of Datasets And Models In Uganda. Benjamin Akera, Jonathan Mukiibi, Lydia Sanyu Naggayi, Claire Babirye, Isaac Owomugisha, Solomon Nsumba, Joyce Nakatumba-Nabende, Engineer Bainomugisha, Ernest Mwebaze, John Quinn. 3rd Workshop on African Natural Language Processing, 2022. [pdf](https://openreview.net/pdf?id=BK-z5qzEU-9) diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/gen_utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/gen_utils.py new file mode 100644 index 0000000000000000000000000000000000000000..6ac703a0d5d0912d38fb624dbba967ed3ffdb734 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/gen_utils.py @@ -0,0 +1,149 @@ +import argparse +import os + +import yaml + + +class FunctionTag: + def __init__(self, value): + self.value = value + + +def prompt_func(mode, lang, lang_dict): + language_column_name = f"{lang}_text" + prompt_map = { + "prompt_1": f"{lang_dict[lang]} sentence: {{{{{language_column_name}}}}} \nEnglish sentence: ", + "prompt_1_reverse": "English sentence: {{eng_source_text}} " + f"\n{lang_dict[lang]} sentence: ", + "prompt_2": f"You are a translation expert. Translate the following {lang_dict[lang]} sentences to English \n" + f"{lang_dict[lang]} sentence: {{{{{language_column_name}}}}}\nEnglish sentence: ", + "prompt_2_reverse": f"You are a translation expert. Translate the following English sentences to " + f"{lang_dict[lang]} " + "\nEnglish sentence: {{eng_source_text}} " + f"\n{lang_dict[lang]} sentence: ", + "prompt_3": f"As a {lang_dict[lang]} and English linguist, translate the following {lang_dict[lang]} sentences " + f"to English. \n{lang_dict[lang]} sentence: {{{{{language_column_name}}}}}\nEnglish sentence: ", + "prompt_3_reverse": f"As a {lang_dict[lang]} and English linguist, translate the following English sentences to " + f"{lang_dict[lang]}. " + "\nEnglish sentence: {{eng_source_text}} " + f"\n{lang_dict[lang]} sentence: ", + } + return prompt_map[mode] + + +def gen_lang_yamls(output_dir: str, overwrite: bool, mode: str, reverse: bool) -> None: + """ + Generate a yaml file for each language. + + :param output_dir: The directory to output the files to. + :param overwrite: Whether to overwrite files if they already exist. + """ + err = [] + languages = { + "eng": "English", + "lug": "Luganda", + "ach": "Acholi", + "lgg": "Lugbara", + "teo": "Ateso", + "nyn": "Runyankole", + "swa": "Swahili", + "ibo": "Igbo", + } + + for lang in languages.keys(): + try: + if lang != "eng": + if not reverse: + file_name = f"salt_{lang}-eng.yaml" + task_name = f"salt_{lang}-eng_{mode}" + yaml_template = "salt" + yaml_details = { + "include": yaml_template, + "task": task_name, + "dataset_name": "text-all", + "doc_to_target": "eng_target_text", + "doc_to_text": prompt_func(mode, lang, languages), + } + os.makedirs(f"{output_dir}/{mode}", exist_ok=True) + with open( + f"{output_dir}/{mode}/{file_name}", + "w" if overwrite else "x", + encoding="utf8", + ) as f: + f.write("# Generated by utils.py\n") + yaml.dump( + yaml_details, + f, + allow_unicode=True, + ) + else: + file_name = f"salt_eng-{lang}.yaml" + task_name = f"salt_eng-{lang}_{mode}" + yaml_template = "salt" + yaml_details = { + "include": yaml_template, + "task": task_name, + "dataset_name": "text-all", + "doc_to_target": f"{lang}_text", + "doc_to_text": prompt_func(f"{mode}_reverse", lang, languages), + } + os.makedirs(f"{output_dir}/{mode}", exist_ok=True) + with open( + f"{output_dir}/{mode}/{file_name}", + "w" if overwrite else "x", + encoding="utf8", + ) as f: + f.write("# Generated by utils.py\n") + yaml.dump( + yaml_details, + f, + allow_unicode=True, + ) + except FileExistsError: + err.append(file_name) + + if len(err) > 0: + raise FileExistsError( + "Files were not created because they already exist (use --overwrite flag):" + f" {', '.join(err)}" + ) + + +def main() -> None: + """Parse CLI args and generate language-specific yaml files.""" + parser = argparse.ArgumentParser() + parser.add_argument( + "--overwrite", + default=True, + action="store_true", + help="Overwrite files if they already exist", + ) + parser.add_argument( + "--output-dir", + default="./", + help="Directory to write yaml files to", + ) + parser.add_argument( + "--mode", + default="prompt_1", + choices=["prompt_1", "prompt_2", "prompt_3"], + help="Prompt number", + ) + parser.add_argument( + "--reverse", + default=True, + choices=[True, False], + help="Reverse the translation direction", + ) + args = parser.parse_args() + + gen_lang_yamls( + output_dir=args.output_dir, + overwrite=args.overwrite, + mode=args.mode, + reverse=args.reverse, + ) + + +if __name__ == "__main__": + main() diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_1/salt b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_1/salt new file mode 100644 index 0000000000000000000000000000000000000000..a07d434a8bfb5e4c85abef6fe556e648c6fe5a00 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_1/salt @@ -0,0 +1,24 @@ +tag: +- salt_tasks +- salt_prompt_1 +- afrobench_MT_tasks +dataset_path: Sunbird/salt +dataset_kwargs: {trust_remote_code: True} +output_type: generate_until +validation_split: dev +fewshot_split: dev +test_split: test +metric_list: + - metric: bleu + aggregation: bleu + higher_is_better: true + - metric: chrf + aggregation: chrf + higher_is_better: true +generation_kwargs: + until: + - "\n" + do_sample: false + temperature: 0.0 +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_1/salt_ach-eng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_1/salt_ach-eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..41731279817637401307fc9f55ecd96cd2a80794 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_1/salt_ach-eng.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: text-all +doc_to_target: eng_target_text +doc_to_text: "Acholi sentence: {{ach_text}} \nEnglish sentence: " +include: salt +task: salt_ach-eng_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_1/salt_eng-ach.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_1/salt_eng-ach.yaml new file mode 100644 index 0000000000000000000000000000000000000000..219e5780634f4812157ea6d2ad70b7b22e72ae49 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_1/salt_eng-ach.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: text-all +doc_to_target: ach_text +doc_to_text: "English sentence: {{eng_source_text}} \nAcholi sentence: " +include: salt +task: salt_eng-ach_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_1/salt_eng-ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_1/salt_eng-ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f90220591f5f7047da6d488740c759c850a95b1e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_1/salt_eng-ibo.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: text-all +doc_to_target: ibo_text +doc_to_text: "English sentence: {{eng_source_text}} \nIgbo sentence: " +include: salt +task: salt_eng-ibo_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_1/salt_eng-lgg.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_1/salt_eng-lgg.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2a038ddb39eb1b171be9e5631e129995ceeed64e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_1/salt_eng-lgg.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: text-all +doc_to_target: lgg_text +doc_to_text: "English sentence: {{eng_source_text}} \nLugbara sentence: " +include: salt +task: salt_eng-lgg_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_1/salt_eng-lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_1/salt_eng-lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4539913786124aec4ea68f16538989a91131ca44 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_1/salt_eng-lug.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: text-all +doc_to_target: lug_text +doc_to_text: "English sentence: {{eng_source_text}} \nLuganda sentence: " +include: salt +task: salt_eng-lug_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_1/salt_eng-nyn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_1/salt_eng-nyn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..448e1101d681d4f31bde8c81418d4f2f64b6eb13 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_1/salt_eng-nyn.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: text-all +doc_to_target: nyn_text +doc_to_text: "English sentence: {{eng_source_text}} \nRunyankole sentence: " +include: salt +task: salt_eng-nyn_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_1/salt_eng-swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_1/salt_eng-swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..792b4840c2551627b66008fdd2c172e3660cc914 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_1/salt_eng-swa.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: text-all +doc_to_target: swa_text +doc_to_text: "English sentence: {{eng_source_text}} \nSwahili sentence: " +include: salt +task: salt_eng-swa_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_1/salt_eng-teo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_1/salt_eng-teo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..810626c6a5ddf8525d45344b5a5eb7a2d65ab34e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_1/salt_eng-teo.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: text-all +doc_to_target: teo_text +doc_to_text: "English sentence: {{eng_source_text}} \nAteso sentence: " +include: salt +task: salt_eng-teo_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_1/salt_ibo-eng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_1/salt_ibo-eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0a98c8648081bc8c3e1fd1c897c41212701f36fa --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_1/salt_ibo-eng.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: text-all +doc_to_target: eng_target_text +doc_to_text: "Igbo sentence: {{ibo_text}} \nEnglish sentence: " +include: salt +task: salt_ibo-eng_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_1/salt_lgg-eng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_1/salt_lgg-eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c8e281ac3189dbaada8d21fbd4896a0c8478dbc6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_1/salt_lgg-eng.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: text-all +doc_to_target: eng_target_text +doc_to_text: "Lugbara sentence: {{lgg_text}} \nEnglish sentence: " +include: salt +task: salt_lgg-eng_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_1/salt_lug-eng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_1/salt_lug-eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f924d5c50f82e1dbbf6be1dd4a138d2c5d61c5ac --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_1/salt_lug-eng.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: text-all +doc_to_target: eng_target_text +doc_to_text: "Luganda sentence: {{lug_text}} \nEnglish sentence: " +include: salt +task: salt_lug-eng_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_1/salt_nyn-eng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_1/salt_nyn-eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bd9363614648969391f20deb49fd2a92afdcfede --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_1/salt_nyn-eng.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: text-all +doc_to_target: eng_target_text +doc_to_text: "Runyankole sentence: {{nyn_text}} \nEnglish sentence: " +include: salt +task: salt_nyn-eng_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_1/salt_swa-eng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_1/salt_swa-eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c2308593e3d54e222d7543403f214dba76719a80 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_1/salt_swa-eng.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: text-all +doc_to_target: eng_target_text +doc_to_text: "Swahili sentence: {{swa_text}} \nEnglish sentence: " +include: salt +task: salt_swa-eng_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_1/salt_teo-eng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_1/salt_teo-eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6efb4ea0807a9a66eb84797503b6bc4762777fd0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_1/salt_teo-eng.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: text-all +doc_to_target: eng_target_text +doc_to_text: "Ateso sentence: {{teo_text}} \nEnglish sentence: " +include: salt +task: salt_teo-eng_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_2/salt b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_2/salt new file mode 100644 index 0000000000000000000000000000000000000000..66355878cbb8354261bd426623d29589ce93383a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_2/salt @@ -0,0 +1,24 @@ +tag: +- salt_tasks +- salt_prompt_2 +- afrobench_MT_tasks +dataset_path: Sunbird/salt +dataset_kwargs: {trust_remote_code: True} +output_type: generate_until +validation_split: dev +fewshot_split: dev +test_split: test +metric_list: + - metric: bleu + aggregation: bleu + higher_is_better: true + - metric: chrf + aggregation: chrf + higher_is_better: true +generation_kwargs: + until: + - "\n" + do_sample: false + temperature: 0.0 +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_2/salt_ach-eng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_2/salt_ach-eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dda717b7942cb37c7f6d821070572cd302717639 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_2/salt_ach-eng.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: text-all +doc_to_target: eng_target_text +doc_to_text: "You are a translation expert. Translate the following Acholi sentences\ + \ to English \nAcholi sentence: {{ach_text}}\nEnglish sentence: " +include: salt +task: salt_ach-eng_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_2/salt_eng-ach.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_2/salt_eng-ach.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1e4a72a5116a41a7d7950cfed80cbd826a37a0dc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_2/salt_eng-ach.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: text-all +doc_to_target: ach_text +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Acholi \nEnglish sentence: {{eng_source_text}} \nAcholi sentence: " +include: salt +task: salt_eng-ach_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_2/salt_eng-ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_2/salt_eng-ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..04649c1287e599a2ecdf376b4b30bc86700dcaca --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_2/salt_eng-ibo.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: text-all +doc_to_target: ibo_text +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Igbo \nEnglish sentence: {{eng_source_text}} \nIgbo sentence: " +include: salt +task: salt_eng-ibo_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_2/salt_eng-lgg.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_2/salt_eng-lgg.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0ac6becbcb7b10890cb1b2cd56dbe43c23742683 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_2/salt_eng-lgg.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: text-all +doc_to_target: lgg_text +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Lugbara \nEnglish sentence: {{eng_source_text}} \nLugbara sentence: " +include: salt +task: salt_eng-lgg_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_2/salt_eng-lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_2/salt_eng-lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1b5f6399cf6ddc5276fb48545e4ad3d1e0e4ab1f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_2/salt_eng-lug.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: text-all +doc_to_target: lug_text +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Luganda \nEnglish sentence: {{eng_source_text}} \nLuganda sentence: " +include: salt +task: salt_eng-lug_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_2/salt_eng-nyn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_2/salt_eng-nyn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..84452d5aed07b2fe13d6836a7656ff85dfa2ae8e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_2/salt_eng-nyn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: text-all +doc_to_target: nyn_text +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Runyankole \nEnglish sentence: {{eng_source_text}} \nRunyankole sentence: " +include: salt +task: salt_eng-nyn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_2/salt_eng-swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_2/salt_eng-swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..523db9fb7e913dff30b80d28ca13b8c613653ad6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_2/salt_eng-swa.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: text-all +doc_to_target: swa_text +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Swahili \nEnglish sentence: {{eng_source_text}} \nSwahili sentence: " +include: salt +task: salt_eng-swa_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_2/salt_eng-teo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_2/salt_eng-teo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..000e8d043bb1897c5647480d6584191181b45c68 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_2/salt_eng-teo.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: text-all +doc_to_target: teo_text +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Ateso \nEnglish sentence: {{eng_source_text}} \nAteso sentence: " +include: salt +task: salt_eng-teo_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_2/salt_ibo-eng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_2/salt_ibo-eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b4ec6601af313b25606c05752715a3dfadf1476e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_2/salt_ibo-eng.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: text-all +doc_to_target: eng_target_text +doc_to_text: "You are a translation expert. Translate the following Igbo sentences\ + \ to English \nIgbo sentence: {{ibo_text}}\nEnglish sentence: " +include: salt +task: salt_ibo-eng_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_2/salt_lgg-eng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_2/salt_lgg-eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2d802c0faa99f895605e00c25aea3197b0fad7d2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_2/salt_lgg-eng.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: text-all +doc_to_target: eng_target_text +doc_to_text: "You are a translation expert. Translate the following Lugbara sentences\ + \ to English \nLugbara sentence: {{lgg_text}}\nEnglish sentence: " +include: salt +task: salt_lgg-eng_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_2/salt_lug-eng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_2/salt_lug-eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..521bbf15c008670a0d71b671be84e58b9ca7290b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_2/salt_lug-eng.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: text-all +doc_to_target: eng_target_text +doc_to_text: "You are a translation expert. Translate the following Luganda sentences\ + \ to English \nLuganda sentence: {{lug_text}}\nEnglish sentence: " +include: salt +task: salt_lug-eng_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_2/salt_nyn-eng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_2/salt_nyn-eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4cc4abfc26505ada2abb05775a6b4b43c67fb139 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_2/salt_nyn-eng.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: text-all +doc_to_target: eng_target_text +doc_to_text: "You are a translation expert. Translate the following Runyankole sentences\ + \ to English \nRunyankole sentence: {{nyn_text}}\nEnglish sentence: " +include: salt +task: salt_nyn-eng_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_2/salt_swa-eng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_2/salt_swa-eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9e80b9087df91df24f619c61c61c553decfcb1bf --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_2/salt_swa-eng.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: text-all +doc_to_target: eng_target_text +doc_to_text: "You are a translation expert. Translate the following Swahili sentences\ + \ to English \nSwahili sentence: {{swa_text}}\nEnglish sentence: " +include: salt +task: salt_swa-eng_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_2/salt_teo-eng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_2/salt_teo-eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d0b0d516de9ae00758cd9fccb45c84d65eb069bc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_2/salt_teo-eng.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: text-all +doc_to_target: eng_target_text +doc_to_text: "You are a translation expert. Translate the following Ateso sentences\ + \ to English \nAteso sentence: {{teo_text}}\nEnglish sentence: " +include: salt +task: salt_teo-eng_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_3/salt b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_3/salt new file mode 100644 index 0000000000000000000000000000000000000000..51dac9c53b42569b2b5c7f19a5b9fa6b83fc68e4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_3/salt @@ -0,0 +1,24 @@ +tag: +- salt_tasks +- salt_prompt_3 +- afrobench_MT_tasks +dataset_path: Sunbird/salt +dataset_kwargs: {trust_remote_code: True} +output_type: generate_until +validation_split: dev +fewshot_split: dev +test_split: test +metric_list: + - metric: bleu + aggregation: bleu + higher_is_better: true + - metric: chrf + aggregation: chrf + higher_is_better: true +generation_kwargs: + until: + - "\n" + do_sample: false + temperature: 0.0 +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_3/salt_ach-eng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_3/salt_ach-eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c198a59f843447475f221823361f7ddf919419c3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_3/salt_ach-eng.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: text-all +doc_to_target: eng_target_text +doc_to_text: "As a Acholi and English linguist, translate the following Acholi sentences\ + \ to English. \nAcholi sentence: {{ach_text}}\nEnglish sentence: " +include: salt +task: salt_ach-eng_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_3/salt_eng-ach.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_3/salt_eng-ach.yaml new file mode 100644 index 0000000000000000000000000000000000000000..636a77d8606343d9de230547b958f8e49b448b5c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_3/salt_eng-ach.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: text-all +doc_to_target: ach_text +doc_to_text: "As a Acholi and English linguist, translate the following English sentences\ + \ to Acholi. \nEnglish sentence: {{eng_source_text}} \nAcholi sentence: " +include: salt +task: salt_eng-ach_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_3/salt_eng-ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_3/salt_eng-ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..44d015d6ca9db85477a082c687a46c7e46276068 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_3/salt_eng-ibo.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: text-all +doc_to_target: ibo_text +doc_to_text: "As a Igbo and English linguist, translate the following English sentences\ + \ to Igbo. \nEnglish sentence: {{eng_source_text}} \nIgbo sentence: " +include: salt +task: salt_eng-ibo_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_3/salt_eng-lgg.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_3/salt_eng-lgg.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8f1e6f43ba7783c2b521ee3a0caeec1d0904790e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_3/salt_eng-lgg.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: text-all +doc_to_target: lgg_text +doc_to_text: "As a Lugbara and English linguist, translate the following English sentences\ + \ to Lugbara. \nEnglish sentence: {{eng_source_text}} \nLugbara sentence: " +include: salt +task: salt_eng-lgg_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_3/salt_eng-lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_3/salt_eng-lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e2065c30df12a680ca08b218ce3e842324313da4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_3/salt_eng-lug.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: text-all +doc_to_target: lug_text +doc_to_text: "As a Luganda and English linguist, translate the following English sentences\ + \ to Luganda. \nEnglish sentence: {{eng_source_text}} \nLuganda sentence: " +include: salt +task: salt_eng-lug_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_3/salt_eng-nyn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_3/salt_eng-nyn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9e48970a8ccb136d4328598224c370076949954b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_3/salt_eng-nyn.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: text-all +doc_to_target: nyn_text +doc_to_text: "As a Runyankole and English linguist, translate the following English\ + \ sentences to Runyankole. \nEnglish sentence: {{eng_source_text}} \nRunyankole\ + \ sentence: " +include: salt +task: salt_eng-nyn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_3/salt_eng-swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_3/salt_eng-swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cfd3f8eadb1ca0ff898595c897a3eebde72f08a8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_3/salt_eng-swa.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: text-all +doc_to_target: swa_text +doc_to_text: "As a Swahili and English linguist, translate the following English sentences\ + \ to Swahili. \nEnglish sentence: {{eng_source_text}} \nSwahili sentence: " +include: salt +task: salt_eng-swa_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_3/salt_eng-teo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_3/salt_eng-teo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f8d280bb41808f8af50287861fa1131b92295e70 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_3/salt_eng-teo.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: text-all +doc_to_target: teo_text +doc_to_text: "As a Ateso and English linguist, translate the following English sentences\ + \ to Ateso. \nEnglish sentence: {{eng_source_text}} \nAteso sentence: " +include: salt +task: salt_eng-teo_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_3/salt_ibo-eng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_3/salt_ibo-eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..13be699cb1dc1255939321205d25921625cdb140 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_3/salt_ibo-eng.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: text-all +doc_to_target: eng_target_text +doc_to_text: "As a Igbo and English linguist, translate the following Igbo sentences\ + \ to English. \nIgbo sentence: {{ibo_text}}\nEnglish sentence: " +include: salt +task: salt_ibo-eng_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_3/salt_lgg-eng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_3/salt_lgg-eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7aa4ffc442c41ea9abd148257b1e49524173eca5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_3/salt_lgg-eng.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: text-all +doc_to_target: eng_target_text +doc_to_text: "As a Lugbara and English linguist, translate the following Lugbara sentences\ + \ to English. \nLugbara sentence: {{lgg_text}}\nEnglish sentence: " +include: salt +task: salt_lgg-eng_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_3/salt_lug-eng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_3/salt_lug-eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..da505f6d7589d9a7bba4ea7be1c73134fc562a20 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_3/salt_lug-eng.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: text-all +doc_to_target: eng_target_text +doc_to_text: "As a Luganda and English linguist, translate the following Luganda sentences\ + \ to English. \nLuganda sentence: {{lug_text}}\nEnglish sentence: " +include: salt +task: salt_lug-eng_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_3/salt_nyn-eng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_3/salt_nyn-eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9edba7c495369e1849106e854100a65d0bda9ee5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_3/salt_nyn-eng.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: text-all +doc_to_target: eng_target_text +doc_to_text: "As a Runyankole and English linguist, translate the following Runyankole\ + \ sentences to English. \nRunyankole sentence: {{nyn_text}}\nEnglish sentence: " +include: salt +task: salt_nyn-eng_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_3/salt_swa-eng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_3/salt_swa-eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3d01c9170c602c7eebdc3b0a5c216d5bdd4bc52a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_3/salt_swa-eng.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: text-all +doc_to_target: eng_target_text +doc_to_text: "As a Swahili and English linguist, translate the following Swahili sentences\ + \ to English. \nSwahili sentence: {{swa_text}}\nEnglish sentence: " +include: salt +task: salt_swa-eng_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_3/salt_teo-eng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_3/salt_teo-eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c81336cac58f12d6dd2118315a6cdb64a913a2af --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/prompt_3/salt_teo-eng.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: text-all +doc_to_target: eng_target_text +doc_to_text: "As a Ateso and English linguist, translate the following Ateso sentences\ + \ to English. \nAteso sentence: {{teo_text}}\nEnglish sentence: " +include: salt +task: salt_teo-eng_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/salt.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/salt.yaml new file mode 100644 index 0000000000000000000000000000000000000000..edd3070d8ba2c24b651038ca7408a38b45e00da3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/salt/salt.yaml @@ -0,0 +1,11 @@ +group: salt +task: + - salt_prompt_1 + - salt_prompt_2 + - salt_prompt_3 +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sample_run_scripts/run_afrobench.sh b/lm-evaluation-harness/lm_eval/tasks/afrobench/sample_run_scripts/run_afrobench.sh new file mode 100644 index 0000000000000000000000000000000000000000..886c94956cc8204ce9fda69e912cec91424a3d92 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sample_run_scripts/run_afrobench.sh @@ -0,0 +1,32 @@ +#!/bin/bash + +batch_size=5 +num_fewshot=0 + +export CUDA_VISIBLE_DEVICES=0,1 + +model_names=( + "google/gemma-1.1-7b-it", + "google/gemma-2-9b-it", + "google/gemma-2-27b-it", + "Jacaranda/AfroLlama_V1", + "LLaMAX/LLaMAX3-8B-Alpaca", + "meta-llama/Llama-2-7b-chat-hf", + "meta-llama/Llama-3.1-8B-Instruct", + "meta-llama/Llama-3.1-70B-Instruct", + "meta-llama/Meta-Llama-3-8B-Instruct", + "CohereForAI/aya-101" +) + +for model_name in "${model_names[@]}" +do + echo "Running model: $model_name" + lm_eval --model hf \ + --model_args pretrained=${model_names},parallelize=true \ + --tasks afrobench\ + --batch_size ${batch_size} \ + --num_fewshot ${num_fewshot} \ + --verbosity DEBUG \ + --output_path 'path_to_results/' \ + --log_samples +done diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sample_run_scripts/run_afrobench_lite.sh b/lm-evaluation-harness/lm_eval/tasks/afrobench/sample_run_scripts/run_afrobench_lite.sh new file mode 100644 index 0000000000000000000000000000000000000000..89291faadb97fa9267d09be80e81a7b480aabcb5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sample_run_scripts/run_afrobench_lite.sh @@ -0,0 +1,32 @@ +#!/bin/bash + +batch_size=5 +num_fewshot=0 + +export CUDA_VISIBLE_DEVICES=0,1 + +model_names=( + "google/gemma-1.1-7b-it", + "google/gemma-2-9b-it", + "google/gemma-2-27b-it", + "Jacaranda/AfroLlama_V1", + "LLaMAX/LLaMAX3-8B-Alpaca", + "meta-llama/Llama-2-7b-chat-hf", + "meta-llama/Llama-3.1-8B-Instruct", + "meta-llama/Llama-3.1-70B-Instruct", + "meta-llama/Meta-Llama-3-8B-Instruct", + "CohereForAI/aya-101" +) + +for model_name in "${model_names[@]}" +do + echo "Running model: $model_name" + lm_eval --model hf \ + --model_args pretrained=${model_name},parallelize=true \ + --tasks afrobench_lite\ + --batch_size ${batch_size} \ + --num_fewshot ${num_fewshot} \ + --verbosity DEBUG \ + --output_path 'path_to_results/' \ + --log_samples +done diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/README.md b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/README.md new file mode 100644 index 0000000000000000000000000000000000000000..732db84b0eb6ad373442692b221e7f97e18e112a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/README.md @@ -0,0 +1,37 @@ +# + +## Paper +Title: `SIB-200: A Simple, Inclusive, and Big Evaluation Dataset for Topic Classification in 200+ Languages and Dialects` + +Paper Link: https://aclanthology.org/2024.eacl-long.14/ + +## Abstract +>Despite the progress in building multilingual language models, evaluation is often limited to a few languages with available datasets which excludes a large number of low-resource languages. In this paper, we create SIB-200—a large-scale open-sourced benchmark dataset for topic classification in 205 languages and dialects to address the lack of evaluation dataset for Natural Language Understanding (NLU). For many of the languages covered in SIB-200, this is the first publicly available evaluation dataset for NLU. The dataset is based on Flores-200 machine translation corpus. We annotated the English portion of the dataset and extended the sentence-level annotation to the remaining 204 languages covered in the corpus. Despite the simplicity of this task, our evaluation in full-supervised setting, cross-lingual transfer setting and prompting of large language model setting show that there is still a large gap between the performance of high-resource and low-resource languages when multilingual evaluation is scaled to numerous world languages. We found that languages unseen during the pre-training of multilingual language models, languages from under-represented families (like Nilotic and Altantic-Congo), and languages from the regions of Africa, Americas, Oceania and South East Asia, often have the lowest performance on our topic classification dataset. We hope our dataset %will encourages a more inclusive evaluation of multilingual language models on a more diverse set of languages. + +HomePage: https://github.com/dadelani/sib-200 + +### Citation + +``` +@inproceedings{adelani-etal-2024-sib, + title = "{SIB}-200: A Simple, Inclusive, and Big Evaluation Dataset for Topic Classification in 200+ Languages and Dialects", + author = "Adelani, David Ifeoluwa and + Liu, Hannah and + Shen, Xiaoyu and + Vassilyev, Nikita and + Alabi, Jesujoba O. and + Mao, Yanke and + Gao, Haonan and + Lee, En-Shiun Annie", + editor = "Graham, Yvette and + Purver, Matthew", + booktitle = "Proceedings of the 18th Conference of the European Chapter of the Association for Computational Linguistics (Volume 1: Long Papers)", + month = mar, + year = "2024", + address = "St. Julian{'}s, Malta", + publisher = "Association for Computational Linguistics", + url = "https://aclanthology.org/2024.eacl-long.14/", + pages = "226--245", + abstract = "Despite the progress in building multilingual language models, evaluation is often limited to a few languages with available datasets which excludes a large number of low-resource languages. In this paper, we create SIB-200{---}a large-scale open-sourced benchmark dataset for topic classification in 205 languages and dialects to address the lack of evaluation dataset for Natural Language Understanding (NLU). For many of the languages covered in SIB-200, this is the first publicly available evaluation dataset for NLU. The dataset is based on Flores-200 machine translation corpus. We annotated the English portion of the dataset and extended the sentence-level annotation to the remaining 204 languages covered in the corpus. Despite the simplicity of this task, our evaluation in full-supervised setting, cross-lingual transfer setting and prompting of large language model setting show that there is still a large gap between the performance of high-resource and low-resource languages when multilingual evaluation is scaled to numerous world languages. We found that languages unseen during the pre-training of multilingual language models, languages from under-represented families (like Nilotic and Altantic-Congo), and languages from the regions of Africa, Americas, Oceania and South East Asia, often have the lowest performance on our topic classification dataset. We hope our dataset {\%}will encourages a more inclusive evaluation of multilingual language models on a more diverse set of languages." +} +``` diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib new file mode 100644 index 0000000000000000000000000000000000000000..37fda5d192dc8b4e1aa115d66858876e6bca3bda --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib @@ -0,0 +1,43 @@ +tag: + - sib_tasks + - sib_prompt_1 + - afrobench_TC_tasks +dataset_path: Davlan/sib200 +dataset_name: null +output_type: multiple_choice +validation_split: validation +test_split: test +fewshot_split: validation +doc_to_target: category +doc_to_choice: + - "science/technology" + - "travel" + - "politics" + - "sports" + - "health" + - "entertainment" + - "geography" +should_decontaminate: true +doc_to_decontamination_query: text +metric_list: + - metric: f1 + aggregation: !function utils.weighted_f1_score + # aggregation: mean + average: weighted + hf_evaluate: true + higher_is_better: True + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" + - metric: acc + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_aeb.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_aeb.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d4116035df2599f79d31293b25abf43191943abd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_aeb.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: aeb_Arab +doc_to_text: "Given the categories science/technology, travel, politics, sports, health,\ + \ entertainment, or geography; what category does the text: '{{text}}' belong to:\ + \ \n\n" +include: sib +task: sib_aeb_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_afr.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_afr.yaml new file mode 100644 index 0000000000000000000000000000000000000000..001eee846bd92a3e1703d64d799e5bc8c066f70e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_afr.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: afr_Latn +doc_to_text: "Given the categories science/technology, travel, politics, sports, health,\ + \ entertainment, or geography; what category does the text: '{{text}}' belong to:\ + \ \n\n" +include: sib +task: sib_afr_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_aka.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_aka.yaml new file mode 100644 index 0000000000000000000000000000000000000000..907977dc638bdfc7aba5ea11324d54667cd21d1e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_aka.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: aka_Latn +doc_to_text: "Given the categories science/technology, travel, politics, sports, health,\ + \ entertainment, or geography; what category does the text: '{{text}}' belong to:\ + \ \n\n" +include: sib +task: sib_aka_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dde5420724bdb678ac877c5ff895df74ba0b08c6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_amh.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: amh_Ethi +doc_to_text: "Given the categories science/technology, travel, politics, sports, health,\ + \ entertainment, or geography; what category does the text: '{{text}}' belong to:\ + \ \n\n" +include: sib +task: sib_amh_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_ary.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_ary.yaml new file mode 100644 index 0000000000000000000000000000000000000000..68347bd51439c95b88403f843fb78a06a3562d39 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_ary.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: ary_Arab +doc_to_text: "Given the categories science/technology, travel, politics, sports, health,\ + \ entertainment, or geography; what category does the text: '{{text}}' belong to:\ + \ \n\n" +include: sib +task: sib_ary_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_arz.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_arz.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2c0328134c766bd56637a2097f1b87bfa03a4973 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_arz.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: arz_Arab +doc_to_text: "Given the categories science/technology, travel, politics, sports, health,\ + \ entertainment, or geography; what category does the text: '{{text}}' belong to:\ + \ \n\n" +include: sib +task: sib_arz_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_bam.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_bam.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5469a8a17ea44b468172c326a148f1185a559015 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_bam.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: bam_Latn +doc_to_text: "Given the categories science/technology, travel, politics, sports, health,\ + \ entertainment, or geography; what category does the text: '{{text}}' belong to:\ + \ \n\n" +include: sib +task: sib_bam_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_bem.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_bem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..01aaa1cbd82342de4ace8c11387f1851a21661d1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_bem.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: bem_Latn +doc_to_text: "Given the categories science/technology, travel, politics, sports, health,\ + \ entertainment, or geography; what category does the text: '{{text}}' belong to:\ + \ \n\n" +include: sib +task: sib_bem_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_cjk.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_cjk.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6deaee753f460189a1fcf47c800239b2242ccf8d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_cjk.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: cjk_Latn +doc_to_text: "Given the categories science/technology, travel, politics, sports, health,\ + \ entertainment, or geography; what category does the text: '{{text}}' belong to:\ + \ \n\n" +include: sib +task: sib_cjk_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_dik.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_dik.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d80d0a080890269475d0133cb4a73cc80ffbe6eb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_dik.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: dik_Latn +doc_to_text: "Given the categories science/technology, travel, politics, sports, health,\ + \ entertainment, or geography; what category does the text: '{{text}}' belong to:\ + \ \n\n" +include: sib +task: sib_dik_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_dyu.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_dyu.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1d72e6321e92d7e8947cce5109d363f7eb51f9de --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_dyu.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: dyu_Latn +doc_to_text: "Given the categories science/technology, travel, politics, sports, health,\ + \ entertainment, or geography; what category does the text: '{{text}}' belong to:\ + \ \n\n" +include: sib +task: sib_dyu_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_eng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1e32469681e4400517131926dff6e8b1a717b69d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_eng.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn +doc_to_text: "Given the categories science/technology, travel, politics, sports, health,\ + \ entertainment, or geography; what category does the text: '{{text}}' belong to:\ + \ \n\n" +include: sib +task: sib_eng_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_ewe.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_ewe.yaml new file mode 100644 index 0000000000000000000000000000000000000000..60cf7db830a1aff9215989f4af5c9a6f8d278985 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_ewe.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: ewe_Latn +doc_to_text: "Given the categories science/technology, travel, politics, sports, health,\ + \ entertainment, or geography; what category does the text: '{{text}}' belong to:\ + \ \n\n" +include: sib +task: sib_ewe_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_fon.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_fon.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7ae765522ccd81ddadd2842bb7e8a346fff18088 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_fon.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: fon_Latn +doc_to_text: "Given the categories science/technology, travel, politics, sports, health,\ + \ entertainment, or geography; what category does the text: '{{text}}' belong to:\ + \ \n\n" +include: sib +task: sib_fon_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_fra.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_fra.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4614e6d27f2d41e5558045d933df41a66a909cfb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_fra.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: fra_Latn +doc_to_text: "Given the categories science/technology, travel, politics, sports, health,\ + \ entertainment, or geography; what category does the text: '{{text}}' belong to:\ + \ \n\n" +include: sib +task: sib_fra_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_fuv.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_fuv.yaml new file mode 100644 index 0000000000000000000000000000000000000000..24f1d28a8f088d383bf7fbbff939dc73b4cf447e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_fuv.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: fuv_Latn +doc_to_text: "Given the categories science/technology, travel, politics, sports, health,\ + \ entertainment, or geography; what category does the text: '{{text}}' belong to:\ + \ \n\n" +include: sib +task: sib_fuv_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_gaz.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_gaz.yaml new file mode 100644 index 0000000000000000000000000000000000000000..df904f957318c08eaf8c2f5cba4d0befa5220fbc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_gaz.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: gaz_Latn +doc_to_text: "Given the categories science/technology, travel, politics, sports, health,\ + \ entertainment, or geography; what category does the text: '{{text}}' belong to:\ + \ \n\n" +include: sib +task: sib_gaz_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b160b8cfc0aa662bfadcc68f2891208e7039c01b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_hau.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: hau_Latn +doc_to_text: "Given the categories science/technology, travel, politics, sports, health,\ + \ entertainment, or geography; what category does the text: '{{text}}' belong to:\ + \ \n\n" +include: sib +task: sib_hau_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e481aeacd5c7d63cbfd11e7efcb3fb1ac738e945 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_ibo.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: ibo_Latn +doc_to_text: "Given the categories science/technology, travel, politics, sports, health,\ + \ entertainment, or geography; what category does the text: '{{text}}' belong to:\ + \ \n\n" +include: sib +task: sib_ibo_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_kab.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_kab.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a914b01cc54c35941cd769dbe6667ee624421b91 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_kab.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: kab_Latn +doc_to_text: "Given the categories science/technology, travel, politics, sports, health,\ + \ entertainment, or geography; what category does the text: '{{text}}' belong to:\ + \ \n\n" +include: sib +task: sib_kab_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_kam.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_kam.yaml new file mode 100644 index 0000000000000000000000000000000000000000..aaa05108b0cc3313932e71a174b0f53e747e42eb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_kam.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: kam_Latn +doc_to_text: "Given the categories science/technology, travel, politics, sports, health,\ + \ entertainment, or geography; what category does the text: '{{text}}' belong to:\ + \ \n\n" +include: sib +task: sib_kam_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_kbp.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_kbp.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d428490863c573b3a757672bc3c074d8fd548c0d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_kbp.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: kbp_Latn +doc_to_text: "Given the categories science/technology, travel, politics, sports, health,\ + \ entertainment, or geography; what category does the text: '{{text}}' belong to:\ + \ \n\n" +include: sib +task: sib_kbp_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_kea.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_kea.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4e458fb225b8c1b4b4ee2823f46b4b4ad7a6dcad --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_kea.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: kea_Latn +doc_to_text: "Given the categories science/technology, travel, politics, sports, health,\ + \ entertainment, or geography; what category does the text: '{{text}}' belong to:\ + \ \n\n" +include: sib +task: sib_kea_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_kik.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_kik.yaml new file mode 100644 index 0000000000000000000000000000000000000000..beb94a8edb7cec7b51c960fe319a98e798a84581 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_kik.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: kik_Latn +doc_to_text: "Given the categories science/technology, travel, politics, sports, health,\ + \ entertainment, or geography; what category does the text: '{{text}}' belong to:\ + \ \n\n" +include: sib +task: sib_kik_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7c16432eba27800bcc8eb927e4a201aac7b3f2e0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_kin.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: kin_Latn +doc_to_text: "Given the categories science/technology, travel, politics, sports, health,\ + \ entertainment, or geography; what category does the text: '{{text}}' belong to:\ + \ \n\n" +include: sib +task: sib_kin_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_kmb.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_kmb.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c46477e31e4639a9b9c1dca0ce59318534e883e4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_kmb.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: kmb_Latn +doc_to_text: "Given the categories science/technology, travel, politics, sports, health,\ + \ entertainment, or geography; what category does the text: '{{text}}' belong to:\ + \ \n\n" +include: sib +task: sib_kmb_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_knc.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_knc.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9b43157e3642dc91b6f04de06ecc622b67fb036e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_knc.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: knc_Latn +doc_to_text: "Given the categories science/technology, travel, politics, sports, health,\ + \ entertainment, or geography; what category does the text: '{{text}}' belong to:\ + \ \n\n" +include: sib +task: sib_knc_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_kon.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_kon.yaml new file mode 100644 index 0000000000000000000000000000000000000000..def4a77def17ff2d11cc00d6c87962a03c4081cd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_kon.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: kon_Latn +doc_to_text: "Given the categories science/technology, travel, politics, sports, health,\ + \ entertainment, or geography; what category does the text: '{{text}}' belong to:\ + \ \n\n" +include: sib +task: sib_kon_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_lin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_lin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bbba95e0cf7217c4385f4601a7867ffc6576b2b7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_lin.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: lin_Latn +doc_to_text: "Given the categories science/technology, travel, politics, sports, health,\ + \ entertainment, or geography; what category does the text: '{{text}}' belong to:\ + \ \n\n" +include: sib +task: sib_lin_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_lua.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_lua.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d4bc665b3f9e6fce1703b2ea53c93bcc52111363 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_lua.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: lua_Latn +doc_to_text: "Given the categories science/technology, travel, politics, sports, health,\ + \ entertainment, or geography; what category does the text: '{{text}}' belong to:\ + \ \n\n" +include: sib +task: sib_lua_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cbf42e1889e695a379d8261bac27d02fa7f4d33d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_lug.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: lug_Latn +doc_to_text: "Given the categories science/technology, travel, politics, sports, health,\ + \ entertainment, or geography; what category does the text: '{{text}}' belong to:\ + \ \n\n" +include: sib +task: sib_lug_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_luo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_luo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a62ea03c7ba534928d5c3c333d631216cf0dd248 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_luo.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: luo_Latn +doc_to_text: "Given the categories science/technology, travel, politics, sports, health,\ + \ entertainment, or geography; what category does the text: '{{text}}' belong to:\ + \ \n\n" +include: sib +task: sib_luo_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_mos.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_mos.yaml new file mode 100644 index 0000000000000000000000000000000000000000..54140a5d1339758f59a3504d3a4a0a5448414b90 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_mos.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: mos_Latn +doc_to_text: "Given the categories science/technology, travel, politics, sports, health,\ + \ entertainment, or geography; what category does the text: '{{text}}' belong to:\ + \ \n\n" +include: sib +task: sib_mos_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_nso.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_nso.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7f7382d58f3071f1dddcab360ecf06b2cf7a427c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_nso.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: nso_Latn +doc_to_text: "Given the categories science/technology, travel, politics, sports, health,\ + \ entertainment, or geography; what category does the text: '{{text}}' belong to:\ + \ \n\n" +include: sib +task: sib_nso_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_nus.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_nus.yaml new file mode 100644 index 0000000000000000000000000000000000000000..28208912f85036d26b494ed495b43f2a57982869 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_nus.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: nus_Latn +doc_to_text: "Given the categories science/technology, travel, politics, sports, health,\ + \ entertainment, or geography; what category does the text: '{{text}}' belong to:\ + \ \n\n" +include: sib +task: sib_nus_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_nya.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_nya.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6ca90a9233e68301406a3303ebbb85cb47da2207 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_nya.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: nya_Latn +doc_to_text: "Given the categories science/technology, travel, politics, sports, health,\ + \ entertainment, or geography; what category does the text: '{{text}}' belong to:\ + \ \n\n" +include: sib +task: sib_nya_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_plt.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_plt.yaml new file mode 100644 index 0000000000000000000000000000000000000000..650b9a4b711f30b59c0aadde797e941d69a17ed6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_plt.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: plt_Latn +doc_to_text: "Given the categories science/technology, travel, politics, sports, health,\ + \ entertainment, or geography; what category does the text: '{{text}}' belong to:\ + \ \n\n" +include: sib +task: sib_plt_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_por.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_por.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7901e924a043d74dadbf8b0dabff2303273d03a5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_por.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: por_Latn +doc_to_text: "Given the categories science/technology, travel, politics, sports, health,\ + \ entertainment, or geography; what category does the text: '{{text}}' belong to:\ + \ \n\n" +include: sib +task: sib_por_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_run.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_run.yaml new file mode 100644 index 0000000000000000000000000000000000000000..510fc5c15841c9c130af5cce0e3e2d8499eb71d0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_run.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: run_Latn +doc_to_text: "Given the categories science/technology, travel, politics, sports, health,\ + \ entertainment, or geography; what category does the text: '{{text}}' belong to:\ + \ \n\n" +include: sib +task: sib_run_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_sag.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_sag.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e7c0bb3148857edab4e8eaef00974fa5e4dfd974 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_sag.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: sag_Latn +doc_to_text: "Given the categories science/technology, travel, politics, sports, health,\ + \ entertainment, or geography; what category does the text: '{{text}}' belong to:\ + \ \n\n" +include: sib +task: sib_sag_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_sna.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_sna.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f4115112c393c0dd424b14bdd66046d58e82eb83 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_sna.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: sna_Latn +doc_to_text: "Given the categories science/technology, travel, politics, sports, health,\ + \ entertainment, or geography; what category does the text: '{{text}}' belong to:\ + \ \n\n" +include: sib +task: sib_sna_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_som.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_som.yaml new file mode 100644 index 0000000000000000000000000000000000000000..be9c19f1039b8093e3c5bcd7573168b23f6e923c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_som.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: som_Latn +doc_to_text: "Given the categories science/technology, travel, politics, sports, health,\ + \ entertainment, or geography; what category does the text: '{{text}}' belong to:\ + \ \n\n" +include: sib +task: sib_som_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_sot.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_sot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..78d0e1f50dc0909475131e7892bbe726f5144412 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_sot.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: sot_Latn +doc_to_text: "Given the categories science/technology, travel, politics, sports, health,\ + \ entertainment, or geography; what category does the text: '{{text}}' belong to:\ + \ \n\n" +include: sib +task: sib_sot_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_ssw.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_ssw.yaml new file mode 100644 index 0000000000000000000000000000000000000000..988f6828cbe84bdf7cec2a03798a452956e1768d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_ssw.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: ssw_Latn +doc_to_text: "Given the categories science/technology, travel, politics, sports, health,\ + \ entertainment, or geography; what category does the text: '{{text}}' belong to:\ + \ \n\n" +include: sib +task: sib_ssw_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d4a92192eb750ed34c647381ae0c8655b141f4a2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_swa.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: swh_Latn +doc_to_text: "Given the categories science/technology, travel, politics, sports, health,\ + \ entertainment, or geography; what category does the text: '{{text}}' belong to:\ + \ \n\n" +include: sib +task: sib_swa_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_taq.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_taq.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a860f019dd0d259ea3fd9eddfb776870ea24b7f2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_taq.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: taq_Latn +doc_to_text: "Given the categories science/technology, travel, politics, sports, health,\ + \ entertainment, or geography; what category does the text: '{{text}}' belong to:\ + \ \n\n" +include: sib +task: sib_taq_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_tir.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_tir.yaml new file mode 100644 index 0000000000000000000000000000000000000000..606755c5c59c01cdd1148437cdfccb4791ebc689 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_tir.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: tir_Ethi +doc_to_text: "Given the categories science/technology, travel, politics, sports, health,\ + \ entertainment, or geography; what category does the text: '{{text}}' belong to:\ + \ \n\n" +include: sib +task: sib_tir_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_tso.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_tso.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c6b2e46369554e76573e3cdec7128b56d9853913 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_tso.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: tso_Latn +doc_to_text: "Given the categories science/technology, travel, politics, sports, health,\ + \ entertainment, or geography; what category does the text: '{{text}}' belong to:\ + \ \n\n" +include: sib +task: sib_tso_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_tzm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_tzm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..10cf4c5b6626fe9ffc3addfbc8197156e23ee45f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_tzm.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: tzm_Tfng +doc_to_text: "Given the categories science/technology, travel, politics, sports, health,\ + \ entertainment, or geography; what category does the text: '{{text}}' belong to:\ + \ \n\n" +include: sib +task: sib_tzm_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_umb.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_umb.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d171c9c6b6fd7f2db5ac205e3adfed2e6e6fb867 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_umb.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: umb_Latn +doc_to_text: "Given the categories science/technology, travel, politics, sports, health,\ + \ entertainment, or geography; what category does the text: '{{text}}' belong to:\ + \ \n\n" +include: sib +task: sib_umb_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c3a6d7e6234c0ac9df872fb3cfcbc1e9f0e4f483 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_wol.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: wol_Latn +doc_to_text: "Given the categories science/technology, travel, politics, sports, health,\ + \ entertainment, or geography; what category does the text: '{{text}}' belong to:\ + \ \n\n" +include: sib +task: sib_wol_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..57ce4d2db833add543832e1798a90b7479d8a360 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_xho.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: xho_Latn +doc_to_text: "Given the categories science/technology, travel, politics, sports, health,\ + \ entertainment, or geography; what category does the text: '{{text}}' belong to:\ + \ \n\n" +include: sib +task: sib_xho_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cab811762f3b61828cb698857e0d46f33855f568 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_yor.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: yor_Latn +doc_to_text: "Given the categories science/technology, travel, politics, sports, health,\ + \ entertainment, or geography; what category does the text: '{{text}}' belong to:\ + \ \n\n" +include: sib +task: sib_yor_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..3e735e2deb1f9c53152c072615aebe8ba3acb90b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/utils.py @@ -0,0 +1 @@ +from lm_eval.utils import weighted_f1_score diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib new file mode 100644 index 0000000000000000000000000000000000000000..27dd7d1f64838b9692fbaa06ea98c6cd7f7db97e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib @@ -0,0 +1,43 @@ +tag: + - sib_tasks + - sib_prompt_2 + - afrobench_TC_tasks +dataset_path: Davlan/sib200 +dataset_name: null +output_type: multiple_choice +validation_split: validation +test_split: test +fewshot_split: validation +doc_to_target: category +doc_to_choice: + - "science/technology" + - "travel" + - "politics" + - "sports" + - "health" + - "entertainment" + - "geography" +should_decontaminate: true +doc_to_decontamination_query: text +metric_list: + - metric: f1 + aggregation: !function utils.weighted_f1_score + # aggregation: mean + average: weighted + hf_evaluate: true + higher_is_better: True + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" + - metric: acc + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_aeb.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_aeb.yaml new file mode 100644 index 0000000000000000000000000000000000000000..32b2443948fd04761dab4331d9421b50e6293397 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_aeb.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: aeb_Arab +doc_to_text: 'Does this Tunisian Arabic topic; ''{{text}}'' belong to one of the following + categories: science/technology, travel, politics, sports, health, entertainment, + or geography? category only + + + ' +include: sib +task: sib_aeb_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_afr.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_afr.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c212b13f1f2cf4cd9a2b5b70fce75429a2dbbd91 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_afr.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: afr_Latn +doc_to_text: 'Does this Afrikaans topic; ''{{text}}'' belong to one of the following + categories: science/technology, travel, politics, sports, health, entertainment, + or geography? category only + + + ' +include: sib +task: sib_afr_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_aka.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_aka.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dacfef07608dfebc67025bc8ff983260ee535f6b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_aka.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: aka_Latn +doc_to_text: 'Does this Akan topic; ''{{text}}'' belong to one of the following categories: + science/technology, travel, politics, sports, health, entertainment, or geography? + category only + + + ' +include: sib +task: sib_aka_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_ary.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_ary.yaml new file mode 100644 index 0000000000000000000000000000000000000000..141a6691de71e7f932970dd3a73c91aa818c45b2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_ary.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: ary_Arab +doc_to_text: 'Does this Moroccan Arabic topic; ''{{text}}'' belong to one of the following + categories: science/technology, travel, politics, sports, health, entertainment, + or geography? category only + + + ' +include: sib +task: sib_ary_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_arz.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_arz.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b2fee5eed9e22ca4448a5aa1efe26e756ed41562 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_arz.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: arz_Arab +doc_to_text: 'Does this Egyptian Arabic topic; ''{{text}}'' belong to one of the following + categories: science/technology, travel, politics, sports, health, entertainment, + or geography? category only + + + ' +include: sib +task: sib_arz_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_bam.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_bam.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0ae5ddd0ea44b3d4c9a90e40125b942cd1919d26 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_bam.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: bam_Latn +doc_to_text: 'Does this Bambara topic; ''{{text}}'' belong to one of the following + categories: science/technology, travel, politics, sports, health, entertainment, + or geography? category only + + + ' +include: sib +task: sib_bam_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_bem.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_bem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1631a349226b60b9de250b4f97db8e474094951e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_bem.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: bem_Latn +doc_to_text: 'Does this Bemba topic; ''{{text}}'' belong to one of the following categories: + science/technology, travel, politics, sports, health, entertainment, or geography? + category only + + + ' +include: sib +task: sib_bem_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_cjk.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_cjk.yaml new file mode 100644 index 0000000000000000000000000000000000000000..85521f131a3532d7791bc3c022572bb624fd653c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_cjk.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: cjk_Latn +doc_to_text: 'Does this Chokwe topic; ''{{text}}'' belong to one of the following + categories: science/technology, travel, politics, sports, health, entertainment, + or geography? category only + + + ' +include: sib +task: sib_cjk_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_dik.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_dik.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c660516f42e0e869c8a266d113e65dcbbbf8f032 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_dik.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: dik_Latn +doc_to_text: 'Does this Southwestern Dinka topic; ''{{text}}'' belong to one of the + following categories: science/technology, travel, politics, sports, health, entertainment, + or geography? category only + + + ' +include: sib +task: sib_dik_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_dyu.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_dyu.yaml new file mode 100644 index 0000000000000000000000000000000000000000..698782fda2a65ea766eef9b91381d497949005ca --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_dyu.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: dyu_Latn +doc_to_text: 'Does this Dyula topic; ''{{text}}'' belong to one of the following categories: + science/technology, travel, politics, sports, health, entertainment, or geography? + category only + + + ' +include: sib +task: sib_dyu_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_eng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..564d86565f8aa47d9944d3a5aedc9555ca29c9a8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_eng.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: eng_Latn +doc_to_text: 'Does this English topic; ''{{text}}'' belong to one of the following + categories: science/technology, travel, politics, sports, health, entertainment, + or geography? category only + + + ' +include: sib +task: sib_eng_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_fon.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_fon.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9bb542dd84452cdd500d01a6e561c408e7a7fcf1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_fon.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: fon_Latn +doc_to_text: 'Does this Fon topic; ''{{text}}'' belong to one of the following categories: + science/technology, travel, politics, sports, health, entertainment, or geography? + category only + + + ' +include: sib +task: sib_fon_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_fra.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_fra.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cf279d611378a2a1981415940a692389728fd339 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_fra.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: fra_Latn +doc_to_text: 'Does this French topic; ''{{text}}'' belong to one of the following + categories: science/technology, travel, politics, sports, health, entertainment, + or geography? category only + + + ' +include: sib +task: sib_fra_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_fuv.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_fuv.yaml new file mode 100644 index 0000000000000000000000000000000000000000..50bb4b824748d070ad7d004efd15d2bab5cd8c0f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_fuv.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: fuv_Latn +doc_to_text: 'Does this Nigerian Fulfulde topic; ''{{text}}'' belong to one of the + following categories: science/technology, travel, politics, sports, health, entertainment, + or geography? category only + + + ' +include: sib +task: sib_fuv_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_gaz.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_gaz.yaml new file mode 100644 index 0000000000000000000000000000000000000000..601d5f79f2605a3c0db8278500ecce1f5987222a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_gaz.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: gaz_Latn +doc_to_text: 'Does this West Central Oromo topic; ''{{text}}'' belong to one of the + following categories: science/technology, travel, politics, sports, health, entertainment, + or geography? category only + + + ' +include: sib +task: sib_gaz_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_kab.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_kab.yaml new file mode 100644 index 0000000000000000000000000000000000000000..22303a3fbb1db518170ee57c258cff95c9f2c134 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_kab.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: kab_Latn +doc_to_text: 'Does this Kabyle topic; ''{{text}}'' belong to one of the following + categories: science/technology, travel, politics, sports, health, entertainment, + or geography? category only + + + ' +include: sib +task: sib_kab_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_kbp.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_kbp.yaml new file mode 100644 index 0000000000000000000000000000000000000000..679d7ccd7a74430df674154fa03af065fc4e23a8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_kbp.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: kbp_Latn +doc_to_text: 'Does this Kabiye topic; ''{{text}}'' belong to one of the following + categories: science/technology, travel, politics, sports, health, entertainment, + or geography? category only + + + ' +include: sib +task: sib_kbp_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_kea.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_kea.yaml new file mode 100644 index 0000000000000000000000000000000000000000..aee33cf27faf2fed8b6873b601a5a437dae11bb7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_kea.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: kea_Latn +doc_to_text: 'Does this Kabuverdianu topic; ''{{text}}'' belong to one of the following + categories: science/technology, travel, politics, sports, health, entertainment, + or geography? category only + + + ' +include: sib +task: sib_kea_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_kik.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_kik.yaml new file mode 100644 index 0000000000000000000000000000000000000000..77c87bc131b912e0564156acf740cb5aa3007615 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_kik.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: kik_Latn +doc_to_text: 'Does this Kikuyu topic; ''{{text}}'' belong to one of the following + categories: science/technology, travel, politics, sports, health, entertainment, + or geography? category only + + + ' +include: sib +task: sib_kik_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5be0643e11f39513363994cc6bbc02ac1604f24c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_kin.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: kin_Latn +doc_to_text: 'Does this Kinyarwanda topic; ''{{text}}'' belong to one of the following + categories: science/technology, travel, politics, sports, health, entertainment, + or geography? category only + + + ' +include: sib +task: sib_kin_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_kmb.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_kmb.yaml new file mode 100644 index 0000000000000000000000000000000000000000..02f4e9d22410d932c345df2eeb1b4de1c3e71c4b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_kmb.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: kmb_Latn +doc_to_text: 'Does this Kimbundu topic; ''{{text}}'' belong to one of the following + categories: science/technology, travel, politics, sports, health, entertainment, + or geography? category only + + + ' +include: sib +task: sib_kmb_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_knc.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_knc.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2623c480235bbf269b082a6604af129bb82e7df4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_knc.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: knc_Latn +doc_to_text: 'Does this Central Kanuri topic; ''{{text}}'' belong to one of the following + categories: science/technology, travel, politics, sports, health, entertainment, + or geography? category only + + + ' +include: sib +task: sib_knc_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_kon.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_kon.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7ec3bcf97652bde14ee764bf961ea49aca088df4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_kon.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: kon_Latn +doc_to_text: 'Does this Kikongo topic; ''{{text}}'' belong to one of the following + categories: science/technology, travel, politics, sports, health, entertainment, + or geography? category only + + + ' +include: sib +task: sib_kon_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_lin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_lin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ec2fa57a8bbc4309bbb44a865568a2cf70b842e4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_lin.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: lin_Latn +doc_to_text: 'Does this Lingala topic; ''{{text}}'' belong to one of the following + categories: science/technology, travel, politics, sports, health, entertainment, + or geography? category only + + + ' +include: sib +task: sib_lin_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_luo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_luo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9d1a438594b1818e3fe34c9ce63e47e0c802e700 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_luo.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: luo_Latn +doc_to_text: 'Does this Luo topic; ''{{text}}'' belong to one of the following categories: + science/technology, travel, politics, sports, health, entertainment, or geography? + category only + + + ' +include: sib +task: sib_luo_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_nus.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_nus.yaml new file mode 100644 index 0000000000000000000000000000000000000000..abca40e85705feeaae8fb118ae9d162c611e0545 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_nus.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: nus_Latn +doc_to_text: 'Does this Nuer topic; ''{{text}}'' belong to one of the following categories: + science/technology, travel, politics, sports, health, entertainment, or geography? + category only + + + ' +include: sib +task: sib_nus_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_plt.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_plt.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b5b385cade643f329904a7a0dab0797e57433581 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_plt.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: plt_Latn +doc_to_text: 'Does this Plateau Malagasy topic; ''{{text}}'' belong to one of the + following categories: science/technology, travel, politics, sports, health, entertainment, + or geography? category only + + + ' +include: sib +task: sib_plt_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_tir.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_tir.yaml new file mode 100644 index 0000000000000000000000000000000000000000..80dcc1bb3d8d6a65ea6dcdf75af3c946a803b071 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_tir.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: tir_Ethi +doc_to_text: 'Does this Tigrinya topic; ''{{text}}'' belong to one of the following + categories: science/technology, travel, politics, sports, health, entertainment, + or geography? category only + + + ' +include: sib +task: sib_tir_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib new file mode 100644 index 0000000000000000000000000000000000000000..fed4e5c5019f791c72cfbe214efb2698943c5b92 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib @@ -0,0 +1,43 @@ +tag: + - sib_tasks + - sib_prompt_3 + - afrobench_TC_tasks +dataset_path: Davlan/sib200 +dataset_name: null +output_type: multiple_choice +validation_split: validation +test_split: test +fewshot_split: validation +doc_to_target: category +doc_to_choice: + - "science/technology" + - "travel" + - "politics" + - "sports" + - "health" + - "entertainment" + - "geography" +should_decontaminate: true +doc_to_decontamination_query: text +metric_list: + - metric: f1 + aggregation: !function utils.weighted_f1_score + # aggregation: mean + average: weighted + hf_evaluate: true + higher_is_better: True + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" + - metric: acc + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_aeb.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_aeb.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7b82cc4ec3cc8cff2dff2818f9238477ea12528a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_aeb.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: aeb_Arab +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories science/technology, travel, politics, sports, health, entertainment,\ + \ or geography; what is the topic of the Tunisian Arabic statement below? Return\ + \ only the category. \n\ntext: {{text}} \\category:\n\n" +include: sib +task: sib_aeb_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_afr.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_afr.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f818759646be15a6c6d1c0193a7b24deb730bb03 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_afr.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: afr_Latn +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories science/technology, travel, politics, sports, health, entertainment,\ + \ or geography; what is the topic of the Afrikaans statement below? Return only\ + \ the category. \n\ntext: {{text}} \\category:\n\n" +include: sib +task: sib_afr_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_aka.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_aka.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6d4ff4e42cf1d7c926015812ea5c716928b697fc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_aka.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: aka_Latn +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories science/technology, travel, politics, sports, health, entertainment,\ + \ or geography; what is the topic of the Akan statement below? Return only the category.\ + \ \n\ntext: {{text}} \\category:\n\n" +include: sib +task: sib_aka_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_ary.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_ary.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ccb9a06880d6aa946ffd750429da3fb650c46eea --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_ary.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: ary_Arab +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories science/technology, travel, politics, sports, health, entertainment,\ + \ or geography; what is the topic of the Moroccan Arabic statement below? Return\ + \ only the category. \n\ntext: {{text}} \\category:\n\n" +include: sib +task: sib_ary_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_arz.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_arz.yaml new file mode 100644 index 0000000000000000000000000000000000000000..19ebbed7b9a441cca520eff58a663354d29a7395 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_arz.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: arz_Arab +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories science/technology, travel, politics, sports, health, entertainment,\ + \ or geography; what is the topic of the Egyptian Arabic statement below? Return\ + \ only the category. \n\ntext: {{text}} \\category:\n\n" +include: sib +task: sib_arz_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_cjk.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_cjk.yaml new file mode 100644 index 0000000000000000000000000000000000000000..470612b51b1f7d4821573764692a04f2a623a42f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_cjk.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: cjk_Latn +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories science/technology, travel, politics, sports, health, entertainment,\ + \ or geography; what is the topic of the Chokwe statement below? Return only the\ + \ category. \n\ntext: {{text}} \\category:\n\n" +include: sib +task: sib_cjk_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/sib.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/sib.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d6935fee28978fd5f7efb02afd1a54dac363d111 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/sib.yaml @@ -0,0 +1,13 @@ +group: sib +task: + - sib_prompt_1 + - sib_prompt_2 + - sib_prompt_3 + - sib_prompt_4 + - sib_prompt_5 +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..d99649e343fa4c491c77cb3167c89cc09907f579 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/utils.py @@ -0,0 +1,227 @@ +import argparse +import os + +import yaml + + +class FunctionTag: + def __init__(self, value): + self.value = value + + +def prompt_func(mode, lang): + prompt_map = { + "prompt_1": "Given the categories science/technology, travel, politics, sports, health, entertainment, or geography; what category does the text: '{{text}}' belong to: \n\n", + "prompt_2": f"Does this {lang} topic; " + "'{{text}}' belong to one of the following categories: science/technology, travel, politics, sports, health, entertainment, or geography? category only\n\n", + "prompt_3": f"You are an assistant able to classify topics in texts. \n\n" + f"Given the categories science/technology, travel, politics, sports, health, entertainment, or geography; what is " + f"the topic of the {lang} statement below? Return only the category. " + "\n\ntext: {{text}} \category:\n\n", + "prompt_4": "Label the following text as science/technology, travel, politics, sports, health, entertainment, or geography. Provide only the category as your " + "response. \n\ntext: {{text}} \category: \n\n", + "prompt_5": f"You are tasked with performing topic classification on the following {lang} text. " + f"For each input, classify the topic as science/technology, travel, politics, sports, health, entertainment, or geography. " + f"Use the following guidelines: \n\n " + f"science/technology: The text discusses scientific discoveries, technological advancements, or related topics. \n" + f"travel: The text describes travel experiences, destinations, or related topics. \n" + f"politics: The text covers political events, policies, or related topics. \n" + f"sports: The text talks about sports events, athletes, or related topics. \n" + f"health: The text addresses health issues, medical advancements, or related topics. \n" + f"entertainment: The text pertains to movies, music, celebrities, or related topics. \n" + f"geography: The text involves geographical information, locations, or related topics. \n\n" + f"If the text contains multiple topics, choose the dominant topic. " + f"For ambiguous or unclear topics, select the category that best reflects the overall content. " + "Please provide a single classification for each input.\n\ntext: {{text}} \category: \n\n", + } + return prompt_map[mode] + + +def gen_lang_yamls(output_dir: str, overwrite: bool, mode: str) -> None: + """ + Generate a yaml file for each language. + + :param output_dir: The directory to output the files to. + :param overwrite: Whether to overwrite files if they already exist. + """ + err = [] + languages = { + "aeb": "Tunisian Arabic", + "afr": "Afrikaans", + "aka": "Akan", + "amh": "Amharic", + "ary": "Moroccan Arabic", + "arz": "Egyptian Arabic", + "bam": "Bambara", + "bem": "Bemba", + "cjk": "Chokwe", + "dik": "Southwestern Dinka", + "dyu": "Dyula", + "eng": "English", + "ewe": "Ewe", + "fon": "Fon", + "fra": "French", + "fuv": "Nigerian Fulfulde", + "gaz": "West Central Oromo", + "hau": "Hausa", + "ibo": "Igbo", + "kab": "Kabyle", + "kam": "Kamba", + "kmb": "Kimbundu", + "kbp": "Kabiye", + "kea": "Kabuverdianu", + "kik": "Kikuyu", + "kin": "Kinyarwanda", + "kon": "Kikongo", + "knc": "Central Kanuri", + "lua": "Luba-Kasai", + "lug": "Luganda", + "luo": "Luo", + "lin": "Lingala", + "mos": "Mossi", + "nus": "Nuer", + "nso": "Northern Sotho", + "nya": "Nyanga", + "plt": "Plateau Malagasy", + "por": "Portuguese", + "run": "Rundi", + "sag": "Sango", + "sna": "Shona", + "som": "Somali", + "sot": "Southern Sotho", + "ssw": "Swazi", + "swa": "Swahili", + "taq": "Tamasheq", + "tir": "Tigrinya", + "tum": "Tumbuka", + "tso": "Tsonga", + "twi": "Twi", + "tzm": "Tamazight", + "umb": "Umbundu", + "wol": "Wolof", + "xho": "Xhosa", + "yor": "Yoruba", + "zul": "Zulu", + } + + lang_2_dataset_lang_code = { + "aeb": "aeb_Arab", + "afr": "afr_Latn", + "aka": "aka_Latn", + "amh": "amh_Ethi", + "ary": "ary_Arab", + "arz": "arz_Arab", + "bam": "bam_Latn", + "bem": "bem_Latn", + "cjk": "cjk_Latn", + "dik": "dik_Latn", + "dyu": "dyu_Latn", + "eng": "eng_Latn", + "ewe": "ewe_Latn", + "fon": "fon_Latn", + "fra": "fra_Latn", + "fuv": "fuv_Latn", + "gaz": "gaz_Latn", + "hau": "hau_Latn", + "ibo": "ibo_Latn", + "kab": "kab_Latn", + "kam": "kam_Latn", + "kmb": "kmb_Latn", + "kbp": "kbp_Latn", + "kea": "kea_Latn", + "kik": "kik_Latn", + "kin": "kin_Latn", + "kon": "kon_Latn", + "knc": "knc_Latn", + "lua": "lua_Latn", + "lug": "lug_Latn", + "luo": "luo_Latn", + "lin": "lin_Latn", + "mos": "mos_Latn", + "nus": "nus_Latn", + "nso": "nso_Latn", + "nya": "nya_Latn", + "plt": "plt_Latn", + "por": "por_Latn", + "run": "run_Latn", + "sag": "sag_Latn", + "sna": "sna_Latn", + "som": "som_Latn", + "sot": "sot_Latn", + "ssw": "ssw_Latn", + "swa": "swh_Latn", + "taq": "taq_Latn", + "tir": "tir_Ethi", + "tum": "tum_Latn", + "tso": "tso_Latn", + "twi": "twi_Latn", + "tzm": "tzm_Tfng", + "umb": "umb_Latn", + "wol": "wol_Latn", + "xho": "xho_Latn", + "yor": "yor_Latn", + "zul": "zul_Latn", + } + + for lang in languages.keys(): + try: + file_name = f"sib_{lang}.yaml" + task_name = f"sib_{lang}_{mode}" + yaml_template = "sib" + yaml_details = { + "include": yaml_template, + "task": task_name, + "dataset_name": lang_2_dataset_lang_code[lang], + "doc_to_text": prompt_func(mode, languages[lang]), + } + file_path = os.path.join(output_dir, mode) + os.makedirs(file_path, exist_ok=True) + + with open( + f"{output_dir}/{mode}/{file_name}", + "w" if overwrite else "x", + encoding="utf8", + ) as f: + f.write("# Generated by utils.py\n") + yaml.dump( + yaml_details, + f, + allow_unicode=True, + ) + except FileExistsError: + err.append(file_name) + + if len(err) > 0: + raise FileExistsError( + "Files were not created because they already exist (use --overwrite flag):" + f" {', '.join(err)}" + ) + + +def main() -> None: + """Parse CLI args and generate language-specific yaml files.""" + parser = argparse.ArgumentParser() + parser.add_argument( + "--overwrite", + default=True, + action="store_true", + help="Overwrite files if they already exist", + ) + parser.add_argument( + "--output-dir", + default="./", + help="Directory to write yaml files to", + ) + parser.add_argument( + "--mode", + default="prompt_3", + choices=["prompt_1", "prompt_2", "prompt_3", "prompt_4", "prompt_5"], + help="Prompt number", + ) + args = parser.parse_args() + + gen_lang_yamls(output_dir=args.output_dir, overwrite=args.overwrite, mode=args.mode) + + +if __name__ == "__main__": + main()