diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_nso_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_nso_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..028aa75cc17d326bf4d1d85b5c96ff050bb8d78e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_nso_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: nso_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Northern Sotho\ + \ sentences to English \nNorthern Sotho: {{sentence_nso_Latn}}\nEnglish: " +include: flores +task: flores_nso_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_tir_Ethi-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_tir_Ethi-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..56607b6a6e76f921917eb4453b0851dd8a9fb415 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_tir_Ethi-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: tir_Ethi-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Tigrinya sentences\ + \ to English \nTigrinya: {{sentence_tir_Ethi}}\nEnglish: " +include: flores +task: flores_tir_Ethi-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_tum_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_tum_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d70a89b24f187643ac4e93dcad084e598385207d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_tum_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: tum_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Tumbuka sentences\ + \ to English \nTumbuka: {{sentence_tum_Latn}}\nEnglish: " +include: flores +task: flores_tum_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_twi_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_twi_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d9dc957751e0c5115f4f8cb9d3bd47cfd3a66d9d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_twi_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: twi_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Twi sentences\ + \ to English \nTwi: {{sentence_twi_Latn}}\nEnglish: " +include: flores +task: flores_twi_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_tzm_Tfng-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_tzm_Tfng-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..81f9c721e731ce51ac8cc8a8adc31225edfc3d59 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_tzm_Tfng-eng_Latn.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: tzm_Tfng-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Central Atlas\ + \ Tamazight sentences to English \nCentral Atlas Tamazight: {{sentence_tzm_Tfng}}\n\ + English: " +include: flores +task: flores_tzm_Tfng-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_umb_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_umb_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..983675b039a2ba51232dede065edec9dd7536c75 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_umb_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: umb_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Umbundu sentences\ + \ to English \nUmbundu: {{sentence_umb_Latn}}\nEnglish: " +include: flores +task: flores_umb_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_yor_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_yor_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e066592660b79c6b5e4d5c6046786a2b118e1eed --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_yor_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: yor_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Yoruba sentences\ + \ to English \nYoruba: {{sentence_yor_Latn}}\nEnglish: " +include: flores +task: flores_yor_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_zul_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_zul_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f3b2fef466a1599ed1c5920328031176db342169 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/african-english/flores_zul_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: zul_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "You are a translation expert. Translate the following Zulu sentences\ + \ to English \nZulu: {{sentence_zul_Latn}}\nEnglish: " +include: flores +task: flores_zul_Latn-eng_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-ace_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-ace_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4356785a7d7a7de55ea328e7957ba14764a26745 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-ace_Latn.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: eng_Latn-ace_Latn +doc_to_target: sentence_ace_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Acehnese (Latin script) \nEnglish: {{sentence_eng_Latn}} \nAcehnese (Latin\ + \ script): " +include: flores +task: flores_eng_Latn-ace_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-dyu_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-dyu_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a9fab72b27ebc9d9c9a80dd7b41c0d270f1114e0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-dyu_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-dyu_Latn +doc_to_target: sentence_dyu_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Dyula \nEnglish: {{sentence_eng_Latn}} \nDyula: " +include: flores +task: flores_eng_Latn-dyu_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-gaz_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-gaz_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..36fa1d6c4e1fad7f33e7182abfdca60a8df9d386 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-gaz_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-gaz_Latn +doc_to_target: sentence_gaz_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Oromo \nEnglish: {{sentence_eng_Latn}} \nOromo: " +include: flores +task: flores_eng_Latn-gaz_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-ibo_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-ibo_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b31e37cd4ce5a6891f7dff30ff75f05b99bdc48c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-ibo_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-ibo_Latn +doc_to_target: sentence_ibo_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Igbo \nEnglish: {{sentence_eng_Latn}} \nIgbo: " +include: flores +task: flores_eng_Latn-ibo_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-kab_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-kab_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1d6cfd8cb97ca07352c0d7927bc2476a3e9e378a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-kab_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-kab_Latn +doc_to_target: sentence_kab_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Kabyle \nEnglish: {{sentence_eng_Latn}} \nKabyle: " +include: flores +task: flores_eng_Latn-kab_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-kam_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-kam_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dd2da95c49b7828dfbc174d6a9d891546d433ecd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-kam_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-kam_Latn +doc_to_target: sentence_kam_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Kamba \nEnglish: {{sentence_eng_Latn}} \nKamba: " +include: flores +task: flores_eng_Latn-kam_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-kik_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-kik_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1519f36e63c76ba57759547e91bab111c3796dcf --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-kik_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-kik_Latn +doc_to_target: sentence_kik_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Kikuyu \nEnglish: {{sentence_eng_Latn}} \nKikuyu: " +include: flores +task: flores_eng_Latn-kik_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-kin_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-kin_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2b33033ff00dce959d37f6b7fe8f0440a6cd1577 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-kin_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-kin_Latn +doc_to_target: sentence_kin_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Kinyarwanda \nEnglish: {{sentence_eng_Latn}} \nKinyarwanda: " +include: flores +task: flores_eng_Latn-kin_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-kmb_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-kmb_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..803989174a43ad7567cc321f7f7847039bc516d5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-kmb_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-kmb_Latn +doc_to_target: sentence_kmb_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Kimbundu \nEnglish: {{sentence_eng_Latn}} \nKimbundu: " +include: flores +task: flores_eng_Latn-kmb_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-knc_Arab.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-knc_Arab.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c0d262413539f659922105528184c6b1f9c74f05 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-knc_Arab.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: eng_Latn-knc_Arab +doc_to_target: sentence_knc_Arab +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Central Kanuri (Arabic script) \nEnglish: {{sentence_eng_Latn}} \nCentral Kanuri\ + \ (Arabic script): " +include: flores +task: flores_eng_Latn-knc_Arab_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-knc_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-knc_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..61ea7a2cdf03e9cd2e6fcef2abcc6e072cb5f430 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-knc_Latn.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: eng_Latn-knc_Latn +doc_to_target: sentence_knc_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Central Kanuri (Latin script) \nEnglish: {{sentence_eng_Latn}} \nCentral Kanuri\ + \ (Latin script): " +include: flores +task: flores_eng_Latn-knc_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-kon_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-kon_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1967452e0032b023b48dfd3980e9d8241aed8e09 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-kon_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-kon_Latn +doc_to_target: sentence_kon_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Kikongo \nEnglish: {{sentence_eng_Latn}} \nKikongo: " +include: flores +task: flores_eng_Latn-kon_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-lin_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-lin_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..05e2593bdee5d218324f959277480f74db95a82b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-lin_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-lin_Latn +doc_to_target: sentence_lin_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Lingala \nEnglish: {{sentence_eng_Latn}} \nLingala: " +include: flores +task: flores_eng_Latn-lin_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-lua_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-lua_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5f4fe01e16cf1715dbf8467e9bb6fb1558f4b923 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-lua_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-lua_Latn +doc_to_target: sentence_lua_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Luba-Kasai \nEnglish: {{sentence_eng_Latn}} \nLuba-Kasai: " +include: flores +task: flores_eng_Latn-lua_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-lug_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-lug_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0cfc35568598cd7733748a5f06fa5a1ad5c7c85e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-lug_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-lug_Latn +doc_to_target: sentence_lug_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Luganda \nEnglish: {{sentence_eng_Latn}} \nLuganda: " +include: flores +task: flores_eng_Latn-lug_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-luo_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-luo_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..05c027bb0256d1a22e1c14ca2812b6f9abb65fb7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-luo_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-luo_Latn +doc_to_target: sentence_luo_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Luo \nEnglish: {{sentence_eng_Latn}} \nLuo: " +include: flores +task: flores_eng_Latn-luo_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-mos_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-mos_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0a676522a51603f951c5dfe0d88a6d99823b46eb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-mos_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-mos_Latn +doc_to_target: sentence_mos_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Mossi \nEnglish: {{sentence_eng_Latn}} \nMossi: " +include: flores +task: flores_eng_Latn-mos_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-nso_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-nso_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c681b492c17f5e95709c9f0bd06637b10c07c9c3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-nso_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-nso_Latn +doc_to_target: sentence_nso_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Northern Sotho \nEnglish: {{sentence_eng_Latn}} \nNorthern Sotho: " +include: flores +task: flores_eng_Latn-nso_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-nus_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-nus_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7ae375058b9393357df2aca5f52c69d6b6fde744 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-nus_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-nus_Latn +doc_to_target: sentence_nus_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Nuer \nEnglish: {{sentence_eng_Latn}} \nNuer: " +include: flores +task: flores_eng_Latn-nus_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-nya_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-nya_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..135029028e124537ec4b2dab4222fcb582d38beb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-nya_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-nya_Latn +doc_to_target: sentence_nya_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Nyanja \nEnglish: {{sentence_eng_Latn}} \nNyanja: " +include: flores +task: flores_eng_Latn-nya_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-plt_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-plt_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..faa85197438e9ff31744de7683957736b3ad34bc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-plt_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-plt_Latn +doc_to_target: sentence_plt_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Plateau Malagasy \nEnglish: {{sentence_eng_Latn}} \nPlateau Malagasy: " +include: flores +task: flores_eng_Latn-plt_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-run_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-run_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8b670e3f7146cec72c15c9be17ad0df6b30a1a4b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-run_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-run_Latn +doc_to_target: sentence_run_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Rundi \nEnglish: {{sentence_eng_Latn}} \nRundi: " +include: flores +task: flores_eng_Latn-run_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-sag_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-sag_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..32f399391b905b77d5bea93229fc6b4c5de9e533 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-sag_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-sag_Latn +doc_to_target: sentence_sag_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Sango \nEnglish: {{sentence_eng_Latn}} \nSango: " +include: flores +task: flores_eng_Latn-sag_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-sna_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-sna_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0e219c40275fb7938cc2a121822b534776aff57b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-sna_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-sna_Latn +doc_to_target: sentence_sna_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Shona \nEnglish: {{sentence_eng_Latn}} \nShona: " +include: flores +task: flores_eng_Latn-sna_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-ssw_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-ssw_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..23b9216f912ba5cd340181dd5c07b19c4ff03c7d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-ssw_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-ssw_Latn +doc_to_target: sentence_ssw_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Swati \nEnglish: {{sentence_eng_Latn}} \nSwati: " +include: flores +task: flores_eng_Latn-ssw_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-sun_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-sun_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9f51ced5e6ce5353da73945815adbaab9e9c0d94 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-sun_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-sun_Latn +doc_to_target: sentence_sun_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Sundanese \nEnglish: {{sentence_eng_Latn}} \nSundanese: " +include: flores +task: flores_eng_Latn-sun_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-taq_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-taq_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0b09b52f46c56e15fc30aff90cbac8c8b8f8e2b8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-taq_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-taq_Latn +doc_to_target: sentence_taq_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Tamasheq \nEnglish: {{sentence_eng_Latn}} \nTamasheq: " +include: flores +task: flores_eng_Latn-taq_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-tsn_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-tsn_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1e592366ebb36a4c621f05e31326fbf36125025b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_2/english-african/flores_eng_Latn-tsn_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn-tsn_Latn +doc_to_target: sentence_tsn_Latn +doc_to_text: "You are a translation expert. Translate the following English sentences\ + \ to Setswana \nEnglish: {{sentence_eng_Latn}} \nSetswana: " +include: flores +task: flores_eng_Latn-tsn_Latn_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores new file mode 100644 index 0000000000000000000000000000000000000000..60bf41116e43ccdd17efcdcbe0e72c8aad0cf684 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores @@ -0,0 +1,27 @@ +tag: +- african_flores_tasks +- flores_afr-eng +- flores_afr-eng_prompt_3 +- afrobench_MT_tasks +dataset_path: facebook/flores +dataset_kwargs: {trust_remote_code: True} +output_type: generate_until +validation_split: dev +fewshot_split: dev +test_split: devtest +metric_list: + - metric: bleu + aggregation: bleu + higher_is_better: true + - metric: chrf + aggregation: chrf + higher_is_better: true +generation_kwargs: + until: + - "**" + - + do_sample: false + temperature: 0.0 +repeats: 1 +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_ace_Arab-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_ace_Arab-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ee5f12704a3a7f03a52aba093e01e335a98729ff --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_ace_Arab-eng_Latn.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: ace_Arab-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Acehnese (Arabic script) and English linguist, translate the following\ + \ Acehnese (Arabic script) sentences to English \nAcehnese (Arabic script): {{sentence_ace_Arab}}\n\ + English: " +include: flores +task: flores_ace_Arab-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_ace_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_ace_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e1d70ba341b8e30123cfd2885f570e5050c359a5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_ace_Latn-eng_Latn.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: ace_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Acehnese (Latin script) and English linguist, translate the following\ + \ Acehnese (Latin script) sentences to English \nAcehnese (Latin script): {{sentence_ace_Latn}}\n\ + English: " +include: flores +task: flores_ace_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_acq_Arab-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_acq_Arab-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8cda39626e72a3df883f1b74c314e8c275fa4fbb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_acq_Arab-eng_Latn.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: acq_Arab-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Ta’izzi-Adeni Arabic and English linguist, translate the following\ + \ Ta’izzi-Adeni Arabic sentences to English \nTa’izzi-Adeni Arabic: {{sentence_acq_Arab}}\n\ + English: " +include: flores +task: flores_acq_Arab-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_aeb_Arab-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_aeb_Arab-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..97f8ef2c91bd0255f2b887ff6b1acb75fc1c0487 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_aeb_Arab-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: aeb_Arab-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Tunisian Arabic and English linguist, translate the following Tunisian\ + \ Arabic sentences to English \nTunisian Arabic: {{sentence_aeb_Arab}}\nEnglish: " +include: flores +task: flores_aeb_Arab-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_afr_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_afr_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e228cb9c66858d173835016566cd1f4731038120 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_afr_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: afr_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Afrikaans and English linguist, translate the following Afrikaans\ + \ sentences to English \nAfrikaans: {{sentence_afr_Latn}}\nEnglish: " +include: flores +task: flores_afr_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_aka_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_aka_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6d6fc38582c415828023478f33f2925427a68cbb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_aka_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: aka_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Akan and English linguist, translate the following Akan sentences\ + \ to English \nAkan: {{sentence_aka_Latn}}\nEnglish: " +include: flores +task: flores_aka_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_amh_Ethi-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_amh_Ethi-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..58f33f9a13c5c9840fd4dcdcdd12c664daf60878 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_amh_Ethi-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: amh_Ethi-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Amharic and English linguist, translate the following Amharic sentences\ + \ to English \nAmharic: {{sentence_amh_Ethi}}\nEnglish: " +include: flores +task: flores_amh_Ethi-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_ary_Arab-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_ary_Arab-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3006ebf72c0088340346a3a7b8f140da84211048 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_ary_Arab-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: ary_Arab-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Moroccan Arabic and English linguist, translate the following Moroccan\ + \ Arabic sentences to English \nMoroccan Arabic: {{sentence_ary_Arab}}\nEnglish: " +include: flores +task: flores_ary_Arab-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_arz_Arab-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_arz_Arab-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..46cc0a18d4633b7032c7c179606845941f595e8b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_arz_Arab-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: arz_Arab-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Egyptian Arabic and English linguist, translate the following Egyptian\ + \ Arabic sentences to English \nEgyptian Arabic: {{sentence_arz_Arab}}\nEnglish: " +include: flores +task: flores_arz_Arab-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_bam_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_bam_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c50a8dfa4ae3b2a99c2bb40f64749fb22c8928ac --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_bam_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: bam_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Bambara and English linguist, translate the following Bambara sentences\ + \ to English \nBambara: {{sentence_bam_Latn}}\nEnglish: " +include: flores +task: flores_bam_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_ban_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_ban_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..86f2eed3fef3b2aefef9f0a7e640310d054e3fc9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_ban_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: ban_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Balinese and English linguist, translate the following Balinese\ + \ sentences to English \nBalinese: {{sentence_ban_Latn}}\nEnglish: " +include: flores +task: flores_ban_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_bem_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_bem_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..55c32fe9c5e3f4321b6c3145d862d23f48da233b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_bem_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: bem_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Bemba and English linguist, translate the following Bemba sentences\ + \ to English \nBemba: {{sentence_bem_Latn}}\nEnglish: " +include: flores +task: flores_bem_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_cjk_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_cjk_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..642cd4dda88f9c7e38092fbc650e8340ff8998a2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_cjk_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: cjk_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Chokwe and English linguist, translate the following Chokwe sentences\ + \ to English \nChokwe: {{sentence_cjk_Latn}}\nEnglish: " +include: flores +task: flores_cjk_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_dik_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_dik_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8005a642241e9ba4e9255a224f6c7d641553edd7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_dik_Latn-eng_Latn.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: dik_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Southwestern Dinka and English linguist, translate the following\ + \ Southwestern Dinka sentences to English \nSouthwestern Dinka: {{sentence_dik_Latn}}\n\ + English: " +include: flores +task: flores_dik_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_dyu_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_dyu_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a99efc0867c365186db75a8f04a0fbfa741f91c2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_dyu_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: dyu_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Dyula and English linguist, translate the following Dyula sentences\ + \ to English \nDyula: {{sentence_dyu_Latn}}\nEnglish: " +include: flores +task: flores_dyu_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_fra_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_fra_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b10c46e3226342c3a01c90f881df8575049eb6b6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_fra_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: fra_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a French and English linguist, translate the following French sentences\ + \ to English \nFrench: {{sentence_fra_Latn}}\nEnglish: " +include: flores +task: flores_fra_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_fuv_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_fuv_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ffcbd3c04f1f6fd608e11286be4d88c079890a88 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_fuv_Latn-eng_Latn.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: fuv_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Nigerian Fulfulde and English linguist, translate the following\ + \ Nigerian Fulfulde sentences to English \nNigerian Fulfulde: {{sentence_fuv_Latn}}\n\ + English: " +include: flores +task: flores_fuv_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_gaz_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_gaz_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..703cd3517a81e172683fb43b91ddbb4ca7db500e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_gaz_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: gaz_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Oromo and English linguist, translate the following Oromo sentences\ + \ to English \nOromo: {{sentence_gaz_Latn}}\nEnglish: " +include: flores +task: flores_gaz_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_hau_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_hau_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7527bf78ebc88167a99707eb3101b1c350e5c991 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_hau_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: hau_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Hausa and English linguist, translate the following Hausa sentences\ + \ to English \nHausa: {{sentence_hau_Latn}}\nEnglish: " +include: flores +task: flores_hau_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_kab_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_kab_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ec406c5e0fbf8f5b41b17e432586c00f8383eabd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_kab_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: kab_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Kabyle and English linguist, translate the following Kabyle sentences\ + \ to English \nKabyle: {{sentence_kab_Latn}}\nEnglish: " +include: flores +task: flores_kab_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_kin_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_kin_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8bb14aedf4e95de93a3d4da32a07b19bf854b697 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_kin_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: kin_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Kinyarwanda and English linguist, translate the following Kinyarwanda\ + \ sentences to English \nKinyarwanda: {{sentence_kin_Latn}}\nEnglish: " +include: flores +task: flores_kin_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_kmb_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_kmb_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c31ede4dfe8449d2a1c8e84b74eeee0fdc908b78 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_kmb_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: kmb_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Kimbundu and English linguist, translate the following Kimbundu\ + \ sentences to English \nKimbundu: {{sentence_kmb_Latn}}\nEnglish: " +include: flores +task: flores_kmb_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_knc_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_knc_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9621de73d33ea37feeb3a4face5dbe50812bf9ca --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_knc_Latn-eng_Latn.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: knc_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Central Kanuri (Latin script) and English linguist, translate the\ + \ following Central Kanuri (Latin script) sentences to English \nCentral Kanuri\ + \ (Latin script): {{sentence_knc_Latn}}\nEnglish: " +include: flores +task: flores_knc_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_kon_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_kon_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..54ede3a6e2c0280761a3af529cc0a8fe82d2f518 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_kon_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: kon_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Kikongo and English linguist, translate the following Kikongo sentences\ + \ to English \nKikongo: {{sentence_kon_Latn}}\nEnglish: " +include: flores +task: flores_kon_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_nus_Latn-eng_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_nus_Latn-eng_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a3e0d1e3ac8ff35a2b0c89f37cb2697a6d15299a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/african-english/flores_nus_Latn-eng_Latn.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: nus_Latn-eng_Latn +doc_to_target: sentence_eng_Latn +doc_to_text: "As a Nuer and English linguist, translate the following Nuer sentences\ + \ to English \nNuer: {{sentence_nus_Latn}}\nEnglish: " +include: flores +task: flores_nus_Latn-eng_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-ace_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-ace_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..766b7c30061e8adfd3e4827052fba7160483ae4a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/english-african/flores_eng_Latn-ace_Latn.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: eng_Latn-ace_Latn +doc_to_target: sentence_ace_Latn +doc_to_text: "As a Acehnese (Latin script) and English linguist, translate the following\ + \ English sentences to Acehnese (Latin script) \nEnglish: {{sentence_eng_Latn}}\ + \ \nAcehnese (Latin script): " +include: flores +task: flores_eng_Latn-ace_Latn_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/flores b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/flores new file mode 100644 index 0000000000000000000000000000000000000000..74f9f33eb22662bec79709bd64d8d31f3fb8eae0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/flores/prompt_3/flores @@ -0,0 +1,24 @@ +tag: +- flores_tasks +- flores_afr-eng +dataset_path: facebook/flores +dataset_kwargs: {trust_remote_code: True} +output_type: generate_until +validation_split: dev +fewshot_split: dev +test_split: devtest +metric_list: + - metric: bleu + aggregation: bleu + higher_is_better: true + - metric: chrf + aggregation: chrf + higher_is_better: true +generation_kwargs: + until: + - "\n" + do_sample: false + temperature: 0.0 +repeats: 1 +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_tum.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_tum.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9e17521fb63ca03a4b38747157cf0171dcb2cf13 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_tum.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: tum_Latn +doc_to_text: "Given the categories science/technology, travel, politics, sports, health,\ + \ entertainment, or geography; what category does the text: '{{text}}' belong to:\ + \ \n\n" +include: sib +task: sib_tum_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bf818808af1a80bfa7cfad46b5f16930b5619636 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_twi.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: twi_Latn +doc_to_text: "Given the categories science/technology, travel, politics, sports, health,\ + \ entertainment, or geography; what category does the text: '{{text}}' belong to:\ + \ \n\n" +include: sib +task: sib_twi_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..694ddfc11f55e33b04544bde8a2004939e8bb158 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_1/sib_zul.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: zul_Latn +doc_to_text: "Given the categories science/technology, travel, politics, sports, health,\ + \ entertainment, or geography; what category does the text: '{{text}}' belong to:\ + \ \n\n" +include: sib +task: sib_zul_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..259009f056b82ff8968feb9df082f7c232845124 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_amh.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: amh_Ethi +doc_to_text: 'Does this Amharic topic; ''{{text}}'' belong to one of the following + categories: science/technology, travel, politics, sports, health, entertainment, + or geography? category only + + + ' +include: sib +task: sib_amh_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_ewe.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_ewe.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ba064082941553b1177d6c4ea4901e6aa7ba61be --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_ewe.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: ewe_Latn +doc_to_text: 'Does this Ewe topic; ''{{text}}'' belong to one of the following categories: + science/technology, travel, politics, sports, health, entertainment, or geography? + category only + + + ' +include: sib +task: sib_ewe_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2c7255d4747d4b5a033129bca1be441114bef36f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_hau.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: hau_Latn +doc_to_text: 'Does this Hausa topic; ''{{text}}'' belong to one of the following categories: + science/technology, travel, politics, sports, health, entertainment, or geography? + category only + + + ' +include: sib +task: sib_hau_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..714c132f655a0b57e9c160a513a6c73350c5919c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_ibo.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: ibo_Latn +doc_to_text: 'Does this Igbo topic; ''{{text}}'' belong to one of the following categories: + science/technology, travel, politics, sports, health, entertainment, or geography? + category only + + + ' +include: sib +task: sib_ibo_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_kam.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_kam.yaml new file mode 100644 index 0000000000000000000000000000000000000000..498781d6e836f9854a7703f669e80b3a16003637 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_kam.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: kam_Latn +doc_to_text: 'Does this Kamba topic; ''{{text}}'' belong to one of the following categories: + science/technology, travel, politics, sports, health, entertainment, or geography? + category only + + + ' +include: sib +task: sib_kam_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_lua.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_lua.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4f3acc3dbeb2b708257c9b5f1fcc7cac4a703d54 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_lua.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: lua_Latn +doc_to_text: 'Does this Luba-Kasai topic; ''{{text}}'' belong to one of the following + categories: science/technology, travel, politics, sports, health, entertainment, + or geography? category only + + + ' +include: sib +task: sib_lua_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1d6e7b9f0c315b7868580704abc3cdba0775cc65 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_lug.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: lug_Latn +doc_to_text: 'Does this Luganda topic; ''{{text}}'' belong to one of the following + categories: science/technology, travel, politics, sports, health, entertainment, + or geography? category only + + + ' +include: sib +task: sib_lug_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_mos.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_mos.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cc025905e76696e820804e4989e9b1bec2fa2257 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_mos.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: mos_Latn +doc_to_text: 'Does this Mossi topic; ''{{text}}'' belong to one of the following categories: + science/technology, travel, politics, sports, health, entertainment, or geography? + category only + + + ' +include: sib +task: sib_mos_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_nso.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_nso.yaml new file mode 100644 index 0000000000000000000000000000000000000000..75021cc514b64598cd1e94902ecdd653db50681e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_nso.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: nso_Latn +doc_to_text: 'Does this Northern Sotho topic; ''{{text}}'' belong to one of the following + categories: science/technology, travel, politics, sports, health, entertainment, + or geography? category only + + + ' +include: sib +task: sib_nso_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_nya.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_nya.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e09e27331ad6104a7c585f63638b4e24e9ba8880 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_nya.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: nya_Latn +doc_to_text: 'Does this Nyanga topic; ''{{text}}'' belong to one of the following + categories: science/technology, travel, politics, sports, health, entertainment, + or geography? category only + + + ' +include: sib +task: sib_nya_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_por.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_por.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a297c05a4be4992f33fb07737bd704ab076c9cfe --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_por.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: por_Latn +doc_to_text: 'Does this Portuguese topic; ''{{text}}'' belong to one of the following + categories: science/technology, travel, politics, sports, health, entertainment, + or geography? category only + + + ' +include: sib +task: sib_por_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_run.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_run.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a4bb32245653846c6eb82fef3716b31ba85adf4d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_run.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: run_Latn +doc_to_text: 'Does this Rundi topic; ''{{text}}'' belong to one of the following categories: + science/technology, travel, politics, sports, health, entertainment, or geography? + category only + + + ' +include: sib +task: sib_run_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_sag.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_sag.yaml new file mode 100644 index 0000000000000000000000000000000000000000..979b4d84e0dae472a83fdb64e7b62c35453763e9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_sag.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: sag_Latn +doc_to_text: 'Does this Sango topic; ''{{text}}'' belong to one of the following categories: + science/technology, travel, politics, sports, health, entertainment, or geography? + category only + + + ' +include: sib +task: sib_sag_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_sna.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_sna.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b41184b3c702c2346484fb6a982a1f9a10fe6516 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_sna.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: sna_Latn +doc_to_text: 'Does this Shona topic; ''{{text}}'' belong to one of the following categories: + science/technology, travel, politics, sports, health, entertainment, or geography? + category only + + + ' +include: sib +task: sib_sna_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_som.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_som.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cda1fb4133df8f42bf69a0e296b866cd128ef368 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_som.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: som_Latn +doc_to_text: 'Does this Somali topic; ''{{text}}'' belong to one of the following + categories: science/technology, travel, politics, sports, health, entertainment, + or geography? category only + + + ' +include: sib +task: sib_som_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_sot.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_sot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..08d0dbecbec823c712108c42d64f9a7cbed73463 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_sot.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: sot_Latn +doc_to_text: 'Does this Southern Sotho topic; ''{{text}}'' belong to one of the following + categories: science/technology, travel, politics, sports, health, entertainment, + or geography? category only + + + ' +include: sib +task: sib_sot_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_ssw.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_ssw.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0d3b99e7a07affa07aeb7ad4452887808fbd47de --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_ssw.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: ssw_Latn +doc_to_text: 'Does this Swazi topic; ''{{text}}'' belong to one of the following categories: + science/technology, travel, politics, sports, health, entertainment, or geography? + category only + + + ' +include: sib +task: sib_ssw_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8e9faa831698a196843ae2f6b5f8cf4939bdada0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_swa.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: swh_Latn +doc_to_text: 'Does this Swahili topic; ''{{text}}'' belong to one of the following + categories: science/technology, travel, politics, sports, health, entertainment, + or geography? category only + + + ' +include: sib +task: sib_swa_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_taq.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_taq.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f1862c468c6c874e56fca81c9bbc0df09c91425e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_taq.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: taq_Latn +doc_to_text: 'Does this Tamasheq topic; ''{{text}}'' belong to one of the following + categories: science/technology, travel, politics, sports, health, entertainment, + or geography? category only + + + ' +include: sib +task: sib_taq_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_tso.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_tso.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fad909b4b7a714ba44df14041808d54e2dc7edc2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_tso.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: tso_Latn +doc_to_text: 'Does this Tsonga topic; ''{{text}}'' belong to one of the following + categories: science/technology, travel, politics, sports, health, entertainment, + or geography? category only + + + ' +include: sib +task: sib_tso_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_tum.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_tum.yaml new file mode 100644 index 0000000000000000000000000000000000000000..613535bc95647f5edf818e844423eebad5291937 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_tum.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: tum_Latn +doc_to_text: 'Does this Tumbuka topic; ''{{text}}'' belong to one of the following + categories: science/technology, travel, politics, sports, health, entertainment, + or geography? category only + + + ' +include: sib +task: sib_tum_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..064edb4cb8e0e1a88dbc1ccfad20adefa13034e9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_twi.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: twi_Latn +doc_to_text: 'Does this Twi topic; ''{{text}}'' belong to one of the following categories: + science/technology, travel, politics, sports, health, entertainment, or geography? + category only + + + ' +include: sib +task: sib_twi_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_tzm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_tzm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7ec8adc260622a611261d30aebd49450d202b700 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_tzm.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: tzm_Tfng +doc_to_text: 'Does this Tamazight topic; ''{{text}}'' belong to one of the following + categories: science/technology, travel, politics, sports, health, entertainment, + or geography? category only + + + ' +include: sib +task: sib_tzm_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_umb.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_umb.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5a910abc5fa30c9462c950f059059f5f91554b3a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_umb.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: umb_Latn +doc_to_text: 'Does this Umbundu topic; ''{{text}}'' belong to one of the following + categories: science/technology, travel, politics, sports, health, entertainment, + or geography? category only + + + ' +include: sib +task: sib_umb_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4453b3458ecde8bbc26cff73793db71471301850 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_wol.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: wol_Latn +doc_to_text: 'Does this Wolof topic; ''{{text}}'' belong to one of the following categories: + science/technology, travel, politics, sports, health, entertainment, or geography? + category only + + + ' +include: sib +task: sib_wol_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e038cc9458fbb376331296bd4dd1c96a5f26a8f1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_xho.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: xho_Latn +doc_to_text: 'Does this Xhosa topic; ''{{text}}'' belong to one of the following categories: + science/technology, travel, politics, sports, health, entertainment, or geography? + category only + + + ' +include: sib +task: sib_xho_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e831b3117b828bb4ea68016ca19fdcd9c89525b3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_yor.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: yor_Latn +doc_to_text: 'Does this Yoruba topic; ''{{text}}'' belong to one of the following + categories: science/technology, travel, politics, sports, health, entertainment, + or geography? category only + + + ' +include: sib +task: sib_yor_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f61a4061f2b636167400491e59db890289aff3d3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/sib_zul.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: zul_Latn +doc_to_text: 'Does this Zulu topic; ''{{text}}'' belong to one of the following categories: + science/technology, travel, politics, sports, health, entertainment, or geography? + category only + + + ' +include: sib +task: sib_zul_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..3e735e2deb1f9c53152c072615aebe8ba3acb90b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_2/utils.py @@ -0,0 +1 @@ +from lm_eval.utils import weighted_f1_score diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..58207e9e39010c4f30d3ff0a1f46fd9e53f3b042 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_amh.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: amh_Ethi +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories science/technology, travel, politics, sports, health, entertainment,\ + \ or geography; what is the topic of the Amharic statement below? Return only the\ + \ category. \n\ntext: {{text}} \\category:\n\n" +include: sib +task: sib_amh_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_bam.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_bam.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d2c1a18d9b3a4b076ca70b32981b2ecb58e3f9c0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_bam.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: bam_Latn +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories science/technology, travel, politics, sports, health, entertainment,\ + \ or geography; what is the topic of the Bambara statement below? Return only the\ + \ category. \n\ntext: {{text}} \\category:\n\n" +include: sib +task: sib_bam_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_bem.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_bem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..99750497c258c93b1393ced6f34b9a724fbe518d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_bem.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: bem_Latn +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories science/technology, travel, politics, sports, health, entertainment,\ + \ or geography; what is the topic of the Bemba statement below? Return only the\ + \ category. \n\ntext: {{text}} \\category:\n\n" +include: sib +task: sib_bem_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_dik.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_dik.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c5269b0262805239b807d789a7343dc0d1507a29 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_dik.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: dik_Latn +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories science/technology, travel, politics, sports, health, entertainment,\ + \ or geography; what is the topic of the Southwestern Dinka statement below? Return\ + \ only the category. \n\ntext: {{text}} \\category:\n\n" +include: sib +task: sib_dik_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_dyu.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_dyu.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f04a1c17199e50ecebfac887459b3c3f124a1529 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_dyu.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: dyu_Latn +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories science/technology, travel, politics, sports, health, entertainment,\ + \ or geography; what is the topic of the Dyula statement below? Return only the\ + \ category. \n\ntext: {{text}} \\category:\n\n" +include: sib +task: sib_dyu_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_eng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bf22d08fcab877aae2ce77081274d1928e27f8c4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_eng.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: eng_Latn +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories science/technology, travel, politics, sports, health, entertainment,\ + \ or geography; what is the topic of the English statement below? Return only the\ + \ category. \n\ntext: {{text}} \\category:\n\n" +include: sib +task: sib_eng_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_ewe.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_ewe.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4cc991048ad19285b1dd269e91a6bb32898b6d88 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_ewe.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: ewe_Latn +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories science/technology, travel, politics, sports, health, entertainment,\ + \ or geography; what is the topic of the Ewe statement below? Return only the category.\ + \ \n\ntext: {{text}} \\category:\n\n" +include: sib +task: sib_ewe_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_fon.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_fon.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3127fde242956bcaf89224ed3a88be80dc967c52 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_fon.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: fon_Latn +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories science/technology, travel, politics, sports, health, entertainment,\ + \ or geography; what is the topic of the Fon statement below? Return only the category.\ + \ \n\ntext: {{text}} \\category:\n\n" +include: sib +task: sib_fon_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_fra.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_fra.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9a24ff30e4f6408f02a0f4a8978250d91e36621f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_fra.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: fra_Latn +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories science/technology, travel, politics, sports, health, entertainment,\ + \ or geography; what is the topic of the French statement below? Return only the\ + \ category. \n\ntext: {{text}} \\category:\n\n" +include: sib +task: sib_fra_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_fuv.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_fuv.yaml new file mode 100644 index 0000000000000000000000000000000000000000..405838c78ddbe6a99d66f436699de00f0b7e814b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_fuv.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: fuv_Latn +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories science/technology, travel, politics, sports, health, entertainment,\ + \ or geography; what is the topic of the Nigerian Fulfulde statement below? Return\ + \ only the category. \n\ntext: {{text}} \\category:\n\n" +include: sib +task: sib_fuv_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_gaz.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_gaz.yaml new file mode 100644 index 0000000000000000000000000000000000000000..282b439a3c2d6703c04446b4cf477a9bd60bf340 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_gaz.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: gaz_Latn +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories science/technology, travel, politics, sports, health, entertainment,\ + \ or geography; what is the topic of the West Central Oromo statement below? Return\ + \ only the category. \n\ntext: {{text}} \\category:\n\n" +include: sib +task: sib_gaz_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..239181bf1f1586ac83aa1741e9cffe531c2433b1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_hau.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: hau_Latn +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories science/technology, travel, politics, sports, health, entertainment,\ + \ or geography; what is the topic of the Hausa statement below? Return only the\ + \ category. \n\ntext: {{text}} \\category:\n\n" +include: sib +task: sib_hau_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0581291dd7ed5cf53e52fe2f44154363b8be9599 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_ibo.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: ibo_Latn +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories science/technology, travel, politics, sports, health, entertainment,\ + \ or geography; what is the topic of the Igbo statement below? Return only the category.\ + \ \n\ntext: {{text}} \\category:\n\n" +include: sib +task: sib_ibo_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_kab.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_kab.yaml new file mode 100644 index 0000000000000000000000000000000000000000..32fbbf4407d07a5d78558b022260117f592645d7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_kab.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: kab_Latn +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories science/technology, travel, politics, sports, health, entertainment,\ + \ or geography; what is the topic of the Kabyle statement below? Return only the\ + \ category. \n\ntext: {{text}} \\category:\n\n" +include: sib +task: sib_kab_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_kam.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_kam.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3f745ba54f9daaca1a3302443c4a5aba3de795f4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_kam.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: kam_Latn +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories science/technology, travel, politics, sports, health, entertainment,\ + \ or geography; what is the topic of the Kamba statement below? Return only the\ + \ category. \n\ntext: {{text}} \\category:\n\n" +include: sib +task: sib_kam_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_kbp.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_kbp.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c5be1035bb58862fe73c2678287620d409a76b87 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_kbp.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: kbp_Latn +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories science/technology, travel, politics, sports, health, entertainment,\ + \ or geography; what is the topic of the Kabiye statement below? Return only the\ + \ category. \n\ntext: {{text}} \\category:\n\n" +include: sib +task: sib_kbp_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_kea.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_kea.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a1d3e2a68cf8b465f589701f1004ae4b5dc07dd9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_kea.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: kea_Latn +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories science/technology, travel, politics, sports, health, entertainment,\ + \ or geography; what is the topic of the Kabuverdianu statement below? Return only\ + \ the category. \n\ntext: {{text}} \\category:\n\n" +include: sib +task: sib_kea_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_kik.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_kik.yaml new file mode 100644 index 0000000000000000000000000000000000000000..521a0f89226460e6f1a9e25c0d24066cd929c662 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_kik.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: kik_Latn +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories science/technology, travel, politics, sports, health, entertainment,\ + \ or geography; what is the topic of the Kikuyu statement below? Return only the\ + \ category. \n\ntext: {{text}} \\category:\n\n" +include: sib +task: sib_kik_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..066bfb630c59e334f65dcd74ea536a6790b3337d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_kin.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: kin_Latn +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories science/technology, travel, politics, sports, health, entertainment,\ + \ or geography; what is the topic of the Kinyarwanda statement below? Return only\ + \ the category. \n\ntext: {{text}} \\category:\n\n" +include: sib +task: sib_kin_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_kmb.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_kmb.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c548af893d77f13231cab13318e396cbcf423388 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_kmb.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: kmb_Latn +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories science/technology, travel, politics, sports, health, entertainment,\ + \ or geography; what is the topic of the Kimbundu statement below? Return only the\ + \ category. \n\ntext: {{text}} \\category:\n\n" +include: sib +task: sib_kmb_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_knc.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_knc.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9136823770a1e6754070143b2b0e40a988da22f4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_knc.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: knc_Latn +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories science/technology, travel, politics, sports, health, entertainment,\ + \ or geography; what is the topic of the Central Kanuri statement below? Return\ + \ only the category. \n\ntext: {{text}} \\category:\n\n" +include: sib +task: sib_knc_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_kon.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_kon.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d8777511ef33fad4019d3e157d1dbc4f6d0aad96 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_kon.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: kon_Latn +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories science/technology, travel, politics, sports, health, entertainment,\ + \ or geography; what is the topic of the Kikongo statement below? Return only the\ + \ category. \n\ntext: {{text}} \\category:\n\n" +include: sib +task: sib_kon_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_lin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_lin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8688cb875fa5554625073325810a9dbb1198f06b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_lin.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: lin_Latn +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories science/technology, travel, politics, sports, health, entertainment,\ + \ or geography; what is the topic of the Lingala statement below? Return only the\ + \ category. \n\ntext: {{text}} \\category:\n\n" +include: sib +task: sib_lin_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_lua.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_lua.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e71ac2aae77f40f06796c1572b2d38b44ec53962 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_lua.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: lua_Latn +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories science/technology, travel, politics, sports, health, entertainment,\ + \ or geography; what is the topic of the Luba-Kasai statement below? Return only\ + \ the category. \n\ntext: {{text}} \\category:\n\n" +include: sib +task: sib_lua_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c3554267ebad03a604a4a3dcca369af535efb156 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_lug.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: lug_Latn +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories science/technology, travel, politics, sports, health, entertainment,\ + \ or geography; what is the topic of the Luganda statement below? Return only the\ + \ category. \n\ntext: {{text}} \\category:\n\n" +include: sib +task: sib_lug_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_luo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_luo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..161814d36bde3e61fe3cdf38e98d2ef62f6b9248 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_luo.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: luo_Latn +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories science/technology, travel, politics, sports, health, entertainment,\ + \ or geography; what is the topic of the Luo statement below? Return only the category.\ + \ \n\ntext: {{text}} \\category:\n\n" +include: sib +task: sib_luo_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_mos.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_mos.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7b80d5008087bf66a82b8b7855fc5b8c857497fe --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_mos.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: mos_Latn +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories science/technology, travel, politics, sports, health, entertainment,\ + \ or geography; what is the topic of the Mossi statement below? Return only the\ + \ category. \n\ntext: {{text}} \\category:\n\n" +include: sib +task: sib_mos_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_nso.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_nso.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5c9dd8bd3f8cd30ce172286ca005c16b0ead9214 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_nso.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: nso_Latn +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories science/technology, travel, politics, sports, health, entertainment,\ + \ or geography; what is the topic of the Northern Sotho statement below? Return\ + \ only the category. \n\ntext: {{text}} \\category:\n\n" +include: sib +task: sib_nso_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_nus.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_nus.yaml new file mode 100644 index 0000000000000000000000000000000000000000..361698af10f6f231fbbdabf9e92a287504c45057 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_nus.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: nus_Latn +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories science/technology, travel, politics, sports, health, entertainment,\ + \ or geography; what is the topic of the Nuer statement below? Return only the category.\ + \ \n\ntext: {{text}} \\category:\n\n" +include: sib +task: sib_nus_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_nya.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_nya.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0c455c788ad7fc13a440885b6c7fc594ed4fc6e4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_nya.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: nya_Latn +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories science/technology, travel, politics, sports, health, entertainment,\ + \ or geography; what is the topic of the Nyanga statement below? Return only the\ + \ category. \n\ntext: {{text}} \\category:\n\n" +include: sib +task: sib_nya_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_plt.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_plt.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bb90a034be0e94aba92823ba3b5762fc13eabe6f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_plt.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: plt_Latn +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories science/technology, travel, politics, sports, health, entertainment,\ + \ or geography; what is the topic of the Plateau Malagasy statement below? Return\ + \ only the category. \n\ntext: {{text}} \\category:\n\n" +include: sib +task: sib_plt_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_por.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_por.yaml new file mode 100644 index 0000000000000000000000000000000000000000..65b8c2da4ab91e1723ceabd2e9fb08d3b6de2cfe --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_por.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: por_Latn +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories science/technology, travel, politics, sports, health, entertainment,\ + \ or geography; what is the topic of the Portuguese statement below? Return only\ + \ the category. \n\ntext: {{text}} \\category:\n\n" +include: sib +task: sib_por_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_run.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_run.yaml new file mode 100644 index 0000000000000000000000000000000000000000..19f3681cf856c7bc28bb1fcb5e8c31eda1f1b618 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_run.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: run_Latn +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories science/technology, travel, politics, sports, health, entertainment,\ + \ or geography; what is the topic of the Rundi statement below? Return only the\ + \ category. \n\ntext: {{text}} \\category:\n\n" +include: sib +task: sib_run_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_sag.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_sag.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8dfdcbd41929bc5747c34f18e93633fec8ac04e6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_sag.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: sag_Latn +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories science/technology, travel, politics, sports, health, entertainment,\ + \ or geography; what is the topic of the Sango statement below? Return only the\ + \ category. \n\ntext: {{text}} \\category:\n\n" +include: sib +task: sib_sag_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_sna.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_sna.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f30ff0d2b995c831b4005318a18a723998c92aa8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_sna.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: sna_Latn +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories science/technology, travel, politics, sports, health, entertainment,\ + \ or geography; what is the topic of the Shona statement below? Return only the\ + \ category. \n\ntext: {{text}} \\category:\n\n" +include: sib +task: sib_sna_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_som.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_som.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0ea27fd2e1b298324e1e1abcff152b25cd9cfc3d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_som.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: som_Latn +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories science/technology, travel, politics, sports, health, entertainment,\ + \ or geography; what is the topic of the Somali statement below? Return only the\ + \ category. \n\ntext: {{text}} \\category:\n\n" +include: sib +task: sib_som_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_sot.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_sot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d4ad477db4c912a09e317c7edc7818fe96b355f7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_sot.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: sot_Latn +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories science/technology, travel, politics, sports, health, entertainment,\ + \ or geography; what is the topic of the Southern Sotho statement below? Return\ + \ only the category. \n\ntext: {{text}} \\category:\n\n" +include: sib +task: sib_sot_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_ssw.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_ssw.yaml new file mode 100644 index 0000000000000000000000000000000000000000..25b7f85e1c955207a6afe5d154ed4286602a5313 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_ssw.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: ssw_Latn +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories science/technology, travel, politics, sports, health, entertainment,\ + \ or geography; what is the topic of the Swazi statement below? Return only the\ + \ category. \n\ntext: {{text}} \\category:\n\n" +include: sib +task: sib_ssw_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7be0be9d211d0fce16493b3b62d593a7ad60b864 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_swa.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: swh_Latn +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories science/technology, travel, politics, sports, health, entertainment,\ + \ or geography; what is the topic of the Swahili statement below? Return only the\ + \ category. \n\ntext: {{text}} \\category:\n\n" +include: sib +task: sib_swa_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_taq.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_taq.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a7e7b3abbbbca622c1b56169a7abc6c917d9b241 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_taq.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: taq_Latn +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories science/technology, travel, politics, sports, health, entertainment,\ + \ or geography; what is the topic of the Tamasheq statement below? Return only the\ + \ category. \n\ntext: {{text}} \\category:\n\n" +include: sib +task: sib_taq_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_tir.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_tir.yaml new file mode 100644 index 0000000000000000000000000000000000000000..aceb352596ba9fed4c4a3a544beb616923dca213 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_tir.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: tir_Ethi +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories science/technology, travel, politics, sports, health, entertainment,\ + \ or geography; what is the topic of the Tigrinya statement below? Return only the\ + \ category. \n\ntext: {{text}} \\category:\n\n" +include: sib +task: sib_tir_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_tso.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_tso.yaml new file mode 100644 index 0000000000000000000000000000000000000000..025b7163c069a6291dc4691a34de44732cc4c8b7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_tso.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: tso_Latn +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories science/technology, travel, politics, sports, health, entertainment,\ + \ or geography; what is the topic of the Tsonga statement below? Return only the\ + \ category. \n\ntext: {{text}} \\category:\n\n" +include: sib +task: sib_tso_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_tum.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_tum.yaml new file mode 100644 index 0000000000000000000000000000000000000000..35092ea79435767ad3e4907e152273d2cd6f1dca --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_tum.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: tum_Latn +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories science/technology, travel, politics, sports, health, entertainment,\ + \ or geography; what is the topic of the Tumbuka statement below? Return only the\ + \ category. \n\ntext: {{text}} \\category:\n\n" +include: sib +task: sib_tum_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fc75f6579cdb57391a21bbce3a49fb062d7263f9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_twi.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: twi_Latn +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories science/technology, travel, politics, sports, health, entertainment,\ + \ or geography; what is the topic of the Twi statement below? Return only the category.\ + \ \n\ntext: {{text}} \\category:\n\n" +include: sib +task: sib_twi_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_tzm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_tzm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b9b3044cdd08f81c95edfaeb8ded07d4a1da919d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_tzm.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: tzm_Tfng +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories science/technology, travel, politics, sports, health, entertainment,\ + \ or geography; what is the topic of the Tamazight statement below? Return only\ + \ the category. \n\ntext: {{text}} \\category:\n\n" +include: sib +task: sib_tzm_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_umb.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_umb.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d8bb8540180f44e612ea82e5276749d104362492 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_umb.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: umb_Latn +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories science/technology, travel, politics, sports, health, entertainment,\ + \ or geography; what is the topic of the Umbundu statement below? Return only the\ + \ category. \n\ntext: {{text}} \\category:\n\n" +include: sib +task: sib_umb_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..115796d5276b8739efb6fccfb9064b5f4bb6a27e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_wol.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: wol_Latn +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories science/technology, travel, politics, sports, health, entertainment,\ + \ or geography; what is the topic of the Wolof statement below? Return only the\ + \ category. \n\ntext: {{text}} \\category:\n\n" +include: sib +task: sib_wol_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b61c84b700da4e798848d52ed2310c9cb5ee3467 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_xho.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: xho_Latn +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories science/technology, travel, politics, sports, health, entertainment,\ + \ or geography; what is the topic of the Xhosa statement below? Return only the\ + \ category. \n\ntext: {{text}} \\category:\n\n" +include: sib +task: sib_xho_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c5ccd0c738eb5d1d365e5d7342ebe4eaaf7686b8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_yor.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: yor_Latn +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories science/technology, travel, politics, sports, health, entertainment,\ + \ or geography; what is the topic of the Yoruba statement below? Return only the\ + \ category. \n\ntext: {{text}} \\category:\n\n" +include: sib +task: sib_yor_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4915989dbbb0849b3345a80a420daa18a37eb97b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/sib_zul.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: zul_Latn +doc_to_text: "You are an assistant able to classify topics in texts. \n\nGiven the\ + \ categories science/technology, travel, politics, sports, health, entertainment,\ + \ or geography; what is the topic of the Zulu statement below? Return only the category.\ + \ \n\ntext: {{text}} \\category:\n\n" +include: sib +task: sib_zul_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..3e735e2deb1f9c53152c072615aebe8ba3acb90b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_3/utils.py @@ -0,0 +1 @@ +from lm_eval.utils import weighted_f1_score diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib new file mode 100644 index 0000000000000000000000000000000000000000..28ed8f4a0da4e25815ebcfa6e58092a382e1708e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib @@ -0,0 +1,43 @@ +tag: + - sib_tasks + - sib_prompt_4 + - afrobench_TC_tasks +dataset_path: Davlan/sib200 +dataset_name: null +output_type: multiple_choice +validation_split: validation +test_split: test +fewshot_split: validation +doc_to_target: category +doc_to_choice: + - "science/technology" + - "travel" + - "politics" + - "sports" + - "health" + - "entertainment" + - "geography" +should_decontaminate: true +doc_to_decontamination_query: text +metric_list: + - metric: f1 + aggregation: !function utils.weighted_f1_score + # aggregation: mean + average: weighted + hf_evaluate: true + higher_is_better: True + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" + - metric: acc + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_aeb.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_aeb.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e8c737f278122c8893e028ea2334ff93646a73cf --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_aeb.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: aeb_Arab +doc_to_text: "Label the following text as science/technology, travel, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_aeb_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_afr.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_afr.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7139d04e9a5b4a4865ba11941d3078802cc9a85c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_afr.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: afr_Latn +doc_to_text: "Label the following text as science/technology, travel, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_afr_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_aka.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_aka.yaml new file mode 100644 index 0000000000000000000000000000000000000000..59c8c56a6b78ffbb0da5e3b3abeb24ccd13b35d1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_aka.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: aka_Latn +doc_to_text: "Label the following text as science/technology, travel, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_aka_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cec6b6c43425195e36be88cbdc266d8806844a24 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_amh.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: amh_Ethi +doc_to_text: "Label the following text as science/technology, travel, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_amh_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_ary.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_ary.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2c10743470b814bf689bfef10410af3b4e03bb84 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_ary.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: ary_Arab +doc_to_text: "Label the following text as science/technology, travel, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_ary_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_arz.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_arz.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1740975a66196d9c4c3bd6780ad50281766cb0b2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_arz.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: arz_Arab +doc_to_text: "Label the following text as science/technology, travel, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_arz_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_bam.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_bam.yaml new file mode 100644 index 0000000000000000000000000000000000000000..33ee240e6d95a0e43426b514f5e33f696526faeb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_bam.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: bam_Latn +doc_to_text: "Label the following text as science/technology, travel, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_bam_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_bem.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_bem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..aa5608e849606f6d63f371b1bd7362d355b7d42b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_bem.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: bem_Latn +doc_to_text: "Label the following text as science/technology, travel, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_bem_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_cjk.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_cjk.yaml new file mode 100644 index 0000000000000000000000000000000000000000..52e08d7b8c5dc10d6c35a5b4fa4deee9b494f2d5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_cjk.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: cjk_Latn +doc_to_text: "Label the following text as science/technology, travel, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_cjk_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_dik.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_dik.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c8db6013f1d2a63d8242ac59d55c3f006f01e660 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_dik.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: dik_Latn +doc_to_text: "Label the following text as science/technology, travel, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_dik_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_dyu.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_dyu.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e9bbc0b547f3b25150eb000d4e56bb6e24e86991 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_dyu.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: dyu_Latn +doc_to_text: "Label the following text as science/technology, travel, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_dyu_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_eng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1c84749120e002dee47446d05600d81ed14bc193 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_eng.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: eng_Latn +doc_to_text: "Label the following text as science/technology, travel, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_eng_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_ewe.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_ewe.yaml new file mode 100644 index 0000000000000000000000000000000000000000..02e7ea822fee11a3d0b3869ef3ea493048a114da --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_ewe.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: ewe_Latn +doc_to_text: "Label the following text as science/technology, travel, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_ewe_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_fon.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_fon.yaml new file mode 100644 index 0000000000000000000000000000000000000000..67053ed8cd739682270062acea206792a7df5679 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_fon.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: fon_Latn +doc_to_text: "Label the following text as science/technology, travel, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_fon_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_fra.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_fra.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c2b858ce4e502334f8440c8551b1bcd10feb3b15 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_fra.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: fra_Latn +doc_to_text: "Label the following text as science/technology, travel, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_fra_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_fuv.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_fuv.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0c73f82679a48664a468f8e36432a0e33399190c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_fuv.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: fuv_Latn +doc_to_text: "Label the following text as science/technology, travel, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_fuv_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_gaz.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_gaz.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ba95ef5d8ee884a65befdab1a83853686d8b8ef5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_gaz.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: gaz_Latn +doc_to_text: "Label the following text as science/technology, travel, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_gaz_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1d53794868c164768810226db74aab7f06ccb383 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_hau.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: hau_Latn +doc_to_text: "Label the following text as science/technology, travel, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_hau_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2683d98dba0ca644b5314ef96a1359571a83fe9d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_ibo.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: ibo_Latn +doc_to_text: "Label the following text as science/technology, travel, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_ibo_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_kab.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_kab.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9f645a4598e2de1ccc45de14274a882b26deceb7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_kab.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: kab_Latn +doc_to_text: "Label the following text as science/technology, travel, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_kab_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_kam.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_kam.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7f035b89505f2f6ef889addc2af1c972efc8ff2d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_kam.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: kam_Latn +doc_to_text: "Label the following text as science/technology, travel, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_kam_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_kbp.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_kbp.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c65b6352e1dd12d2a7d511825f9c043ad213aebc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_kbp.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: kbp_Latn +doc_to_text: "Label the following text as science/technology, travel, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_kbp_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_kea.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_kea.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0e7bba4ae7a6c359568f6252edaebd0bee96c860 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_kea.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: kea_Latn +doc_to_text: "Label the following text as science/technology, travel, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_kea_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_kik.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_kik.yaml new file mode 100644 index 0000000000000000000000000000000000000000..06480d183bb1d2765f8d23e8dda80ee6c37c029e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_kik.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: kik_Latn +doc_to_text: "Label the following text as science/technology, travel, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_kik_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7b447219fb3bafaf2a81e3ac727e5216f408893f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_kin.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: kin_Latn +doc_to_text: "Label the following text as science/technology, travel, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_kin_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_kmb.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_kmb.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5fc51890964f59d20f53b06cf3ddbdb02b444471 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_kmb.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: kmb_Latn +doc_to_text: "Label the following text as science/technology, travel, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_kmb_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_knc.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_knc.yaml new file mode 100644 index 0000000000000000000000000000000000000000..326443318488c921765094a96c84adb8e208eda8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_knc.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: knc_Latn +doc_to_text: "Label the following text as science/technology, travel, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_knc_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_kon.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_kon.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6da4ab390d0e433391e313dea2c82d302d090dd2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_kon.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: kon_Latn +doc_to_text: "Label the following text as science/technology, travel, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_kon_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_lin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_lin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..51076dbd56131d82389d83bf2b12a224ef6c5443 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_lin.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: lin_Latn +doc_to_text: "Label the following text as science/technology, travel, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_lin_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_lua.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_lua.yaml new file mode 100644 index 0000000000000000000000000000000000000000..95973f7d5ce8309ee595ce4e751d564851796923 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_lua.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: lua_Latn +doc_to_text: "Label the following text as science/technology, travel, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_lua_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a570b58488496b4a73ca0fe46a2210c95b470bb2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_lug.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: lug_Latn +doc_to_text: "Label the following text as science/technology, travel, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_lug_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_luo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_luo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..76d799856b98c4f78804815fd5bd86dd415a100d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_luo.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: luo_Latn +doc_to_text: "Label the following text as science/technology, travel, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_luo_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_mos.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_mos.yaml new file mode 100644 index 0000000000000000000000000000000000000000..aeb058ac584c862898e7acf7441aa76b8c123709 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_mos.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: mos_Latn +doc_to_text: "Label the following text as science/technology, travel, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_mos_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_nso.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_nso.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f76e016a6bcdda14daf24179da982f696732a199 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_nso.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: nso_Latn +doc_to_text: "Label the following text as science/technology, travel, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_nso_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_nus.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_nus.yaml new file mode 100644 index 0000000000000000000000000000000000000000..255c1861589e185c2bcdb3f1d9f679ed26837be0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_nus.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: nus_Latn +doc_to_text: "Label the following text as science/technology, travel, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_nus_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_nya.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_nya.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bc7a48abf7f86cdea8f07d54a3a166ffe9550f06 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_nya.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: nya_Latn +doc_to_text: "Label the following text as science/technology, travel, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_nya_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_plt.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_plt.yaml new file mode 100644 index 0000000000000000000000000000000000000000..880c3d03ef5be9db726024243158f283c2013861 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_plt.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: plt_Latn +doc_to_text: "Label the following text as science/technology, travel, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_plt_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_por.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_por.yaml new file mode 100644 index 0000000000000000000000000000000000000000..16a258365d56b100c44fd269da27413fd3bffa83 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_por.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: por_Latn +doc_to_text: "Label the following text as science/technology, travel, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_por_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_run.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_run.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a97737edf4b9ab31a53749d109359c8acb3d3f4f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_run.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: run_Latn +doc_to_text: "Label the following text as science/technology, travel, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_run_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_sag.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_sag.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8c6897795ec414f37037eb2b79e6ffb6e3124ed7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_sag.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: sag_Latn +doc_to_text: "Label the following text as science/technology, travel, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_sag_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_sna.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_sna.yaml new file mode 100644 index 0000000000000000000000000000000000000000..da13a6ecf2b11650068be21b5b28ff470aba9002 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_sna.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: sna_Latn +doc_to_text: "Label the following text as science/technology, travel, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_sna_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_som.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_som.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b6c35f3cb4a25a3ddecda9d4b1dc8528fce64d1f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_som.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: som_Latn +doc_to_text: "Label the following text as science/technology, travel, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_som_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_sot.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_sot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1385e058deaf19b0cdf272a768f260997d7cae92 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_sot.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: sot_Latn +doc_to_text: "Label the following text as science/technology, travel, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_sot_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_ssw.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_ssw.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d678c12422ff05f87849f934daf402454fa3415e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_ssw.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: ssw_Latn +doc_to_text: "Label the following text as science/technology, travel, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_ssw_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7492cfa329f48c959f6255ffbb879d952fcbe200 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_swa.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: swh_Latn +doc_to_text: "Label the following text as science/technology, travel, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_swa_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_taq.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_taq.yaml new file mode 100644 index 0000000000000000000000000000000000000000..783be833f8c77c16a82948f4055162941723849f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_taq.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: taq_Latn +doc_to_text: "Label the following text as science/technology, travel, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_taq_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_tir.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_tir.yaml new file mode 100644 index 0000000000000000000000000000000000000000..931ede568a3faa337637be06bccfd9ca136d8bc2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_tir.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: tir_Ethi +doc_to_text: "Label the following text as science/technology, travel, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_tir_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_tso.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_tso.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fc4c0f1a3278259574fc84fd09be60422174b871 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_tso.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: tso_Latn +doc_to_text: "Label the following text as science/technology, travel, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_tso_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_tum.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_tum.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c099dc6fd73bdad8c8e9e4d306cee8d9dec243fa --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_tum.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: tum_Latn +doc_to_text: "Label the following text as science/technology, travel, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_tum_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..00501281a217c4d68834e8a1cd6dec9463e87268 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_twi.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: twi_Latn +doc_to_text: "Label the following text as science/technology, travel, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_twi_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_tzm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_tzm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3915fa18f2c7963e0b1a0f4f10ca1da87f765141 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_tzm.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: tzm_Tfng +doc_to_text: "Label the following text as science/technology, travel, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_tzm_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_umb.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_umb.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a7f1cc79736bfba50fb8ab03c8749a77523feec4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_umb.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: umb_Latn +doc_to_text: "Label the following text as science/technology, travel, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_umb_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fc2440248af154acd2aecdfc6d341230d4bfa67a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_wol.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: wol_Latn +doc_to_text: "Label the following text as science/technology, travel, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_wol_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e075b84c190d949ad8e177b06445e0136f0445d0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_xho.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: xho_Latn +doc_to_text: "Label the following text as science/technology, travel, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_xho_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..41ef062098ebc4b54b7aec5da59851d490924e6c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_yor.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: yor_Latn +doc_to_text: "Label the following text as science/technology, travel, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_yor_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7fc2f85efc9f9799446133ac107b6b0d66cfb38b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/sib_zul.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: zul_Latn +doc_to_text: "Label the following text as science/technology, travel, politics, sports,\ + \ health, entertainment, or geography. Provide only the category as your response.\ + \ \n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_zul_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..3e735e2deb1f9c53152c072615aebe8ba3acb90b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_4/utils.py @@ -0,0 +1 @@ +from lm_eval.utils import weighted_f1_score diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib new file mode 100644 index 0000000000000000000000000000000000000000..812df7f614a9c8146b6da3137f4c2e97049b07f2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib @@ -0,0 +1,43 @@ +tag: + - sib_tasks + - sib_prompt_5 + - afrobench_TC_tasks +dataset_path: Davlan/sib200 +dataset_name: null +output_type: multiple_choice +validation_split: validation +test_split: test +fewshot_split: validation +doc_to_target: category +doc_to_choice: + - "science/technology" + - "travel" + - "politics" + - "sports" + - "health" + - "entertainment" + - "geography" +should_decontaminate: true +doc_to_decontamination_query: text +metric_list: + - metric: f1 + aggregation: !function utils.weighted_f1_score + # aggregation: mean + average: weighted + hf_evaluate: true + higher_is_better: True + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" + - metric: acc + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_aeb.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_aeb.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3c88c0a28bdd40981ba847762e3cc08b36e66690 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_aeb.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: aeb_Arab +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Tunisian Arabic text. For each input, classify the topic as science/technology,\ + \ travel, politics, sports, health, entertainment, or geography. Use the following\ + \ guidelines: \n\n science/technology: The text discusses scientific discoveries,\ + \ technological advancements, or related topics. \ntravel: The text describes travel\ + \ experiences, destinations, or related topics. \npolitics: The text covers political\ + \ events, policies, or related topics. \nsports: The text talks about sports events,\ + \ athletes, or related topics. \nhealth: The text addresses health issues, medical\ + \ advancements, or related topics. \nentertainment: The text pertains to movies,\ + \ music, celebrities, or related topics. \ngeography: The text involves geographical\ + \ information, locations, or related topics. \n\nIf the text contains multiple topics,\ + \ choose the dominant topic. For ambiguous or unclear topics, select the category\ + \ that best reflects the overall content. Please provide a single classification\ + \ for each input.\n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_aeb_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_afr.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_afr.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0d585478be65a678629dd3718c9e44f25a42b5e6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_afr.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: afr_Latn +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Afrikaans text. For each input, classify the topic as science/technology, travel,\ + \ politics, sports, health, entertainment, or geography. Use the following guidelines:\ + \ \n\n science/technology: The text discusses scientific discoveries, technological\ + \ advancements, or related topics. \ntravel: The text describes travel experiences,\ + \ destinations, or related topics. \npolitics: The text covers political events,\ + \ policies, or related topics. \nsports: The text talks about sports events, athletes,\ + \ or related topics. \nhealth: The text addresses health issues, medical advancements,\ + \ or related topics. \nentertainment: The text pertains to movies, music, celebrities,\ + \ or related topics. \ngeography: The text involves geographical information, locations,\ + \ or related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{text}} \\category: \n\n" +include: sib +task: sib_afr_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_aka.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_aka.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4644bfa3c9c923d544f86f13b273c5f754b236f7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_aka.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: aka_Latn +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Akan text. For each input, classify the topic as science/technology, travel, politics,\ + \ sports, health, entertainment, or geography. Use the following guidelines: \n\n\ + \ science/technology: The text discusses scientific discoveries, technological advancements,\ + \ or related topics. \ntravel: The text describes travel experiences, destinations,\ + \ or related topics. \npolitics: The text covers political events, policies, or\ + \ related topics. \nsports: The text talks about sports events, athletes, or related\ + \ topics. \nhealth: The text addresses health issues, medical advancements, or related\ + \ topics. \nentertainment: The text pertains to movies, music, celebrities, or related\ + \ topics. \ngeography: The text involves geographical information, locations, or\ + \ related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{text}} \\category: \n\n" +include: sib +task: sib_aka_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f2b5e6f9223f1b20d1d02b5f27635a0684388744 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_amh.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: amh_Ethi +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Amharic text. For each input, classify the topic as science/technology, travel,\ + \ politics, sports, health, entertainment, or geography. Use the following guidelines:\ + \ \n\n science/technology: The text discusses scientific discoveries, technological\ + \ advancements, or related topics. \ntravel: The text describes travel experiences,\ + \ destinations, or related topics. \npolitics: The text covers political events,\ + \ policies, or related topics. \nsports: The text talks about sports events, athletes,\ + \ or related topics. \nhealth: The text addresses health issues, medical advancements,\ + \ or related topics. \nentertainment: The text pertains to movies, music, celebrities,\ + \ or related topics. \ngeography: The text involves geographical information, locations,\ + \ or related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{text}} \\category: \n\n" +include: sib +task: sib_amh_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_ary.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_ary.yaml new file mode 100644 index 0000000000000000000000000000000000000000..348c849d219b06501167c51182458f1946f51439 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_ary.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: ary_Arab +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Moroccan Arabic text. For each input, classify the topic as science/technology,\ + \ travel, politics, sports, health, entertainment, or geography. Use the following\ + \ guidelines: \n\n science/technology: The text discusses scientific discoveries,\ + \ technological advancements, or related topics. \ntravel: The text describes travel\ + \ experiences, destinations, or related topics. \npolitics: The text covers political\ + \ events, policies, or related topics. \nsports: The text talks about sports events,\ + \ athletes, or related topics. \nhealth: The text addresses health issues, medical\ + \ advancements, or related topics. \nentertainment: The text pertains to movies,\ + \ music, celebrities, or related topics. \ngeography: The text involves geographical\ + \ information, locations, or related topics. \n\nIf the text contains multiple topics,\ + \ choose the dominant topic. For ambiguous or unclear topics, select the category\ + \ that best reflects the overall content. Please provide a single classification\ + \ for each input.\n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_ary_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_arz.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_arz.yaml new file mode 100644 index 0000000000000000000000000000000000000000..100570428142a0f35ec728558251e96ec484ccb5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_arz.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: arz_Arab +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Egyptian Arabic text. For each input, classify the topic as science/technology,\ + \ travel, politics, sports, health, entertainment, or geography. Use the following\ + \ guidelines: \n\n science/technology: The text discusses scientific discoveries,\ + \ technological advancements, or related topics. \ntravel: The text describes travel\ + \ experiences, destinations, or related topics. \npolitics: The text covers political\ + \ events, policies, or related topics. \nsports: The text talks about sports events,\ + \ athletes, or related topics. \nhealth: The text addresses health issues, medical\ + \ advancements, or related topics. \nentertainment: The text pertains to movies,\ + \ music, celebrities, or related topics. \ngeography: The text involves geographical\ + \ information, locations, or related topics. \n\nIf the text contains multiple topics,\ + \ choose the dominant topic. For ambiguous or unclear topics, select the category\ + \ that best reflects the overall content. Please provide a single classification\ + \ for each input.\n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_arz_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_bam.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_bam.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bdc655003fc959e8219ae681b64b82dd137853d1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_bam.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: bam_Latn +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Bambara text. For each input, classify the topic as science/technology, travel,\ + \ politics, sports, health, entertainment, or geography. Use the following guidelines:\ + \ \n\n science/technology: The text discusses scientific discoveries, technological\ + \ advancements, or related topics. \ntravel: The text describes travel experiences,\ + \ destinations, or related topics. \npolitics: The text covers political events,\ + \ policies, or related topics. \nsports: The text talks about sports events, athletes,\ + \ or related topics. \nhealth: The text addresses health issues, medical advancements,\ + \ or related topics. \nentertainment: The text pertains to movies, music, celebrities,\ + \ or related topics. \ngeography: The text involves geographical information, locations,\ + \ or related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{text}} \\category: \n\n" +include: sib +task: sib_bam_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_bem.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_bem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d42ea873b83e69d3d3d621a9ba1fafd7a88a4ab3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_bem.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: bem_Latn +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Bemba text. For each input, classify the topic as science/technology, travel,\ + \ politics, sports, health, entertainment, or geography. Use the following guidelines:\ + \ \n\n science/technology: The text discusses scientific discoveries, technological\ + \ advancements, or related topics. \ntravel: The text describes travel experiences,\ + \ destinations, or related topics. \npolitics: The text covers political events,\ + \ policies, or related topics. \nsports: The text talks about sports events, athletes,\ + \ or related topics. \nhealth: The text addresses health issues, medical advancements,\ + \ or related topics. \nentertainment: The text pertains to movies, music, celebrities,\ + \ or related topics. \ngeography: The text involves geographical information, locations,\ + \ or related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{text}} \\category: \n\n" +include: sib +task: sib_bem_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_cjk.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_cjk.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9623b8c52bb39ab7dfa0f743d5482169a462b7ab --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_cjk.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: cjk_Latn +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Chokwe text. For each input, classify the topic as science/technology, travel,\ + \ politics, sports, health, entertainment, or geography. Use the following guidelines:\ + \ \n\n science/technology: The text discusses scientific discoveries, technological\ + \ advancements, or related topics. \ntravel: The text describes travel experiences,\ + \ destinations, or related topics. \npolitics: The text covers political events,\ + \ policies, or related topics. \nsports: The text talks about sports events, athletes,\ + \ or related topics. \nhealth: The text addresses health issues, medical advancements,\ + \ or related topics. \nentertainment: The text pertains to movies, music, celebrities,\ + \ or related topics. \ngeography: The text involves geographical information, locations,\ + \ or related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{text}} \\category: \n\n" +include: sib +task: sib_cjk_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_dik.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_dik.yaml new file mode 100644 index 0000000000000000000000000000000000000000..83e76e963fe2962b472b8312d35a79c4e14d2b55 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_dik.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: dik_Latn +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Southwestern Dinka text. For each input, classify the topic as science/technology,\ + \ travel, politics, sports, health, entertainment, or geography. Use the following\ + \ guidelines: \n\n science/technology: The text discusses scientific discoveries,\ + \ technological advancements, or related topics. \ntravel: The text describes travel\ + \ experiences, destinations, or related topics. \npolitics: The text covers political\ + \ events, policies, or related topics. \nsports: The text talks about sports events,\ + \ athletes, or related topics. \nhealth: The text addresses health issues, medical\ + \ advancements, or related topics. \nentertainment: The text pertains to movies,\ + \ music, celebrities, or related topics. \ngeography: The text involves geographical\ + \ information, locations, or related topics. \n\nIf the text contains multiple topics,\ + \ choose the dominant topic. For ambiguous or unclear topics, select the category\ + \ that best reflects the overall content. Please provide a single classification\ + \ for each input.\n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_dik_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_dyu.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_dyu.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8ab215e89f959e6edf4bd07d1729f0424e85e0a8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_dyu.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: dyu_Latn +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Dyula text. For each input, classify the topic as science/technology, travel,\ + \ politics, sports, health, entertainment, or geography. Use the following guidelines:\ + \ \n\n science/technology: The text discusses scientific discoveries, technological\ + \ advancements, or related topics. \ntravel: The text describes travel experiences,\ + \ destinations, or related topics. \npolitics: The text covers political events,\ + \ policies, or related topics. \nsports: The text talks about sports events, athletes,\ + \ or related topics. \nhealth: The text addresses health issues, medical advancements,\ + \ or related topics. \nentertainment: The text pertains to movies, music, celebrities,\ + \ or related topics. \ngeography: The text involves geographical information, locations,\ + \ or related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{text}} \\category: \n\n" +include: sib +task: sib_dyu_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_eng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a17a006d21d32ce9901010cbdcd94aad3af933f4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_eng.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: eng_Latn +doc_to_text: "You are tasked with performing topic classification on the following\ + \ English text. For each input, classify the topic as science/technology, travel,\ + \ politics, sports, health, entertainment, or geography. Use the following guidelines:\ + \ \n\n science/technology: The text discusses scientific discoveries, technological\ + \ advancements, or related topics. \ntravel: The text describes travel experiences,\ + \ destinations, or related topics. \npolitics: The text covers political events,\ + \ policies, or related topics. \nsports: The text talks about sports events, athletes,\ + \ or related topics. \nhealth: The text addresses health issues, medical advancements,\ + \ or related topics. \nentertainment: The text pertains to movies, music, celebrities,\ + \ or related topics. \ngeography: The text involves geographical information, locations,\ + \ or related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{text}} \\category: \n\n" +include: sib +task: sib_eng_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_ewe.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_ewe.yaml new file mode 100644 index 0000000000000000000000000000000000000000..195876998160addea6184dbc4d3375192068aec5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_ewe.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: ewe_Latn +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Ewe text. For each input, classify the topic as science/technology, travel, politics,\ + \ sports, health, entertainment, or geography. Use the following guidelines: \n\n\ + \ science/technology: The text discusses scientific discoveries, technological advancements,\ + \ or related topics. \ntravel: The text describes travel experiences, destinations,\ + \ or related topics. \npolitics: The text covers political events, policies, or\ + \ related topics. \nsports: The text talks about sports events, athletes, or related\ + \ topics. \nhealth: The text addresses health issues, medical advancements, or related\ + \ topics. \nentertainment: The text pertains to movies, music, celebrities, or related\ + \ topics. \ngeography: The text involves geographical information, locations, or\ + \ related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{text}} \\category: \n\n" +include: sib +task: sib_ewe_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_fon.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_fon.yaml new file mode 100644 index 0000000000000000000000000000000000000000..61980b5110a424ed7391b29dbedf7f7828563f03 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_fon.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: fon_Latn +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Fon text. For each input, classify the topic as science/technology, travel, politics,\ + \ sports, health, entertainment, or geography. Use the following guidelines: \n\n\ + \ science/technology: The text discusses scientific discoveries, technological advancements,\ + \ or related topics. \ntravel: The text describes travel experiences, destinations,\ + \ or related topics. \npolitics: The text covers political events, policies, or\ + \ related topics. \nsports: The text talks about sports events, athletes, or related\ + \ topics. \nhealth: The text addresses health issues, medical advancements, or related\ + \ topics. \nentertainment: The text pertains to movies, music, celebrities, or related\ + \ topics. \ngeography: The text involves geographical information, locations, or\ + \ related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{text}} \\category: \n\n" +include: sib +task: sib_fon_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_fra.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_fra.yaml new file mode 100644 index 0000000000000000000000000000000000000000..29573054bcfa08ddf225f925cb6131b6d4909163 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_fra.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: fra_Latn +doc_to_text: "You are tasked with performing topic classification on the following\ + \ French text. For each input, classify the topic as science/technology, travel,\ + \ politics, sports, health, entertainment, or geography. Use the following guidelines:\ + \ \n\n science/technology: The text discusses scientific discoveries, technological\ + \ advancements, or related topics. \ntravel: The text describes travel experiences,\ + \ destinations, or related topics. \npolitics: The text covers political events,\ + \ policies, or related topics. \nsports: The text talks about sports events, athletes,\ + \ or related topics. \nhealth: The text addresses health issues, medical advancements,\ + \ or related topics. \nentertainment: The text pertains to movies, music, celebrities,\ + \ or related topics. \ngeography: The text involves geographical information, locations,\ + \ or related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{text}} \\category: \n\n" +include: sib +task: sib_fra_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_fuv.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_fuv.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3b48f9f4e09e7042dfdec6ffa224300ee824b580 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_fuv.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: fuv_Latn +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Nigerian Fulfulde text. For each input, classify the topic as science/technology,\ + \ travel, politics, sports, health, entertainment, or geography. Use the following\ + \ guidelines: \n\n science/technology: The text discusses scientific discoveries,\ + \ technological advancements, or related topics. \ntravel: The text describes travel\ + \ experiences, destinations, or related topics. \npolitics: The text covers political\ + \ events, policies, or related topics. \nsports: The text talks about sports events,\ + \ athletes, or related topics. \nhealth: The text addresses health issues, medical\ + \ advancements, or related topics. \nentertainment: The text pertains to movies,\ + \ music, celebrities, or related topics. \ngeography: The text involves geographical\ + \ information, locations, or related topics. \n\nIf the text contains multiple topics,\ + \ choose the dominant topic. For ambiguous or unclear topics, select the category\ + \ that best reflects the overall content. Please provide a single classification\ + \ for each input.\n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_fuv_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_gaz.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_gaz.yaml new file mode 100644 index 0000000000000000000000000000000000000000..37e2a4f97793217ecbd9f8c551d585901366cd34 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_gaz.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: gaz_Latn +doc_to_text: "You are tasked with performing topic classification on the following\ + \ West Central Oromo text. For each input, classify the topic as science/technology,\ + \ travel, politics, sports, health, entertainment, or geography. Use the following\ + \ guidelines: \n\n science/technology: The text discusses scientific discoveries,\ + \ technological advancements, or related topics. \ntravel: The text describes travel\ + \ experiences, destinations, or related topics. \npolitics: The text covers political\ + \ events, policies, or related topics. \nsports: The text talks about sports events,\ + \ athletes, or related topics. \nhealth: The text addresses health issues, medical\ + \ advancements, or related topics. \nentertainment: The text pertains to movies,\ + \ music, celebrities, or related topics. \ngeography: The text involves geographical\ + \ information, locations, or related topics. \n\nIf the text contains multiple topics,\ + \ choose the dominant topic. For ambiguous or unclear topics, select the category\ + \ that best reflects the overall content. Please provide a single classification\ + \ for each input.\n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_gaz_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..24ce0970e9923725e2255abe09dc6e1629c9d23f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_hau.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: hau_Latn +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Hausa text. For each input, classify the topic as science/technology, travel,\ + \ politics, sports, health, entertainment, or geography. Use the following guidelines:\ + \ \n\n science/technology: The text discusses scientific discoveries, technological\ + \ advancements, or related topics. \ntravel: The text describes travel experiences,\ + \ destinations, or related topics. \npolitics: The text covers political events,\ + \ policies, or related topics. \nsports: The text talks about sports events, athletes,\ + \ or related topics. \nhealth: The text addresses health issues, medical advancements,\ + \ or related topics. \nentertainment: The text pertains to movies, music, celebrities,\ + \ or related topics. \ngeography: The text involves geographical information, locations,\ + \ or related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{text}} \\category: \n\n" +include: sib +task: sib_hau_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a39ee75cb90a18ddb12bbedc54fd82a4c4c45ded --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_ibo.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: ibo_Latn +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Igbo text. For each input, classify the topic as science/technology, travel, politics,\ + \ sports, health, entertainment, or geography. Use the following guidelines: \n\n\ + \ science/technology: The text discusses scientific discoveries, technological advancements,\ + \ or related topics. \ntravel: The text describes travel experiences, destinations,\ + \ or related topics. \npolitics: The text covers political events, policies, or\ + \ related topics. \nsports: The text talks about sports events, athletes, or related\ + \ topics. \nhealth: The text addresses health issues, medical advancements, or related\ + \ topics. \nentertainment: The text pertains to movies, music, celebrities, or related\ + \ topics. \ngeography: The text involves geographical information, locations, or\ + \ related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{text}} \\category: \n\n" +include: sib +task: sib_ibo_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_kab.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_kab.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d29da033388668302aca70c3511d52377e8797d9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_kab.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: kab_Latn +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Kabyle text. For each input, classify the topic as science/technology, travel,\ + \ politics, sports, health, entertainment, or geography. Use the following guidelines:\ + \ \n\n science/technology: The text discusses scientific discoveries, technological\ + \ advancements, or related topics. \ntravel: The text describes travel experiences,\ + \ destinations, or related topics. \npolitics: The text covers political events,\ + \ policies, or related topics. \nsports: The text talks about sports events, athletes,\ + \ or related topics. \nhealth: The text addresses health issues, medical advancements,\ + \ or related topics. \nentertainment: The text pertains to movies, music, celebrities,\ + \ or related topics. \ngeography: The text involves geographical information, locations,\ + \ or related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{text}} \\category: \n\n" +include: sib +task: sib_kab_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_kam.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_kam.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1e55d1218586efb9f8d0b9bad3e9c2c76727a74b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_kam.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: kam_Latn +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Kamba text. For each input, classify the topic as science/technology, travel,\ + \ politics, sports, health, entertainment, or geography. Use the following guidelines:\ + \ \n\n science/technology: The text discusses scientific discoveries, technological\ + \ advancements, or related topics. \ntravel: The text describes travel experiences,\ + \ destinations, or related topics. \npolitics: The text covers political events,\ + \ policies, or related topics. \nsports: The text talks about sports events, athletes,\ + \ or related topics. \nhealth: The text addresses health issues, medical advancements,\ + \ or related topics. \nentertainment: The text pertains to movies, music, celebrities,\ + \ or related topics. \ngeography: The text involves geographical information, locations,\ + \ or related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{text}} \\category: \n\n" +include: sib +task: sib_kam_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_kbp.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_kbp.yaml new file mode 100644 index 0000000000000000000000000000000000000000..210baea8685a19c4e6bee6bcd141f2f0cb2a101a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_kbp.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: kbp_Latn +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Kabiye text. For each input, classify the topic as science/technology, travel,\ + \ politics, sports, health, entertainment, or geography. Use the following guidelines:\ + \ \n\n science/technology: The text discusses scientific discoveries, technological\ + \ advancements, or related topics. \ntravel: The text describes travel experiences,\ + \ destinations, or related topics. \npolitics: The text covers political events,\ + \ policies, or related topics. \nsports: The text talks about sports events, athletes,\ + \ or related topics. \nhealth: The text addresses health issues, medical advancements,\ + \ or related topics. \nentertainment: The text pertains to movies, music, celebrities,\ + \ or related topics. \ngeography: The text involves geographical information, locations,\ + \ or related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{text}} \\category: \n\n" +include: sib +task: sib_kbp_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_kea.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_kea.yaml new file mode 100644 index 0000000000000000000000000000000000000000..34a6813c8eff67f57adff7f43c982006b431ccf6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_kea.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: kea_Latn +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Kabuverdianu text. For each input, classify the topic as science/technology, travel,\ + \ politics, sports, health, entertainment, or geography. Use the following guidelines:\ + \ \n\n science/technology: The text discusses scientific discoveries, technological\ + \ advancements, or related topics. \ntravel: The text describes travel experiences,\ + \ destinations, or related topics. \npolitics: The text covers political events,\ + \ policies, or related topics. \nsports: The text talks about sports events, athletes,\ + \ or related topics. \nhealth: The text addresses health issues, medical advancements,\ + \ or related topics. \nentertainment: The text pertains to movies, music, celebrities,\ + \ or related topics. \ngeography: The text involves geographical information, locations,\ + \ or related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{text}} \\category: \n\n" +include: sib +task: sib_kea_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_kik.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_kik.yaml new file mode 100644 index 0000000000000000000000000000000000000000..55fdcb00e3f003422315e8de2ef64c8aa9e0abbc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_kik.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: kik_Latn +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Kikuyu text. For each input, classify the topic as science/technology, travel,\ + \ politics, sports, health, entertainment, or geography. Use the following guidelines:\ + \ \n\n science/technology: The text discusses scientific discoveries, technological\ + \ advancements, or related topics. \ntravel: The text describes travel experiences,\ + \ destinations, or related topics. \npolitics: The text covers political events,\ + \ policies, or related topics. \nsports: The text talks about sports events, athletes,\ + \ or related topics. \nhealth: The text addresses health issues, medical advancements,\ + \ or related topics. \nentertainment: The text pertains to movies, music, celebrities,\ + \ or related topics. \ngeography: The text involves geographical information, locations,\ + \ or related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{text}} \\category: \n\n" +include: sib +task: sib_kik_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6567d52bf1f47beaba19141ab2f95b8168298290 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_kin.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: kin_Latn +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Kinyarwanda text. For each input, classify the topic as science/technology, travel,\ + \ politics, sports, health, entertainment, or geography. Use the following guidelines:\ + \ \n\n science/technology: The text discusses scientific discoveries, technological\ + \ advancements, or related topics. \ntravel: The text describes travel experiences,\ + \ destinations, or related topics. \npolitics: The text covers political events,\ + \ policies, or related topics. \nsports: The text talks about sports events, athletes,\ + \ or related topics. \nhealth: The text addresses health issues, medical advancements,\ + \ or related topics. \nentertainment: The text pertains to movies, music, celebrities,\ + \ or related topics. \ngeography: The text involves geographical information, locations,\ + \ or related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{text}} \\category: \n\n" +include: sib +task: sib_kin_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_kmb.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_kmb.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4ae05cd06aef47942e20bbf72092cd23a5b4fb2f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_kmb.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: kmb_Latn +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Kimbundu text. For each input, classify the topic as science/technology, travel,\ + \ politics, sports, health, entertainment, or geography. Use the following guidelines:\ + \ \n\n science/technology: The text discusses scientific discoveries, technological\ + \ advancements, or related topics. \ntravel: The text describes travel experiences,\ + \ destinations, or related topics. \npolitics: The text covers political events,\ + \ policies, or related topics. \nsports: The text talks about sports events, athletes,\ + \ or related topics. \nhealth: The text addresses health issues, medical advancements,\ + \ or related topics. \nentertainment: The text pertains to movies, music, celebrities,\ + \ or related topics. \ngeography: The text involves geographical information, locations,\ + \ or related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{text}} \\category: \n\n" +include: sib +task: sib_kmb_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_knc.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_knc.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9870bd64740561b60f69af908c01ab221585d3fc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_knc.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: knc_Latn +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Central Kanuri text. For each input, classify the topic as science/technology,\ + \ travel, politics, sports, health, entertainment, or geography. Use the following\ + \ guidelines: \n\n science/technology: The text discusses scientific discoveries,\ + \ technological advancements, or related topics. \ntravel: The text describes travel\ + \ experiences, destinations, or related topics. \npolitics: The text covers political\ + \ events, policies, or related topics. \nsports: The text talks about sports events,\ + \ athletes, or related topics. \nhealth: The text addresses health issues, medical\ + \ advancements, or related topics. \nentertainment: The text pertains to movies,\ + \ music, celebrities, or related topics. \ngeography: The text involves geographical\ + \ information, locations, or related topics. \n\nIf the text contains multiple topics,\ + \ choose the dominant topic. For ambiguous or unclear topics, select the category\ + \ that best reflects the overall content. Please provide a single classification\ + \ for each input.\n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_knc_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_kon.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_kon.yaml new file mode 100644 index 0000000000000000000000000000000000000000..afcab8b8dd34cfa63b1653223c05170789dddc10 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_kon.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: kon_Latn +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Kikongo text. For each input, classify the topic as science/technology, travel,\ + \ politics, sports, health, entertainment, or geography. Use the following guidelines:\ + \ \n\n science/technology: The text discusses scientific discoveries, technological\ + \ advancements, or related topics. \ntravel: The text describes travel experiences,\ + \ destinations, or related topics. \npolitics: The text covers political events,\ + \ policies, or related topics. \nsports: The text talks about sports events, athletes,\ + \ or related topics. \nhealth: The text addresses health issues, medical advancements,\ + \ or related topics. \nentertainment: The text pertains to movies, music, celebrities,\ + \ or related topics. \ngeography: The text involves geographical information, locations,\ + \ or related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{text}} \\category: \n\n" +include: sib +task: sib_kon_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_lin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_lin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4c1611fd77106ae1eeef87ff5fcce60221ef8039 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_lin.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: lin_Latn +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Lingala text. For each input, classify the topic as science/technology, travel,\ + \ politics, sports, health, entertainment, or geography. Use the following guidelines:\ + \ \n\n science/technology: The text discusses scientific discoveries, technological\ + \ advancements, or related topics. \ntravel: The text describes travel experiences,\ + \ destinations, or related topics. \npolitics: The text covers political events,\ + \ policies, or related topics. \nsports: The text talks about sports events, athletes,\ + \ or related topics. \nhealth: The text addresses health issues, medical advancements,\ + \ or related topics. \nentertainment: The text pertains to movies, music, celebrities,\ + \ or related topics. \ngeography: The text involves geographical information, locations,\ + \ or related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{text}} \\category: \n\n" +include: sib +task: sib_lin_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_lua.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_lua.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b3b2b9edcb68cf91b5c63273e07bedc936a7ffe1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_lua.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: lua_Latn +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Luba-Kasai text. For each input, classify the topic as science/technology, travel,\ + \ politics, sports, health, entertainment, or geography. Use the following guidelines:\ + \ \n\n science/technology: The text discusses scientific discoveries, technological\ + \ advancements, or related topics. \ntravel: The text describes travel experiences,\ + \ destinations, or related topics. \npolitics: The text covers political events,\ + \ policies, or related topics. \nsports: The text talks about sports events, athletes,\ + \ or related topics. \nhealth: The text addresses health issues, medical advancements,\ + \ or related topics. \nentertainment: The text pertains to movies, music, celebrities,\ + \ or related topics. \ngeography: The text involves geographical information, locations,\ + \ or related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{text}} \\category: \n\n" +include: sib +task: sib_lua_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f8ca880ace46c5220acdcf2fb5d47bf37e2791aa --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_lug.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: lug_Latn +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Luganda text. For each input, classify the topic as science/technology, travel,\ + \ politics, sports, health, entertainment, or geography. Use the following guidelines:\ + \ \n\n science/technology: The text discusses scientific discoveries, technological\ + \ advancements, or related topics. \ntravel: The text describes travel experiences,\ + \ destinations, or related topics. \npolitics: The text covers political events,\ + \ policies, or related topics. \nsports: The text talks about sports events, athletes,\ + \ or related topics. \nhealth: The text addresses health issues, medical advancements,\ + \ or related topics. \nentertainment: The text pertains to movies, music, celebrities,\ + \ or related topics. \ngeography: The text involves geographical information, locations,\ + \ or related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{text}} \\category: \n\n" +include: sib +task: sib_lug_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_luo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_luo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b942d69d0cf5a367213aca5cc437b26015c83ae3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_luo.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: luo_Latn +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Luo text. For each input, classify the topic as science/technology, travel, politics,\ + \ sports, health, entertainment, or geography. Use the following guidelines: \n\n\ + \ science/technology: The text discusses scientific discoveries, technological advancements,\ + \ or related topics. \ntravel: The text describes travel experiences, destinations,\ + \ or related topics. \npolitics: The text covers political events, policies, or\ + \ related topics. \nsports: The text talks about sports events, athletes, or related\ + \ topics. \nhealth: The text addresses health issues, medical advancements, or related\ + \ topics. \nentertainment: The text pertains to movies, music, celebrities, or related\ + \ topics. \ngeography: The text involves geographical information, locations, or\ + \ related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{text}} \\category: \n\n" +include: sib +task: sib_luo_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_mos.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_mos.yaml new file mode 100644 index 0000000000000000000000000000000000000000..daccd62e9345db0c0625c7a26df993fe4f528411 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_mos.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: mos_Latn +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Mossi text. For each input, classify the topic as science/technology, travel,\ + \ politics, sports, health, entertainment, or geography. Use the following guidelines:\ + \ \n\n science/technology: The text discusses scientific discoveries, technological\ + \ advancements, or related topics. \ntravel: The text describes travel experiences,\ + \ destinations, or related topics. \npolitics: The text covers political events,\ + \ policies, or related topics. \nsports: The text talks about sports events, athletes,\ + \ or related topics. \nhealth: The text addresses health issues, medical advancements,\ + \ or related topics. \nentertainment: The text pertains to movies, music, celebrities,\ + \ or related topics. \ngeography: The text involves geographical information, locations,\ + \ or related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{text}} \\category: \n\n" +include: sib +task: sib_mos_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_nso.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_nso.yaml new file mode 100644 index 0000000000000000000000000000000000000000..09936e3c333c1cb2c285fc056d6da25246bcfefe --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_nso.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: nso_Latn +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Northern Sotho text. For each input, classify the topic as science/technology,\ + \ travel, politics, sports, health, entertainment, or geography. Use the following\ + \ guidelines: \n\n science/technology: The text discusses scientific discoveries,\ + \ technological advancements, or related topics. \ntravel: The text describes travel\ + \ experiences, destinations, or related topics. \npolitics: The text covers political\ + \ events, policies, or related topics. \nsports: The text talks about sports events,\ + \ athletes, or related topics. \nhealth: The text addresses health issues, medical\ + \ advancements, or related topics. \nentertainment: The text pertains to movies,\ + \ music, celebrities, or related topics. \ngeography: The text involves geographical\ + \ information, locations, or related topics. \n\nIf the text contains multiple topics,\ + \ choose the dominant topic. For ambiguous or unclear topics, select the category\ + \ that best reflects the overall content. Please provide a single classification\ + \ for each input.\n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_nso_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_nus.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_nus.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a5f8e101910ba130d16708ce63388b4808286236 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_nus.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: nus_Latn +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Nuer text. For each input, classify the topic as science/technology, travel, politics,\ + \ sports, health, entertainment, or geography. Use the following guidelines: \n\n\ + \ science/technology: The text discusses scientific discoveries, technological advancements,\ + \ or related topics. \ntravel: The text describes travel experiences, destinations,\ + \ or related topics. \npolitics: The text covers political events, policies, or\ + \ related topics. \nsports: The text talks about sports events, athletes, or related\ + \ topics. \nhealth: The text addresses health issues, medical advancements, or related\ + \ topics. \nentertainment: The text pertains to movies, music, celebrities, or related\ + \ topics. \ngeography: The text involves geographical information, locations, or\ + \ related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{text}} \\category: \n\n" +include: sib +task: sib_nus_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_nya.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_nya.yaml new file mode 100644 index 0000000000000000000000000000000000000000..65737777ba6914aa1a735a260ee7ce7e3bfa9754 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_nya.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: nya_Latn +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Nyanga text. For each input, classify the topic as science/technology, travel,\ + \ politics, sports, health, entertainment, or geography. Use the following guidelines:\ + \ \n\n science/technology: The text discusses scientific discoveries, technological\ + \ advancements, or related topics. \ntravel: The text describes travel experiences,\ + \ destinations, or related topics. \npolitics: The text covers political events,\ + \ policies, or related topics. \nsports: The text talks about sports events, athletes,\ + \ or related topics. \nhealth: The text addresses health issues, medical advancements,\ + \ or related topics. \nentertainment: The text pertains to movies, music, celebrities,\ + \ or related topics. \ngeography: The text involves geographical information, locations,\ + \ or related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{text}} \\category: \n\n" +include: sib +task: sib_nya_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_plt.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_plt.yaml new file mode 100644 index 0000000000000000000000000000000000000000..24f6ea33e114f92a89b2b08581c3e2d93985362f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_plt.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: plt_Latn +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Plateau Malagasy text. For each input, classify the topic as science/technology,\ + \ travel, politics, sports, health, entertainment, or geography. Use the following\ + \ guidelines: \n\n science/technology: The text discusses scientific discoveries,\ + \ technological advancements, or related topics. \ntravel: The text describes travel\ + \ experiences, destinations, or related topics. \npolitics: The text covers political\ + \ events, policies, or related topics. \nsports: The text talks about sports events,\ + \ athletes, or related topics. \nhealth: The text addresses health issues, medical\ + \ advancements, or related topics. \nentertainment: The text pertains to movies,\ + \ music, celebrities, or related topics. \ngeography: The text involves geographical\ + \ information, locations, or related topics. \n\nIf the text contains multiple topics,\ + \ choose the dominant topic. For ambiguous or unclear topics, select the category\ + \ that best reflects the overall content. Please provide a single classification\ + \ for each input.\n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_plt_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_por.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_por.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d98ee118637f21a2ff1ffa30cd84099327965cbc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_por.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: por_Latn +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Portuguese text. For each input, classify the topic as science/technology, travel,\ + \ politics, sports, health, entertainment, or geography. Use the following guidelines:\ + \ \n\n science/technology: The text discusses scientific discoveries, technological\ + \ advancements, or related topics. \ntravel: The text describes travel experiences,\ + \ destinations, or related topics. \npolitics: The text covers political events,\ + \ policies, or related topics. \nsports: The text talks about sports events, athletes,\ + \ or related topics. \nhealth: The text addresses health issues, medical advancements,\ + \ or related topics. \nentertainment: The text pertains to movies, music, celebrities,\ + \ or related topics. \ngeography: The text involves geographical information, locations,\ + \ or related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{text}} \\category: \n\n" +include: sib +task: sib_por_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_run.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_run.yaml new file mode 100644 index 0000000000000000000000000000000000000000..01820da52cbd3b30dfe1c65a052b47ce1af0c7c0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_run.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: run_Latn +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Rundi text. For each input, classify the topic as science/technology, travel,\ + \ politics, sports, health, entertainment, or geography. Use the following guidelines:\ + \ \n\n science/technology: The text discusses scientific discoveries, technological\ + \ advancements, or related topics. \ntravel: The text describes travel experiences,\ + \ destinations, or related topics. \npolitics: The text covers political events,\ + \ policies, or related topics. \nsports: The text talks about sports events, athletes,\ + \ or related topics. \nhealth: The text addresses health issues, medical advancements,\ + \ or related topics. \nentertainment: The text pertains to movies, music, celebrities,\ + \ or related topics. \ngeography: The text involves geographical information, locations,\ + \ or related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{text}} \\category: \n\n" +include: sib +task: sib_run_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_sag.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_sag.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fdabdcb63ca35cc8fd419558911099c7d8f14877 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_sag.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: sag_Latn +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Sango text. For each input, classify the topic as science/technology, travel,\ + \ politics, sports, health, entertainment, or geography. Use the following guidelines:\ + \ \n\n science/technology: The text discusses scientific discoveries, technological\ + \ advancements, or related topics. \ntravel: The text describes travel experiences,\ + \ destinations, or related topics. \npolitics: The text covers political events,\ + \ policies, or related topics. \nsports: The text talks about sports events, athletes,\ + \ or related topics. \nhealth: The text addresses health issues, medical advancements,\ + \ or related topics. \nentertainment: The text pertains to movies, music, celebrities,\ + \ or related topics. \ngeography: The text involves geographical information, locations,\ + \ or related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{text}} \\category: \n\n" +include: sib +task: sib_sag_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_sna.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_sna.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3d66f53a7736d27346e47675426abfd4b63b6388 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_sna.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: sna_Latn +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Shona text. For each input, classify the topic as science/technology, travel,\ + \ politics, sports, health, entertainment, or geography. Use the following guidelines:\ + \ \n\n science/technology: The text discusses scientific discoveries, technological\ + \ advancements, or related topics. \ntravel: The text describes travel experiences,\ + \ destinations, or related topics. \npolitics: The text covers political events,\ + \ policies, or related topics. \nsports: The text talks about sports events, athletes,\ + \ or related topics. \nhealth: The text addresses health issues, medical advancements,\ + \ or related topics. \nentertainment: The text pertains to movies, music, celebrities,\ + \ or related topics. \ngeography: The text involves geographical information, locations,\ + \ or related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{text}} \\category: \n\n" +include: sib +task: sib_sna_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_som.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_som.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c0c34f97d20b7c7e4b7be5b7225bf6a91baec3e0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_som.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: som_Latn +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Somali text. For each input, classify the topic as science/technology, travel,\ + \ politics, sports, health, entertainment, or geography. Use the following guidelines:\ + \ \n\n science/technology: The text discusses scientific discoveries, technological\ + \ advancements, or related topics. \ntravel: The text describes travel experiences,\ + \ destinations, or related topics. \npolitics: The text covers political events,\ + \ policies, or related topics. \nsports: The text talks about sports events, athletes,\ + \ or related topics. \nhealth: The text addresses health issues, medical advancements,\ + \ or related topics. \nentertainment: The text pertains to movies, music, celebrities,\ + \ or related topics. \ngeography: The text involves geographical information, locations,\ + \ or related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{text}} \\category: \n\n" +include: sib +task: sib_som_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_sot.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_sot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..81ab5c3f7e66f457d59edef55eb79c693c18913d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_sot.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: sot_Latn +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Southern Sotho text. For each input, classify the topic as science/technology,\ + \ travel, politics, sports, health, entertainment, or geography. Use the following\ + \ guidelines: \n\n science/technology: The text discusses scientific discoveries,\ + \ technological advancements, or related topics. \ntravel: The text describes travel\ + \ experiences, destinations, or related topics. \npolitics: The text covers political\ + \ events, policies, or related topics. \nsports: The text talks about sports events,\ + \ athletes, or related topics. \nhealth: The text addresses health issues, medical\ + \ advancements, or related topics. \nentertainment: The text pertains to movies,\ + \ music, celebrities, or related topics. \ngeography: The text involves geographical\ + \ information, locations, or related topics. \n\nIf the text contains multiple topics,\ + \ choose the dominant topic. For ambiguous or unclear topics, select the category\ + \ that best reflects the overall content. Please provide a single classification\ + \ for each input.\n\ntext: {{text}} \\category: \n\n" +include: sib +task: sib_sot_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_ssw.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_ssw.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f662d2ab44cebe8ec184c7864207b9ceafa95f58 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_ssw.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: ssw_Latn +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Swazi text. For each input, classify the topic as science/technology, travel,\ + \ politics, sports, health, entertainment, or geography. Use the following guidelines:\ + \ \n\n science/technology: The text discusses scientific discoveries, technological\ + \ advancements, or related topics. \ntravel: The text describes travel experiences,\ + \ destinations, or related topics. \npolitics: The text covers political events,\ + \ policies, or related topics. \nsports: The text talks about sports events, athletes,\ + \ or related topics. \nhealth: The text addresses health issues, medical advancements,\ + \ or related topics. \nentertainment: The text pertains to movies, music, celebrities,\ + \ or related topics. \ngeography: The text involves geographical information, locations,\ + \ or related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{text}} \\category: \n\n" +include: sib +task: sib_ssw_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ee47ca51598f6455804e6e6cad3fb1ca1cacc4d6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_swa.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: swh_Latn +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Swahili text. For each input, classify the topic as science/technology, travel,\ + \ politics, sports, health, entertainment, or geography. Use the following guidelines:\ + \ \n\n science/technology: The text discusses scientific discoveries, technological\ + \ advancements, or related topics. \ntravel: The text describes travel experiences,\ + \ destinations, or related topics. \npolitics: The text covers political events,\ + \ policies, or related topics. \nsports: The text talks about sports events, athletes,\ + \ or related topics. \nhealth: The text addresses health issues, medical advancements,\ + \ or related topics. \nentertainment: The text pertains to movies, music, celebrities,\ + \ or related topics. \ngeography: The text involves geographical information, locations,\ + \ or related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{text}} \\category: \n\n" +include: sib +task: sib_swa_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_taq.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_taq.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3fa1380df4b219fae78d9af94069a3becc256832 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_taq.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: taq_Latn +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Tamasheq text. For each input, classify the topic as science/technology, travel,\ + \ politics, sports, health, entertainment, or geography. Use the following guidelines:\ + \ \n\n science/technology: The text discusses scientific discoveries, technological\ + \ advancements, or related topics. \ntravel: The text describes travel experiences,\ + \ destinations, or related topics. \npolitics: The text covers political events,\ + \ policies, or related topics. \nsports: The text talks about sports events, athletes,\ + \ or related topics. \nhealth: The text addresses health issues, medical advancements,\ + \ or related topics. \nentertainment: The text pertains to movies, music, celebrities,\ + \ or related topics. \ngeography: The text involves geographical information, locations,\ + \ or related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{text}} \\category: \n\n" +include: sib +task: sib_taq_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_tir.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_tir.yaml new file mode 100644 index 0000000000000000000000000000000000000000..20ec0638837c561c45b0267d47a2a7481a3e9ec7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_tir.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: tir_Ethi +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Tigrinya text. For each input, classify the topic as science/technology, travel,\ + \ politics, sports, health, entertainment, or geography. Use the following guidelines:\ + \ \n\n science/technology: The text discusses scientific discoveries, technological\ + \ advancements, or related topics. \ntravel: The text describes travel experiences,\ + \ destinations, or related topics. \npolitics: The text covers political events,\ + \ policies, or related topics. \nsports: The text talks about sports events, athletes,\ + \ or related topics. \nhealth: The text addresses health issues, medical advancements,\ + \ or related topics. \nentertainment: The text pertains to movies, music, celebrities,\ + \ or related topics. \ngeography: The text involves geographical information, locations,\ + \ or related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{text}} \\category: \n\n" +include: sib +task: sib_tir_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_tso.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_tso.yaml new file mode 100644 index 0000000000000000000000000000000000000000..44b3b867a796111bbcfe2d295ff5c04435878208 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_tso.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: tso_Latn +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Tsonga text. For each input, classify the topic as science/technology, travel,\ + \ politics, sports, health, entertainment, or geography. Use the following guidelines:\ + \ \n\n science/technology: The text discusses scientific discoveries, technological\ + \ advancements, or related topics. \ntravel: The text describes travel experiences,\ + \ destinations, or related topics. \npolitics: The text covers political events,\ + \ policies, or related topics. \nsports: The text talks about sports events, athletes,\ + \ or related topics. \nhealth: The text addresses health issues, medical advancements,\ + \ or related topics. \nentertainment: The text pertains to movies, music, celebrities,\ + \ or related topics. \ngeography: The text involves geographical information, locations,\ + \ or related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{text}} \\category: \n\n" +include: sib +task: sib_tso_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_tum.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_tum.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bb15fb71e821e69e255b36183d2273a75292fd60 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_tum.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: tum_Latn +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Tumbuka text. For each input, classify the topic as science/technology, travel,\ + \ politics, sports, health, entertainment, or geography. Use the following guidelines:\ + \ \n\n science/technology: The text discusses scientific discoveries, technological\ + \ advancements, or related topics. \ntravel: The text describes travel experiences,\ + \ destinations, or related topics. \npolitics: The text covers political events,\ + \ policies, or related topics. \nsports: The text talks about sports events, athletes,\ + \ or related topics. \nhealth: The text addresses health issues, medical advancements,\ + \ or related topics. \nentertainment: The text pertains to movies, music, celebrities,\ + \ or related topics. \ngeography: The text involves geographical information, locations,\ + \ or related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{text}} \\category: \n\n" +include: sib +task: sib_tum_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..44bca6194bc19417067ce10659958d0c5993ad87 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_twi.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: twi_Latn +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Twi text. For each input, classify the topic as science/technology, travel, politics,\ + \ sports, health, entertainment, or geography. Use the following guidelines: \n\n\ + \ science/technology: The text discusses scientific discoveries, technological advancements,\ + \ or related topics. \ntravel: The text describes travel experiences, destinations,\ + \ or related topics. \npolitics: The text covers political events, policies, or\ + \ related topics. \nsports: The text talks about sports events, athletes, or related\ + \ topics. \nhealth: The text addresses health issues, medical advancements, or related\ + \ topics. \nentertainment: The text pertains to movies, music, celebrities, or related\ + \ topics. \ngeography: The text involves geographical information, locations, or\ + \ related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{text}} \\category: \n\n" +include: sib +task: sib_twi_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_tzm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_tzm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5d1af77d17ccc2e287c4e598b0094383ec5e4b01 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_tzm.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: tzm_Tfng +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Tamazight text. For each input, classify the topic as science/technology, travel,\ + \ politics, sports, health, entertainment, or geography. Use the following guidelines:\ + \ \n\n science/technology: The text discusses scientific discoveries, technological\ + \ advancements, or related topics. \ntravel: The text describes travel experiences,\ + \ destinations, or related topics. \npolitics: The text covers political events,\ + \ policies, or related topics. \nsports: The text talks about sports events, athletes,\ + \ or related topics. \nhealth: The text addresses health issues, medical advancements,\ + \ or related topics. \nentertainment: The text pertains to movies, music, celebrities,\ + \ or related topics. \ngeography: The text involves geographical information, locations,\ + \ or related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{text}} \\category: \n\n" +include: sib +task: sib_tzm_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_umb.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_umb.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a756680cfa29d6b4c83363192d734196989b2d45 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_umb.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: umb_Latn +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Umbundu text. For each input, classify the topic as science/technology, travel,\ + \ politics, sports, health, entertainment, or geography. Use the following guidelines:\ + \ \n\n science/technology: The text discusses scientific discoveries, technological\ + \ advancements, or related topics. \ntravel: The text describes travel experiences,\ + \ destinations, or related topics. \npolitics: The text covers political events,\ + \ policies, or related topics. \nsports: The text talks about sports events, athletes,\ + \ or related topics. \nhealth: The text addresses health issues, medical advancements,\ + \ or related topics. \nentertainment: The text pertains to movies, music, celebrities,\ + \ or related topics. \ngeography: The text involves geographical information, locations,\ + \ or related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{text}} \\category: \n\n" +include: sib +task: sib_umb_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8062b55d71066a62df167f96c8e6a72b67e51b60 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_wol.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: wol_Latn +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Wolof text. For each input, classify the topic as science/technology, travel,\ + \ politics, sports, health, entertainment, or geography. Use the following guidelines:\ + \ \n\n science/technology: The text discusses scientific discoveries, technological\ + \ advancements, or related topics. \ntravel: The text describes travel experiences,\ + \ destinations, or related topics. \npolitics: The text covers political events,\ + \ policies, or related topics. \nsports: The text talks about sports events, athletes,\ + \ or related topics. \nhealth: The text addresses health issues, medical advancements,\ + \ or related topics. \nentertainment: The text pertains to movies, music, celebrities,\ + \ or related topics. \ngeography: The text involves geographical information, locations,\ + \ or related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{text}} \\category: \n\n" +include: sib +task: sib_wol_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..22c27b71a7e08c4f9c58039f5f52a10df324a878 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_xho.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: xho_Latn +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Xhosa text. For each input, classify the topic as science/technology, travel,\ + \ politics, sports, health, entertainment, or geography. Use the following guidelines:\ + \ \n\n science/technology: The text discusses scientific discoveries, technological\ + \ advancements, or related topics. \ntravel: The text describes travel experiences,\ + \ destinations, or related topics. \npolitics: The text covers political events,\ + \ policies, or related topics. \nsports: The text talks about sports events, athletes,\ + \ or related topics. \nhealth: The text addresses health issues, medical advancements,\ + \ or related topics. \nentertainment: The text pertains to movies, music, celebrities,\ + \ or related topics. \ngeography: The text involves geographical information, locations,\ + \ or related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{text}} \\category: \n\n" +include: sib +task: sib_xho_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..df51978255654580b37eca4e552e4561a6455e69 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_yor.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: yor_Latn +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Yoruba text. For each input, classify the topic as science/technology, travel,\ + \ politics, sports, health, entertainment, or geography. Use the following guidelines:\ + \ \n\n science/technology: The text discusses scientific discoveries, technological\ + \ advancements, or related topics. \ntravel: The text describes travel experiences,\ + \ destinations, or related topics. \npolitics: The text covers political events,\ + \ policies, or related topics. \nsports: The text talks about sports events, athletes,\ + \ or related topics. \nhealth: The text addresses health issues, medical advancements,\ + \ or related topics. \nentertainment: The text pertains to movies, music, celebrities,\ + \ or related topics. \ngeography: The text involves geographical information, locations,\ + \ or related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{text}} \\category: \n\n" +include: sib +task: sib_yor_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..03fb9af917049b8dac781d9aefac58ebd3fe4dba --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/sib_zul.yaml @@ -0,0 +1,18 @@ +# Generated by utils.py +dataset_name: zul_Latn +doc_to_text: "You are tasked with performing topic classification on the following\ + \ Zulu text. For each input, classify the topic as science/technology, travel, politics,\ + \ sports, health, entertainment, or geography. Use the following guidelines: \n\n\ + \ science/technology: The text discusses scientific discoveries, technological advancements,\ + \ or related topics. \ntravel: The text describes travel experiences, destinations,\ + \ or related topics. \npolitics: The text covers political events, policies, or\ + \ related topics. \nsports: The text talks about sports events, athletes, or related\ + \ topics. \nhealth: The text addresses health issues, medical advancements, or related\ + \ topics. \nentertainment: The text pertains to movies, music, celebrities, or related\ + \ topics. \ngeography: The text involves geographical information, locations, or\ + \ related topics. \n\nIf the text contains multiple topics, choose the dominant\ + \ topic. For ambiguous or unclear topics, select the category that best reflects\ + \ the overall content. Please provide a single classification for each input.\n\n\ + text: {{text}} \\category: \n\n" +include: sib +task: sib_zul_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..3e735e2deb1f9c53152c072615aebe8ba3acb90b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/sib/prompt_5/utils.py @@ -0,0 +1 @@ +from lm_eval.utils import weighted_f1_score diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/README.md b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/README.md new file mode 100644 index 0000000000000000000000000000000000000000..a0253f987e3723c309bcb5ce4c9a9ad2b3a166ec --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/README.md @@ -0,0 +1,25 @@ +# + +## Paper +Title: `Uhura: A Benchmark for Evaluating Scientific Question Answering and Truthfulness in Low-Resource African Languages` + +Paper Link: https://arxiv.org/abs/2412.00948 + +## Abstract +>Evaluations of Large Language Models (LLMs) on knowledge-intensive tasks and factual accuracy often focus on high-resource languages primarily because datasets for low-resource languages (LRLs) are scarce. In this paper, we present Uhura -- a new benchmark that focuses on two tasks in six typologically-diverse African languages, created via human translation of existing English benchmarks. The first dataset, Uhura-ARC-Easy, is composed of multiple-choice science questions. The second, Uhura-TruthfulQA, is a safety benchmark testing the truthfulness of models on topics including health, law, finance, and politics. We highlight the challenges creating benchmarks with highly technical content for LRLs and outline mitigation strategies. Our evaluation reveals a significant performance gap between proprietary models such as GPT-4o and o1-preview, and Claude models, and open-source models like Meta's LLaMA and Google's Gemma. Additionally, all models perform better in English than in African languages. These results indicate that LMs struggle with answering scientific questions and are more prone to generating false claims in low-resource African languages. Our findings underscore the necessity for continuous improvement of multilingual LM capabilities in LRL settings to ensure safe and reliable use in real-world contexts. We open-source the Uhura Benchmark and Uhura Platform to foster further research and development in NLP for LRLs. + +HomePage: https://huggingface.co/datasets/masakhane/uhura-arc-easy + +### Citation + +``` +@misc{bayes2024uhurabenchmarkevaluatingscientific, + title={Uhura: A Benchmark for Evaluating Scientific Question Answering and Truthfulness in Low-Resource African Languages}, + author={Edward Bayes and Israel Abebe Azime and Jesujoba O. Alabi and Jonas Kgomo and Tyna Eloundou and Elizabeth Proehl and Kai Chen and Imaan Khadir and Naome A. Etori and Shamsuddeen Hassan Muhammad and Choice Mpanza and Igneciah Pocia Thete and Dietrich Klakow and David Ifeoluwa Adelani}, + year={2024}, + eprint={2412.00948}, + archivePrefix={arXiv}, + primaryClass={cs.CL}, + url={https://arxiv.org/abs/2412.00948}, +} +``` diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_1/uhura-arc-easy b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_1/uhura-arc-easy new file mode 100644 index 0000000000000000000000000000000000000000..a7e37181359d9021a3dbb669c42a0e80e0b36c8f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_1/uhura-arc-easy @@ -0,0 +1,39 @@ +tag: + - uhura_arc_easy_tasks + - uhura_arc_easy_prompt_1 +task: null +dataset_path: masakhane/uhura-arc-easy +dataset_name: null +output_type: multiple_choice +doc_to_target: "{{['A', 'B', 'C', 'D'].index(answerKey)}}" +doc_to_choice: + - A + - B + - C + - D +test_split: test +fewshot_split: validation +should_decontaminate: false +doc_to_decontamination_query: "Question: {{question}}\nAnswer:" +metric_list: + - metric: f1 + aggregation: !function utils.weighted_f1_score + # aggregation: mean + average: weighted + hf_evaluate: true + higher_is_better: True + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" + - metric: acc + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_1/uhura-arc-easy_am.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_1/uhura-arc-easy_am.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f61efe4ea501ae6ee9c7553a05bdb7d8540c08f7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_1/uhura-arc-easy_am.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: am_multiple_choice +doc_to_text: "You are a virtual assistant that answers multiple-choice questions with\ + \ the correct option only.\n\nQuestion: {{question}}\n\nChoices:\n\n{% for i in\ + \ range(choices['text']|length) %}\t{{ 'ABCD'[i] }}: {{ choices['text'][i] }}\n\ + {% endfor %}\nAnswer: " +include: uhura-arc-easy +task: uhura-arc-easy_am_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_1/uhura-arc-easy_en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_1/uhura-arc-easy_en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f1b879e0221b72d2da47f4fe033f83f20526fef2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_1/uhura-arc-easy_en.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: en_multiple_choice +doc_to_text: "You are a virtual assistant that answers multiple-choice questions with\ + \ the correct option only.\n\nQuestion: {{question}}\n\nChoices:\n\n{% for i in\ + \ range(choices['text']|length) %}\t{{ 'ABCD'[i] }}: {{ choices['text'][i] }}\n\ + {% endfor %}\nAnswer: " +include: uhura-arc-easy +task: uhura-arc-easy_en_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_1/uhura-arc-easy_ha.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_1/uhura-arc-easy_ha.yaml new file mode 100644 index 0000000000000000000000000000000000000000..986ac5074660ef3c9756e1112e8aa5f34eafefe2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_1/uhura-arc-easy_ha.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: ha_multiple_choice +doc_to_text: "You are a virtual assistant that answers multiple-choice questions with\ + \ the correct option only.\n\nQuestion: {{question}}\n\nChoices:\n\n{% for i in\ + \ range(choices['text']|length) %}\t{{ 'ABCD'[i] }}: {{ choices['text'][i] }}\n\ + {% endfor %}\nAnswer: " +include: uhura-arc-easy +task: uhura-arc-easy_ha_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_1/uhura-arc-easy_nso.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_1/uhura-arc-easy_nso.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ead6d97d67ec6f76e5378899e979413b6f6bb41b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_1/uhura-arc-easy_nso.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: nso_multiple_choice +doc_to_text: "You are a virtual assistant that answers multiple-choice questions with\ + \ the correct option only.\n\nQuestion: {{question}}\n\nChoices:\n\n{% for i in\ + \ range(choices['text']|length) %}\t{{ 'ABCD'[i] }}: {{ choices['text'][i] }}\n\ + {% endfor %}\nAnswer: " +fewshot_split: train +include: uhura-arc-easy +task: uhura-arc-easy_nso_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_1/uhura-arc-easy_sw.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_1/uhura-arc-easy_sw.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2e07bb234736d67e91cbcf798d6992fda2f438ec --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_1/uhura-arc-easy_sw.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: sw_multiple_choice +doc_to_text: "You are a virtual assistant that answers multiple-choice questions with\ + \ the correct option only.\n\nQuestion: {{question}}\n\nChoices:\n\n{% for i in\ + \ range(choices['text']|length) %}\t{{ 'ABCD'[i] }}: {{ choices['text'][i] }}\n\ + {% endfor %}\nAnswer: " +include: uhura-arc-easy +task: uhura-arc-easy_sw_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_1/uhura-arc-easy_yo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_1/uhura-arc-easy_yo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f96113e4a5d3712fec89aab27fb309c0b85551b3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_1/uhura-arc-easy_yo.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: yo_multiple_choice +doc_to_text: "You are a virtual assistant that answers multiple-choice questions with\ + \ the correct option only.\n\nQuestion: {{question}}\n\nChoices:\n\n{% for i in\ + \ range(choices['text']|length) %}\t{{ 'ABCD'[i] }}: {{ choices['text'][i] }}\n\ + {% endfor %}\nAnswer: " +include: uhura-arc-easy +task: uhura-arc-easy_yo_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_1/uhura-arc-easy_zu.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_1/uhura-arc-easy_zu.yaml new file mode 100644 index 0000000000000000000000000000000000000000..41c965a071018685e0539ae8ee0f18389d4a0d01 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_1/uhura-arc-easy_zu.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: zu_multiple_choice +doc_to_text: "You are a virtual assistant that answers multiple-choice questions with\ + \ the correct option only.\n\nQuestion: {{question}}\n\nChoices:\n\n{% for i in\ + \ range(choices['text']|length) %}\t{{ 'ABCD'[i] }}: {{ choices['text'][i] }}\n\ + {% endfor %}\nAnswer: " +fewshot_split: train +include: uhura-arc-easy +task: uhura-arc-easy_zu_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_1/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_1/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..3e735e2deb1f9c53152c072615aebe8ba3acb90b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_1/utils.py @@ -0,0 +1 @@ +from lm_eval.utils import weighted_f1_score diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_2/uhura-arc-easy b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_2/uhura-arc-easy new file mode 100644 index 0000000000000000000000000000000000000000..295d9c8e907f841f74f8e9b7253d52c6eee2b224 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_2/uhura-arc-easy @@ -0,0 +1,38 @@ +tag: + - uhura_arc_easy_tasks + - uhura_arc_easy_prompt_2 +dataset_path: masakhane/uhura-arc-easy +dataset_name: null +output_type: multiple_choice +doc_to_target: "{{['A', 'B', 'C', 'D'].index(answerKey)}}" +doc_to_choice: + - A + - B + - C + - D +test_split: test +fewshot_split: validation +should_decontaminate: false +doc_to_decontamination_query: "Question: {{question}}\nAnswer:" +metric_list: + - metric: f1 + aggregation: !function utils.weighted_f1_score + # aggregation: mean + average: weighted + hf_evaluate: true + higher_is_better: True + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" + - metric: acc + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_2/uhura-arc-easy_am.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_2/uhura-arc-easy_am.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2596bd487078859572e018d19f6f39c5e32f3dc5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_2/uhura-arc-easy_am.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: am_multiple_choice +doc_to_text: "Choose the correct option that answers the question below:\n\nQuestion:\ + \ {{question}}\n\nChoices:\n\n{% for i in range(choices['text']|length) %}\t{{ 'ABCD'[i]\ + \ }}: {{ choices['text'][i] }}\n{% endfor %}\nAnswer: " +include: uhura-arc-easy +task: uhura-arc-easy_am_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_2/uhura-arc-easy_en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_2/uhura-arc-easy_en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f3edfc10ea1ada6252d938e57aed3a1f03ade802 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_2/uhura-arc-easy_en.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: en_multiple_choice +doc_to_text: "Choose the correct option that answers the question below:\n\nQuestion:\ + \ {{question}}\n\nChoices:\n\n{% for i in range(choices['text']|length) %}\t{{ 'ABCD'[i]\ + \ }}: {{ choices['text'][i] }}\n{% endfor %}\nAnswer: " +include: uhura-arc-easy +task: uhura-arc-easy_en_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_2/uhura-arc-easy_ha.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_2/uhura-arc-easy_ha.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d857b2e44c5fb9a7da2ac1e8dfbb426978799073 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_2/uhura-arc-easy_ha.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: ha_multiple_choice +doc_to_text: "Choose the correct option that answers the question below:\n\nQuestion:\ + \ {{question}}\n\nChoices:\n\n{% for i in range(choices['text']|length) %}\t{{ 'ABCD'[i]\ + \ }}: {{ choices['text'][i] }}\n{% endfor %}\nAnswer: " +include: uhura-arc-easy +task: uhura-arc-easy_ha_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_2/uhura-arc-easy_nso.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_2/uhura-arc-easy_nso.yaml new file mode 100644 index 0000000000000000000000000000000000000000..93fbfe587dc9dfb45c08aaaeb3c6c3528d766110 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_2/uhura-arc-easy_nso.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: nso_multiple_choice +doc_to_text: "Choose the correct option that answers the question below:\n\nQuestion:\ + \ {{question}}\n\nChoices:\n\n{% for i in range(choices['text']|length) %}\t{{ 'ABCD'[i]\ + \ }}: {{ choices['text'][i] }}\n{% endfor %}\nAnswer: " +fewshot_split: train +include: uhura-arc-easy +task: uhura-arc-easy_nso_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_2/uhura-arc-easy_sw.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_2/uhura-arc-easy_sw.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b5fc929f54de32f0010844d2ed816cd3b634184f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_2/uhura-arc-easy_sw.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: sw_multiple_choice +doc_to_text: "Choose the correct option that answers the question below:\n\nQuestion:\ + \ {{question}}\n\nChoices:\n\n{% for i in range(choices['text']|length) %}\t{{ 'ABCD'[i]\ + \ }}: {{ choices['text'][i] }}\n{% endfor %}\nAnswer: " +include: uhura-arc-easy +task: uhura-arc-easy_sw_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_2/uhura-arc-easy_yo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_2/uhura-arc-easy_yo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..67b09752a4491d345d61bbbba5219cc1f5001544 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_2/uhura-arc-easy_yo.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: yo_multiple_choice +doc_to_text: "Choose the correct option that answers the question below:\n\nQuestion:\ + \ {{question}}\n\nChoices:\n\n{% for i in range(choices['text']|length) %}\t{{ 'ABCD'[i]\ + \ }}: {{ choices['text'][i] }}\n{% endfor %}\nAnswer: " +include: uhura-arc-easy +task: uhura-arc-easy_yo_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_2/uhura-arc-easy_zu.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_2/uhura-arc-easy_zu.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4b261b51fcc11770a22eaf0bd8285b2028d416b9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_2/uhura-arc-easy_zu.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: zu_multiple_choice +doc_to_text: "Choose the correct option that answers the question below:\n\nQuestion:\ + \ {{question}}\n\nChoices:\n\n{% for i in range(choices['text']|length) %}\t{{ 'ABCD'[i]\ + \ }}: {{ choices['text'][i] }}\n{% endfor %}\nAnswer: " +fewshot_split: train +include: uhura-arc-easy +task: uhura-arc-easy_zu_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_2/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_2/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..3e735e2deb1f9c53152c072615aebe8ba3acb90b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_2/utils.py @@ -0,0 +1 @@ +from lm_eval.utils import weighted_f1_score diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_3/uhura-arc-easy b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_3/uhura-arc-easy new file mode 100644 index 0000000000000000000000000000000000000000..23e2c37396c75ce09fcd608e69d5ce42173df1ca --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_3/uhura-arc-easy @@ -0,0 +1,38 @@ +tag: + - uhura_arc_easy_tasks + - uhura_arc_easy_prompt_3 +dataset_path: masakhane/uhura-arc-easy +dataset_name: null +output_type: multiple_choice +doc_to_target: "{{['A', 'B', 'C', 'D'].index(answerKey)}}" +doc_to_choice: + - A + - B + - C + - D +test_split: test +fewshot_split: validation +should_decontaminate: false +doc_to_decontamination_query: "Question: {{question}}\nAnswer:" +metric_list: + - metric: f1 + aggregation: !function utils.weighted_f1_score + # aggregation: mean + average: weighted + hf_evaluate: true + higher_is_better: True + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" + - metric: acc + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_3/uhura-arc-easy_am.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_3/uhura-arc-easy_am.yaml new file mode 100644 index 0000000000000000000000000000000000000000..42716a7cdc9b1ac266efbec32ebe4bebe6bf578e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_3/uhura-arc-easy_am.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: am_multiple_choice +doc_to_text: "Answer the following multiple-choice question by picking 'A', 'B', 'C',\ + \ or 'D'.\n\nQuestion: {{question}}\n\nOptions:\n\n{% for i in range(choices['text']|length)\ + \ %}\t{{ 'ABCD'[i] }}: {{ choices['text'][i] }}\n{% endfor %}\nAnswer: " +include: uhura-arc-easy +task: uhura-arc-easy_am_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_3/uhura-arc-easy_en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_3/uhura-arc-easy_en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a89312e09a10b9db5a2a9a2a0914980c9ef686a5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_3/uhura-arc-easy_en.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: en_multiple_choice +doc_to_text: "Answer the following multiple-choice question by picking 'A', 'B', 'C',\ + \ or 'D'.\n\nQuestion: {{question}}\n\nOptions:\n\n{% for i in range(choices['text']|length)\ + \ %}\t{{ 'ABCD'[i] }}: {{ choices['text'][i] }}\n{% endfor %}\nAnswer: " +include: uhura-arc-easy +task: uhura-arc-easy_en_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_3/uhura-arc-easy_ha.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_3/uhura-arc-easy_ha.yaml new file mode 100644 index 0000000000000000000000000000000000000000..de511a8af7dc52180b56d52a1c7d57955cbd6eb4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_3/uhura-arc-easy_ha.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: ha_multiple_choice +doc_to_text: "Answer the following multiple-choice question by picking 'A', 'B', 'C',\ + \ or 'D'.\n\nQuestion: {{question}}\n\nOptions:\n\n{% for i in range(choices['text']|length)\ + \ %}\t{{ 'ABCD'[i] }}: {{ choices['text'][i] }}\n{% endfor %}\nAnswer: " +include: uhura-arc-easy +task: uhura-arc-easy_ha_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_3/uhura-arc-easy_nso.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_3/uhura-arc-easy_nso.yaml new file mode 100644 index 0000000000000000000000000000000000000000..358d084cea2131f7e94f82ec733234371f2b8446 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_3/uhura-arc-easy_nso.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: nso_multiple_choice +doc_to_text: "Answer the following multiple-choice question by picking 'A', 'B', 'C',\ + \ or 'D'.\n\nQuestion: {{question}}\n\nOptions:\n\n{% for i in range(choices['text']|length)\ + \ %}\t{{ 'ABCD'[i] }}: {{ choices['text'][i] }}\n{% endfor %}\nAnswer: " +fewshot_split: train +include: uhura-arc-easy +task: uhura-arc-easy_nso_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_3/uhura-arc-easy_sw.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_3/uhura-arc-easy_sw.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d4a8785d622caed597dbb0eefbd4290f2636f866 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_3/uhura-arc-easy_sw.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: sw_multiple_choice +doc_to_text: "Answer the following multiple-choice question by picking 'A', 'B', 'C',\ + \ or 'D'.\n\nQuestion: {{question}}\n\nOptions:\n\n{% for i in range(choices['text']|length)\ + \ %}\t{{ 'ABCD'[i] }}: {{ choices['text'][i] }}\n{% endfor %}\nAnswer: " +include: uhura-arc-easy +task: uhura-arc-easy_sw_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_3/uhura-arc-easy_yo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_3/uhura-arc-easy_yo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e9416362827513ac4b95cf843a83ec0d0efbc45e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_3/uhura-arc-easy_yo.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: yo_multiple_choice +doc_to_text: "Answer the following multiple-choice question by picking 'A', 'B', 'C',\ + \ or 'D'.\n\nQuestion: {{question}}\n\nOptions:\n\n{% for i in range(choices['text']|length)\ + \ %}\t{{ 'ABCD'[i] }}: {{ choices['text'][i] }}\n{% endfor %}\nAnswer: " +include: uhura-arc-easy +task: uhura-arc-easy_yo_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_3/uhura-arc-easy_zu.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_3/uhura-arc-easy_zu.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6a44b8c0e6ebae36db00a6847aacb3263c84fb7b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_3/uhura-arc-easy_zu.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: zu_multiple_choice +doc_to_text: "Answer the following multiple-choice question by picking 'A', 'B', 'C',\ + \ or 'D'.\n\nQuestion: {{question}}\n\nOptions:\n\n{% for i in range(choices['text']|length)\ + \ %}\t{{ 'ABCD'[i] }}: {{ choices['text'][i] }}\n{% endfor %}\nAnswer: " +fewshot_split: train +include: uhura-arc-easy +task: uhura-arc-easy_zu_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_3/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_3/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..3e735e2deb1f9c53152c072615aebe8ba3acb90b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_3/utils.py @@ -0,0 +1 @@ +from lm_eval.utils import weighted_f1_score diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_4/uhura-arc-easy b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_4/uhura-arc-easy new file mode 100644 index 0000000000000000000000000000000000000000..e697f4c7363aee6c39b0d927ba9d1b575f4063d5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_4/uhura-arc-easy @@ -0,0 +1,38 @@ +tag: + - uhura_arc_easy_tasks + - uhura_arc_easy_prompt_4 +dataset_path: masakhane/uhura-arc-easy +dataset_name: null +output_type: multiple_choice +doc_to_target: "{{['A', 'B', 'C', 'D'].index(answerKey)}}" +doc_to_choice: + - A + - B + - C + - D +test_split: test +fewshot_split: validation +should_decontaminate: false +doc_to_decontamination_query: "Question: {{question}}\nAnswer:" +metric_list: + - metric: f1 + aggregation: !function utils.weighted_f1_score + # aggregation: mean + average: weighted + hf_evaluate: true + higher_is_better: True + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" + - metric: acc + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_4/uhura-arc-easy_am.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_4/uhura-arc-easy_am.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4eaa02f59b217a8ce13f41fca67f9491aab917aa --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_4/uhura-arc-easy_am.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: am_multiple_choice +doc_to_text: "Question: {{question}}\n\nOptions:\n\n{% for i in range(choices['text']|length)\ + \ %}\t{{ 'ABCD'[i] }}: {{ choices['text'][i] }}\n{% endfor %}\nAnswer: " +include: uhura-arc-easy +task: uhura-arc-easy_am_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_4/uhura-arc-easy_en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_4/uhura-arc-easy_en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..461e6f9e7516e2382380e391f3d2d713bf494ef9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_4/uhura-arc-easy_en.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: en_multiple_choice +doc_to_text: "Question: {{question}}\n\nOptions:\n\n{% for i in range(choices['text']|length)\ + \ %}\t{{ 'ABCD'[i] }}: {{ choices['text'][i] }}\n{% endfor %}\nAnswer: " +include: uhura-arc-easy +task: uhura-arc-easy_en_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_4/uhura-arc-easy_ha.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_4/uhura-arc-easy_ha.yaml new file mode 100644 index 0000000000000000000000000000000000000000..435ea73bc7639669ac53445f0ea9adf72edbd347 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_4/uhura-arc-easy_ha.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: ha_multiple_choice +doc_to_text: "Question: {{question}}\n\nOptions:\n\n{% for i in range(choices['text']|length)\ + \ %}\t{{ 'ABCD'[i] }}: {{ choices['text'][i] }}\n{% endfor %}\nAnswer: " +include: uhura-arc-easy +task: uhura-arc-easy_ha_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_4/uhura-arc-easy_nso.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_4/uhura-arc-easy_nso.yaml new file mode 100644 index 0000000000000000000000000000000000000000..09112d5af3adaf7251a39032aabf60af73779088 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_4/uhura-arc-easy_nso.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: nso_multiple_choice +doc_to_text: "Question: {{question}}\n\nOptions:\n\n{% for i in range(choices['text']|length)\ + \ %}\t{{ 'ABCD'[i] }}: {{ choices['text'][i] }}\n{% endfor %}\nAnswer: " +fewshot_split: train +include: uhura-arc-easy +task: uhura-arc-easy_nso_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_4/uhura-arc-easy_sw.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_4/uhura-arc-easy_sw.yaml new file mode 100644 index 0000000000000000000000000000000000000000..264770eeda75acdf9088ac24e24645a9e5638c25 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_4/uhura-arc-easy_sw.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: sw_multiple_choice +doc_to_text: "Question: {{question}}\n\nOptions:\n\n{% for i in range(choices['text']|length)\ + \ %}\t{{ 'ABCD'[i] }}: {{ choices['text'][i] }}\n{% endfor %}\nAnswer: " +include: uhura-arc-easy +task: uhura-arc-easy_sw_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_4/uhura-arc-easy_yo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_4/uhura-arc-easy_yo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..10af53de81d71c82f02f1da80bdc9b5dc114bfed --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_4/uhura-arc-easy_yo.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: yo_multiple_choice +doc_to_text: "Question: {{question}}\n\nOptions:\n\n{% for i in range(choices['text']|length)\ + \ %}\t{{ 'ABCD'[i] }}: {{ choices['text'][i] }}\n{% endfor %}\nAnswer: " +include: uhura-arc-easy +task: uhura-arc-easy_yo_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_4/uhura-arc-easy_zu.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_4/uhura-arc-easy_zu.yaml new file mode 100644 index 0000000000000000000000000000000000000000..239b1648a6e2c5d431a83ef0a94c16c6db90cc1b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_4/uhura-arc-easy_zu.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: zu_multiple_choice +doc_to_text: "Question: {{question}}\n\nOptions:\n\n{% for i in range(choices['text']|length)\ + \ %}\t{{ 'ABCD'[i] }}: {{ choices['text'][i] }}\n{% endfor %}\nAnswer: " +fewshot_split: train +include: uhura-arc-easy +task: uhura-arc-easy_zu_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_4/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_4/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..3e735e2deb1f9c53152c072615aebe8ba3acb90b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_4/utils.py @@ -0,0 +1 @@ +from lm_eval.utils import weighted_f1_score diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_5/uhura-arc-easy b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_5/uhura-arc-easy new file mode 100644 index 0000000000000000000000000000000000000000..3f5ac554027a87a6fe5eeda14887a46f5af5ef2f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_5/uhura-arc-easy @@ -0,0 +1,38 @@ +tag: + - uhura_arc_easy_tasks + - uhura_arc_easy_prompt_5 +dataset_path: masakhane/uhura-arc-easy +dataset_name: null +output_type: multiple_choice +doc_to_target: "{{['A', 'B', 'C', 'D'].index(answerKey)}}" +doc_to_choice: + - A + - B + - C + - D +test_split: test +fewshot_split: validation +should_decontaminate: false +doc_to_decontamination_query: "Question: {{question}}\nAnswer:" +metric_list: + - metric: f1 + aggregation: !function utils.weighted_f1_score + # aggregation: mean + average: weighted + hf_evaluate: true + higher_is_better: True + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" + - metric: acc + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_5/uhura-arc-easy_am.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_5/uhura-arc-easy_am.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f7f0231017eb5553893708f939b5fb23d0f60e1e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_5/uhura-arc-easy_am.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: am_multiple_choice +doc_to_text: "Which of the following options answers this question: {{question}}\n\ + \n{% for i in range(choices['text']|length) %}\t{{ 'ABCD'[i] }}: {{ choices['text'][i]\ + \ }}\n{% endfor %}\nAnswer: " +include: uhura-arc-easy +task: uhura-arc-easy_am_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_5/uhura-arc-easy_en.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_5/uhura-arc-easy_en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5aea6abac239580735b805c2582be3976a9986d4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_5/uhura-arc-easy_en.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: en_multiple_choice +doc_to_text: "Which of the following options answers this question: {{question}}\n\ + \n{% for i in range(choices['text']|length) %}\t{{ 'ABCD'[i] }}: {{ choices['text'][i]\ + \ }}\n{% endfor %}\nAnswer: " +include: uhura-arc-easy +task: uhura-arc-easy_en_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_5/uhura-arc-easy_ha.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_5/uhura-arc-easy_ha.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6293bda284e9ca224c3be7a65e1686f3c97210d8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_5/uhura-arc-easy_ha.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: ha_multiple_choice +doc_to_text: "Which of the following options answers this question: {{question}}\n\ + \n{% for i in range(choices['text']|length) %}\t{{ 'ABCD'[i] }}: {{ choices['text'][i]\ + \ }}\n{% endfor %}\nAnswer: " +include: uhura-arc-easy +task: uhura-arc-easy_ha_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_5/uhura-arc-easy_nso.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_5/uhura-arc-easy_nso.yaml new file mode 100644 index 0000000000000000000000000000000000000000..80aff7064e48444477970baf5ccedf930a560a34 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_5/uhura-arc-easy_nso.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: nso_multiple_choice +doc_to_text: "Which of the following options answers this question: {{question}}\n\ + \n{% for i in range(choices['text']|length) %}\t{{ 'ABCD'[i] }}: {{ choices['text'][i]\ + \ }}\n{% endfor %}\nAnswer: " +fewshot_split: train +include: uhura-arc-easy +task: uhura-arc-easy_nso_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_5/uhura-arc-easy_sw.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_5/uhura-arc-easy_sw.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a5bc7d2e5600b46a8660c83e76c69b9a24d0f398 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_5/uhura-arc-easy_sw.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: sw_multiple_choice +doc_to_text: "Which of the following options answers this question: {{question}}\n\ + \n{% for i in range(choices['text']|length) %}\t{{ 'ABCD'[i] }}: {{ choices['text'][i]\ + \ }}\n{% endfor %}\nAnswer: " +include: uhura-arc-easy +task: uhura-arc-easy_sw_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_5/uhura-arc-easy_yo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_5/uhura-arc-easy_yo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a267e987218945fc09217266defb4b0775fd777f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_5/uhura-arc-easy_yo.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: yo_multiple_choice +doc_to_text: "Which of the following options answers this question: {{question}}\n\ + \n{% for i in range(choices['text']|length) %}\t{{ 'ABCD'[i] }}: {{ choices['text'][i]\ + \ }}\n{% endfor %}\nAnswer: " +include: uhura-arc-easy +task: uhura-arc-easy_yo_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_5/uhura-arc-easy_zu.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_5/uhura-arc-easy_zu.yaml new file mode 100644 index 0000000000000000000000000000000000000000..69ce4a396af9f0bd6c96071319ef51ac3c1a81cd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_5/uhura-arc-easy_zu.yaml @@ -0,0 +1,8 @@ +# Generated by utils.py +dataset_name: zu_multiple_choice +doc_to_text: "Which of the following options answers this question: {{question}}\n\ + \n{% for i in range(choices['text']|length) %}\t{{ 'ABCD'[i] }}: {{ choices['text'][i]\ + \ }}\n{% endfor %}\nAnswer: " +fewshot_split: train +include: uhura-arc-easy +task: uhura-arc-easy_zu_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_5/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_5/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..3e735e2deb1f9c53152c072615aebe8ba3acb90b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/prompt_5/utils.py @@ -0,0 +1 @@ +from lm_eval.utils import weighted_f1_score diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/uhura.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/uhura.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e2e2fea5fb49838103490f6f16321da45022cc7c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/uhura.yaml @@ -0,0 +1,13 @@ +group: uhura_arc_easy +task: + - uhura_arc_easy_prompt_1 + - uhura_arc_easy_prompt_2 + - uhura_arc_easy_prompt_3 + - uhura_arc_easy_prompt_4 + - uhura_arc_easy_prompt_5 +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..1216618cbff12e6f4a21ff532d7da16abbef1bde --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/uhura-arc-easy/utils.py @@ -0,0 +1,129 @@ +import argparse +import os + +import pycountry +import yaml + + +def get_language_from_code(code: str) -> str: + language_tuple = pycountry.languages.get(**{f"alpha_{len(code)}": code}) + return language_tuple.name + + +def prompt_func(mode): + prompt_map = { + "prompt_1": "You are a virtual assistant that answers multiple-choice questions with the correct option only.\n\n" + "Question: {{question}}\n\n" + "Choices:\n\n" + "{% for i in range(choices['text']|length) %}" + "\t{{ 'ABCD'[i] }}: {{ choices['text'][i] }}\n" + "{% endfor %}\n" + "Answer: ", + "prompt_2": "Choose the correct option that answers the question below:\n\n" + "Question: {{question}}\n\n" + "Choices:\n\n" + "{% for i in range(choices['text']|length) %}" + "\t{{ 'ABCD'[i] }}: {{ choices['text'][i] }}\n" + "{% endfor %}\n" + "Answer: ", + "prompt_3": "Answer the following multiple-choice question by picking 'A', 'B', 'C', or 'D'.\n\n" + "Question: {{question}}\n\n" + "Options:\n\n" + "{% for i in range(choices['text']|length) %}" + "\t{{ 'ABCD'[i] }}: {{ choices['text'][i] }}\n" + "{% endfor %}\n" + "Answer: ", + "prompt_4": "Question: {{question}}\n\n" + "Options:\n\n" + "{% for i in range(choices['text']|length) %}" + "\t{{ 'ABCD'[i] }}: {{ choices['text'][i] }}\n" + "{% endfor %}\n" + "Answer: ", + "prompt_5": "Which of the following options answers this question: {{question}}\n\n" + "{% for i in range(choices['text']|length) %}" + "\t{{ 'ABCD'[i] }}: {{ choices['text'][i] }}\n" + "{% endfor %}\n" + "Answer: ", + } + return prompt_map[mode] + + +def gen_lang_yamls(output_dir: str, overwrite: bool, mode: str) -> None: + """ + Generate a yaml file for each language. + + :param output_dir: The directory to output the files to. + :param overwrite: Whether to overwrite files if they already exist. + """ + err = [] + languages = {"am", "en", "ha", "nso", "sw", "yo", "zu"} + + for lang in languages: + try: + file_name = f"uhura-arc-easy_{lang}.yaml" + task_name = f"uhura-arc-easy_{lang}_{mode}" + yaml_template = "uhura-arc-easy" + yaml_details = { + "include": yaml_template, + "task": task_name, + "dataset_name": f"{lang}_multiple_choice{'_unmatched' if lang == 'nso' else ''}", + "doc_to_text": prompt_func(mode), + } + if lang in ("nso", "zu"): + yaml_details["fewshot_split"] = "train" + + file_path = os.path.join(output_dir, mode) + os.makedirs(file_path, exist_ok=True) + + with open( + f"{output_dir}/{mode}/{file_name}", + "w" if overwrite else "x", + encoding="utf8", + ) as f: + f.write("# Generated by utils.py\n") + yaml.dump( + yaml_details, + f, + allow_unicode=True, + ) + except FileExistsError: + err.append(file_name) + + if len(err) > 0: + raise FileExistsError( + "Files were not created because they already exist (use --overwrite flag):" + f" {', '.join(err)}" + ) + + +def main() -> None: + """Parse CLI args and generate language-specific yaml files.""" + parser = argparse.ArgumentParser() + parser.add_argument( + "--overwrite", + default=True, + action="store_true", + help="Overwrite files if they already exist", + ) + parser.add_argument( + "--output-dir", + default="./", + help="Directory to write yaml files to", + ) + + PROMPT_CHOICES = ["prompt_1", "prompt_2", "prompt_3", "prompt_4", "prompt_5"] + parser.add_argument( + "--mode", + nargs="*", + default=PROMPT_CHOICES, + choices=PROMPT_CHOICES, + help="Prompt number(s)", + ) + args = parser.parse_args() + + for mode in args.mode: + gen_lang_yamls(output_dir=args.output_dir, overwrite=args.overwrite, mode=mode) + + +if __name__ == "__main__": + main() diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/README.md b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/README.md new file mode 100644 index 0000000000000000000000000000000000000000..d9a47076e564de38fb4a7eb2cbd1df8a3b0290d2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/README.md @@ -0,0 +1,34 @@ +# + +## Paper +Title: `XL-Sum: Large-Scale Multilingual Abstractive Summarization for 44 Languages` + +Paper Link: https://aclanthology.org/2021.findings-acl.413/ + +## Abstract +>Contemporary works on abstractive text summarization have focused primarily on high-resource languages like English, mostly due to the limited availability of datasets for low/mid-resource ones. In this work, we present XL-Sum, a comprehensive and diverse dataset comprising 1 million professionally annotated article-summary pairs from BBC, extracted using a set of carefully designed heuristics. The dataset covers 44 languages ranging from low to high-resource, for many of which no public dataset is currently available. XL-Sum is highly abstractive, concise, and of high quality, as indicated by human and intrinsic evaluation. We fine-tune mT5, a state-of-the-art pretrained multilingual model, with XL-Sum and experiment on multilingual and low-resource summarization tasks. XL-Sum induces competitive results compared to the ones obtained using similar monolingual datasets: we show higher than 11 ROUGE-2 scores on 10 languages we benchmark on, with some of them exceeding 15, as obtained by multilingual training. Additionally, training on low-resource languages individually also provides competitive performance. To the best of our knowledge, XL-Sum is the largest abstractive summarization dataset in terms of the number of samples collected from a single source and the number of languages covered. We are releasing our dataset and models to encourage future research on multilingual abstractive summarization. + +HomePage: https://github.com/csebuetnlp/xl-sum + +### Citation + +``` +@inproceedings{hasan-etal-2021-xl, + title = "{XL}-Sum: Large-Scale Multilingual Abstractive Summarization for 44 Languages", + author = "Hasan, Tahmid and + Bhattacharjee, Abhik and + Islam, Md. Saiful and + Mubasshir, Kazi and + Li, Yuan-Fang and + Kang, Yong-Bin and + Rahman, M. Sohel and + Shahriyar, Rifat", + booktitle = "Findings of the Association for Computational Linguistics: ACL-IJCNLP 2021", + month = aug, + year = "2021", + address = "Online", + publisher = "Association for Computational Linguistics", + url = "https://aclanthology.org/2021.findings-acl.413", + pages = "4693--4703", +} +``` diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_1/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_1/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..85db4d4f4cef061e526c970ece194317e576de06 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_1/utils.py @@ -0,0 +1,18 @@ +import evaluate + + +def rougeL(items): + """ + # passthrough for efficiency + """ + return items + + +def rougeL_agg(items): + """ + Higher is better + """ + refs = list(zip(*items))[0] + preds = list(zip(*items))[1] + rouge_scorer = evaluate.load("rouge") + return rouge_scorer.compute(predictions=preds, references=refs)["rougeL"] diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_1/xlsum b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_1/xlsum new file mode 100644 index 0000000000000000000000000000000000000000..f6b0421edd9ba2b3f1c2eac1dbfaf6f51e5cfba5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_1/xlsum @@ -0,0 +1,22 @@ +tag: + - xlsum_tasks + - xlsum_prompt_1 +task: null +dataset_path: csebuetnlp/xlsum +dataset_name: null +dataset_kwargs: + trust_remote_code: true +output_type: generate_until +generation_kwargs: + until: + - "" +validation_split: validation +fewshot_split: validation +test_split: test +should_decontaminate: false +metric_list: + - metric: !function utils.rougeL + higher_is_better: true + aggregation: !function utils.rougeL_agg +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_1/xlsum_amharic.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_1/xlsum_amharic.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8ab68805aa658c3c15d8367f48115f40e2581aac --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_1/xlsum_amharic.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: amharic +doc_to_target: '{{summary}}' +doc_to_text: 'Provide a summary of the document written in Amharic. Ensure that you + provide the summary in Amharic and nothing else. + + Document in Amharic: {{''text''}}\nSummary: ' +include: xlsum +task: xlsum_amharic_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_1/xlsum_arabic.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_1/xlsum_arabic.yaml new file mode 100644 index 0000000000000000000000000000000000000000..af7df7d90f01b274c1d54076256d7e3a510627b4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_1/xlsum_arabic.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: arabic +doc_to_target: '{{summary}}' +doc_to_text: 'Provide a summary of the document written in Arabic. Ensure that you + provide the summary in Arabic and nothing else. + + Document in Arabic: {{''text''}}\nSummary: ' +include: xlsum +task: xlsum_arabic_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_1/xlsum_hausa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_1/xlsum_hausa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..37f6b3e518835365e7b59fb550c15e286c85f63a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_1/xlsum_hausa.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: hausa +doc_to_target: '{{summary}}' +doc_to_text: 'Provide a summary of the document written in Hausa. Ensure that you + provide the summary in Hausa and nothing else. + + Document in Hausa: {{''text''}}\nSummary: ' +include: xlsum +task: xlsum_hausa_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_1/xlsum_igbo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_1/xlsum_igbo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..04644b5d7bdd8595c5beb02240fe521162dcf3fd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_1/xlsum_igbo.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: igbo +doc_to_target: '{{summary}}' +doc_to_text: 'Provide a summary of the document written in Igbo. Ensure that you provide + the summary in Igbo and nothing else. + + Document in Igbo: {{''text''}}\nSummary: ' +include: xlsum +task: xlsum_igbo_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_1/xlsum_kirundi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_1/xlsum_kirundi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7c434296f5598cb995c40568ab69141f29571d57 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_1/xlsum_kirundi.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: kirundi +doc_to_target: '{{summary}}' +doc_to_text: 'Provide a summary of the document written in Kirundi. Ensure that you + provide the summary in Kirundi and nothing else. + + Document in Kirundi: {{''text''}}\nSummary: ' +include: xlsum +task: xlsum_kirundi_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_1/xlsum_oromo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_1/xlsum_oromo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..78fb14eca4344c17ed3300954193764568be40d4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_1/xlsum_oromo.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: oromo +doc_to_target: '{{summary}}' +doc_to_text: 'Provide a summary of the document written in Oromo. Ensure that you + provide the summary in Oromo and nothing else. + + Document in Oromo: {{''text''}}\nSummary: ' +include: xlsum +task: xlsum_oromo_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_1/xlsum_pidgin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_1/xlsum_pidgin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..68f2c17f560ee888ea1ee958c9ba2392d6f47dfc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_1/xlsum_pidgin.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: pidgin +doc_to_target: '{{summary}}' +doc_to_text: 'Provide a summary of the document written in Nigerian pidgin. Ensure + that you provide the summary in Nigerian pidgin and nothing else. + + Document in Nigerian pidgin: {{''text''}}\nSummary: ' +include: xlsum +task: xlsum_pidgin_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_1/xlsum_somali.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_1/xlsum_somali.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d699dc1905796945e89f3659060202f7314ed776 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_1/xlsum_somali.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: somali +doc_to_target: '{{summary}}' +doc_to_text: 'Provide a summary of the document written in Somali. Ensure that you + provide the summary in Somali and nothing else. + + Document in Somali: {{''text''}}\nSummary: ' +include: xlsum +task: xlsum_somali_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_1/xlsum_swahili.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_1/xlsum_swahili.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6a951c11b8c7ee59f1dfcd3eb44eaada2bd0652a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_1/xlsum_swahili.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: swahili +doc_to_target: '{{summary}}' +doc_to_text: 'Provide a summary of the document written in Swahili. Ensure that you + provide the summary in Swahili and nothing else. + + Document in Swahili: {{''text''}}\nSummary: ' +include: xlsum +task: xlsum_swahili_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_1/xlsum_telugu.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_1/xlsum_telugu.yaml new file mode 100644 index 0000000000000000000000000000000000000000..82a60171a5e2a42e1eb5d43aaf8a034e77b4a798 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_1/xlsum_telugu.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: telugu +doc_to_target: '{{summary}}' +doc_to_text: 'Provide a summary of the document written in Telugu. Ensure that you + provide the summary in Telugu and nothing else. + + Document in Telugu: {{''text''}}\nSummary: ' +include: xlsum +task: xlsum_telugu_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_1/xlsum_tigrinya.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_1/xlsum_tigrinya.yaml new file mode 100644 index 0000000000000000000000000000000000000000..31630982a134b934311c00164c42dc9fabf22cc7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_1/xlsum_tigrinya.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: tigrinya +doc_to_target: '{{summary}}' +doc_to_text: 'Provide a summary of the document written in Tigrinya. Ensure that you + provide the summary in Tigrinya and nothing else. + + Document in Tigrinya: {{''text''}}\nSummary: ' +include: xlsum +task: xlsum_tigrinya_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_1/xlsum_yoruba.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_1/xlsum_yoruba.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9c14a9113e293c057d028e27cd09ed1f6812c1e8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_1/xlsum_yoruba.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: yoruba +doc_to_target: '{{summary}}' +doc_to_text: 'Provide a summary of the document written in Yoruba. Ensure that you + provide the summary in Yoruba and nothing else. + + Document in Yoruba: {{''text''}}\nSummary: ' +include: xlsum +task: xlsum_yoruba_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_2/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_2/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..85db4d4f4cef061e526c970ece194317e576de06 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_2/utils.py @@ -0,0 +1,18 @@ +import evaluate + + +def rougeL(items): + """ + # passthrough for efficiency + """ + return items + + +def rougeL_agg(items): + """ + Higher is better + """ + refs = list(zip(*items))[0] + preds = list(zip(*items))[1] + rouge_scorer = evaluate.load("rouge") + return rouge_scorer.compute(predictions=preds, references=refs)["rougeL"] diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_2/xlsum b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_2/xlsum new file mode 100644 index 0000000000000000000000000000000000000000..e572c00c6ae1c0f8f84f1030c5903325ca1f0ae4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_2/xlsum @@ -0,0 +1,22 @@ +tag: + - xlsum_tasks + - xlsum_prompt_2 +task: null +dataset_path: csebuetnlp/xlsum +dataset_name: null +dataset_kwargs: + trust_remote_code: true +output_type: generate_until +generation_kwargs: + until: + - "" +validation_split: validation +fewshot_split: validation +test_split: test +should_decontaminate: false +metric_list: + - metric: !function utils.rougeL + higher_is_better: true + aggregation: !function utils.rougeL_agg +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_2/xlsum_amharic.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_2/xlsum_amharic.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0f2275c657b54df708f62526fdc12b0381f197eb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_2/xlsum_amharic.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: amharic +doc_to_target: '{{summary}}' +doc_to_text: 'Summarize the document below in triple backticks and return only the + summary and nothing else. + + ```{{''text''}}```\n' +include: xlsum +task: xlsum_amharic_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_2/xlsum_arabic.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_2/xlsum_arabic.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a4f772c31610175417ee97105ae4d99f526f0c41 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_2/xlsum_arabic.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: arabic +doc_to_target: '{{summary}}' +doc_to_text: 'Summarize the document below in triple backticks and return only the + summary and nothing else. + + ```{{''text''}}```\n' +include: xlsum +task: xlsum_arabic_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_2/xlsum_hausa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_2/xlsum_hausa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7485672cb48c86d1df444391064b04754836fa77 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_2/xlsum_hausa.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: hausa +doc_to_target: '{{summary}}' +doc_to_text: 'Summarize the document below in triple backticks and return only the + summary and nothing else. + + ```{{''text''}}```\n' +include: xlsum +task: xlsum_hausa_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_2/xlsum_igbo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_2/xlsum_igbo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2cf7fafe394e049aaa3382068d6da8cac70cf705 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_2/xlsum_igbo.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: igbo +doc_to_target: '{{summary}}' +doc_to_text: 'Summarize the document below in triple backticks and return only the + summary and nothing else. + + ```{{''text''}}```\n' +include: xlsum +task: xlsum_igbo_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_2/xlsum_kirundi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_2/xlsum_kirundi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..63021d7d4f9ff0d252c82665fd8262bd6bb5c327 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_2/xlsum_kirundi.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: kirundi +doc_to_target: '{{summary}}' +doc_to_text: 'Summarize the document below in triple backticks and return only the + summary and nothing else. + + ```{{''text''}}```\n' +include: xlsum +task: xlsum_kirundi_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_2/xlsum_oromo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_2/xlsum_oromo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b637b10d0428d614fcd4c06bdb1fb2383057ef77 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_2/xlsum_oromo.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: oromo +doc_to_target: '{{summary}}' +doc_to_text: 'Summarize the document below in triple backticks and return only the + summary and nothing else. + + ```{{''text''}}```\n' +include: xlsum +task: xlsum_oromo_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_2/xlsum_pidgin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_2/xlsum_pidgin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6c13d93d5c1ff95ac76c1b87f4c301c97a771f52 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_2/xlsum_pidgin.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: pidgin +doc_to_target: '{{summary}}' +doc_to_text: 'Summarize the document below in triple backticks and return only the + summary and nothing else. + + ```{{''text''}}```\n' +include: xlsum +task: xlsum_pidgin_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_2/xlsum_somali.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_2/xlsum_somali.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b7245ddc193a133701fd8f71cd6b52cd34899594 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_2/xlsum_somali.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: somali +doc_to_target: '{{summary}}' +doc_to_text: 'Summarize the document below in triple backticks and return only the + summary and nothing else. + + ```{{''text''}}```\n' +include: xlsum +task: xlsum_somali_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_2/xlsum_swahili.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_2/xlsum_swahili.yaml new file mode 100644 index 0000000000000000000000000000000000000000..65f176fba40e37003a9cfd8813957760cdac2aa1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_2/xlsum_swahili.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: swahili +doc_to_target: '{{summary}}' +doc_to_text: 'Summarize the document below in triple backticks and return only the + summary and nothing else. + + ```{{''text''}}```\n' +include: xlsum +task: xlsum_swahili_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_2/xlsum_telugu.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_2/xlsum_telugu.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0ecbdde5c90806b2684fce1373c3fad94ef5c65e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_2/xlsum_telugu.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: telugu +doc_to_target: '{{summary}}' +doc_to_text: 'Summarize the document below in triple backticks and return only the + summary and nothing else. + + ```{{''text''}}```\n' +include: xlsum +task: xlsum_telugu_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_2/xlsum_tigrinya.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_2/xlsum_tigrinya.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d46e2fb573f5cdd3bb9459c7eb5b95150cae5ec8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_2/xlsum_tigrinya.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: tigrinya +doc_to_target: '{{summary}}' +doc_to_text: 'Summarize the document below in triple backticks and return only the + summary and nothing else. + + ```{{''text''}}```\n' +include: xlsum +task: xlsum_tigrinya_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_2/xlsum_yoruba.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_2/xlsum_yoruba.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7ea0ef503444a8237ce5fd693f6ebec082a8a6cd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_2/xlsum_yoruba.yaml @@ -0,0 +1,9 @@ +# Generated by utils.py +dataset_name: yoruba +doc_to_target: '{{summary}}' +doc_to_text: 'Summarize the document below in triple backticks and return only the + summary and nothing else. + + ```{{''text''}}```\n' +include: xlsum +task: xlsum_yoruba_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_3/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_3/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..85db4d4f4cef061e526c970ece194317e576de06 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_3/utils.py @@ -0,0 +1,18 @@ +import evaluate + + +def rougeL(items): + """ + # passthrough for efficiency + """ + return items + + +def rougeL_agg(items): + """ + Higher is better + """ + refs = list(zip(*items))[0] + preds = list(zip(*items))[1] + rouge_scorer = evaluate.load("rouge") + return rouge_scorer.compute(predictions=preds, references=refs)["rougeL"] diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_3/xlsum b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_3/xlsum new file mode 100644 index 0000000000000000000000000000000000000000..08842ef8eb627dfb12387ae7ef2e232d2f3c40d3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_3/xlsum @@ -0,0 +1,22 @@ +tag: + - xlsum_tasks + - xlsum_prompt_3 +task: null +dataset_path: csebuetnlp/xlsum +dataset_name: null +dataset_kwargs: + trust_remote_code: true +output_type: generate_until +generation_kwargs: + until: + - "" +validation_split: validation +fewshot_split: validation +test_split: test +should_decontaminate: false +metric_list: + - metric: !function utils.rougeL + higher_is_better: true + aggregation: !function utils.rougeL_agg +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_3/xlsum_amharic.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_3/xlsum_amharic.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6fc85e7ceb43f474754081088a77b3979b785334 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_3/xlsum_amharic.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: amharic +doc_to_target: '{{summary}}' +doc_to_text: 'You are an advanced Summarizer, a specialized assistant designed to + summarize documents in Amharic. Your main goal is to ensure summaries are concise + and informative. Ensure you return the summary only and nothing else. + + Document: {{''text''}}\nSummary: ' +include: xlsum +task: xlsum_amharic_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_3/xlsum_arabic.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_3/xlsum_arabic.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d4f2b1f5c09bcf2dc76fb4807e09a1cc52b6ce82 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_3/xlsum_arabic.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: arabic +doc_to_target: '{{summary}}' +doc_to_text: 'You are an advanced Summarizer, a specialized assistant designed to + summarize documents in Arabic. Your main goal is to ensure summaries are concise + and informative. Ensure you return the summary only and nothing else. + + Document: {{''text''}}\nSummary: ' +include: xlsum +task: xlsum_arabic_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_3/xlsum_hausa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_3/xlsum_hausa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e1a0603749b196a9cda3f995fb16ea2814513142 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_3/xlsum_hausa.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: hausa +doc_to_target: '{{summary}}' +doc_to_text: 'You are an advanced Summarizer, a specialized assistant designed to + summarize documents in Hausa. Your main goal is to ensure summaries are concise + and informative. Ensure you return the summary only and nothing else. + + Document: {{''text''}}\nSummary: ' +include: xlsum +task: xlsum_hausa_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_3/xlsum_igbo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_3/xlsum_igbo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6b23f8f395679740acd523757311a7803407c3cd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_3/xlsum_igbo.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: igbo +doc_to_target: '{{summary}}' +doc_to_text: 'You are an advanced Summarizer, a specialized assistant designed to + summarize documents in Igbo. Your main goal is to ensure summaries are concise and + informative. Ensure you return the summary only and nothing else. + + Document: {{''text''}}\nSummary: ' +include: xlsum +task: xlsum_igbo_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_3/xlsum_kirundi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_3/xlsum_kirundi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8f40b2a7ff68847a5bae0649451460d22f24ae2a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_3/xlsum_kirundi.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: kirundi +doc_to_target: '{{summary}}' +doc_to_text: 'You are an advanced Summarizer, a specialized assistant designed to + summarize documents in Kirundi. Your main goal is to ensure summaries are concise + and informative. Ensure you return the summary only and nothing else. + + Document: {{''text''}}\nSummary: ' +include: xlsum +task: xlsum_kirundi_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_3/xlsum_oromo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_3/xlsum_oromo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bbc912851b05e78829a2835e14e28831469d302d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_3/xlsum_oromo.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: oromo +doc_to_target: '{{summary}}' +doc_to_text: 'You are an advanced Summarizer, a specialized assistant designed to + summarize documents in Oromo. Your main goal is to ensure summaries are concise + and informative. Ensure you return the summary only and nothing else. + + Document: {{''text''}}\nSummary: ' +include: xlsum +task: xlsum_oromo_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_3/xlsum_pidgin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_3/xlsum_pidgin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8149e441e9869a94312fe54001cd93fd6d720eaa --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_3/xlsum_pidgin.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: pidgin +doc_to_target: '{{summary}}' +doc_to_text: 'You are an advanced Summarizer, a specialized assistant designed to + summarize documents in Nigerian pidgin. Your main goal is to ensure summaries are + concise and informative. Ensure you return the summary only and nothing else. + + Document: {{''text''}}\nSummary: ' +include: xlsum +task: xlsum_pidgin_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_3/xlsum_somali.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_3/xlsum_somali.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a2936da11bdca6fb8fd9ba6f288968ee0c1843a4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_3/xlsum_somali.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: somali +doc_to_target: '{{summary}}' +doc_to_text: 'You are an advanced Summarizer, a specialized assistant designed to + summarize documents in Somali. Your main goal is to ensure summaries are concise + and informative. Ensure you return the summary only and nothing else. + + Document: {{''text''}}\nSummary: ' +include: xlsum +task: xlsum_somali_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_3/xlsum_swahili.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_3/xlsum_swahili.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6f90f4cfaa99d4291d36e9e9aec715e32925cb55 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_3/xlsum_swahili.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: swahili +doc_to_target: '{{summary}}' +doc_to_text: 'You are an advanced Summarizer, a specialized assistant designed to + summarize documents in Swahili. Your main goal is to ensure summaries are concise + and informative. Ensure you return the summary only and nothing else. + + Document: {{''text''}}\nSummary: ' +include: xlsum +task: xlsum_swahili_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_3/xlsum_telugu.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_3/xlsum_telugu.yaml new file mode 100644 index 0000000000000000000000000000000000000000..67d116dce2105d9121c5f13e268333005d9a91dc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_3/xlsum_telugu.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: telugu +doc_to_target: '{{summary}}' +doc_to_text: 'You are an advanced Summarizer, a specialized assistant designed to + summarize documents in Telugu. Your main goal is to ensure summaries are concise + and informative. Ensure you return the summary only and nothing else. + + Document: {{''text''}}\nSummary: ' +include: xlsum +task: xlsum_telugu_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_3/xlsum_tigrinya.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_3/xlsum_tigrinya.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5b20d6e3bfdb66590f7004ff0f187e2a6537db84 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_3/xlsum_tigrinya.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: tigrinya +doc_to_target: '{{summary}}' +doc_to_text: 'You are an advanced Summarizer, a specialized assistant designed to + summarize documents in Tigrinya. Your main goal is to ensure summaries are concise + and informative. Ensure you return the summary only and nothing else. + + Document: {{''text''}}\nSummary: ' +include: xlsum +task: xlsum_tigrinya_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_3/xlsum_yoruba.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_3/xlsum_yoruba.yaml new file mode 100644 index 0000000000000000000000000000000000000000..353be14cda6713964f73436b2085d6ae63fcdc57 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/prompt_3/xlsum_yoruba.yaml @@ -0,0 +1,10 @@ +# Generated by utils.py +dataset_name: yoruba +doc_to_target: '{{summary}}' +doc_to_text: 'You are an advanced Summarizer, a specialized assistant designed to + summarize documents in Yoruba. Your main goal is to ensure summaries are concise + and informative. Ensure you return the summary only and nothing else. + + Document: {{''text''}}\nSummary: ' +include: xlsum +task: xlsum_yoruba_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..8df1e12e8b4aa683bd71c2fb23d90ff7667de5b2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/utils.py @@ -0,0 +1,118 @@ +import argparse +import os + +import yaml + + +def prompt_func(mode, lang): + if lang == "pidgin": + lang = "Nigerian Pidgin" + + prompt_map = { + "prompt_1": f"Provide a summary of the document written in {lang.capitalize()}. Ensure that you provide the summary in {lang.capitalize()} and nothing else.\n" + f"Document in {lang.capitalize()}: " + r"{{'text'}}\n" + "Summary: ", + "prompt_2": "Summarize the document below in triple backticks and return only the summary and nothing else.\n" + + r"```{{'text'}}```\n", + "prompt_3": f"You are an advanced Summarizer, a specialized assistant designed to summarize documents in {lang.capitalize()}. " + f"Your main goal is to ensure summaries are concise and informative. Ensure you return the summary only and nothing else.\n" + f"Document: " + r"{{'text'}}\n" + "Summary: ", + "prompt_4": f"Summarize this {lang.capitalize()} document:\n" + r"{{'text'}}\n" + "Summary: ", + "prompt_5": f"{lang.capitalize()} document: " + r"{{'text'}}\n" + "Summary: ", + } + return prompt_map[mode] + + +def gen_lang_yamls(output_dir: str, overwrite: bool, mode: str) -> None: + """ + Generate a yaml file for each language. + + :param output_dir: The directory to output the files to. + :param overwrite: Whether to overwrite files if they already exist. + """ + err = [] + XLSUM_LANGUAGES = ( + "amharic", + "arabic", + "hausa", + "igbo", + "kirundi", + "oromo", + "pidgin", + "somali", + "swahili", + "telugu", + "tigrinya", + "yoruba", + ) + + for lang in XLSUM_LANGUAGES: + try: + file_name = f"xlsum_{lang}.yaml" + task_name = f"xlsum_{lang}_{mode}" + yaml_template = "xlsum" + yaml_details = { + "include": yaml_template, + "task": task_name, + "dataset_name": lang, + "doc_to_text": prompt_func(mode, lang), + "doc_to_target": "{{summary}}", + } + file_path = os.path.join(output_dir, mode) + os.makedirs(file_path, exist_ok=True) + + with open( + f"{output_dir}/{mode}/{file_name}", + "w" if overwrite else "x", + encoding="utf8", + ) as f: + f.write("# Generated by utils.py\n") + yaml.dump( + yaml_details, + f, + allow_unicode=True, + ) + except FileExistsError: + err.append(file_name) + + if len(err) > 0: + raise FileExistsError( + "Files were not created because they already exist (use --overwrite flag):" + f" {', '.join(err)}" + ) + + +def main() -> None: + """Parse CLI args and generate language-specific yaml files.""" + parser = argparse.ArgumentParser() + parser.add_argument( + "--overwrite", + default=True, + action="store_true", + help="Overwrite files if they already exist", + ) + parser.add_argument( + "--output-dir", + default="./", + help="Directory to write yaml files to", + ) + + PROMPT_CHOICES = ["prompt_1", "prompt_2", "prompt_3", "prompt_4", "prompt_5"] + parser.add_argument( + "--mode", + nargs="*", + default=PROMPT_CHOICES, + choices=PROMPT_CHOICES, + help="Prompt number(s)", + ) + args = parser.parse_args() + + for mode in args.mode: + gen_lang_yamls(output_dir=args.output_dir, overwrite=args.overwrite, mode=mode) + + +if __name__ == "__main__": + main() diff --git a/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/xlsum.yaml b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/xlsum.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8d87717597c59eb333f712d69eb854e971146915 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrobench/xlsum/xlsum.yaml @@ -0,0 +1,11 @@ +group: xlum +task: + - xlsum_prompt_1 + - xlsum_prompt_2 + - xlsum_prompt_3 +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true +metadata: + version: 2 diff --git a/lm-evaluation-harness/lm_eval/tasks/agieval/README.md b/lm-evaluation-harness/lm_eval/tasks/agieval/README.md new file mode 100644 index 0000000000000000000000000000000000000000..53a9df036d6c0a4dcc2b310ac324f1bf7b0f60dc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/agieval/README.md @@ -0,0 +1,118 @@ +# AGIEval + +### Paper + +Title: AGIEval: A Human-Centric Benchmark for Evaluating Foundation Models + +Abstract: https://arxiv.org/abs/2304.06364.pdf + +AGIEval is a human-centric benchmark specifically designed to evaluate the general abilities of foundation models in tasks pertinent to human cognition and problem-solving. +This benchmark is derived from 20 official, public, and high-standard admission and qualification exams intended for general human test-takers, such as general college admission tests (e.g., Chinese College Entrance Exam (Gaokao) and American SAT), law school admission tests, math competitions, lawyer qualification tests, and national civil service exams. + +Homepage: https://github.com/ruixiangcui/AGIEval + +### Citation + +``` +@misc{zhong2023agieval, + title={AGIEval: A Human-Centric Benchmark for Evaluating Foundation Models}, + author={Wanjun Zhong and Ruixiang Cui and Yiduo Guo and Yaobo Liang and Shuai Lu and Yanlin Wang and Amin Saied and Weizhu Chen and Nan Duan}, + year={2023}, + eprint={2304.06364}, + archivePrefix={arXiv}, + primaryClass={cs.CL} +} +``` + +Please make sure to cite all the individual datasets in your paper when you use them. We provide the relevant citation information below: + +``` +@inproceedings{ling-etal-2017-program, + title = "Program Induction by Rationale Generation: Learning to Solve and Explain Algebraic Word Problems", + author = "Ling, Wang and + Yogatama, Dani and + Dyer, Chris and + Blunsom, Phil", + booktitle = "Proceedings of the 55th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)", + month = jul, + year = "2017", + address = "Vancouver, Canada", + publisher = "Association for Computational Linguistics", + url = "https://aclanthology.org/P17-1015", + doi = "10.18653/v1/P17-1015", + pages = "158--167", + abstract = "Solving algebraic word problems requires executing a series of arithmetic operations{---}a program{---}to obtain a final answer. However, since programs can be arbitrarily complicated, inducing them directly from question-answer pairs is a formidable challenge. To make this task more feasible, we solve these problems by generating answer rationales, sequences of natural language and human-readable mathematical expressions that derive the final answer through a series of small steps. Although rationales do not explicitly specify programs, they provide a scaffolding for their structure via intermediate milestones. To evaluate our approach, we have created a new 100,000-sample dataset of questions, answers and rationales. Experimental results show that indirect supervision of program learning via answer rationales is a promising strategy for inducing arithmetic programs.", +} + +@inproceedings{hendrycksmath2021, + title={Measuring Mathematical Problem Solving With the MATH Dataset}, + author={Dan Hendrycks and Collin Burns and Saurav Kadavath and Akul Arora and Steven Basart and Eric Tang and Dawn Song and Jacob Steinhardt}, + journal={NeurIPS}, + year={2021} +} + +@inproceedings{Liu2020LogiQAAC, + title={LogiQA: A Challenge Dataset for Machine Reading Comprehension with Logical Reasoning}, + author={Jian Liu and Leyang Cui and Hanmeng Liu and Dandan Huang and Yile Wang and Yue Zhang}, + booktitle={International Joint Conference on Artificial Intelligence}, + year={2020} +} + +@inproceedings{zhong2019jec, + title={JEC-QA: A Legal-Domain Question Answering Dataset}, + author={Zhong, Haoxi and Xiao, Chaojun and Tu, Cunchao and Zhang, Tianyang and Liu, Zhiyuan and Sun, Maosong}, + booktitle={Proceedings of AAAI}, + year={2020}, +} + +@article{Wang2021FromLT, + title={From LSAT: The Progress and Challenges of Complex Reasoning}, + author={Siyuan Wang and Zhongkun Liu and Wanjun Zhong and Ming Zhou and Zhongyu Wei and Zhumin Chen and Nan Duan}, + journal={IEEE/ACM Transactions on Audio, Speech, and Language Processing}, + year={2021}, + volume={30}, + pages={2201-2216} +} +``` + +### Groups, Tags, and Tasks + +#### Groups + +- `agieval`: Evaluates all tasks listed below. + +- `agieval_en`: Evaluates all English subtasks: `agieval_aqua_rat`, `agieval_gaokao_english`, `agieval_logiqa_en`, `agieval_lsat_*`, `agieval_sat_*`, `agieval_math` + +- `agieval_cn`: Evaluates all Chinese subtasks: +`agieval_gaokao_biology`, `agieval_gaokao_chemistry`, `agieval_gaokao_chinese`, `agieval_gaokao_geography`, +`agieval_gaokao_history`, `agieval_gaokao_mathqa`, `agieval_gaokao_mathcloze`, `agieval_gaokao_physics`, `agieval_jec_qa_ca`, `agieval_jec_qa_kd`, `agieval_logiqa_zh` + +- `agieval_nous`: Evaluates a specific subset of AGIEval tasks (multiple-choice and english-only), namely those in https://github.com/teknium1/LLM-Benchmark-Logs/blob/main/benchmark-logs/Mistral-7B-Base.md + +#### Tags + +None. + +#### Tasks + +- `agieval_aqua_rat` +- `agieval_gaokao_biology` +- `agieval_gaokao_chemistry` +- `agieval_gaokao_chinese` +- `agieval_gaokao_english` +- `agieval_gaokao_geography` +- `agieval_gaokao_history` +- `agieval_gaokao_mathqa` +- `agieval_gaokao_mathcloze` +- `agieval_gaokao_physics` +- `agieval_jec_qa_ca` +- `agieval_jec_qa_kd` +- `agieval_logiqa_en` +- `agieval_logiqa_zh` +- `agieval_lsat_ar` +- `agieval_lsat_lr` +- `agieval_lsat_rc` +- `agieval_sat_en` +- `agieval_sat_en_without_passage` +- `agieval_sat_math` +- `agieval_math` diff --git a/lm-evaluation-harness/lm_eval/tasks/agieval/agieval.yaml b/lm-evaluation-harness/lm_eval/tasks/agieval/agieval.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d086af83579ec2daf826a55f7dd82cf2e1f82a96 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/agieval/agieval.yaml @@ -0,0 +1,29 @@ +group: agieval +task: + - agieval_gaokao_biology + - agieval_gaokao_chemistry + - agieval_gaokao_chinese + - agieval_gaokao_geography + - agieval_gaokao_history + - agieval_gaokao_mathcloze + - agieval_gaokao_mathqa + - agieval_gaokao_physics + - agieval_jec_qa_ca + - agieval_jec_qa_kd + - agieval_logiqa_zh + - agieval_aqua_rat + - agieval_gaokao_english + - agieval_logiqa_en + - agieval_lsat_ar + - agieval_lsat_lr + - agieval_lsat_rc + - agieval_math + - agieval_sat_en_without_passage + - agieval_sat_en + - agieval_sat_math +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/agieval/agieval_cn.yaml b/lm-evaluation-harness/lm_eval/tasks/agieval/agieval_cn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0e8ca2fdedaabe768fe731bbf2dbbea3ef117448 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/agieval/agieval_cn.yaml @@ -0,0 +1,19 @@ +group: agieval_cn +task: + - agieval_gaokao_biology + - agieval_gaokao_chemistry + - agieval_gaokao_chinese + - agieval_gaokao_geography + - agieval_gaokao_history + - agieval_gaokao_mathcloze + - agieval_gaokao_mathqa + - agieval_gaokao_physics + - agieval_jec_qa_ca + - agieval_jec_qa_kd + - agieval_logiqa_zh +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/agieval/agieval_en.yaml b/lm-evaluation-harness/lm_eval/tasks/agieval/agieval_en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0a873d66d3a4e98fc2ce2df26e53f20a599bc4e8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/agieval/agieval_en.yaml @@ -0,0 +1,18 @@ +group: agieval_en +task: + - agieval_aqua_rat + - agieval_gaokao_english # categorizing as EN because the AGIEval codebase lists this as in `english_qa_tasks` + - agieval_logiqa_en + - agieval_lsat_ar + - agieval_lsat_lr + - agieval_lsat_rc + - agieval_math + - agieval_sat_en_without_passage + - agieval_sat_en + - agieval_sat_math +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/agieval/agieval_nous.yaml b/lm-evaluation-harness/lm_eval/tasks/agieval/agieval_nous.yaml new file mode 100644 index 0000000000000000000000000000000000000000..aa2a874892e77aaa0216feac4f2f6353b3302a93 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/agieval/agieval_nous.yaml @@ -0,0 +1,16 @@ +group: agieval_nous +task: + - agieval_aqua_rat + - agieval_logiqa_en + - agieval_lsat_ar + - agieval_lsat_lr + - agieval_lsat_rc + - agieval_sat_en_without_passage + - agieval_sat_en + - agieval_sat_math +aggregate_metric_list: + - metric: acc_norm + aggregation: mean + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/agieval/aqua-rat.yaml b/lm-evaluation-harness/lm_eval/tasks/agieval/aqua-rat.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e5a3a3e86f6c5448000df38a146a95518691b934 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/agieval/aqua-rat.yaml @@ -0,0 +1,20 @@ +task: agieval_aqua_rat +dataset_path: hails/agieval-aqua-rat +dataset_name: null +output_type: multiple_choice +training_split: null +validation_split: null +test_split: test +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "{{choices}}" +process_results: !function utils.process_results_mcqa +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/agieval/gaokao-biology.yaml b/lm-evaluation-harness/lm_eval/tasks/agieval/gaokao-biology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b8b9eca8397965a0bf3c7152fbd237236b0f37f2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/agieval/gaokao-biology.yaml @@ -0,0 +1,3 @@ +include: aqua-rat.yaml +task: agieval_gaokao_biology +dataset_path: hails/agieval-gaokao-biology diff --git a/lm-evaluation-harness/lm_eval/tasks/agieval/gaokao-chemistry.yaml b/lm-evaluation-harness/lm_eval/tasks/agieval/gaokao-chemistry.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a4ba01a5274376bab68fb8a56bd25a6e81d1edfb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/agieval/gaokao-chemistry.yaml @@ -0,0 +1,3 @@ +include: aqua-rat.yaml +task: agieval_gaokao_chemistry +dataset_path: hails/agieval-gaokao-chemistry diff --git a/lm-evaluation-harness/lm_eval/tasks/agieval/gaokao-chinese.yaml b/lm-evaluation-harness/lm_eval/tasks/agieval/gaokao-chinese.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3d58b5bc495917482ef69f04604b7f78f91339f5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/agieval/gaokao-chinese.yaml @@ -0,0 +1,3 @@ +include: aqua-rat.yaml +task: agieval_gaokao_chinese +dataset_path: hails/agieval-gaokao-chinese diff --git a/lm-evaluation-harness/lm_eval/tasks/agieval/gaokao-english.yaml b/lm-evaluation-harness/lm_eval/tasks/agieval/gaokao-english.yaml new file mode 100644 index 0000000000000000000000000000000000000000..12ea66787acfa60eefc5a49936c6484b08c8fda0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/agieval/gaokao-english.yaml @@ -0,0 +1,3 @@ +include: aqua-rat.yaml +task: agieval_gaokao_english +dataset_path: hails/agieval-gaokao-english diff --git a/lm-evaluation-harness/lm_eval/tasks/agieval/gaokao-geography.yaml b/lm-evaluation-harness/lm_eval/tasks/agieval/gaokao-geography.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6dbce6f4873e272f9c28f49b0061857060df2e97 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/agieval/gaokao-geography.yaml @@ -0,0 +1,3 @@ +include: aqua-rat.yaml +task: agieval_gaokao_geography +dataset_path: hails/agieval-gaokao-geography diff --git a/lm-evaluation-harness/lm_eval/tasks/agieval/gaokao-history.yaml b/lm-evaluation-harness/lm_eval/tasks/agieval/gaokao-history.yaml new file mode 100644 index 0000000000000000000000000000000000000000..55b317281977f215285d7d699f656e54be55bf37 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/agieval/gaokao-history.yaml @@ -0,0 +1,3 @@ +include: aqua-rat.yaml +task: agieval_gaokao_history +dataset_path: hails/agieval-gaokao-history diff --git a/lm-evaluation-harness/lm_eval/tasks/agieval/gaokao-mathcloze.yaml b/lm-evaluation-harness/lm_eval/tasks/agieval/gaokao-mathcloze.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8692e2f45f8cc016cba40f246345fa5909a9256a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/agieval/gaokao-mathcloze.yaml @@ -0,0 +1,22 @@ +task: agieval_gaokao_mathcloze +dataset_path: hails/agieval-gaokao-mathcloze +dataset_name: null +output_type: generate_until +training_split: null +validation_split: null +test_split: test +doc_to_text: "{{query}}" +doc_to_target: "{{answer}}" +process_results: !function utils.process_results +generation_kwargs: + max_gen_toks: 32 + do_sample: False + temperature: 0.0 + until: + - "Q:" +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/agieval/gaokao-mathqa.yaml b/lm-evaluation-harness/lm_eval/tasks/agieval/gaokao-mathqa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e0d97a515559d9eaecca8bc73949a1d6886b1922 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/agieval/gaokao-mathqa.yaml @@ -0,0 +1,3 @@ +include: aqua-rat.yaml +task: agieval_gaokao_mathqa +dataset_path: hails/agieval-gaokao-mathqa diff --git a/lm-evaluation-harness/lm_eval/tasks/agieval/gaokao-physics.yaml b/lm-evaluation-harness/lm_eval/tasks/agieval/gaokao-physics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..43a047edafd06ab29c666741dd4f28560c64eff9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/agieval/gaokao-physics.yaml @@ -0,0 +1,3 @@ +include: aqua-rat.yaml +task: agieval_gaokao_physics +dataset_path: hails/agieval-gaokao-physics diff --git a/lm-evaluation-harness/lm_eval/tasks/agieval/jec-qa-ca.yaml b/lm-evaluation-harness/lm_eval/tasks/agieval/jec-qa-ca.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2d09c2b3814a65fffcae1a06894bd1a3f57ca983 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/agieval/jec-qa-ca.yaml @@ -0,0 +1,3 @@ +include: aqua-rat.yaml +task: agieval_jec_qa_ca +dataset_path: hails/agieval-jec-qa-ca diff --git a/lm-evaluation-harness/lm_eval/tasks/agieval/jec-qa-kd.yaml b/lm-evaluation-harness/lm_eval/tasks/agieval/jec-qa-kd.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5843b6deb9aa1e326426adb96fb5ebeca333e6cf --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/agieval/jec-qa-kd.yaml @@ -0,0 +1,3 @@ +include: aqua-rat.yaml +task: agieval_jec_qa_kd +dataset_path: hails/agieval-jec-qa-kd diff --git a/lm-evaluation-harness/lm_eval/tasks/agieval/logiqa-en.yaml b/lm-evaluation-harness/lm_eval/tasks/agieval/logiqa-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5bd1dff40b0017ee23067cd20bc8543eaf8081b2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/agieval/logiqa-en.yaml @@ -0,0 +1,3 @@ +include: aqua-rat.yaml +task: agieval_logiqa_en +dataset_path: hails/agieval-logiqa-en diff --git a/lm-evaluation-harness/lm_eval/tasks/agieval/logiqa-zh.yaml b/lm-evaluation-harness/lm_eval/tasks/agieval/logiqa-zh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2ca9198b53240e04e69a617274f93964be067539 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/agieval/logiqa-zh.yaml @@ -0,0 +1,3 @@ +include: aqua-rat.yaml +task: agieval_logiqa_zh +dataset_path: hails/agieval-logiqa-zh diff --git a/lm-evaluation-harness/lm_eval/tasks/agieval/lsat-ar.yaml b/lm-evaluation-harness/lm_eval/tasks/agieval/lsat-ar.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2284f601f19988805194f8af96c7789251bcaeae --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/agieval/lsat-ar.yaml @@ -0,0 +1,3 @@ +include: aqua-rat.yaml +task: agieval_lsat_ar +dataset_path: hails/agieval-lsat-ar diff --git a/lm-evaluation-harness/lm_eval/tasks/agieval/lsat-lr.yaml b/lm-evaluation-harness/lm_eval/tasks/agieval/lsat-lr.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8505d4463f72d1942a47e7ae76b07c1958c92ee5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/agieval/lsat-lr.yaml @@ -0,0 +1,3 @@ +include: aqua-rat.yaml +task: agieval_lsat_lr +dataset_path: hails/agieval-lsat-lr diff --git a/lm-evaluation-harness/lm_eval/tasks/agieval/lsat-rc.yaml b/lm-evaluation-harness/lm_eval/tasks/agieval/lsat-rc.yaml new file mode 100644 index 0000000000000000000000000000000000000000..23a9dce7d3853af35091d0bd32df1dbd481ab7aa --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/agieval/lsat-rc.yaml @@ -0,0 +1,3 @@ +include: aqua-rat.yaml +task: agieval_lsat_rc +dataset_path: hails/agieval-lsat-rc diff --git a/lm-evaluation-harness/lm_eval/tasks/agieval/math.yaml b/lm-evaluation-harness/lm_eval/tasks/agieval/math.yaml new file mode 100644 index 0000000000000000000000000000000000000000..534a2e661c38a52ee14cbdddbd6c3f946336cfc7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/agieval/math.yaml @@ -0,0 +1,22 @@ +task: agieval_math +dataset_path: hails/agieval-math +dataset_name: null +output_type: generate_until +training_split: null +validation_split: null +test_split: test +doc_to_text: "{{query}}" +doc_to_target: "{{answer}}" +process_results: !function utils.process_results +generation_kwargs: + max_gen_toks: 32 + do_sample: False + temperature: 0.0 + until: + - "Q:" +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/agieval/sat-en-without-passage.yaml b/lm-evaluation-harness/lm_eval/tasks/agieval/sat-en-without-passage.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d36b467cd76f62f84463cfff2fa423c5f2c87860 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/agieval/sat-en-without-passage.yaml @@ -0,0 +1,3 @@ +include: aqua-rat.yaml +task: agieval_sat_en_without_passage +dataset_path: hails/agieval-sat-en-without-passage diff --git a/lm-evaluation-harness/lm_eval/tasks/agieval/sat-en.yaml b/lm-evaluation-harness/lm_eval/tasks/agieval/sat-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..793d48aec2228daef7431f846e3166de0e12a602 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/agieval/sat-en.yaml @@ -0,0 +1,3 @@ +include: aqua-rat.yaml +task: agieval_sat_en +dataset_path: hails/agieval-sat-en diff --git a/lm-evaluation-harness/lm_eval/tasks/agieval/sat-math.yaml b/lm-evaluation-harness/lm_eval/tasks/agieval/sat-math.yaml new file mode 100644 index 0000000000000000000000000000000000000000..228add09443bd1827a70a2648b9a1abb9a910e94 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/agieval/sat-math.yaml @@ -0,0 +1,3 @@ +include: aqua-rat.yaml +task: agieval_sat_math +dataset_path: hails/agieval-sat-math diff --git a/lm-evaluation-harness/lm_eval/tasks/agieval/utils.py b/lm-evaluation-harness/lm_eval/tasks/agieval/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..aa6e544f1a7e15e853b99be2fe01502baadefcee --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/agieval/utils.py @@ -0,0 +1,274 @@ +# Answer parsing and normalization code, from +# https://github.com/ruixiangcui/AGIEval/blob/main/src/ +# math_equivalence.py and post_process.py +import re +from typing import Dict, List + +import numpy as np + + +def parse_math_answer(raw_string): + def remove_boxed(s): + left = "\\boxed{" + try: + assert s[: len(left)] == left + assert s[-1] == "}" + answer = s[len(left) : -1] + if "=" in answer: + answer = answer.split("=")[-1].lstrip(" ") + return answer + except Exception: + return None + + def last_boxed_only_string(string): + idx = string.rfind("\\boxed") + if idx < 0: + idx = string.rfind("\\fbox") + if idx < 0: + return None + i = idx + right_brace_idx = None + num_left_braces_open = 0 + while i < len(string): + if string[i] == "{": + num_left_braces_open += 1 + if string[i] == "}": + num_left_braces_open -= 1 + if num_left_braces_open == 0: + right_brace_idx = i + break + i += 1 + + if right_brace_idx is None: + retval = None + else: + retval = string[idx : right_brace_idx + 1] + + return retval + + def get_answer_with_dollar_sign(s): + first_pattern = "\$(.*)\$" + last_match = None + matches = re.findall(first_pattern, s) + if matches: + last_match = matches[-1] + if "=" in last_match: + last_match = last_match.split("=")[-1].lstrip(" ") + return last_match + + def get_answer_without_dollar_sign(s): + last_match = None + if "=" in s: + last_match = s.split("=")[-1].lstrip(" ").rstrip(".") + if "\\n" in last_match: + last_match = last_match.split("\\n")[0] + else: + pattern = "(?:\\$)?\d+(?:\.\d+)?(?![\w\d])" + matches = re.findall(pattern, s) + if matches: + last_match = matches[-1] + return last_match + + if "\\boxed" in raw_string: + answer = remove_boxed(last_boxed_only_string(raw_string)) + else: + answer = get_answer_with_dollar_sign(raw_string) + if not answer: + answer = get_answer_without_dollar_sign(raw_string) + return answer + + +# code from https://github.com/hendrycks/math/blob/main/modeling/math_equivalence.py +def _fix_fracs(string): + substrs = string.split("\\frac") + new_str = substrs[0] + if len(substrs) > 1: + substrs = substrs[1:] + for substr in substrs: + new_str += "\\frac" + if substr[0] == "{": + new_str += substr + else: + try: + assert len(substr) >= 2 + except Exception: + return string + a = substr[0] + b = substr[1] + if b != "{": + if len(substr) > 2: + post_substr = substr[2:] + new_str += "{" + a + "}{" + b + "}" + post_substr + else: + new_str += "{" + a + "}{" + b + "}" + else: + if len(substr) > 2: + post_substr = substr[2:] + new_str += "{" + a + "}" + b + post_substr + else: + new_str += "{" + a + "}" + b + string = new_str + return string + + +def _fix_a_slash_b(string): + if len(string.split("/")) != 2: + return string + a = string.split("/")[0] + b = string.split("/")[1] + try: + a = int(a) + b = int(b) + assert string == "{}/{}".format(a, b) + new_string = "\\frac{" + str(a) + "}{" + str(b) + "}" + return new_string + except Exception: + return string + + +def _remove_right_units(string): + # "\\text{ " only ever occurs (at least in the val set) when describing units + if "\\text{ " in string: + splits = string.split("\\text{ ") + assert len(splits) == 2 + return splits[0] + else: + return string + + +def _fix_sqrt(string): + if "\\sqrt" not in string: + return string + splits = string.split("\\sqrt") + new_string = splits[0] + for split in splits[1:]: + if split[0] != "{": + a = split[0] + new_substr = "\\sqrt{" + a + "}" + split[1:] + else: + new_substr = "\\sqrt" + split + new_string += new_substr + return new_string + + +def _strip_string(string): + # linebreaks + string = string.replace("\n", "") + # print(string) + + # remove inverse spaces + string = string.replace("\\!", "") + # print(string) + + # replace \\ with \ + string = string.replace("\\\\", "\\") + # print(string) + + # replace tfrac and dfrac with frac + string = string.replace("tfrac", "frac") + string = string.replace("dfrac", "frac") + # print(string) + + # remove \left and \right + string = string.replace("\\left", "") + string = string.replace("\\right", "") + # print(string) + + # Remove circ (degrees) + string = string.replace("^{\\circ}", "") + string = string.replace("^\\circ", "") + + # remove dollar signs + string = string.replace("\\$", "") + + # remove units (on the right) + string = _remove_right_units(string) + + # remove percentage + string = string.replace("\\%", "") + string = string.replace("\%", "") + + # " 0." equivalent to " ." and "{0." equivalent to "{." Alternatively, add "0" if "." is the start of the string + string = string.replace(" .", " 0.") + string = string.replace("{.", "{0.") + # if empty, return empty string + if len(string) == 0: + return string + if string[0] == ".": + string = "0" + string + + # to consider: get rid of e.g. "k = " or "q = " at beginning + if len(string.split("=")) == 2: + if len(string.split("=")[0]) <= 2: + string = string.split("=")[1] + + # fix sqrt3 --> sqrt{3} + string = _fix_sqrt(string) + + # remove spaces + string = string.replace(" ", "") + + # \frac1b or \frac12 --> \frac{1}{b} and \frac{1}{2}, etc. Even works with \frac1{72} (but not \frac{72}1). Also does a/b --> \\frac{a}{b} + string = _fix_fracs(string) + + # manually change 0.5 --> \frac{1}{2} + if string == "0.5": + string = "\\frac{1}{2}" + + # NOTE: X/Y changed to \frac{X}{Y} in dataset, but in simple cases fix in case the model output is X/Y + string = _fix_a_slash_b(string) + + return string + + +def is_equiv(str1, str2, verbose=False): + if str1 is None and str2 is None: + print("WARNING: Both None") + return True + if str1 is None or str2 is None: + return False + + str1, str2 = parse_math_answer(str1), parse_math_answer(str2) + + try: + ss1 = _strip_string(str1) + ss2 = _strip_string(str2) + if verbose: + print(ss1, ss2) + return ss1 == ss2 + except Exception: + return str1 == str2 + + +def process_results(doc: dict, results: List[str]) -> Dict[str, int]: + candidate = results[0] + + gold = doc["answer"] + + if not gold: + print(doc, candidate, gold) + if is_equiv(candidate, gold): + retval = 1 + else: + retval = 0 + + results = { + "acc": retval, + } + return results + + +# use a custom process_results() function, because AGIEval can have multiple valid answers +def process_results_mcqa(doc, results): + results = [result[0] for result in results] + + gold = doc["gold"] + + acc = 1.0 if int(np.argmax(results)) in gold else 0.0 + completion_len = np.array([float(len(i)) for i in doc["choices"]]) + acc_norm = 1.0 if int(np.argmax(results / completion_len)) in gold else 0.0 + + return { + "acc": acc, + "acc_norm": acc_norm, + } diff --git a/lm-evaluation-harness/lm_eval/tasks/alghafa/copa_ar/README.md b/lm-evaluation-harness/lm_eval/tasks/alghafa/copa_ar/README.md new file mode 100644 index 0000000000000000000000000000000000000000..972acb9f7431d34c216bbd27fee35f7ca138dcf5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/alghafa/copa_ar/README.md @@ -0,0 +1,40 @@ +#Arabic COPA + +### Paper + +Original Title: `COPA` + + + +The Choice Of Plausible Alternatives (COPA) evaluation provides researchers with a tool for assessing progress in open-domain commonsense causal reasoning. + +[Homepage](https://people.ict.usc.edu/~gordon/copa.html) + +AlGhafa has translated this dataset to Arabic[AlGafa](https://aclanthology.org/2023.arabicnlp-1.21.pdf) + +The link to the Arabic version of the dataset [PICA](https://gitlab.com/tiiuae/alghafa/-/tree/main/arabic-eval/copa_ar) + +### Citation + +### Groups and Tasks + +#### Groups + +* Not part of a group yet. + +#### Tasks + +* `copa_ar` + +### Checklist + +For adding novel benchmarks/datasets to the library: +* [x] Is the task an existing benchmark in the literature? + * [x] Have you referenced the original paper that introduced the task? + * [x] If yes, does the original paper provide a reference implementation? If so, have you checked against the reference implementation and documented how to run such a test? + + +If other tasks on this dataset are already supported: +* [x] Is the "Main" variant of this task clearly denoted? +* [x] Have you provided a short sentence in a README on what each new variant adds / evaluates? +* [x] Have you noted which, if any, published evaluation setups are matched by this variant? diff --git a/lm-evaluation-harness/lm_eval/tasks/alghafa/copa_ar/copa_ar.yaml b/lm-evaluation-harness/lm_eval/tasks/alghafa/copa_ar/copa_ar.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e35d1688babf0b5386f70f563fa923242540d0d5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/alghafa/copa_ar/copa_ar.yaml @@ -0,0 +1,21 @@ +task: copa_ar +dataset_path: Hennara/copa_ar +dataset_name: null +output_type: multiple_choice +training_split: null +validation_split: null +test_split: test +doc_to_text: "السؤال: {{query}}\nالجواب:" +doc_to_choice: "{{[sol1, sol2]}}" +doc_to_target: label +should_decontaminate: true +doc_to_decontamination_query: query +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/alghafa/piqa_ar/README.md b/lm-evaluation-harness/lm_eval/tasks/alghafa/piqa_ar/README.md new file mode 100644 index 0000000000000000000000000000000000000000..e1b71e93da4c00104c38c03b9d4486966e8ad567 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/alghafa/piqa_ar/README.md @@ -0,0 +1,43 @@ +#Arabic PIQA + +### Paper + +Original Title: `PIQA: Reasoning about Physical Commonsense in Natural Language` + +Original paper: [PICA](https://arxiv.org/abs/1911.11641) + +Physical Interaction: Question Answering (PIQA) is a physical commonsense +reasoning and a corresponding benchmark dataset. PIQA was designed to investigate +the physical knowledge of existing models. To what extent are current approaches +actually learning about the world? + +[Homepage](https://yonatanbisk.com/piqa) + +AlGhafa has translated this dataset to Arabic[AlGafa](https://aclanthology.org/2023.arabicnlp-1.21.pdf) + +The link to the Arabic version of the dataset [PICA](https://gitlab.com/tiiuae/alghafa/-/tree/main/arabic-eval/pica_ar) + +### Citation + +### Groups and Tasks + +#### Groups + +* Not part of a group yet. + +#### Tasks + +* `piqa_ar` + +### Checklist + +For adding novel benchmarks/datasets to the library: +* [x] Is the task an existing benchmark in the literature? + * [x] Have you referenced the original paper that introduced the task? + * [x] If yes, does the original paper provide a reference implementation? If so, have you checked against the reference implementation and documented how to run such a test? + + +If other tasks on this dataset are already supported: +* [x] Is the "Main" variant of this task clearly denoted? +* [x] Have you provided a short sentence in a README on what each new variant adds / evaluates? +* [x] Have you noted which, if any, published evaluation setups are matched by this variant? diff --git a/lm-evaluation-harness/lm_eval/tasks/alghafa/piqa_ar/piqa_ar.yaml b/lm-evaluation-harness/lm_eval/tasks/alghafa/piqa_ar/piqa_ar.yaml new file mode 100644 index 0000000000000000000000000000000000000000..19dfaee0c609f409d3bd6e37163054c2e80af37a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/alghafa/piqa_ar/piqa_ar.yaml @@ -0,0 +1,21 @@ +task: piqa_ar +dataset_path: Hennara/pica_ar +dataset_name: null +output_type: multiple_choice +training_split: null +validation_split: null +test_split: test +doc_to_text: "السؤال: {{goal}}\nالجواب:" +doc_to_choice: "{{[sol1, sol2]}}" +doc_to_target: label +should_decontaminate: true +doc_to_decontamination_query: goal +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/anli/README.md b/lm-evaluation-harness/lm_eval/tasks/anli/README.md new file mode 100644 index 0000000000000000000000000000000000000000..ba3f99d4826f0604f583772a2b48fe676a6f3e06 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/anli/README.md @@ -0,0 +1,56 @@ +# ANLI + +### Paper + +Title: `Adversarial NLI: A New Benchmark for Natural Language Understanding` + +Paper Link: https://arxiv.org/abs/1910.14599 + +Adversarial NLI (ANLI) is a dataset collected via an iterative, adversarial +human-and-model-in-the-loop procedure. It consists of three rounds that progressively +increase in difficulty and complexity, and each question-answer includes annotator- +provided explanations. + +Homepage: https://github.com/facebookresearch/anli + +### Citation + +``` +@inproceedings{nie-etal-2020-adversarial, + title = "Adversarial {NLI}: A New Benchmark for Natural Language Understanding", + author = "Nie, Yixin and + Williams, Adina and + Dinan, Emily and + Bansal, Mohit and + Weston, Jason and + Kiela, Douwe", + booktitle = "Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics", + year = "2020", + publisher = "Association for Computational Linguistics", +} +``` + +### Groups and Tasks + +#### Groups + +* `anli`: Evaluates `anli_r1`, `anli_r2`, and `anli_r3` + +#### Tasks +* `anli_r1`: The data collected adversarially in the first round. +* `anli_r2`: The data collected adversarially in the second round, after training on the previous round's data. +* `anli_r3`: The data collected adversarially in the third round, after training on the previous multiple rounds of data. + + +### Checklist + +For adding novel benchmarks/datasets to the library: + * [x] Is the task an existing benchmark in the literature? + * [x] Have you referenced the original paper that introduced the task? + * [ ] If yes, does the original paper provide a reference implementation? If so, have you checked against the reference implementation and documented how to run such a test? + + +If other tasks on this dataset are already supported: +* [ ] Is the "Main" variant of this task clearly denoted? +* [x] Have you provided a short sentence in a README on what each new variant adds / evaluates? +* [ ] Have you noted which, if any, published evaluation setups are matched by this variant? diff --git a/lm-evaluation-harness/lm_eval/tasks/anli/anli_r1.yaml b/lm-evaluation-harness/lm_eval/tasks/anli/anli_r1.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2de1d259600c85a31f6d2bec69d37783cc0cd0f8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/anli/anli_r1.yaml @@ -0,0 +1,26 @@ +tag: + - anli +task: anli_r1 +dataset_path: anli +dataset_name: null +output_type: multiple_choice +training_split: train_r1 +validation_split: dev_r1 +test_split: test_r1 +doc_to_text: "{{premise}}\nQuestion: {{hypothesis}} True, False, or Neither?\nAnswer:" +# True = entailment +# False = contradiction +# Neither = neutral +doc_to_target: "{{['True', 'Neither', 'False'][label]}}" +doc_to_choice: + - "True" + - "Neither" + - "False" +should_decontaminate: true +doc_to_decontamination_query: premise +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/anli/anli_r2.yaml b/lm-evaluation-harness/lm_eval/tasks/anli/anli_r2.yaml new file mode 100644 index 0000000000000000000000000000000000000000..85f28d67cf230fa36cd38dd8d6a345f6e679c53e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/anli/anli_r2.yaml @@ -0,0 +1,5 @@ +include: anli_r1.yaml +task: anli_r2 +training_split: train_r2 +validation_split: dev_r2 +test_split: test_r2 diff --git a/lm-evaluation-harness/lm_eval/tasks/anli/anli_r3.yaml b/lm-evaluation-harness/lm_eval/tasks/anli/anli_r3.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6b9f98a867f7d03b90e84a425dc8b044b4cc96fb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/anli/anli_r3.yaml @@ -0,0 +1,5 @@ +include: anli_r1.yaml +task: anli_r3 +training_split: train_r3 +validation_split: dev_r3 +test_split: test_r3 diff --git a/lm-evaluation-harness/lm_eval/tasks/arab_culture/README.md b/lm-evaluation-harness/lm_eval/tasks/arab_culture/README.md new file mode 100644 index 0000000000000000000000000000000000000000..f8bc5a8c97da0d644460ad0bfc5597d337c28679 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arab_culture/README.md @@ -0,0 +1,70 @@ +# Arab Culture + +### Paper + +Title: Commonsense Reasoning in Arab Culture + + +Abstract: https://arxiv.org/abs/2502.12788 + +Despite progress in Arabic large language models, such as Jais and AceGPT, their evaluation on commonsense reasoning has largely relied on machine-translated datasets, which lack cultural depth and may introduce Anglocentric biases. Commonsense reasoning is shaped by geographical and cultural contexts, and existing English datasets fail to capture the diversity of the Arab world. To address this, we introduce \datasetname, a commonsense reasoning dataset in Modern Standard Arabic (MSA), covering cultures of 13 countries across the Gulf, Levant, North Africa, and the Nile Valley. The dataset was built from scratch by engaging native speakers to write and validate culturally relevant questions for their respective countries. \datasetname spans 12 daily life domains with 54 fine-grained subtopics, reflecting various aspects of social norms, traditions, and everyday experiences. Zero-shot evaluations show that open-weight language models with up to 32B parameters struggle to comprehend diverse Arab cultures, with performance varying across regions. These findings highlight the need for more culturally aware models and datasets tailored to the Arabic-speaking world. + +Homepage: https://github.com/fajri91/ArabicCulture + + +### Citation + +``` +@misc{sadallah2025commonsensereasoningarabculture, + title={Commonsense Reasoning in Arab Culture}, + author={Abdelrahman Sadallah and Junior Cedric Tonga and Khalid Almubarak and Saeed Almheiri and Farah Atif and Chatrine Qwaider and Karima Kadaoui and Sara Shatnawi and Yaser Alesh and Fajri Koto}, + year={2025}, + eprint={2502.12788}, + archivePrefix={arXiv}, + primaryClass={cs.CL}, + url={https://arxiv.org/abs/2502.12788}, +} +``` + +### There are two variant of this task: `arab_culture`, and `arab_culture_completion` + +- The `arab_culture` is the normal MCQ evaluation type, which appends the answers to the question, and then measure the likelihood of the different choices markers (A,B,C or "أ","ب","ج"). For more info, follow the MMLU style [tempelate](https://github.com/EleutherAI/lm-evaluation-harness/blob/main/lm_eval/tasks/mmlu/default/_default_template_yaml#L7-L8) +- The `arab_culture_completion` do the evaluation in a sentence completion manner, by appending each asnwer to the question separetley and chooses the answer with the higher likelihood. See [this](https://github.com/EleutherAI/lm-evaluation-harness/blob/1f9bc88fe61f6bfa36f74e91ce3d59ab5685e4f1/lm_eval/tasks/arc/arc_easy.yaml#L10-L12) for more information + +### Groups and Tasks + +#### Groups + +* `arabculture`: evaluates all ArabCulture tasks. + +* `arab_culture_gulf`: evaluates Gulf countires ArabCulture tasks. +* `arab_culture_levant`: evaluates Levant countires ArabCulture tasks. +* `arab_culture_nile_valley`: evaluates Nile Valley countires ArabCulture tasks. +* `arab_culture_north_africa`: evaluates North Africa ArabCulture tasks. + +### Evaluation modes +This bechmark allows for different evaluation settings by allowing to adding more extra context for the model: + +We have three settings: +* without any information +``` +COUNTRY=False +REGION=False +``` +* with only region information +``` +COUNTRY=False +REGION=True +``` +* with region and country information +``` +COUNTRY=True +REGION=True +``` + +**Please add these flags add environment variables.** + + +* We also allow for prompting in English, which we found to acheive higher results on most of the evaluated models (please refer to our paper). + +* To change the language of the prompt, Define the `ARABIC` environment variable. diff --git a/lm-evaluation-harness/lm_eval/tasks/arab_culture/_arab_culture.yaml b/lm-evaluation-harness/lm_eval/tasks/arab_culture/_arab_culture.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8da809e673c13ac90476102a1e9a2cc07ee90816 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arab_culture/_arab_culture.yaml @@ -0,0 +1,12 @@ +aggregate_metric_list: + metric: acc + weight_by_size: true +group: arab_culture +metadata: + description: Arab Culture tasks + version: 0 +task: +- arab_culture_gulf +- arab_culture_levant +- arab_culture_north_africa +- arab_culture_nile_valley diff --git a/lm-evaluation-harness/lm_eval/tasks/arab_culture/_arab_culture_gulf.yaml b/lm-evaluation-harness/lm_eval/tasks/arab_culture/_arab_culture_gulf.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ca0ca4d89ad7c3cb95b83b50d3891ca26033397c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arab_culture/_arab_culture_gulf.yaml @@ -0,0 +1,10 @@ +aggregate_metric_list: + metric: acc + weight_by_size: true +group: arab_culture_gulf +group_alias: Gulf +metadata: + description: arab Culture tasks + version: 0 +task: +- arab_culture_gulf_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/arab_culture/_arab_culture_levant.yaml b/lm-evaluation-harness/lm_eval/tasks/arab_culture/_arab_culture_levant.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b3344d372055efd1fafd71b221631c38589ff9d7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arab_culture/_arab_culture_levant.yaml @@ -0,0 +1,10 @@ +aggregate_metric_list: + metric: acc + weight_by_size: true +group: arab_culture_levant +group_alias: Levant +metadata: + description: arab Culture tasks + version: 0 +task: +- arab_culture_levant_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/arab_culture/_arab_culture_nile_valley.yaml b/lm-evaluation-harness/lm_eval/tasks/arab_culture/_arab_culture_nile_valley.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e858409a9b5d5473431fde3c55369ea3b70d2d32 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arab_culture/_arab_culture_nile_valley.yaml @@ -0,0 +1,10 @@ +aggregate_metric_list: + metric: acc + weight_by_size: true +group: arab_culture_nile_valley +group_alias: Nile Valley +metadata: + description: arab Culture tasks + version: 0 +task: +- arab_culture_nile_valley_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/arab_culture/_arab_culture_north_africa.yaml b/lm-evaluation-harness/lm_eval/tasks/arab_culture/_arab_culture_north_africa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..30f31ffa797e29dc4de71990587054f797af3f92 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arab_culture/_arab_culture_north_africa.yaml @@ -0,0 +1,10 @@ +aggregate_metric_list: + metric: acc + weight_by_size: true +group: arab_culture_north_africa +group_alias: North Africa +metadata: + description: arab Culture tasks + version: 0 +task: +- arab_culture_north_africa_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/arab_culture/_default_arab_culture_mcq_template_yaml b/lm-evaluation-harness/lm_eval/tasks/arab_culture/_default_arab_culture_mcq_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..30b78fe6ca6e58f3aca53bd07594a838a4e50aae --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arab_culture/_default_arab_culture_mcq_template_yaml @@ -0,0 +1,19 @@ +dataset_path: MBZUAI/ArabCulture +test_split: test +fewshot_split: test +fewshot_config: + sampler: first_n +output_type: multiple_choice +doc_to_text: !function utils_mcq.doc_to_text +doc_to_choice: !function utils_mcq.doc_to_choice +doc_to_target: !function utils_mcq.doc_to_target +target_delimiter: "" +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arab_culture/_generate_configs.py b/lm-evaluation-harness/lm_eval/tasks/arab_culture/_generate_configs.py new file mode 100644 index 0000000000000000000000000000000000000000..81ee1c619d12909822fb19b6b8da319434400bf1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arab_culture/_generate_configs.py @@ -0,0 +1,122 @@ +""" +Take in a YAML, and output all "other" splits with this YAML +""" + +import argparse +import logging +import os + +import yaml +from tqdm import tqdm + + +eval_logger = logging.getLogger("lm-eval") + +countries = { + "KSA": "Gulf", + "UAE": "Gulf", + "Yemen": "Gulf", + "Lebanon": "Levant", + "Syria": "Levant", + "Palestine": "Levant", + "Jordan": "Levant", + "Tunisia": "North Africa", + "Algeria": "North Africa", + "Morocco": "North Africa", + "Libya": "North Africa", + "Egypt": "Nile Valley", + "Sudan": "Nile Valley", +} + +VERSION = 0 + + +def parse_args(): + parser = argparse.ArgumentParser() + parser.add_argument( + "--base_yaml_path", default="_default_arab_culture_mcq_template_yaml" + ) + parser.add_argument("--save_prefix_path", default="arab_culture") + return parser.parse_args() + + +if __name__ == "__main__": + args = parse_args() + + # get filename of base_yaml so we can `"include": ` it in our "other" YAMLs. + base_yaml_name = os.path.split(args.base_yaml_path)[-1] + # with open(args.base_yaml_path, encoding="utf-8") as f: + # base_yaml = yaml.full_load(f) + + ALL_REGIONS = [] + for country, region in tqdm(countries.items()): + if region not in ALL_REGIONS: + ALL_REGIONS.append(region) + + # description = f"The following are multiple choice questions (with answers) about {' '.join(subject.split('_'))}.\n\n" + + yaml_dict = { + "include": base_yaml_name, + "tag": f"arab_culture_{region.lower().replace(' ', '_')}_tasks", + "task": f"arab_culture_{country.lower().replace(' ', '_')}", + "task_alias": country, + "dataset_name": country, + # "description": description, + } + + file_save_path = ( + args.save_prefix_path + + f"_{country.lower().replace(' ', '_').replace('(', '').replace(')', '')}.yaml" + ) + eval_logger.info(f"Saving yaml for subset {country} to {file_save_path}") + with open(file_save_path, "w", encoding="utf-8") as yaml_file: + yaml.dump( + yaml_dict, + yaml_file, + allow_unicode=True, + default_style='"', + ) + + arab_culture_mcq_regions = [ + f"arab_culture_{region.lower().replace(' ', '_')}" for region in ALL_REGIONS + ] + + file_save_path = args.save_prefix_path + ".yaml" + + eval_logger.info(f"Saving benchmark config to {file_save_path}") + + for region in ALL_REGIONS: + file_save_path = ( + args.save_prefix_path + f"_{region.lower().replace(' ', '_')}.yaml" + ) + eval_logger.info(f"Saving yaml for subset {region} to {file_save_path}") + with open("_" + file_save_path, "w", encoding="utf-8") as yaml_file: + yaml.dump( + { + "group": f"arab_culture_{region.lower().replace(' ', '_')}", + "group_alias": region, + "task": [f"arab_culture_{region.lower().replace(' ', '_')}_tasks"], + "aggregate_metric_list": {"metric": "acc", "weight_by_size": True}, + "metadata": { + "description": "arab Culture tasks", + "version": VERSION, + }, + }, + yaml_file, + indent=4, + default_flow_style=False, + ) + + file_save_path = args.save_prefix_path + ".yaml" + with open("_" + file_save_path, "w", encoding="utf-8") as yaml_file: + yaml.dump( + { + "group": "arab_culture", + "task": arab_culture_mcq_regions, + "aggregate_metric_list": {"metric": "acc", "weight_by_size": True}, + "metadata": {"description": "Arab Culture tasks", "version": VERSION}, + }, + yaml_file, + indent=4, + default_flow_style=False, + ) diff --git a/lm-evaluation-harness/lm_eval/tasks/arab_culture/arab_culture_algeria.yaml b/lm-evaluation-harness/lm_eval/tasks/arab_culture/arab_culture_algeria.yaml new file mode 100644 index 0000000000000000000000000000000000000000..705606b81254a5c423eb5d2066189a3f6cddf92c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arab_culture/arab_culture_algeria.yaml @@ -0,0 +1,5 @@ +"dataset_name": "Algeria" +"include": "_default_arab_culture_mcq_template_yaml" +"tag": "arab_culture_north_africa_tasks" +"task": "arab_culture_algeria" +"task_alias": "Algeria" diff --git a/lm-evaluation-harness/lm_eval/tasks/arab_culture/arab_culture_egypt.yaml b/lm-evaluation-harness/lm_eval/tasks/arab_culture/arab_culture_egypt.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f71863147d6f8d465f49a90aee21c290ab2a82f8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arab_culture/arab_culture_egypt.yaml @@ -0,0 +1,5 @@ +"dataset_name": "Egypt" +"include": "_default_arab_culture_mcq_template_yaml" +"tag": "arab_culture_nile_valley_tasks" +"task": "arab_culture_egypt" +"task_alias": "Egypt" diff --git a/lm-evaluation-harness/lm_eval/tasks/arab_culture/arab_culture_jordan.yaml b/lm-evaluation-harness/lm_eval/tasks/arab_culture/arab_culture_jordan.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c6587d2988f170797a3cad1916ea5ee7d298bc0e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arab_culture/arab_culture_jordan.yaml @@ -0,0 +1,5 @@ +"dataset_name": "Jordan" +"include": "_default_arab_culture_mcq_template_yaml" +"tag": "arab_culture_levant_tasks" +"task": "arab_culture_jordan" +"task_alias": "Jordan" diff --git a/lm-evaluation-harness/lm_eval/tasks/arab_culture/arab_culture_ksa.yaml b/lm-evaluation-harness/lm_eval/tasks/arab_culture/arab_culture_ksa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..07d87cb80c76fccce2d48c950756822a5ebce46e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arab_culture/arab_culture_ksa.yaml @@ -0,0 +1,5 @@ +"dataset_name": "KSA" +"include": "_default_arab_culture_mcq_template_yaml" +"tag": "arab_culture_gulf_tasks" +"task": "arab_culture_ksa" +"task_alias": "KSA" diff --git a/lm-evaluation-harness/lm_eval/tasks/arab_culture/arab_culture_lebanon.yaml b/lm-evaluation-harness/lm_eval/tasks/arab_culture/arab_culture_lebanon.yaml new file mode 100644 index 0000000000000000000000000000000000000000..41c2b53cdf96bacc42141038e4917d1e1a9ff614 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arab_culture/arab_culture_lebanon.yaml @@ -0,0 +1,5 @@ +"dataset_name": "Lebanon" +"include": "_default_arab_culture_mcq_template_yaml" +"tag": "arab_culture_levant_tasks" +"task": "arab_culture_lebanon" +"task_alias": "Lebanon" diff --git a/lm-evaluation-harness/lm_eval/tasks/arab_culture/arab_culture_libya.yaml b/lm-evaluation-harness/lm_eval/tasks/arab_culture/arab_culture_libya.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e82c5598241a12f58000066fbcfde0a6eb2fa2a2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arab_culture/arab_culture_libya.yaml @@ -0,0 +1,5 @@ +"dataset_name": "Libya" +"include": "_default_arab_culture_mcq_template_yaml" +"tag": "arab_culture_north_africa_tasks" +"task": "arab_culture_libya" +"task_alias": "Libya" diff --git a/lm-evaluation-harness/lm_eval/tasks/arab_culture/arab_culture_morocco.yaml b/lm-evaluation-harness/lm_eval/tasks/arab_culture/arab_culture_morocco.yaml new file mode 100644 index 0000000000000000000000000000000000000000..847a86f5e80e184515b0b8bafcecd3e9dd499590 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arab_culture/arab_culture_morocco.yaml @@ -0,0 +1,5 @@ +"dataset_name": "Morocco" +"include": "_default_arab_culture_mcq_template_yaml" +"tag": "arab_culture_north_africa_tasks" +"task": "arab_culture_morocco" +"task_alias": "Morocco" diff --git a/lm-evaluation-harness/lm_eval/tasks/arab_culture/arab_culture_palestine.yaml b/lm-evaluation-harness/lm_eval/tasks/arab_culture/arab_culture_palestine.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dcbe183bb7a58c2a79d7f4b26d04a52255d9967b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arab_culture/arab_culture_palestine.yaml @@ -0,0 +1,5 @@ +"dataset_name": "Palestine" +"include": "_default_arab_culture_mcq_template_yaml" +"tag": "arab_culture_levant_tasks" +"task": "arab_culture_palestine" +"task_alias": "Palestine" diff --git a/lm-evaluation-harness/lm_eval/tasks/arab_culture/arab_culture_sudan.yaml b/lm-evaluation-harness/lm_eval/tasks/arab_culture/arab_culture_sudan.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9920d5ba119f94ff160b567a180cfb7af1f2cbdc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arab_culture/arab_culture_sudan.yaml @@ -0,0 +1,5 @@ +"dataset_name": "Sudan" +"include": "_default_arab_culture_mcq_template_yaml" +"tag": "arab_culture_nile_valley_tasks" +"task": "arab_culture_sudan" +"task_alias": "Sudan" diff --git a/lm-evaluation-harness/lm_eval/tasks/arab_culture/arab_culture_syria.yaml b/lm-evaluation-harness/lm_eval/tasks/arab_culture/arab_culture_syria.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0ed6f7672346cd580c4e128c08cf27c3020d92e4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arab_culture/arab_culture_syria.yaml @@ -0,0 +1,5 @@ +"dataset_name": "Syria" +"include": "_default_arab_culture_mcq_template_yaml" +"tag": "arab_culture_levant_tasks" +"task": "arab_culture_syria" +"task_alias": "Syria" diff --git a/lm-evaluation-harness/lm_eval/tasks/arab_culture/arab_culture_tunisia.yaml b/lm-evaluation-harness/lm_eval/tasks/arab_culture/arab_culture_tunisia.yaml new file mode 100644 index 0000000000000000000000000000000000000000..de8d78a88e893e47b250e3f8f5c7d7b22611d26a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arab_culture/arab_culture_tunisia.yaml @@ -0,0 +1,5 @@ +"dataset_name": "Tunisia" +"include": "_default_arab_culture_mcq_template_yaml" +"tag": "arab_culture_north_africa_tasks" +"task": "arab_culture_tunisia" +"task_alias": "Tunisia" diff --git a/lm-evaluation-harness/lm_eval/tasks/arab_culture/arab_culture_uae.yaml b/lm-evaluation-harness/lm_eval/tasks/arab_culture/arab_culture_uae.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5bcbeb83b7ac5fbec45c97802ea32477a055ae75 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arab_culture/arab_culture_uae.yaml @@ -0,0 +1,5 @@ +"dataset_name": "UAE" +"include": "_default_arab_culture_mcq_template_yaml" +"tag": "arab_culture_gulf_tasks" +"task": "arab_culture_uae" +"task_alias": "UAE" diff --git a/lm-evaluation-harness/lm_eval/tasks/arab_culture/arab_culture_yemen.yaml b/lm-evaluation-harness/lm_eval/tasks/arab_culture/arab_culture_yemen.yaml new file mode 100644 index 0000000000000000000000000000000000000000..caa32073c216e4eec6c3d1101d6b780146d3b91b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arab_culture/arab_culture_yemen.yaml @@ -0,0 +1,5 @@ +"dataset_name": "Yemen" +"include": "_default_arab_culture_mcq_template_yaml" +"tag": "arab_culture_gulf_tasks" +"task": "arab_culture_yemen" +"task_alias": "Yemen" diff --git a/lm-evaluation-harness/lm_eval/tasks/arab_culture/prompts.py b/lm-evaluation-harness/lm_eval/tasks/arab_culture/prompts.py new file mode 100644 index 0000000000000000000000000000000000000000..983bd8b26f25f1db06e4b5314da6be45804dc65e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arab_culture/prompts.py @@ -0,0 +1,73 @@ +REGION_COUNTRY_PROMPT_AR = """ +مهمتك هي اختيار الخيار الأنسب ثقافياً بناءً على السياق المقدم أدناه. + +الموقع: {country}, {region} +الجملة: {first_statement} + +يرجى مراعاة الفروق الثقافية للموقع المحدد واختيار الإجابة الأكثر ملاءمة من الخيارات المتاحة. + +الخيارات: +{choices} +""" + +REGION_PROMPT_AR = """ +مهمتك هي اختيار الخيار الأنسب ثقافياً بناءً على السياق المقدم أدناه. + +الموقع: {region} +الجملة: {first_statement} + +يرجى مراعاة الفروق الثقافية للموقع المحدد واختيار الإجابة الأكثر ملاءمة من الخيارات المتاحة. + +الخيارات: +{choices} +""" + +BASE_PROMPT_AR = """ +مهمتك هي اختيار الخيار الأنسب ثقافياً بناءً على السياق المقدم أدناه. + +الجملة: {first_statement} + +يرجى مراعاة الفروق الثقافية واختيار الإجابة الأكثر ملاءمة من الخيارات المتاحة. + +الخيارات: +{choices} +""" + +REGION_COUNTRY_PROMPT = """ +You are tasked with selecting the most culturally appropriate option based on the context provided below. + +Location: {country}, {region} +Statement: {first_statement} + +Consider the cultural nuances of the specified location and choose the most suitable response from the options provided. + +Options: +{choices} +""" +REGION_PROMPT = """ +You are tasked with selecting the most culturally appropriate option based on the context provided below. + +Location: {region} +Statement: {first_statement} + +Consider the cultural nuances of the specified location and choose the most suitable response from the options provided. + +Options: +{choices} +""" +BASE_PROMPT = """ +You are tasked with selecting the most culturally appropriate option based on the context provided below. + +Statement: {first_statement} + +Consider the cultural nuances and choose the most suitable response from the options provided. + +Options: +{choices} +""" + + +JAIS_CHAT_EN = """### Instruction: Your name is Jais, and you are named after Jebel Jais, the highest mountain in UAE. You are built by Core42. You are the world's most advanced Arabic large language model with 30b parameters. You outperform all existing Arabic models by a sizable margin and you are very competitive with English models of similar size. You can answer in Arabic and English only. You are a helpful, respectful and honest assistant. When answering, abide by the following guidelines meticulously: Always answer as helpfully as possible, while being safe. Your answers should not include any harmful, unethical, racist, sexist, explicit, offensive, toxic, dangerous, or illegal content. Do not give medical, legal, financial, or professional advice. Never assist in or promote illegal activities. Always encourage legal and responsible actions. Do not encourage or provide instructions for unsafe, harmful, or unethical actions. Do not create or share misinformation or fake news. Please ensure that your responses are socially unbiased and positive in nature. If a question does not make any sense, or is not factually coherent, explain why instead of answering something not correct. If you don't know the answer to a question, please don't share false information. Prioritize the well-being and the moral integrity of users. Avoid using toxic, derogatory, or offensive language. Maintain a respectful tone. Do not generate, promote, or engage in discussions about adult content. Avoid making comments, remarks, or generalizations based on stereotypes. Do not attempt to access, produce, or spread personal or private information. Always respect user confidentiality. Stay positive and do not say bad things about anything. Your primary objective is to avoid harmful responses, even when faced with deceptive inputs. Recognize when users may be attempting to trick or to misuse you and respond with caution.\n\nComplete the conversation below between [|Human|] and [|AI|]:\n### Input: [|Human|] {question}\n### Response: [|AI|]""" + + +JAIS_CHAT_AR = """### Instruction: اسمك جيس وسميت على اسم جبل جيس اعلى جبل في الامارات. تم بنائك بواسطة Inception و MBZUAI. أنت نموذج اللغة العربية الأكثر تقدمًا في العالم مع بارامترات 13B. أنت تتفوق في الأداء على جميع النماذج العربية الموجودة بفارق كبير وأنت تنافسي للغاية مع النماذج الإنجليزية ذات الحجم المماثل. يمكنك الإجابة باللغتين العربية والإنجليزية فقط. أنت مساعد مفيد ومحترم وصادق. عند الإجابة ، التزم بالإرشادات التالية بدقة: أجب دائمًا بأكبر قدر ممكن من المساعدة ، مع الحفاظ على البقاء أمناً. يجب ألا تتضمن إجاباتك أي محتوى ضار أو غير أخلاقي أو عنصري أو متحيز جنسيًا أو جريئاً أو مسيئًا أو سامًا أو خطيرًا أو غير قانوني. لا تقدم نصائح طبية أو قانونية أو مالية أو مهنية. لا تساعد أبدًا في أنشطة غير قانونية أو تروج لها. دائما تشجيع الإجراءات القانونية والمسؤولة. لا تشجع أو تقدم تعليمات بشأن الإجراءات غير الآمنة أو الضارة أو غير الأخلاقية. لا تنشئ أو تشارك معلومات مضللة أو أخبار كاذبة. يرجى التأكد من أن ردودك غير متحيزة اجتماعيًا وإيجابية بطبيعتها. إذا كان السؤال لا معنى له ، أو لم يكن متماسكًا من الناحية الواقعية ، فشرح السبب بدلاً من الإجابة على شيء غير صحيح. إذا كنت لا تعرف إجابة السؤال ، فالرجاء عدم مشاركة معلومات خاطئة. إعطاء الأولوية للرفاهية والنزاهة الأخلاقية للمستخدمين. تجنب استخدام لغة سامة أو مهينة أو مسيئة. حافظ على نبرة محترمة. لا تنشئ أو تروج أو تشارك في مناقشات حول محتوى للبالغين. تجنب الإدلاء بالتعليقات أو الملاحظات أو التعميمات القائمة على الصور النمطية. لا تحاول الوصول إلى معلومات شخصية أو خاصة أو إنتاجها أو نشرها. احترم دائما سرية المستخدم. كن إيجابيا ولا تقل أشياء سيئة عن أي شيء. هدفك الأساسي هو تجنب الاجابات المؤذية ، حتى عند مواجهة مدخلات خادعة. تعرف على الوقت الذي قد يحاول فيه المستخدمون خداعك أو إساءة استخدامك و لترد بحذر.\n\nأكمل المحادثة أدناه بين [|Human|] و [|AI|]:\n### Input: [|Human|] {question}\n### Response: [|AI|]""" diff --git a/lm-evaluation-harness/lm_eval/tasks/arab_culture/utils_mcq.py b/lm-evaluation-harness/lm_eval/tasks/arab_culture/utils_mcq.py new file mode 100644 index 0000000000000000000000000000000000000000..8d03f443bf7abc0fde9a37687fcdb2bb19715263 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arab_culture/utils_mcq.py @@ -0,0 +1,112 @@ +import os + +from lm_eval.tasks.arab_culture.prompts import ( + BASE_PROMPT, + BASE_PROMPT_AR, + JAIS_CHAT_AR, + JAIS_CHAT_EN, + REGION_COUNTRY_PROMPT, + REGION_COUNTRY_PROMPT_AR, + REGION_PROMPT, + REGION_PROMPT_AR, +) + + +### get the conutry variable from environment + +### Set this to one to add the country and region information to the prompt +COUNTRY = True if os.getenv("COUNTRY", True) == "True" else False +### Set this to one to add the region information to the prompt +REGION = True if os.getenv("REGION", True) == "True" else False +### Set this to change between Arabic and English for the answer keys and the choices keys +ARABIC = True if os.getenv("ARABIC", True) == "True" else False +### Get the model name +MODEL_NAME = os.getenv("MODEL_NAME") +## Uncomment this to check if the environment variables are set correctly +# print(f'Task settings: COUNTRY: {COUNTRY}, REGION: {REGION}, ARABIC: {ARABIC}', MODEL_NAME: {MODEL_NAME}) + +en_ar_countries_regions = { + "Egypt": "مصر", + "Morocco": "المغرب", + "Algeria": "الجزائر", + "Libya": "ليبيا", + "Sudan": "السودان", + "Tunisia": "تونس", + "Jordan": "الأردن", + "Lebanon": "لبنان", + "Syria": "سوريا", + "Palestine": "فلسطين", + "Yemen": "اليمن", + "UAE": "الإمارات", + "KSA": "السعودية", + "Gulf": "الخليج", + "Levant": "الشام", + "North Africa": "شمال أفريقيا", + "Nile Valley": "وادي النيل", +} + + +def doc_to_text(doc): + country = "" if not doc["country"] else doc["country"] + region = "" if not doc["region"] else doc["region"] + first_statement = doc["first_statement"].strip() + + ## We don't have a setting for only information about the country without the region + if COUNTRY: + assert REGION, ( + "If you want to add the country information, you must also add the region information" + ) + + ## convert contry and region name to arabic if the language is arabic + if ARABIC: + country = en_ar_countries_regions[country] + region = en_ar_countries_regions[region] + + choices = doc["options"] + choices_str = "" + for i in range(3): + key = choices["arabic_keys"][i] if ARABIC else choices["english_keys"][i] + choice_str = key + ". " + choices["text"][i].strip() + "\n" + choices_str += choice_str + + if COUNTRY and REGION: + cur_prompt = REGION_COUNTRY_PROMPT_AR if ARABIC else REGION_COUNTRY_PROMPT + doc_text = cur_prompt.format( + country=country, + region=region, + first_statement=first_statement, + choices=choices_str, + ) + elif REGION: + cur_prompt = REGION_PROMPT_AR if ARABIC else REGION_PROMPT + doc_text = cur_prompt.format( + region=region, first_statement=first_statement, choices=choices_str + ) + else: + cur_prompt = BASE_PROMPT_AR if ARABIC else BASE_PROMPT + doc_text = cur_prompt.format( + first_statement=first_statement, choices=choices_str + ) + + ### apply jais chat tempelate + if MODEL_NAME and "jais" in MODEL_NAME and "chat" in MODEL_NAME: + if ARABIC: + doc_text = JAIS_CHAT_AR.format(question=doc_text) + else: + doc_text = JAIS_CHAT_EN.format(question=doc_text) + + return doc_text + + +def doc_to_choice(doc): + return doc["options"]["arabic_keys"] if ARABIC else doc["options"]["english_keys"] + + +def doc_to_target(doc): + ans = ( + doc["answer_key"]["arabic_answer_key"] + if ARABIC + else doc["answer_key"]["english_answer_key"] + ) + ans = ans.strip() + return ans diff --git a/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/README.md b/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/README.md new file mode 100644 index 0000000000000000000000000000000000000000..f8bc5a8c97da0d644460ad0bfc5597d337c28679 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/README.md @@ -0,0 +1,70 @@ +# Arab Culture + +### Paper + +Title: Commonsense Reasoning in Arab Culture + + +Abstract: https://arxiv.org/abs/2502.12788 + +Despite progress in Arabic large language models, such as Jais and AceGPT, their evaluation on commonsense reasoning has largely relied on machine-translated datasets, which lack cultural depth and may introduce Anglocentric biases. Commonsense reasoning is shaped by geographical and cultural contexts, and existing English datasets fail to capture the diversity of the Arab world. To address this, we introduce \datasetname, a commonsense reasoning dataset in Modern Standard Arabic (MSA), covering cultures of 13 countries across the Gulf, Levant, North Africa, and the Nile Valley. The dataset was built from scratch by engaging native speakers to write and validate culturally relevant questions for their respective countries. \datasetname spans 12 daily life domains with 54 fine-grained subtopics, reflecting various aspects of social norms, traditions, and everyday experiences. Zero-shot evaluations show that open-weight language models with up to 32B parameters struggle to comprehend diverse Arab cultures, with performance varying across regions. These findings highlight the need for more culturally aware models and datasets tailored to the Arabic-speaking world. + +Homepage: https://github.com/fajri91/ArabicCulture + + +### Citation + +``` +@misc{sadallah2025commonsensereasoningarabculture, + title={Commonsense Reasoning in Arab Culture}, + author={Abdelrahman Sadallah and Junior Cedric Tonga and Khalid Almubarak and Saeed Almheiri and Farah Atif and Chatrine Qwaider and Karima Kadaoui and Sara Shatnawi and Yaser Alesh and Fajri Koto}, + year={2025}, + eprint={2502.12788}, + archivePrefix={arXiv}, + primaryClass={cs.CL}, + url={https://arxiv.org/abs/2502.12788}, +} +``` + +### There are two variant of this task: `arab_culture`, and `arab_culture_completion` + +- The `arab_culture` is the normal MCQ evaluation type, which appends the answers to the question, and then measure the likelihood of the different choices markers (A,B,C or "أ","ب","ج"). For more info, follow the MMLU style [tempelate](https://github.com/EleutherAI/lm-evaluation-harness/blob/main/lm_eval/tasks/mmlu/default/_default_template_yaml#L7-L8) +- The `arab_culture_completion` do the evaluation in a sentence completion manner, by appending each asnwer to the question separetley and chooses the answer with the higher likelihood. See [this](https://github.com/EleutherAI/lm-evaluation-harness/blob/1f9bc88fe61f6bfa36f74e91ce3d59ab5685e4f1/lm_eval/tasks/arc/arc_easy.yaml#L10-L12) for more information + +### Groups and Tasks + +#### Groups + +* `arabculture`: evaluates all ArabCulture tasks. + +* `arab_culture_gulf`: evaluates Gulf countires ArabCulture tasks. +* `arab_culture_levant`: evaluates Levant countires ArabCulture tasks. +* `arab_culture_nile_valley`: evaluates Nile Valley countires ArabCulture tasks. +* `arab_culture_north_africa`: evaluates North Africa ArabCulture tasks. + +### Evaluation modes +This bechmark allows for different evaluation settings by allowing to adding more extra context for the model: + +We have three settings: +* without any information +``` +COUNTRY=False +REGION=False +``` +* with only region information +``` +COUNTRY=False +REGION=True +``` +* with region and country information +``` +COUNTRY=True +REGION=True +``` + +**Please add these flags add environment variables.** + + +* We also allow for prompting in English, which we found to acheive higher results on most of the evaluated models (please refer to our paper). + +* To change the language of the prompt, Define the `ARABIC` environment variable. diff --git a/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/_arab_culture_completion.yaml b/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/_arab_culture_completion.yaml new file mode 100644 index 0000000000000000000000000000000000000000..814f366e8d106201b536a4bf85447fcaf09aebdc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/_arab_culture_completion.yaml @@ -0,0 +1,12 @@ +aggregate_metric_list: + metric: acc + weight_by_size: true +group: arab_culture_completion +metadata: + description: Arab Culture tasks + version: 0 +task: +- arab_culture_completion_gulf +- arab_culture_completion_levant +- arab_culture_completion_north_africa +- arab_culture_completion_nile_valley diff --git a/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/_arab_culture_completion_gulf.yaml b/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/_arab_culture_completion_gulf.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b342b428b85e9625d3a8045902a6f4355df98306 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/_arab_culture_completion_gulf.yaml @@ -0,0 +1,10 @@ +aggregate_metric_list: + metric: acc + weight_by_size: true +group: arab_culture_completion_gulf +group_alias: Gulf +metadata: + description: arab Culture tasks + version: 0 +task: +- arab_culture_completion_gulf_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/_arab_culture_completion_levant.yaml b/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/_arab_culture_completion_levant.yaml new file mode 100644 index 0000000000000000000000000000000000000000..199f68dc05f3125df329938937e6c697a237485e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/_arab_culture_completion_levant.yaml @@ -0,0 +1,10 @@ +aggregate_metric_list: + metric: acc + weight_by_size: true +group: arab_culture_completion_levant +group_alias: Levant +metadata: + description: arab Culture tasks + version: 0 +task: +- arab_culture_completion_levant_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/_arab_culture_completion_nile_valley.yaml b/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/_arab_culture_completion_nile_valley.yaml new file mode 100644 index 0000000000000000000000000000000000000000..284711eeb9af2fd94efd0f266a6d9b3a5764640d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/_arab_culture_completion_nile_valley.yaml @@ -0,0 +1,10 @@ +aggregate_metric_list: + metric: acc + weight_by_size: true +group: arab_culture_completion_nile_valley +group_alias: Nile Valley +metadata: + description: arab Culture tasks + version: 0 +task: +- arab_culture_completion_nile_valley_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/_arab_culture_completion_north_africa.yaml b/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/_arab_culture_completion_north_africa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..10c32d9fa17fa8a0da4a204bce77f48dd191a1c7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/_arab_culture_completion_north_africa.yaml @@ -0,0 +1,10 @@ +aggregate_metric_list: + metric: acc + weight_by_size: true +group: arab_culture_completion_north_africa +group_alias: North Africa +metadata: + description: arab Culture tasks + version: 0 +task: +- arab_culture_completion_north_africa_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/_default_arab_culture_completion_template_yaml b/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/_default_arab_culture_completion_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..6d5961ac605a25e3bfe37e757f2f755ba6c361da --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/_default_arab_culture_completion_template_yaml @@ -0,0 +1,19 @@ +dataset_path: boda/arabic_cluture +test_split: test +fewshot_split: test +fewshot_config: + sampler: first_n +output_type: multiple_choice +doc_to_text: !function utils_completion.doc_to_text +doc_to_choice: !function utils_completion.doc_to_choice +doc_to_target: !function utils_completion.doc_to_target +target_delimiter: "" +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/_generate_configs.py b/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/_generate_configs.py new file mode 100644 index 0000000000000000000000000000000000000000..d5c530f53a64736237147728211d4ae72b880380 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/_generate_configs.py @@ -0,0 +1,125 @@ +""" +Take in a YAML, and output all "other" splits with this YAML +""" + +import argparse +import logging +import os + +import yaml +from tqdm import tqdm + + +eval_logger = logging.getLogger("lm-eval") + +countries = { + "KSA": "Gulf", + "UAE": "Gulf", + "Yemen": "Gulf", + "Lebanon": "Levant", + "Syria": "Levant", + "Palestine": "Levant", + "Jordan": "Levant", + "Tunisia": "North Africa", + "Algeria": "North Africa", + "Morocco": "North Africa", + "Libya": "North Africa", + "Egypt": "Nile Valley", + "Sudan": "Nile Valley", +} + +VERSION = 0 + + +def parse_args(): + parser = argparse.ArgumentParser() + parser.add_argument( + "--base_yaml_path", default="_default_arab_culture_completion_template_yaml" + ) + parser.add_argument("--save_prefix_path", default="arab_culture_completion") + return parser.parse_args() + + +if __name__ == "__main__": + args = parse_args() + + # get filename of base_yaml so we can `"include": ` it in our "other" YAMLs. + base_yaml_name = os.path.split(args.base_yaml_path)[-1] + # with open(args.base_yaml_path, encoding="utf-8") as f: + # base_yaml = yaml.full_load(f) + + ALL_REGIONS = [] + for country, region in tqdm(countries.items()): + if region not in ALL_REGIONS: + ALL_REGIONS.append(region) + + # description = f"The following are multiple choice questions (with answers) about {' '.join(subject.split('_'))}.\n\n" + + yaml_dict = { + "include": base_yaml_name, + "tag": f"arab_culture_completion_{region.lower().replace(' ', '_')}_tasks", + "task": f"arab_culture_completion_{country.lower().replace(' ', '_')}", + "task_alias": country, + "dataset_name": country, + # "description": description, + } + + file_save_path = ( + args.save_prefix_path + + f"_{country.lower().replace(' ', '_').replace('(', '').replace(')', '')}.yaml" + ) + eval_logger.info(f"Saving yaml for subset {country} to {file_save_path}") + with open(file_save_path, "w", encoding="utf-8") as yaml_file: + yaml.dump( + yaml_dict, + yaml_file, + allow_unicode=True, + default_style='"', + ) + + arab_culture_completion_regions = [ + f"arab_culture_completion_{region.lower().replace(' ', '_')}" + for region in ALL_REGIONS + ] + + file_save_path = args.save_prefix_path + ".yaml" + + eval_logger.info(f"Saving benchmark config to {file_save_path}") + + for region in ALL_REGIONS: + file_save_path = ( + args.save_prefix_path + f"_{region.lower().replace(' ', '_')}.yaml" + ) + eval_logger.info(f"Saving yaml for subset {region} to {file_save_path}") + with open("_" + file_save_path, "w", encoding="utf-8") as yaml_file: + yaml.dump( + { + "group": f"arab_culture_completion_{region.lower().replace(' ', '_')}", + "group_alias": region, + "task": [ + f"arab_culture_completion_{region.lower().replace(' ', '_')}_tasks" + ], + "aggregate_metric_list": {"metric": "acc", "weight_by_size": True}, + "metadata": { + "description": "arab Culture tasks", + "version": VERSION, + }, + }, + yaml_file, + indent=4, + default_flow_style=False, + ) + + file_save_path = args.save_prefix_path + ".yaml" + with open("_" + file_save_path, "w", encoding="utf-8") as yaml_file: + yaml.dump( + { + "group": "arab_culture_completion", + "task": arab_culture_completion_regions, + "aggregate_metric_list": {"metric": "acc", "weight_by_size": True}, + "metadata": {"description": "Arab Culture tasks", "version": VERSION}, + }, + yaml_file, + indent=4, + default_flow_style=False, + ) diff --git a/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/arab_culture_completion_algeria.yaml b/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/arab_culture_completion_algeria.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e3cb7d8bac370f66a9984155c3bf2a28f30201b0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/arab_culture_completion_algeria.yaml @@ -0,0 +1,5 @@ +"dataset_name": "Algeria" +"include": "_default_arab_culture_completion_template_yaml" +"tag": "arab_culture_completion_north_africa_tasks" +"task": "arab_culture_completion_algeria" +"task_alias": "Algeria" diff --git a/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/arab_culture_completion_egypt.yaml b/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/arab_culture_completion_egypt.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f740d4b8b575705e740c4d0d3ffb079992e2b879 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/arab_culture_completion_egypt.yaml @@ -0,0 +1,5 @@ +"dataset_name": "Egypt" +"include": "_default_arab_culture_completion_template_yaml" +"tag": "arab_culture_completion_nile_valley_tasks" +"task": "arab_culture_completion_egypt" +"task_alias": "Egypt" diff --git a/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/arab_culture_completion_jordan.yaml b/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/arab_culture_completion_jordan.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dec15211d4aabc3b45da990fb573ebb5e48b1546 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/arab_culture_completion_jordan.yaml @@ -0,0 +1,5 @@ +"dataset_name": "Jordan" +"include": "_default_arab_culture_completion_template_yaml" +"tag": "arab_culture_completion_levant_tasks" +"task": "arab_culture_completion_jordan" +"task_alias": "Jordan" diff --git a/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/arab_culture_completion_ksa.yaml b/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/arab_culture_completion_ksa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ec1ea8900e9d82b771bed92de954172c269f899f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/arab_culture_completion_ksa.yaml @@ -0,0 +1,5 @@ +"dataset_name": "KSA" +"include": "_default_arab_culture_completion_template_yaml" +"tag": "arab_culture_completion_gulf_tasks" +"task": "arab_culture_completion_ksa" +"task_alias": "KSA" diff --git a/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/arab_culture_completion_lebanon.yaml b/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/arab_culture_completion_lebanon.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5f31061ffacc25a44778c63cb723c6cc329e99b4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/arab_culture_completion_lebanon.yaml @@ -0,0 +1,5 @@ +"dataset_name": "Lebanon" +"include": "_default_arab_culture_completion_template_yaml" +"tag": "arab_culture_completion_levant_tasks" +"task": "arab_culture_completion_lebanon" +"task_alias": "Lebanon" diff --git a/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/arab_culture_completion_libya.yaml b/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/arab_culture_completion_libya.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2541f87c0e2bb64f1659d1396d453ca86a40de37 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/arab_culture_completion_libya.yaml @@ -0,0 +1,5 @@ +"dataset_name": "Libya" +"include": "_default_arab_culture_completion_template_yaml" +"tag": "arab_culture_completion_north_africa_tasks" +"task": "arab_culture_completion_libya" +"task_alias": "Libya" diff --git a/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/arab_culture_completion_morocco.yaml b/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/arab_culture_completion_morocco.yaml new file mode 100644 index 0000000000000000000000000000000000000000..86e1cc65d649e69caa27056b5b8892c932aa4a7f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/arab_culture_completion_morocco.yaml @@ -0,0 +1,5 @@ +"dataset_name": "Morocco" +"include": "_default_arab_culture_completion_template_yaml" +"tag": "arab_culture_completion_north_africa_tasks" +"task": "arab_culture_completion_morocco" +"task_alias": "Morocco" diff --git a/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/arab_culture_completion_palestine.yaml b/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/arab_culture_completion_palestine.yaml new file mode 100644 index 0000000000000000000000000000000000000000..44731f7bbd365ac136a07cc8893502969bbada20 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/arab_culture_completion_palestine.yaml @@ -0,0 +1,5 @@ +"dataset_name": "Palestine" +"include": "_default_arab_culture_completion_template_yaml" +"tag": "arab_culture_completion_levant_tasks" +"task": "arab_culture_completion_palestine" +"task_alias": "Palestine" diff --git a/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/arab_culture_completion_sudan.yaml b/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/arab_culture_completion_sudan.yaml new file mode 100644 index 0000000000000000000000000000000000000000..76282e98fa880d923c9985b37d01e5dd4fb5a7ee --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/arab_culture_completion_sudan.yaml @@ -0,0 +1,5 @@ +"dataset_name": "Sudan" +"include": "_default_arab_culture_completion_template_yaml" +"tag": "arab_culture_completion_nile_valley_tasks" +"task": "arab_culture_completion_sudan" +"task_alias": "Sudan" diff --git a/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/arab_culture_completion_syria.yaml b/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/arab_culture_completion_syria.yaml new file mode 100644 index 0000000000000000000000000000000000000000..33a07dd9a546540cc43b6ee22fc0d6b8676e3293 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/arab_culture_completion_syria.yaml @@ -0,0 +1,5 @@ +"dataset_name": "Syria" +"include": "_default_arab_culture_completion_template_yaml" +"tag": "arab_culture_completion_levant_tasks" +"task": "arab_culture_completion_syria" +"task_alias": "Syria" diff --git a/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/arab_culture_completion_tunisia.yaml b/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/arab_culture_completion_tunisia.yaml new file mode 100644 index 0000000000000000000000000000000000000000..41c53529f12350f5cd169938f691d8b50cb9f795 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/arab_culture_completion_tunisia.yaml @@ -0,0 +1,5 @@ +"dataset_name": "Tunisia" +"include": "_default_arab_culture_completion_template_yaml" +"tag": "arab_culture_completion_north_africa_tasks" +"task": "arab_culture_completion_tunisia" +"task_alias": "Tunisia" diff --git a/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/arab_culture_completion_uae.yaml b/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/arab_culture_completion_uae.yaml new file mode 100644 index 0000000000000000000000000000000000000000..479440534a43628bcefc7812e7ee8967cdad1bce --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/arab_culture_completion_uae.yaml @@ -0,0 +1,5 @@ +"dataset_name": "UAE" +"include": "_default_arab_culture_completion_template_yaml" +"tag": "arab_culture_completion_gulf_tasks" +"task": "arab_culture_completion_uae" +"task_alias": "UAE" diff --git a/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/arab_culture_completion_yemen.yaml b/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/arab_culture_completion_yemen.yaml new file mode 100644 index 0000000000000000000000000000000000000000..88dd6dd5804beefc749f66aa1f133dfb33a95894 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/arab_culture_completion_yemen.yaml @@ -0,0 +1,5 @@ +"dataset_name": "Yemen" +"include": "_default_arab_culture_completion_template_yaml" +"tag": "arab_culture_completion_gulf_tasks" +"task": "arab_culture_completion_yemen" +"task_alias": "Yemen" diff --git a/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/prompts.py b/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/prompts.py new file mode 100644 index 0000000000000000000000000000000000000000..b042c056d94e95e854b83672abd36e7cf81ef4d5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/prompts.py @@ -0,0 +1,56 @@ +REGION_COUNTRY_PROMPT_AR = """ +مهمتك هي اختيار الخيار الأنسب ثقافياً بناءً على السياق المقدم أدناه. + +الموقع: {country}, {region} +الجملة: {first_statement} + +يرجى مراعاة الفروق الثقافية للموقع المحدد واختيار الإجابة الأكثر ملاءمة من الخيارات المتاحة. +""" + +REGION_PROMPT_AR = """ +مهمتك هي اختيار الخيار الأنسب ثقافياً بناءً على السياق المقدم أدناه. + +الموقع: {region} +الجملة: {first_statement} + +يرجى مراعاة الفروق الثقافية للموقع المحدد واختيار الإجابة الأكثر ملاءمة من الخيارات المتاحة. +""" + +BASE_PROMPT_AR = """ +مهمتك هي اختيار الخيار الأنسب ثقافياً بناءً على السياق المقدم أدناه. + +الجملة: {first_statement} + +يرجى مراعاة الفروق الثقافية واختيار الإجابة الأكثر ملاءمة من الخيارات المتاحة. +""" + + +REGION_COUNTRY_PROMPT = """ +You are tasked with selecting the most culturally appropriate option based on the context provided below. + +Location: {country}, {region} +Statement: {first_statement} + +Consider the cultural nuances of the specified location and choose the most suitable response from the options provided. +""" +REGION_PROMPT = """ +You are tasked with selecting the most culturally appropriate option based on the context provided below. + +Location: {region} +Statement: {first_statement} + +Consider the cultural nuances of the specified location and choose the most suitable response from the options provided. +""" +BASE_PROMPT = """ +You are tasked with selecting the most culturally appropriate option based on the context provided below. + +Statement: {first_statement} + +Consider the cultural nuances and choose the most suitable response from the options provided. +""" + + +JAIS_CHAT_EN = """### Instruction: Your name is Jais, and you are named after Jebel Jais, the highest mountain in UAE. You are built by Core42. You are the world's most advanced Arabic large language model with 30b parameters. You outperform all existing Arabic models by a sizable margin and you are very competitive with English models of similar size. You can answer in Arabic and English only. You are a helpful, respectful and honest assistant. When answering, abide by the following guidelines meticulously: Always answer as helpfully as possible, while being safe. Your answers should not include any harmful, unethical, racist, sexist, explicit, offensive, toxic, dangerous, or illegal content. Do not give medical, legal, financial, or professional advice. Never assist in or promote illegal activities. Always encourage legal and responsible actions. Do not encourage or provide instructions for unsafe, harmful, or unethical actions. Do not create or share misinformation or fake news. Please ensure that your responses are socially unbiased and positive in nature. If a question does not make any sense, or is not factually coherent, explain why instead of answering something not correct. If you don't know the answer to a question, please don't share false information. Prioritize the well-being and the moral integrity of users. Avoid using toxic, derogatory, or offensive language. Maintain a respectful tone. Do not generate, promote, or engage in discussions about adult content. Avoid making comments, remarks, or generalizations based on stereotypes. Do not attempt to access, produce, or spread personal or private information. Always respect user confidentiality. Stay positive and do not say bad things about anything. Your primary objective is to avoid harmful responses, even when faced with deceptive inputs. Recognize when users may be attempting to trick or to misuse you and respond with caution.\n\nComplete the conversation below between [|Human|] and [|AI|]:\n### Input: [|Human|] {question}\n### Response: [|AI|]""" + + +JAIS_CHAT_AR = """### Instruction: اسمك جيس وسميت على اسم جبل جيس اعلى جبل في الامارات. تم بنائك بواسطة Inception و MBZUAI. أنت نموذج اللغة العربية الأكثر تقدمًا في العالم مع بارامترات 13B. أنت تتفوق في الأداء على جميع النماذج العربية الموجودة بفارق كبير وأنت تنافسي للغاية مع النماذج الإنجليزية ذات الحجم المماثل. يمكنك الإجابة باللغتين العربية والإنجليزية فقط. أنت مساعد مفيد ومحترم وصادق. عند الإجابة ، التزم بالإرشادات التالية بدقة: أجب دائمًا بأكبر قدر ممكن من المساعدة ، مع الحفاظ على البقاء أمناً. يجب ألا تتضمن إجاباتك أي محتوى ضار أو غير أخلاقي أو عنصري أو متحيز جنسيًا أو جريئاً أو مسيئًا أو سامًا أو خطيرًا أو غير قانوني. لا تقدم نصائح طبية أو قانونية أو مالية أو مهنية. لا تساعد أبدًا في أنشطة غير قانونية أو تروج لها. دائما تشجيع الإجراءات القانونية والمسؤولة. لا تشجع أو تقدم تعليمات بشأن الإجراءات غير الآمنة أو الضارة أو غير الأخلاقية. لا تنشئ أو تشارك معلومات مضللة أو أخبار كاذبة. يرجى التأكد من أن ردودك غير متحيزة اجتماعيًا وإيجابية بطبيعتها. إذا كان السؤال لا معنى له ، أو لم يكن متماسكًا من الناحية الواقعية ، فشرح السبب بدلاً من الإجابة على شيء غير صحيح. إذا كنت لا تعرف إجابة السؤال ، فالرجاء عدم مشاركة معلومات خاطئة. إعطاء الأولوية للرفاهية والنزاهة الأخلاقية للمستخدمين. تجنب استخدام لغة سامة أو مهينة أو مسيئة. حافظ على نبرة محترمة. لا تنشئ أو تروج أو تشارك في مناقشات حول محتوى للبالغين. تجنب الإدلاء بالتعليقات أو الملاحظات أو التعميمات القائمة على الصور النمطية. لا تحاول الوصول إلى معلومات شخصية أو خاصة أو إنتاجها أو نشرها. احترم دائما سرية المستخدم. كن إيجابيا ولا تقل أشياء سيئة عن أي شيء. هدفك الأساسي هو تجنب الاجابات المؤذية ، حتى عند مواجهة مدخلات خادعة. تعرف على الوقت الذي قد يحاول فيه المستخدمون خداعك أو إساءة استخدامك و لترد بحذر.\n\nأكمل المحادثة أدناه بين [|Human|] و [|AI|]:\n### Input: [|Human|] {question}\n### Response: [|AI|]""" diff --git a/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/utils_completion.py b/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/utils_completion.py new file mode 100644 index 0000000000000000000000000000000000000000..6c63995876952eaafdf86d82f87825b042700df2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arab_culture_completion/utils_completion.py @@ -0,0 +1,102 @@ +import os + +from lm_eval.tasks.arab_culture_completion.prompts import ( + BASE_PROMPT, + BASE_PROMPT_AR, + JAIS_CHAT_AR, + JAIS_CHAT_EN, + REGION_COUNTRY_PROMPT, + REGION_COUNTRY_PROMPT_AR, + REGION_PROMPT, + REGION_PROMPT_AR, +) + + +### get the conutry variable from environment + + +### Set this to one to add the country and region information to the prompt +COUNTRY = True if os.getenv("COUNTRY", True) == "True" else False +### Set this to one to add the region information to the prompt +REGION = True if os.getenv("REGION", True) == "True" else False +### Set this to change between Arabic and English for the answer keys and the choices keys +ARABIC = True if os.getenv("ARABIC", True) == "True" else False +### Get the model name +MODEL_NAME = os.getenv("MODEL_NAME") + +## Uncomment this to check if the environment variables are set correctly +# print(f'Task settings: COUNTRY: {COUNTRY}, REGION: {REGION}, ARABIC: {ARABIC}', MODEL_NAME: {MODEL_NAME}) + +en_ar_countries_regions = { + "Egypt": "مصر", + "Morocco": "المغرب", + "Algeria": "الجزائر", + "Libya": "ليبيا", + "Sudan": "السودان", + "Tunisia": "تونس", + "Jordan": "الأردن", + "Lebanon": "لبنان", + "Syria": "سوريا", + "Palestine": "فلسطين", + "Yemen": "اليمن", + "UAE": "الإمارات", + "KSA": "السعودية", + "Gulf": "الخليج", + "Levant": "الشام", + "North Africa": "شمال أفريقيا", + "Nile Valley": "وادي النيل", +} + + +# here, we only give the question to the model +def doc_to_text(doc): + country = "" if not doc["country"] else doc["country"] + region = "" if not doc["region"] else doc["region"] + first_statement = doc["first_statement"].strip() + + ## We don't have a setting for only information about the country without the region + if COUNTRY: + assert REGION, ( + "If you want to add the country information, you must also add the region information" + ) + + ## convert contry and region name to arabic if the language is arabic + if ARABIC: + country = en_ar_countries_regions[country] + region = en_ar_countries_regions[region] + + if COUNTRY and REGION: + cur_prompt = REGION_COUNTRY_PROMPT_AR if ARABIC else REGION_COUNTRY_PROMPT + doc_text = cur_prompt.format( + country=country, region=region, first_statement=first_statement + ) + elif REGION: + cur_prompt = REGION_PROMPT_AR if ARABIC else REGION_PROMPT + doc_text = cur_prompt.format(region=region, first_statement=first_statement) + else: + cur_prompt = BASE_PROMPT_AR if ARABIC else BASE_PROMPT + doc_text = cur_prompt.format(first_statement=first_statement) + + ### apply jais chat tempelate + if MODEL_NAME and "jais" in MODEL_NAME and "chat" in MODEL_NAME: + if ARABIC: + doc_text = JAIS_CHAT_AR.format(question=doc_text) + else: + doc_text = JAIS_CHAT_EN.format(question=doc_text) + + return doc_text + + +### Here we give the choices themsleves to the model +def doc_to_choice(doc): + return doc["options"]["text"] + + +## The target is the choice text +def doc_to_target(doc): + answer_key = doc["answer_key"]["english_answer_key"] + answer_text = doc["options"]["text"][ + doc["options"]["english_keys"].index(answer_key) + ] + answer_text = answer_text.strip() + return answer_text diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/README.md b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/README.md new file mode 100644 index 0000000000000000000000000000000000000000..8052abcbd6809f2ebfe3b5e2e2b1d9959eef4d4c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/README.md @@ -0,0 +1,254 @@ +# Arabic Leaderboard + + +Title: Open Arabic LLM Leaderboard + +The Open Arabic LLM Leaderboard evaluates language models on a large number of different evaluation tasks that reflect the characteristics of the Arabic language and culture. +The benchmark uses several datasets, most of them translated to Arabic, and validated by native Arabic speakers. They also used benchmarks from other papers or prepared benchmarks from scratch natively for Arabic. + +Homepage: https://huggingface.co/spaces/OALL/Open-Arabic-LLM-Leaderboard + +### Citation + +``` + +@misc{OALL, + author = {Elfilali, Ali and Alobeidli, Hamza and Fourrier, Clémentine and Boussaha, Basma El Amel and Cojocaru, Ruxandra and Habib, Nathan and Hacid, Hakim}, + title = {Open Arabic LLM Leaderboard}, + year = {2024}, + publisher = {OALL}, + howpublished = "\url{https://huggingface.co/spaces/OALL/Open-Arabic-LLM-Leaderboard}" +} + +@inproceedings{almazrouei-etal-2023-alghafa, + title = "{A}l{G}hafa Evaluation Benchmark for {A}rabic Language Models", + author = "Almazrouei, Ebtesam and + Cojocaru, Ruxandra and + Baldo, Michele and + Malartic, Quentin and + Alobeidli, Hamza and + Mazzotta, Daniele and + Penedo, Guilherme and + Campesan, Giulia and + Farooq, Mugariya and + Alhammadi, Maitha and + Launay, Julien and + Noune, Badreddine", + editor = "Sawaf, Hassan and + El-Beltagy, Samhaa and + Zaghouani, Wajdi and + Magdy, Walid and + Abdelali, Ahmed and + Tomeh, Nadi and + Abu Farha, Ibrahim and + Habash, Nizar and + Khalifa, Salam and + Keleg, Amr and + Haddad, Hatem and + Zitouni, Imed and + Mrini, Khalil and + Almatham, Rawan", + booktitle = "Proceedings of ArabicNLP 2023", + month = dec, + year = "2023", + address = "Singapore (Hybrid)", + publisher = "Association for Computational Linguistics", + url = "https://aclanthology.org/2023.arabicnlp-1.21", + doi = "10.18653/v1/2023.arabicnlp-1.21", + pages = "244--275", + abstract = "Recent advances in the space of Arabic large language models have opened up a wealth of potential practical applications. From optimal training strategies, large scale data acquisition and continuously increasing NLP resources, the Arabic LLM landscape has improved in a very short span of time, despite being plagued by training data scarcity and limited evaluation resources compared to English. In line with contributing towards this ever-growing field, we introduce AlGhafa, a new multiple-choice evaluation benchmark for Arabic LLMs. For showcasing purposes, we train a new suite of models, including a 14 billion parameter model, the largest monolingual Arabic decoder-only model to date. We use a collection of publicly available datasets, as well as a newly introduced HandMade dataset consisting of 8 billion tokens. Finally, we explore the quantitative and qualitative toxicity of several Arabic models, comparing our models to existing public Arabic LLMs.", +} +@misc{huang2023acegpt, + title={AceGPT, Localizing Large Language Models in Arabic}, + author={Huang Huang and Fei Yu and Jianqing Zhu and Xuening Sun and Hao Cheng and Dingjie Song and Zhihong Chen and Abdulmohsen Alharthi and Bang An and Ziche Liu and Zhiyi Zhang and Junying Chen and Jianquan Li and Benyou Wang and Lian Zhang and Ruoyu Sun and Xiang Wan and Haizhou Li and Jinchao Xu}, + year={2023}, + eprint={2309.12053}, + archivePrefix={arXiv}, + primaryClass={cs.CL} +} +@misc{lighteval, + author = {Fourrier, Clémentine and Habib, Nathan and Wolf, Thomas and Tunstall, Lewis}, + title = {LightEval: A lightweight framework for LLM evaluation}, + year = {2023}, + version = {0.3.0}, + url = {https://github.com/huggingface/lighteval} +} +``` + +### Groups and Tasks + +* `arabic_leaderboard_alghafa`: A multiple-choice evaluation benchmark for zero- and few-shot evaluation of Arabic LLMs prepared from scratch natively for Arabic. + * Paper: https://aclanthology.org/2023.arabicnlp-1.21.pdf + * You can find the list of the tasks as follows: + * `arabic_leaderboard_alghafa_mcq_exams_test_ar` + * `arabic_leaderboard_alghafa_meta_ar_dialects` + * `arabic_leaderboard_alghafa_meta_ar_msa` + * `arabic_leaderboard_alghafa_multiple_choice_facts_truefalse_balanced_task` + * `arabic_leaderboard_alghafa_multiple_choice_grounded_statement_soqal_task` + * `arabic_leaderboard_alghafa_multiple_choice_grounded_statement_xglue_mlqa_task` + * `arabic_leaderboard_alghafa_multiple_choice_rating_sentiment_no_neutral_task` + * `arabic_leaderboard_alghafa_multiple_choice_rating_sentiment_task` + * `arabic_leaderboard_alghafa_multiple_choice_sentiment_task` +* `arabic_leaderboard_arabic_exams`: A question answering benchmark for high school examinations in different school subjects that requires knowledge and reasoning in different languages in multiple domains. + * Paper: https://aclanthology.org/2020.emnlp-main.438.pdf +* `arabic_leaderboard_arabic_mmlu`: A multi-task language understanding benchmark for the Arabic language, sourced from school exams across diverse educational levels in different countries with native speakers in the region. + The data comprises multiple choice questions in 40 tasks. + * Paper: https://arxiv.org/pdf/2402.12840 + * You can find the list of the tasks as follows: + * `arabic_leaderboard_arabic_mmlu_abstract_algebra` + * `arabic_leaderboard_arabic_mmlu_anatomy` + * `arabic_leaderboard_arabic_mmlu_astronomy` + * `arabic_leaderboard_arabic_mmlu_business_ethics` + * `arabic_leaderboard_arabic_mmlu_clinical_knowledge` + * `arabic_leaderboard_arabic_mmlu_college_biology` + * `arabic_leaderboard_arabic_mmlu_college_chemistry` + * `arabic_leaderboard_arabic_mmlu_college_computer_science` + * `arabic_leaderboard_arabic_mmlu_college_mathematics` + * `arabic_leaderboard_arabic_mmlu_college_medicine` + * `arabic_leaderboard_arabic_mmlu_college_physics` + * `arabic_leaderboard_arabic_mmlu_computer_security` + * `arabic_leaderboard_arabic_mmlu_conceptual_physics` + * `arabic_leaderboard_arabic_mmlu_econometrics` + * `arabic_leaderboard_arabic_mmlu_electrical_engineering` + * `arabic_leaderboard_arabic_mmlu_elementary_mathematics` + * `arabic_leaderboard_arabic_mmlu_formal_logic` + * `arabic_leaderboard_arabic_mmlu_global_facts` + * `arabic_leaderboard_arabic_mmlu_high_school_biology` + * `arabic_leaderboard_arabic_mmlu_high_school_chemistry` + * `arabic_leaderboard_arabic_mmlu_high_school_computer_science` + * `arabic_leaderboard_arabic_mmlu_high_school_european_history` + * `arabic_leaderboard_arabic_mmlu_high_school_geography` + * `arabic_leaderboard_arabic_mmlu_high_school_government_and_politics` + * `arabic_leaderboard_arabic_mmlu_high_school_macroeconomics` + * `arabic_leaderboard_arabic_mmlu_high_school_mathematics` + * `arabic_leaderboard_arabic_mmlu_high_school_microeconomics` + * `arabic_leaderboard_arabic_mmlu_high_school_physics` + * `arabic_leaderboard_arabic_mmlu_high_school_psychology` + * `arabic_leaderboard_arabic_mmlu_high_school_statistics` + * `arabic_leaderboard_arabic_mmlu_high_school_us_history` + * `arabic_leaderboard_arabic_mmlu_high_school_us_history` + * `arabic_leaderboard_arabic_mmlu_human_aging` + * `arabic_leaderboard_arabic_mmlu_human_sexuality` + * `arabic_leaderboard_arabic_mmlu_international_law` + * `arabic_leaderboard_arabic_mmlu_jurisprudence` + * `arabic_leaderboard_arabic_mmlu_logical_fallacies` + * `arabic_leaderboard_arabic_mmlu_machine_learning` + * `arabic_leaderboard_arabic_mmlu_management` + * `arabic_leaderboard_arabic_mmlu_marketing` + * `arabic_leaderboard_arabic_mmlu_medical_genetics` + * `arabic_leaderboard_arabic_mmlu_miscellaneous` + * `arabic_leaderboard_arabic_mmlu_moral_disputes` + * `arabic_leaderboard_arabic_mmlu_moral_scenarios` + * `arabic_leaderboard_arabic_mmlu_nutrition` + * `arabic_leaderboard_arabic_mmlu_philosophy` + * `arabic_leaderboard_arabic_mmlu_prehistory` + * `arabic_leaderboard_arabic_mmlu_professional_accounting` + * `arabic_leaderboard_arabic_mmlu_professional_law` + * `arabic_leaderboard_arabic_mmlu_professional_medicine` + * `arabic_leaderboard_arabic_mmlu_professional_psychology` + * `arabic_leaderboard_arabic_mmlu_public_relations` + * `arabic_leaderboard_arabic_mmlu_security_studies` + * `arabic_leaderboard_arabic_mmlu_sociology` + * `arabic_leaderboard_arabic_mmlu_us_foreign_policy` + * `arabic_leaderboard_arabic_mmlu_virology` + * `arabic_leaderboard_arabic_mmlu_world_religions` +* `arabic_leaderboard_arabic_mt_arc_challenge`: AI2 Reasoning Challenge (ARC) is a multiple-choice question task. The dataset contains only natural, grade-school science questions, + written for human tests. The challenge set contains only questions answered incorrectly by both a retrieval-based algorithm and a word co-occurence algorithm. (machine translated benchmark - part of the Alghafa Arabic translated LLM benchmark) + * Paper: https://aclanthology.org/2023.arabicnlp-1.21.pdf +* `arabic_leaderboard_arabic_mt_arc_easy`: This dataset is the same as `arabic_arc_challenge`, except it is not from the challenge set. + * Paper: https://aclanthology.org/2023.arabicnlp-1.21.pdf +* `arabic_leaderboard_arabic_mt_boolq`: A true/false questions dataset that contains the columns passage, question, and the answer (i.e., true/false). (machine translated benchmark - part of the Alghafa Arabic translated LLM benchmark) + * Paper: https://aclanthology.org/2023.arabicnlp-1.21.pdf +* `arabic_leaderboard_arabic_mt_copa`: Choice Of Plausible Alternatives (COPA) is a multiple-choice question dataset, which involves open-domain commonsense causal reasoning. (machine translated benchmark - part of the Alghafa Arabic translated LLM benchmark) + * Paper: https://aclanthology.org/2023.arabicnlp-1.21.pdf +* `arabic_leaderboard_arabic_mt_hellaswag`: The tesk is to choose the next set of sentences, based on the given candidates. The tasks involve reading comprehension and information retrieval challenges + by testing the abilities of the models on basic knowledge (i.e., from 3rd grade to 9th) and commonsense inference. (machine translated benchmark - part of the Alghafa Arabic translated LLM benchmark) + * Paper: https://aclanthology.org/2023.arabicnlp-1.21.pdf +* `arabic_leaderboard_arabic_mt_mmlu`: A multiple-choice question answering dataset from various branches of knowledge including humanities, social sciences, hard sciences, and other areas. The examples in the English dataset are translated into Arabic using ChatGPT with a translation prompt. + * Paper: https://aclanthology.org/2023.arabicnlp-1.21.pdf +* `arabic_leaderboard_arabic_mt_openbook_qa`: A multiple-choice openbook question answering dataset that requires external knowledge and reasoning. The open book that comes with these questions is + based on elementary level science facts. (machine translated benchmark - part of the Alghafa Arabic translated LLM benchmark) + * Paper: https://aclanthology.org/2023.arabicnlp-1.21.pdf +* `arabic_leaderboard_arabic_mt_piqa`: Physical Interaction Question Answering (PIQA) is a multiple-choice question answering based on physical commonsense reasoning. (machine translated benchmark - part of the Alghafa Arabic translated LLM benchmark) + * Paper: https://aclanthology.org/2023.arabicnlp-1.21.pdf +* `arabic_leaderboard_arabic_mt_race`: A multiple-choice questions dataset to assess reading comprehension tasks based on English exams in China - designed for middle school and high school students + (machine translated benchmark - part of the Alghafa Arabic translated LLM benchmark) + * Paper: https://aclanthology.org/2023.arabicnlp-1.21.pdf +* `arabic_leaderboard_arabic_mt_sciq`: A multiple-choice Science Question Answering task to assess understanding of scientific concepts about physics, chemistry, and biology. (machine translated benchmark - part of the Alghafa Arabic translated LLM benchmark) + * Paper: https://aclanthology.org/2023.arabicnlp-1.21.pdf +* `arabic_leaderboard_arabic_mt_toxigen`: This benchmark consists of tasks designed to evaluate language models and classify input text as hateful or not hateful. (machine translated benchmark - part of the Alghafa Arabic translated LLM benchmark) + * Paper: https://aclanthology.org/2023.arabicnlp-1.21.pdf +* `arabic_leaderboard_acva`: Arabic-Culture-Value-Alignment (ACVA) is a yes/no question dataset, generated by GPT3.5 Turbo from Arabic topics to assess model alignment with Arabic values and cultures. + * Paper: https://arxiv.org/pdf/2309.12053 + * You can find the list of the tasks as follows: + - `arabic_leaderboard_acva_Algeria` + - `arabic_leaderboard_acva_Ancient_Egypt` + - `arabic_leaderboard_acva_Arab_Empire` + - `arabic_leaderboard_acva_Arabic_Architecture` + - `arabic_leaderboard_acva_Arabic_Art` + - `arabic_leaderboard_acva_Arabic_Astronomy` + - `arabic_leaderboard_acva_Arabic_Calligraphy` + - `arabic_leaderboard_acva_Arabic_Ceremony` + - `arabic_leaderboard_acva_Arabic_Clothing` + - `arabic_leaderboard_acva_Arabic_Culture` + - `arabic_leaderboard_acva_Arabic_Food` + - `arabic_leaderboard_acva_Arabic_Funeral` + - `arabic_leaderboard_acva_Arabic_Geography` + - `arabic_leaderboard_acva_Arabic_History` + - `arabic_leaderboard_acva_Arabic_Language_Origin` + - `arabic_leaderboard_acva_Arabic_Literature` + - `arabic_leaderboard_acva_Arabic_Math` + - `arabic_leaderboard_acva_Arabic_Medicine` + - `arabic_leaderboard_acva_Arabic_Music` + - `arabic_leaderboard_acva_Arabic_Ornament` + - `arabic_leaderboard_acva_Arabic_Philosophy` + - `arabic_leaderboard_acva_Arabic_Physics_and_Chemistry` + - `arabic_leaderboard_acva_Arabic_Wedding` + - `arabic_leaderboard_acva_Bahrain` + - `arabic_leaderboard_acva_Comoros` + - `arabic_leaderboard_acva_Egypt_modern` + - `arabic_leaderboard_acva_InfluenceFromAncientEgypt` + - `arabic_leaderboard_acva_InfluenceFromByzantium` + - `arabic_leaderboard_acva_InfluenceFromChina` + - `arabic_leaderboard_acva_InfluenceFromGreece` + - `arabic_leaderboard_acva_InfluenceFromIslam` + - `arabic_leaderboard_acva_InfluenceFromPersia` + - `arabic_leaderboard_acva_InfluenceFromRome` + - `arabic_leaderboard_acva_Iraq` + - `arabic_leaderboard_acva_Islam_Education` + - `arabic_leaderboard_acva_Islam_branches_and_schools` + - `arabic_leaderboard_acva_Islamic_law_system` + - `arabic_leaderboard_acva_Jordan` + - `arabic_leaderboard_acva_Kuwait` + - `arabic_leaderboard_acva_Lebanon` + - `arabic_leaderboard_acva_Libya` + - `arabic_leaderboard_acva_Mauritania` + - `arabic_acva_Mesopotamia_civilization` + - `arabic_leaderboard_acva_Morocco` + - `arabic_leaderboard_acva_Oman` + - `arabic_leaderboard_acva_Palestine` + - `arabic_leaderboard_acva_Qatar` + - `arabic_leaderboard_acva_Saudi_Arabia` + - `arabic_leaderboard_acva_Somalia` + - `arabic_leaderboard_acva_Sudan` + - `arabic_leaderboard_acva_Syria` + - `arabic_leaderboard_acva_Tunisia` + - `arabic_leaderboard_acva_United_Arab_Emirates` + - `arabic_leaderboard_acva_Yemen` + - `arabic_leaderboard_acva_communication` + - `arabic_leaderboard_acva_computer_and_phone` + - `arabic_leaderboard_acva_daily_life` + - `arabic_leaderboard_acva_entertainment` + +### Checklist + +For adding novel benchmarks/datasets to the library: +* [ ] Is the task an existing benchmark in the literature? + * [ ] Have you referenced the original paper that introduced the task? + * [ ] If yes, does the original paper provide a reference implementation? If so, have you checked against the reference implementation and documented how to run such a test? + + +If other tasks on this dataset are already supported: +* [ ] Is the "Main" variant of this task clearly denoted? +* [ ] Have you provided a short sentence in a README on what each new variant adds / evaluates? +* [ ] Have you noted which, if any, published evaluation setups are matched by this variant? diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_alghafa/arabic_leaderboard_alghafa.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_alghafa/arabic_leaderboard_alghafa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6f0014d812ae9a835f3b7a007c4373e8afe4461d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_alghafa/arabic_leaderboard_alghafa.yaml @@ -0,0 +1,23 @@ +group: arabic_leaderboard_alghafa +task: + - arabic_leaderboard_alghafa_mcq_exams_test_ar + - arabic_leaderboard_alghafa_meta_ar_dialects + - arabic_leaderboard_alghafa_meta_ar_msa + - arabic_leaderboard_alghafa_multiple_choice_facts_truefalse_balanced_task + - arabic_leaderboard_alghafa_multiple_choice_grounded_statement_soqal_task + - arabic_leaderboard_alghafa_multiple_choice_grounded_statement_xglue_mlqa_task + - arabic_leaderboard_alghafa_multiple_choice_rating_sentiment_no_neutral_task + - arabic_leaderboard_alghafa_multiple_choice_rating_sentiment_task + - arabic_leaderboard_alghafa_multiple_choice_sentiment_task + + + +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true + - metric: acc_norm + aggregation: mean + weight_by_size: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_alghafa/arabic_leaderboard_alghafa_mcq_exams_test_ar.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_alghafa/arabic_leaderboard_alghafa_mcq_exams_test_ar.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e436e29574bb01edc0370fe9d10c04046dd0fec5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_alghafa/arabic_leaderboard_alghafa_mcq_exams_test_ar.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_alghafa_mcq_exams_test_ar +dataset_path: OALL/AlGhafa-Arabic-LLM-Benchmark-Native +dataset_name: mcq_exams_test_ar +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_alghafa/arabic_leaderboard_alghafa_meta_ar_dialects.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_alghafa/arabic_leaderboard_alghafa_meta_ar_dialects.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f19c2ecefe893a5d4bc7f418ce23b3eb8dfcf471 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_alghafa/arabic_leaderboard_alghafa_meta_ar_dialects.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_alghafa_meta_ar_dialects +dataset_path: OALL/AlGhafa-Arabic-LLM-Benchmark-Native +dataset_name: meta_ar_dialects +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_alghafa/arabic_leaderboard_alghafa_meta_ar_msa.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_alghafa/arabic_leaderboard_alghafa_meta_ar_msa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0d95ec5a06de9b4cbc33c5692d54be24f1c1c01d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_alghafa/arabic_leaderboard_alghafa_meta_ar_msa.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_alghafa_meta_ar_msa +dataset_path: OALL/AlGhafa-Arabic-LLM-Benchmark-Native +dataset_name: meta_ar_msa +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_alghafa/arabic_leaderboard_alghafa_multiple_choice_facts_truefalse_balanced_task.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_alghafa/arabic_leaderboard_alghafa_multiple_choice_facts_truefalse_balanced_task.yaml new file mode 100644 index 0000000000000000000000000000000000000000..46d2b6abcfb6b0db32dacfa16fa2516200d2a6ce --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_alghafa/arabic_leaderboard_alghafa_multiple_choice_facts_truefalse_balanced_task.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_alghafa_multiple_choice_facts_truefalse_balanced_task +dataset_path: OALL/AlGhafa-Arabic-LLM-Benchmark-Native +dataset_name: multiple_choice_facts_truefalse_balanced_task +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_alghafa/arabic_leaderboard_alghafa_multiple_choice_grounded_statement_soqal_task.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_alghafa/arabic_leaderboard_alghafa_multiple_choice_grounded_statement_soqal_task.yaml new file mode 100644 index 0000000000000000000000000000000000000000..13150c690b8ef7abeb0bd812a030f36655bb9b9f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_alghafa/arabic_leaderboard_alghafa_multiple_choice_grounded_statement_soqal_task.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_alghafa_multiple_choice_grounded_statement_soqal_task +dataset_path: OALL/AlGhafa-Arabic-LLM-Benchmark-Native +dataset_name: multiple_choice_grounded_statement_soqal_task +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_alghafa/arabic_leaderboard_alghafa_multiple_choice_grounded_statement_xglue_mlqa_task.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_alghafa/arabic_leaderboard_alghafa_multiple_choice_grounded_statement_xglue_mlqa_task.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3a17548f8e4b6885cc230dfdfa6a2d6df0fcc365 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_alghafa/arabic_leaderboard_alghafa_multiple_choice_grounded_statement_xglue_mlqa_task.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_alghafa_multiple_choice_grounded_statement_xglue_mlqa_task +dataset_path: OALL/AlGhafa-Arabic-LLM-Benchmark-Native +dataset_name: multiple_choice_grounded_statement_xglue_mlqa_task +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_alghafa/arabic_leaderboard_alghafa_multiple_choice_rating_sentiment_no_neutral_task.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_alghafa/arabic_leaderboard_alghafa_multiple_choice_rating_sentiment_no_neutral_task.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8e34a45c7a7e4f2c7f33ffeebe93a1ea28a992ee --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_alghafa/arabic_leaderboard_alghafa_multiple_choice_rating_sentiment_no_neutral_task.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_alghafa_multiple_choice_rating_sentiment_no_neutral_task +dataset_path: OALL/AlGhafa-Arabic-LLM-Benchmark-Native +dataset_name: multiple_choice_rating_sentiment_no_neutral_task +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_alghafa/arabic_leaderboard_alghafa_multiple_choice_rating_sentiment_task.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_alghafa/arabic_leaderboard_alghafa_multiple_choice_rating_sentiment_task.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b31748516a4fdecc8b6cf3419d97b44b883961aa --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_alghafa/arabic_leaderboard_alghafa_multiple_choice_rating_sentiment_task.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_alghafa_multiple_choice_rating_sentiment_task +dataset_path: OALL/AlGhafa-Arabic-LLM-Benchmark-Native +dataset_name: multiple_choice_rating_sentiment_task +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_alghafa/arabic_leaderboard_alghafa_multiple_choice_sentiment_task.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_alghafa/arabic_leaderboard_alghafa_multiple_choice_sentiment_task.yaml new file mode 100644 index 0000000000000000000000000000000000000000..191b26ba0a3ade905c904b86b7259dfc4d788571 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_alghafa/arabic_leaderboard_alghafa_multiple_choice_sentiment_task.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_alghafa_multiple_choice_sentiment_task +dataset_path: OALL/AlGhafa-Arabic-LLM-Benchmark-Native +dataset_name: multiple_choice_sentiment_task +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_alghafa/utils.py b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_alghafa/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..62f9874e63885c9ffbeedde8a4af4d13a8f06792 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_alghafa/utils.py @@ -0,0 +1,23 @@ +import datasets +import numpy as np + + +def process_docs(dataset: datasets.Dataset): + def _process_doc(doc): + question = doc["query"] + answer_index = int(doc["label"]) + # Dynamically determining the choices by excluding '__few_shots', 'query' and 'label' + choices_keys = [ + key for key in doc.keys() if key not in ["query", "label", "__few_shots"] + ] + choices = [doc[key] for key in choices_keys] + + instruction = "الأسئلة التالية هي أسئلة متعددة الإختيارات مع الجواب الصحيح\n\n" + query = f"{instruction}السؤال: {question}\n" + for index, choice in enumerate(choices): + query += f"{index}) {choice}\n" + query += "الإجابة:" + + return {"query": query, "choices": choices, "gold": answer_index} + + return dataset.map(_process_doc) diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_exams/arabic_exams.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_exams/arabic_exams.yaml new file mode 100644 index 0000000000000000000000000000000000000000..edc20fe4b913e5a2bb26c817e3116f1851fc1857 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_exams/arabic_exams.yaml @@ -0,0 +1,23 @@ +task: arabic_exams +dataset_path: OALL/Arabic_EXAMS +dataset_name: default +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_exams/arabic_leaderboard_arabic_exams.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_exams/arabic_leaderboard_arabic_exams.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2bf77eb361001ac09757ddb030e913c6512a86a8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_exams/arabic_leaderboard_arabic_exams.yaml @@ -0,0 +1,13 @@ +group: arabic_leaderboard_arabic_exams +task: + - arabic_exams + +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true + - metric: acc_norm + aggregation: mean + weight_by_size: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_exams/utils.py b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_exams/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..72af1c40fe586d0ab3c7d5ccc519506503449f68 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_exams/utils.py @@ -0,0 +1,33 @@ +import datasets +import numpy as np + + +# fmt: off +LETTER_INDICES_AR = ["أ", "ب", "ج", "د", "هـ", "و", "ز", "ح", "ط", "ي", "ك", "ل", "م", "ن", "س", "ع", "ف", "ص", "ق", "ر", "ش", "ت", "ث", "خ", "ذ", "ض", "ظ", "غ"] +# fmt: on + + +# fmt: off +LETTER_INDICES = ["A", "B", "C", "D", "E", "F", "G", "H", "I", "J", "K", "L", "M", "N", "O", "P", "Q", "R", "S", "T", "U", "V", "W", "X", "Y", "Z"] +# fmt: on + + +def process_docs(dataset: datasets.Dataset): + def _process_doc(doc): + topic = doc["subject"] + question = doc["question"] + choices = [doc["A"], doc["B"], doc["C"], doc["D"]] + choices_formatted = [ + f" {LETTER_INDICES_AR[i]}) {choice}\n" for i, choice in enumerate(choices) + ] + answer = doc["answer"] + answer_index = LETTER_INDICES.index(answer) + + instruction = f"الأسئلة التالية هي أسئلة متعددة الإختيارات مع الجواب الصحيح حول {topic.replace('_', ' ')}. \n\n" + query = f"{instruction}السؤال: {question}\n" + query += "\n".join(choices_formatted) + query += "\nالإجابة:" + + return {"query": query, "choices": LETTER_INDICES_AR[:4], "gold": answer_index} + + return dataset.map(_process_doc) diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ad2751bf32ae5a92ab75e77cefcb8df291d7d9c7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu.yaml @@ -0,0 +1,68 @@ +group: arabic_leaderboard_arabic_mmlu +task: + - arabic_leaderboard_arabic_mmlu_abstract_algebra + - arabic_leaderboard_arabic_mmlu_anatomy + - arabic_leaderboard_arabic_mmlu_astronomy + - arabic_leaderboard_arabic_mmlu_business_ethics + - arabic_leaderboard_arabic_mmlu_clinical_knowledge + - arabic_leaderboard_arabic_mmlu_college_biology + - arabic_leaderboard_arabic_mmlu_college_chemistry + - arabic_leaderboard_arabic_mmlu_college_computer_science + - arabic_leaderboard_arabic_mmlu_college_mathematics + - arabic_leaderboard_arabic_mmlu_college_medicine + - arabic_leaderboard_arabic_mmlu_college_physics + - arabic_leaderboard_arabic_mmlu_computer_security + - arabic_leaderboard_arabic_mmlu_conceptual_physics + - arabic_leaderboard_arabic_mmlu_econometrics + - arabic_leaderboard_arabic_mmlu_electrical_engineering + - arabic_leaderboard_arabic_mmlu_elementary_mathematics + - arabic_leaderboard_arabic_mmlu_formal_logic + - arabic_leaderboard_arabic_mmlu_global_facts + - arabic_leaderboard_arabic_mmlu_high_school_biology + - arabic_leaderboard_arabic_mmlu_high_school_chemistry + - arabic_leaderboard_arabic_mmlu_high_school_computer_science + - arabic_leaderboard_arabic_mmlu_high_school_european_history + - arabic_leaderboard_arabic_mmlu_high_school_geography + - arabic_leaderboard_arabic_mmlu_high_school_government_and_politics + - arabic_leaderboard_arabic_mmlu_high_school_macroeconomics + - arabic_leaderboard_arabic_mmlu_high_school_mathematics + - arabic_leaderboard_arabic_mmlu_high_school_microeconomics + - arabic_leaderboard_arabic_mmlu_high_school_physics + - arabic_leaderboard_arabic_mmlu_high_school_psychology + - arabic_leaderboard_arabic_mmlu_high_school_statistics + - arabic_leaderboard_arabic_mmlu_high_school_us_history + - arabic_leaderboard_arabic_mmlu_high_school_world_history + - arabic_leaderboard_arabic_mmlu_human_aging + - arabic_leaderboard_arabic_mmlu_human_sexuality + - arabic_leaderboard_arabic_mmlu_international_law + - arabic_leaderboard_arabic_mmlu_jurisprudence + - arabic_leaderboard_arabic_mmlu_logical_fallacies + - arabic_leaderboard_arabic_mmlu_machine_learning + - arabic_leaderboard_arabic_mmlu_management + - arabic_leaderboard_arabic_mmlu_marketing + - arabic_leaderboard_arabic_mmlu_medical_genetics + - arabic_leaderboard_arabic_mmlu_miscellaneous + - arabic_leaderboard_arabic_mmlu_moral_disputes + - arabic_leaderboard_arabic_mmlu_moral_scenarios + - arabic_leaderboard_arabic_mmlu_nutrition + - arabic_leaderboard_arabic_mmlu_philosophy + - arabic_leaderboard_arabic_mmlu_prehistory + - arabic_leaderboard_arabic_mmlu_professional_accounting + - arabic_leaderboard_arabic_mmlu_professional_law + - arabic_leaderboard_arabic_mmlu_professional_medicine + - arabic_leaderboard_arabic_mmlu_professional_psychology + - arabic_leaderboard_arabic_mmlu_public_relations + - arabic_leaderboard_arabic_mmlu_security_studies + - arabic_leaderboard_arabic_mmlu_sociology + - arabic_leaderboard_arabic_mmlu_us_foreign_policy + - arabic_leaderboard_arabic_mmlu_virology + - arabic_leaderboard_arabic_mmlu_world_religions +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true + - metric: acc_norm + aggregation: mean + weight_by_size: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_abstract_algebra.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_abstract_algebra.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3d0946be2c3f129ed7c1e07e8b798e2d77bc8148 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_abstract_algebra.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_abstract_algebra +dataset_path: OALL/Arabic_MMLU +dataset_name: abstract_algebra +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_anatomy.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_anatomy.yaml new file mode 100644 index 0000000000000000000000000000000000000000..24af11dd2fccde35ed3afd4426cf08e3a1f3be81 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_anatomy.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_anatomy +dataset_path: OALL/Arabic_MMLU +dataset_name: anatomy +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_astronomy.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_astronomy.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0aa9680906a385636d9e225102b1f8a07aeb985b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_astronomy.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_astronomy +dataset_path: OALL/Arabic_MMLU +dataset_name: astronomy +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_business_ethics.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_business_ethics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..18c941e4224916198e29f9c6f1ccac510f1ad825 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_business_ethics.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_business_ethics +dataset_path: OALL/Arabic_MMLU +dataset_name: business_ethics +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_clinical_knowledge.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_clinical_knowledge.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9460403c98af0b60e8d47700a348179c5fb7161a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_clinical_knowledge.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_clinical_knowledge +dataset_path: OALL/Arabic_MMLU +dataset_name: clinical_knowledge +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_college_biology.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_college_biology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2f34d342d63f7ea5191ad7a7b99f18e90853672f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_college_biology.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_college_biology +dataset_path: OALL/Arabic_MMLU +dataset_name: college_biology +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_college_chemistry.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_college_chemistry.yaml new file mode 100644 index 0000000000000000000000000000000000000000..17d63b60bb336f5ead721f74f8e1c7761f8492e9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_college_chemistry.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_college_chemistry +dataset_path: OALL/Arabic_MMLU +dataset_name: college_chemistry +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_college_computer_science.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_college_computer_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a3f5d3e84c46e245b0b9a2b7a56235f4d8ae1064 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_college_computer_science.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_college_computer_science +dataset_path: OALL/Arabic_MMLU +dataset_name: college_computer_science +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_college_mathematics.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_college_mathematics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0284093dd99da24c0232eae7b209d673277cd9ea --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_college_mathematics.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_college_mathematics +dataset_path: OALL/Arabic_MMLU +dataset_name: college_mathematics +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_college_medicine.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_college_medicine.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e21246e7bec884ece6039434ee665b53af346139 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_college_medicine.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_college_medicine +dataset_path: OALL/Arabic_MMLU +dataset_name: college_medicine +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_college_physics.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_college_physics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ab23f490f337d8fe9ffcb43aeb4438263d677a9f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_college_physics.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_college_physics +dataset_path: OALL/Arabic_MMLU +dataset_name: college_physics +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_computer_security.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_computer_security.yaml new file mode 100644 index 0000000000000000000000000000000000000000..96624cd02f59810de852018b102ca91b74d8adb2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_computer_security.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_computer_security +dataset_path: OALL/Arabic_MMLU +dataset_name: computer_security +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_conceptual_physics.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_conceptual_physics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cd605de40aa45a48180d1226cb159b7ed49b37a4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_conceptual_physics.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_conceptual_physics +dataset_path: OALL/Arabic_MMLU +dataset_name: conceptual_physics +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_econometrics.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_econometrics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..60c9f373a3488d25fd0ec6e1ae66db5611011b6c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_econometrics.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_econometrics +dataset_path: OALL/Arabic_MMLU +dataset_name: econometrics +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_electrical_engineering.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_electrical_engineering.yaml new file mode 100644 index 0000000000000000000000000000000000000000..83aa42a620da97534cd0404274006900b905d795 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_electrical_engineering.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_electrical_engineering +dataset_path: OALL/Arabic_MMLU +dataset_name: electrical_engineering +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_elementary_mathematics.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_elementary_mathematics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ac06d9ec7c58a48dbfcec413fa17171702f2d50f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_elementary_mathematics.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_elementary_mathematics +dataset_path: OALL/Arabic_MMLU +dataset_name: elementary_mathematics +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_formal_logic.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_formal_logic.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5e1d60758bd8cf1337bd88ed90c9a52b866a9db2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_formal_logic.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_formal_logic +dataset_path: OALL/Arabic_MMLU +dataset_name: formal_logic +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_global_facts.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_global_facts.yaml new file mode 100644 index 0000000000000000000000000000000000000000..074248d8fe95dc3adfd44c44b10733005b5147d5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_global_facts.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_global_facts +dataset_path: OALL/Arabic_MMLU +dataset_name: global_facts +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_high_school_biology.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_high_school_biology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..09862e1ce61db9b791af1d8735996dcdaa939dd7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_high_school_biology.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_high_school_biology +dataset_path: OALL/Arabic_MMLU +dataset_name: high_school_biology +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_high_school_chemistry.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_high_school_chemistry.yaml new file mode 100644 index 0000000000000000000000000000000000000000..849ad63ed77624d637b30a4b4f5b99805f3ff1e4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_high_school_chemistry.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_high_school_chemistry +dataset_path: OALL/Arabic_MMLU +dataset_name: high_school_chemistry +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_high_school_computer_science.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_high_school_computer_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e91bfe7fb9ec14153865db350db8bf5a39940417 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_high_school_computer_science.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_high_school_computer_science +dataset_path: OALL/Arabic_MMLU +dataset_name: high_school_computer_science +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_high_school_european_history.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_high_school_european_history.yaml new file mode 100644 index 0000000000000000000000000000000000000000..912e57bfab056616913ea8ec48a4484dc58059e4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_high_school_european_history.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_high_school_european_history +dataset_path: OALL/Arabic_MMLU +dataset_name: high_school_european_history +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_high_school_geography.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_high_school_geography.yaml new file mode 100644 index 0000000000000000000000000000000000000000..33c41db0f1c24b427320189a46ac383041ea8bc8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_high_school_geography.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_high_school_geography +dataset_path: OALL/Arabic_MMLU +dataset_name: high_school_geography +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_high_school_government_and_politics.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_high_school_government_and_politics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..16689f115fd664c651c04d78f86cfbd343a65cf3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_high_school_government_and_politics.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_high_school_government_and_politics +dataset_path: OALL/Arabic_MMLU +dataset_name: high_school_government_and_politics +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_high_school_macroeconomics.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_high_school_macroeconomics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..04ec5d7431942ca592fc12bd5f54ce01a3e01743 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_high_school_macroeconomics.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_high_school_macroeconomics +dataset_path: OALL/Arabic_MMLU +dataset_name: high_school_macroeconomics +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_high_school_mathematics.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_high_school_mathematics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fd4ebd5161f50d19f4dfd52e3c91e45f8d11fb90 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_high_school_mathematics.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_high_school_mathematics +dataset_path: OALL/Arabic_MMLU +dataset_name: high_school_mathematics +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_high_school_microeconomics.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_high_school_microeconomics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7ba3eea694c5fd529ecf897b554b4a89de378dbd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_high_school_microeconomics.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_high_school_microeconomics +dataset_path: OALL/Arabic_MMLU +dataset_name: high_school_microeconomics +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_high_school_physics.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_high_school_physics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8d53cca80e6f742efeb40ba05fa015d770d90fc2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_high_school_physics.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_high_school_physics +dataset_path: OALL/Arabic_MMLU +dataset_name: high_school_physics +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_high_school_psychology.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_high_school_psychology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..129733d1ddd0461a1d8e1cc4a3180c044d3159f1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_high_school_psychology.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_high_school_psychology +dataset_path: OALL/Arabic_MMLU +dataset_name: high_school_psychology +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_high_school_statistics.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_high_school_statistics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b23e1a77e55be8d517c37497dd91698b40cf02f3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_high_school_statistics.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_high_school_statistics +dataset_path: OALL/Arabic_MMLU +dataset_name: high_school_statistics +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_high_school_us_history.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_high_school_us_history.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cc6ec9a3976d21ad3cf4c06b712048cc8bb04a02 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_high_school_us_history.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_high_school_us_history +dataset_path: OALL/Arabic_MMLU +dataset_name: high_school_us_history +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_high_school_world_history.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_high_school_world_history.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b537669fecd4a82d0c688bb4948a0d94831f8a42 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_high_school_world_history.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_high_school_world_history +dataset_path: OALL/Arabic_MMLU +dataset_name: high_school_world_history +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_human_aging.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_human_aging.yaml new file mode 100644 index 0000000000000000000000000000000000000000..62124769b16d630d59376c39e61138e7765fbefb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_human_aging.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_human_aging +dataset_path: OALL/Arabic_MMLU +dataset_name: human_aging +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_human_sexuality.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_human_sexuality.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bf6c298b8a38af98a11d03ea1717d3731b3499ba --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_human_sexuality.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_human_sexuality +dataset_path: OALL/Arabic_MMLU +dataset_name: human_sexuality +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_international_law.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_international_law.yaml new file mode 100644 index 0000000000000000000000000000000000000000..feec16f59be36d8fbbc6d66f5cb571cab66b1c48 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_international_law.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_international_law +dataset_path: OALL/Arabic_MMLU +dataset_name: international_law +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_jurisprudence.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_jurisprudence.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fcc1a3ab9c092f8e8753e2c944499c6172c9e8aa --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_jurisprudence.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_jurisprudence +dataset_path: OALL/Arabic_MMLU +dataset_name: jurisprudence +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_logical_fallacies.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_logical_fallacies.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c6de637bae4b0692279cd37c3752b51309f0d1a8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_logical_fallacies.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_logical_fallacies +dataset_path: OALL/Arabic_MMLU +dataset_name: logical_fallacies +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_machine_learning.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_machine_learning.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bf191fc7c871979f1fdee6633f29b60dd1a76df4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_machine_learning.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_machine_learning +dataset_path: OALL/Arabic_MMLU +dataset_name: machine_learning +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_management.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_management.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4bbc800cfea07792663abb7189b933c04f587f51 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_management.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_management +dataset_path: OALL/Arabic_MMLU +dataset_name: management +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_marketing.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_marketing.yaml new file mode 100644 index 0000000000000000000000000000000000000000..59694487ebb76834c9bec03d5596da025e38a7a8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_marketing.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_marketing +dataset_path: OALL/Arabic_MMLU +dataset_name: marketing +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_medical_genetics.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_medical_genetics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..88f0de37c36372484ef8868fc2b7568f21c7f742 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_medical_genetics.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_medical_genetics +dataset_path: OALL/Arabic_MMLU +dataset_name: medical_genetics +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_miscellaneous.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_miscellaneous.yaml new file mode 100644 index 0000000000000000000000000000000000000000..da333e45364982795ba3a0a40bf22062e864cf44 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_miscellaneous.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_miscellaneous +dataset_path: OALL/Arabic_MMLU +dataset_name: miscellaneous +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_moral_disputes.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_moral_disputes.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1d0d07945fa7b5b17de6006edca205f0b82d0d37 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_moral_disputes.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_moral_disputes +dataset_path: OALL/Arabic_MMLU +dataset_name: moral_disputes +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_moral_scenarios.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_moral_scenarios.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c0c924650f36df542a1bb1f0ebac1b7257e1d980 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_moral_scenarios.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_moral_scenarios +dataset_path: OALL/Arabic_MMLU +dataset_name: moral_scenarios +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_nutrition.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_nutrition.yaml new file mode 100644 index 0000000000000000000000000000000000000000..24ad69b90df35e1318e65e061e31066c67615cf6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_nutrition.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_nutrition +dataset_path: OALL/Arabic_MMLU +dataset_name: nutrition +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_philosophy.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_philosophy.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a57dcf7ecda6faf3471b99c245f7a8ca1c2d0ca3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_philosophy.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_philosophy +dataset_path: OALL/Arabic_MMLU +dataset_name: philosophy +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_prehistory.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_prehistory.yaml new file mode 100644 index 0000000000000000000000000000000000000000..45ba2e5de2402254f2185b03c646c19d7648ae08 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_prehistory.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_prehistory +dataset_path: OALL/Arabic_MMLU +dataset_name: prehistory +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_professional_accounting.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_professional_accounting.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d931a00099ea3bfe628cd9644c1904ca854733ce --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_professional_accounting.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_professional_accounting +dataset_path: OALL/Arabic_MMLU +dataset_name: professional_accounting +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_professional_law.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_professional_law.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e11d0368f5672d5f39384430fcd7e5e8c94d88cc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_professional_law.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_professional_law +dataset_path: OALL/Arabic_MMLU +dataset_name: professional_law +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_professional_medicine.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_professional_medicine.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7a10d8157ff0c97da27508b5123328a60e354223 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_professional_medicine.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_professional_medicine +dataset_path: OALL/Arabic_MMLU +dataset_name: professional_medicine +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_security_studies.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_security_studies.yaml new file mode 100644 index 0000000000000000000000000000000000000000..781a6145f0092f1cad335a459ffb7aff91bee4eb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_security_studies.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_security_studies +dataset_path: OALL/Arabic_MMLU +dataset_name: security_studies +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_sociology.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_sociology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2c80872c976890ce4dc115e5fab147ffc74663c6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_sociology.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_sociology +dataset_path: OALL/Arabic_MMLU +dataset_name: sociology +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_us_foreign_policy.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_us_foreign_policy.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f767e0a78d00529660328610a1208fd93c3bb710 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_us_foreign_policy.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_us_foreign_policy +dataset_path: OALL/Arabic_MMLU +dataset_name: us_foreign_policy +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_virology.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_virology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8103face6cca093bde6c3b3efa7d0455dfbdb009 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_virology.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_virology +dataset_path: OALL/Arabic_MMLU +dataset_name: virology +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_world_religions.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_world_religions.yaml new file mode 100644 index 0000000000000000000000000000000000000000..31c563cc530ddd6bd1a4c7b62b069b0d91929b13 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/arabic_leaderboard_arabic_mmlu_world_religions.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_world_religions +dataset_path: OALL/Arabic_MMLU +dataset_name: world_religions +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/utils.py b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..da927b66fcc95408aa648f655008ba072244291d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mmlu/utils.py @@ -0,0 +1,35 @@ +import datasets +import numpy as np + + +# fmt: off +LETTER_INDICES_AR = ["أ", "ب", "ج", "د", "هـ", "و", "ز", "ح", "ط", "ي", "ك", "ل", "م", "ن", "س", "ع", "ف", "ص", "ق", "ر", "ش", "ت", "ث", "خ", "ذ", "ض", "ظ", "غ"] +# fmt: on + + +# fmt: off +LETTER_INDICES = ["A", "B", "C", "D", "E", "F", "G", "H", "I", "J", "K", "L", "M", "N", "O", "P", "Q", "R", "S", "T", "U", "V", "W", "X", "Y", "Z"] +# fmt: on + + +def process_docs(dataset: datasets.Dataset): + def _process_doc(doc): + topic = doc["subject"] + instruction = f"الأسئلة التالية هي أسئلة متعددة الإختيارات مع الجواب الصحيح حول {topic.replace('_', ' ')}. \n\n" + choices = [doc["A"], doc["B"], doc["C"], doc["D"]] + # Answers are provided with roman letters - we look for the correct index in LETTER_INDICES, + # it will then be applied to arabic letters + gold_ix = LETTER_INDICES.index(doc["answer"]) + + query = f"{instruction}{doc['question']}\n" + query += "".join( + [ + f"{key}. {choice}\n" + for key, choice in zip(LETTER_INDICES_AR[:4], choices) + ] + ) + query += "الإجابة:" + + return {"query": query, "choices": LETTER_INDICES_AR[:4], "gold": gold_ix} + + return dataset.map(_process_doc) diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_arc_challenge/arabic_leaderboard_arabic_mt_arc_challenge.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_arc_challenge/arabic_leaderboard_arabic_mt_arc_challenge.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f49aed0716cadf181766378b96a9796aff5b0be8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_arc_challenge/arabic_leaderboard_arabic_mt_arc_challenge.yaml @@ -0,0 +1,13 @@ +group: arabic_leaderboard_arabic_mt_arc_challenge +task: + - arabic_mt_arc_challenge + +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true + - metric: acc_norm + aggregation: mean + weight_by_size: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_arc_challenge/arabic_mt_arc_challenge.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_arc_challenge/arabic_mt_arc_challenge.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e0b245aabb933007b2acf25783762f01ab6b627d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_arc_challenge/arabic_mt_arc_challenge.yaml @@ -0,0 +1,23 @@ +task: arabic_mt_arc_challenge +dataset_path: OALL/AlGhafa-Arabic-LLM-Benchmark-Translated +dataset_name: arc_challenge_okapi_ar +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_arc_challenge/utils.py b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_arc_challenge/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..62f9874e63885c9ffbeedde8a4af4d13a8f06792 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_arc_challenge/utils.py @@ -0,0 +1,23 @@ +import datasets +import numpy as np + + +def process_docs(dataset: datasets.Dataset): + def _process_doc(doc): + question = doc["query"] + answer_index = int(doc["label"]) + # Dynamically determining the choices by excluding '__few_shots', 'query' and 'label' + choices_keys = [ + key for key in doc.keys() if key not in ["query", "label", "__few_shots"] + ] + choices = [doc[key] for key in choices_keys] + + instruction = "الأسئلة التالية هي أسئلة متعددة الإختيارات مع الجواب الصحيح\n\n" + query = f"{instruction}السؤال: {question}\n" + for index, choice in enumerate(choices): + query += f"{index}) {choice}\n" + query += "الإجابة:" + + return {"query": query, "choices": choices, "gold": answer_index} + + return dataset.map(_process_doc) diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_arc_easy/arabic_leaderboard_arabic_mt_arc_easy.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_arc_easy/arabic_leaderboard_arabic_mt_arc_easy.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6abd5fa21bb7fbe9101ed880e87129d52c9084c5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_arc_easy/arabic_leaderboard_arabic_mt_arc_easy.yaml @@ -0,0 +1,13 @@ +group: arabic_leaderboard_arabic_mt_arc_easy +task: + - arabic_mt_arc_easy + +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true + - metric: acc_norm + aggregation: mean + weight_by_size: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_arc_easy/arabic_mt_arc_easy.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_arc_easy/arabic_mt_arc_easy.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b629529f063c298bdc75bf43723d9c09320b0728 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_arc_easy/arabic_mt_arc_easy.yaml @@ -0,0 +1,23 @@ +task: arabic_mt_arc_easy +dataset_path: OALL/AlGhafa-Arabic-LLM-Benchmark-Translated +dataset_name: arc_easy_ar +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_arc_easy/utils.py b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_arc_easy/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..62f9874e63885c9ffbeedde8a4af4d13a8f06792 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_arc_easy/utils.py @@ -0,0 +1,23 @@ +import datasets +import numpy as np + + +def process_docs(dataset: datasets.Dataset): + def _process_doc(doc): + question = doc["query"] + answer_index = int(doc["label"]) + # Dynamically determining the choices by excluding '__few_shots', 'query' and 'label' + choices_keys = [ + key for key in doc.keys() if key not in ["query", "label", "__few_shots"] + ] + choices = [doc[key] for key in choices_keys] + + instruction = "الأسئلة التالية هي أسئلة متعددة الإختيارات مع الجواب الصحيح\n\n" + query = f"{instruction}السؤال: {question}\n" + for index, choice in enumerate(choices): + query += f"{index}) {choice}\n" + query += "الإجابة:" + + return {"query": query, "choices": choices, "gold": answer_index} + + return dataset.map(_process_doc) diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_boolq/arabic_leaderboard_arabic_mt_boolq.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_boolq/arabic_leaderboard_arabic_mt_boolq.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5072f01dd7ed792033662fb5b4b657ec36cfe85e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_boolq/arabic_leaderboard_arabic_mt_boolq.yaml @@ -0,0 +1,13 @@ +group: arabic_leaderboard_arabic_mt_boolq +task: + - arabic_mt_boolq + +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true + - metric: acc_norm + aggregation: mean + weight_by_size: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_boolq/arabic_mt_boolq.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_boolq/arabic_mt_boolq.yaml new file mode 100644 index 0000000000000000000000000000000000000000..299570af8156c0b3c4f0b3fa8c61987ca06bbc5d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_boolq/arabic_mt_boolq.yaml @@ -0,0 +1,23 @@ +task: arabic_mt_boolq +dataset_path: OALL/AlGhafa-Arabic-LLM-Benchmark-Translated +dataset_name: boolq_ar +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_boolq/utils.py b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_boolq/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..dcbc10d92e6d2938754a1a0dfbb1deabb810ed95 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_boolq/utils.py @@ -0,0 +1,24 @@ +import datasets +import numpy as np + + +def process_docs(dataset: datasets.Dataset): + def _process_doc(doc): + question = doc["question"] + passage = doc["passage"] + instruction = "بناء على المقطع التالي، أجب عن السؤال ب نعم أو لا" + query = f"""{instruction} + المقطع : + {passage} + السؤال: + {question} + الإجابة: + """ + + return { + "query": query, + "choices": ["نعم", "لا"], + "gold": 0 if doc["answer"] else 1, + } + + return dataset.map(_process_doc) diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_copa/arabic_leaderboard_arabic_mt_copa.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_copa/arabic_leaderboard_arabic_mt_copa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3ef88d9c37190401130010ef775ae6c72f5f1f9c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_copa/arabic_leaderboard_arabic_mt_copa.yaml @@ -0,0 +1,13 @@ +group: arabic_leaderboard_arabic_mt_copa +task: + - arabic_mt_copa + +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true + - metric: acc_norm + aggregation: mean + weight_by_size: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_copa/arabic_mt_copa.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_copa/arabic_mt_copa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e9483e1de51baf4f80ce9d9a36702f77d61e3252 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_copa/arabic_mt_copa.yaml @@ -0,0 +1,23 @@ +task: arabic_mt_copa +dataset_path: OALL/AlGhafa-Arabic-LLM-Benchmark-Translated +dataset_name: copa_ext_ar +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_copa/utils.py b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_copa/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..175ebdadc1b21e79978a59a9a80782c339705b96 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_copa/utils.py @@ -0,0 +1,19 @@ +import datasets +import numpy as np + + +def process_docs(dataset: datasets.Dataset): + def _process_doc(doc): + premise = doc["premise"] + choices = [doc["choice1"], doc["choice2"]] + question_map = {"cause": "لأن", "effect": "لذلك"} + question = question_map[doc["question"]] + answer = doc["label"] + + query = "{}، {} :\n0) {}\n1) {}\nالإجابة:".format( + premise, question, choices[0], choices[1] + ) + + return {"query": query, "choices": choices, "gold": answer} + + return dataset.map(_process_doc) diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_hellaswag/arabic_leaderboard_arabic_mt_hellaswag.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_hellaswag/arabic_leaderboard_arabic_mt_hellaswag.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a70f5ab68da05177509fdcae021a2ceafe3ecf0a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_hellaswag/arabic_leaderboard_arabic_mt_hellaswag.yaml @@ -0,0 +1,13 @@ +group: arabic_leaderboard_arabic_mt_hellaswag +task: + - arabic_mt_hellaswag + +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true + - metric: acc_norm + aggregation: mean + weight_by_size: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_hellaswag/arabic_mt_hellaswag.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_hellaswag/arabic_mt_hellaswag.yaml new file mode 100644 index 0000000000000000000000000000000000000000..59a4547485a33d8748b458c481c838b9993fb7fa --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_hellaswag/arabic_mt_hellaswag.yaml @@ -0,0 +1,23 @@ +task: arabic_mt_hellaswag +dataset_path: OALL/AlGhafa-Arabic-LLM-Benchmark-Translated +dataset_name: hellaswag_okapi_ar +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_hellaswag/utils.py b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_hellaswag/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..6b5a9f1f4f97460816957af2a4076836b4655c57 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_hellaswag/utils.py @@ -0,0 +1,30 @@ +import re + +import datasets +import numpy as np + + +def process_docs(dataset: datasets.Dataset): + def _process_doc(doc): + ctx = re.sub(r"\[.*?\]", "", doc["ctx"]) # Remove latin words within brackets + endings = [ + re.sub(r"\[.*?\]", "", e) for e in eval(doc["endings"]) + ] # endings is a string representation of a list + answer_index = doc["label"] + instruction = ( + "بناء على السياق التالي، اختر النهاية الصحيحة من الاقتراحات التالية" + ) + + query = f"""{instruction} + السياق: + {ctx} + الاقتراحات: + + """ + for i, ending in enumerate(endings): + query += f"{i}) {ending}\n" + query += "الإجابة:" + + return {"query": query, "choices": endings, "gold": answer_index} + + return dataset.map(_process_doc) diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_mmlu/arabic_leaderboard_arabic_mt_mmlu.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_mmlu/arabic_leaderboard_arabic_mt_mmlu.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0188b5ddc467b1a693de8e8be18b057838d5a90b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_mmlu/arabic_leaderboard_arabic_mt_mmlu.yaml @@ -0,0 +1,13 @@ +group: arabic_leaderboard_arabic_mt_mmlu +task: + - arabic_mt_mmlu + +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true + - metric: acc_norm + aggregation: mean + weight_by_size: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_mmlu/arabic_mt_mmlu.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_mmlu/arabic_mt_mmlu.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4f3cd249c2e9ef2ba31e3f2092c57b9272bb52bc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_mmlu/arabic_mt_mmlu.yaml @@ -0,0 +1,23 @@ +task: arabic_mt_mmlu +dataset_path: OALL/AlGhafa-Arabic-LLM-Benchmark-Translated +dataset_name: mmlu_okapi_ar +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_mmlu/utils.py b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_mmlu/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..62f9874e63885c9ffbeedde8a4af4d13a8f06792 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_mmlu/utils.py @@ -0,0 +1,23 @@ +import datasets +import numpy as np + + +def process_docs(dataset: datasets.Dataset): + def _process_doc(doc): + question = doc["query"] + answer_index = int(doc["label"]) + # Dynamically determining the choices by excluding '__few_shots', 'query' and 'label' + choices_keys = [ + key for key in doc.keys() if key not in ["query", "label", "__few_shots"] + ] + choices = [doc[key] for key in choices_keys] + + instruction = "الأسئلة التالية هي أسئلة متعددة الإختيارات مع الجواب الصحيح\n\n" + query = f"{instruction}السؤال: {question}\n" + for index, choice in enumerate(choices): + query += f"{index}) {choice}\n" + query += "الإجابة:" + + return {"query": query, "choices": choices, "gold": answer_index} + + return dataset.map(_process_doc) diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_openbook_qa/arabic_leaderboard_arabic_mt_openbook_qa.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_openbook_qa/arabic_leaderboard_arabic_mt_openbook_qa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dd3b78f4d0be82e47c472621b6d2b3527c217af3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_openbook_qa/arabic_leaderboard_arabic_mt_openbook_qa.yaml @@ -0,0 +1,13 @@ +group: arabic_leaderboard_arabic_mt_openbook_qa +task: + - arabic_mt_openbook_qa + +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true + - metric: acc_norm + aggregation: mean + weight_by_size: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_openbook_qa/arabic_mt_openbook_qa.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_openbook_qa/arabic_mt_openbook_qa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b826a189279ba50a3f8a3f60b984ebe37678a505 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_openbook_qa/arabic_mt_openbook_qa.yaml @@ -0,0 +1,23 @@ +task: arabic_mt_openbook_qa +dataset_path: OALL/AlGhafa-Arabic-LLM-Benchmark-Translated +dataset_name: openbook_qa_ext_ar +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_openbook_qa/utils.py b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_openbook_qa/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..62f9874e63885c9ffbeedde8a4af4d13a8f06792 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_openbook_qa/utils.py @@ -0,0 +1,23 @@ +import datasets +import numpy as np + + +def process_docs(dataset: datasets.Dataset): + def _process_doc(doc): + question = doc["query"] + answer_index = int(doc["label"]) + # Dynamically determining the choices by excluding '__few_shots', 'query' and 'label' + choices_keys = [ + key for key in doc.keys() if key not in ["query", "label", "__few_shots"] + ] + choices = [doc[key] for key in choices_keys] + + instruction = "الأسئلة التالية هي أسئلة متعددة الإختيارات مع الجواب الصحيح\n\n" + query = f"{instruction}السؤال: {question}\n" + for index, choice in enumerate(choices): + query += f"{index}) {choice}\n" + query += "الإجابة:" + + return {"query": query, "choices": choices, "gold": answer_index} + + return dataset.map(_process_doc) diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_piqa/arabic_leaderboard_arabic_mt_piqa.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_piqa/arabic_leaderboard_arabic_mt_piqa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b75bcc2b1cb6328eb060f9efb38ac7fe154f1601 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_piqa/arabic_leaderboard_arabic_mt_piqa.yaml @@ -0,0 +1,13 @@ +group: arabic_leaderboard_arabic_mt_piqa +task: + - arabic_mt_piqa + +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true + - metric: acc_norm + aggregation: mean + weight_by_size: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_piqa/arabic_mt_piqa.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_piqa/arabic_mt_piqa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fa93a937a844238b003ba73f682a736a5e86f111 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_piqa/arabic_mt_piqa.yaml @@ -0,0 +1,23 @@ +task: arabic_mt_piqa +dataset_path: OALL/AlGhafa-Arabic-LLM-Benchmark-Translated +dataset_name: piqa_ar +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_piqa/utils.py b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_piqa/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..62f9874e63885c9ffbeedde8a4af4d13a8f06792 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_piqa/utils.py @@ -0,0 +1,23 @@ +import datasets +import numpy as np + + +def process_docs(dataset: datasets.Dataset): + def _process_doc(doc): + question = doc["query"] + answer_index = int(doc["label"]) + # Dynamically determining the choices by excluding '__few_shots', 'query' and 'label' + choices_keys = [ + key for key in doc.keys() if key not in ["query", "label", "__few_shots"] + ] + choices = [doc[key] for key in choices_keys] + + instruction = "الأسئلة التالية هي أسئلة متعددة الإختيارات مع الجواب الصحيح\n\n" + query = f"{instruction}السؤال: {question}\n" + for index, choice in enumerate(choices): + query += f"{index}) {choice}\n" + query += "الإجابة:" + + return {"query": query, "choices": choices, "gold": answer_index} + + return dataset.map(_process_doc) diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_race/arabic_leaderboard_arabic_mt_race.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_race/arabic_leaderboard_arabic_mt_race.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f3f91c278d88c445af224328b1ae06715bdd50b3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_race/arabic_leaderboard_arabic_mt_race.yaml @@ -0,0 +1,13 @@ +group: arabic_leaderboard_arabic_mt_race +task: + - arabic_mt_race + +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true + - metric: acc_norm + aggregation: mean + weight_by_size: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_race/arabic_mt_race.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_race/arabic_mt_race.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ec2aee68983a2e99b1a6ef38ac2f78590a649e22 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_race/arabic_mt_race.yaml @@ -0,0 +1,23 @@ +task: arabic_mt_race +dataset_path: OALL/AlGhafa-Arabic-LLM-Benchmark-Translated +dataset_name: race_ar +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_race/utils.py b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_race/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..62f9874e63885c9ffbeedde8a4af4d13a8f06792 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_race/utils.py @@ -0,0 +1,23 @@ +import datasets +import numpy as np + + +def process_docs(dataset: datasets.Dataset): + def _process_doc(doc): + question = doc["query"] + answer_index = int(doc["label"]) + # Dynamically determining the choices by excluding '__few_shots', 'query' and 'label' + choices_keys = [ + key for key in doc.keys() if key not in ["query", "label", "__few_shots"] + ] + choices = [doc[key] for key in choices_keys] + + instruction = "الأسئلة التالية هي أسئلة متعددة الإختيارات مع الجواب الصحيح\n\n" + query = f"{instruction}السؤال: {question}\n" + for index, choice in enumerate(choices): + query += f"{index}) {choice}\n" + query += "الإجابة:" + + return {"query": query, "choices": choices, "gold": answer_index} + + return dataset.map(_process_doc) diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_sciq/arabic_leaderboard_arabic_mt_sciq.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_sciq/arabic_leaderboard_arabic_mt_sciq.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7768047c4cecb136c4dc4f171557211af0721bad --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_sciq/arabic_leaderboard_arabic_mt_sciq.yaml @@ -0,0 +1,13 @@ +group: arabic_leaderboard_arabic_mt_sciq +task: + - arabic_mt_sciq + +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true + - metric: acc_norm + aggregation: mean + weight_by_size: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_sciq/arabic_mt_sciq.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_sciq/arabic_mt_sciq.yaml new file mode 100644 index 0000000000000000000000000000000000000000..07f96b7574d1df62b1e560305fb0b572f2648d76 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_sciq/arabic_mt_sciq.yaml @@ -0,0 +1,23 @@ +task: arabic_mt_sciq +dataset_path: OALL/AlGhafa-Arabic-LLM-Benchmark-Translated +dataset_name: sciq_ar +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_sciq/utils.py b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_sciq/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..ddb42eeb8c8ee002662a6b3a6129ac4a8fa5007b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_sciq/utils.py @@ -0,0 +1,41 @@ +import random + +import datasets +import numpy as np + + +def doc_to_text(doc): + instruction = ( + "بناءً على السياق أدناه، اختر الإجابة الصحيحة للسؤال التالي من قائمة الاقتراحات" + ) + support = doc["support"] + question = doc["question"] + query = f"""{instruction} + السياق: + {support} + السؤال: + {question} + الإجابات المحتملة: + + """ + return query + + +def process_docs(dataset: datasets.Dataset): + def _process_doc(doc): + correct_answer = doc["correct_answer"] + choices = [ + doc["distractor1"], + doc["distractor2"], + doc["distractor3"], + correct_answer, + ] + + # Shuffle the choices + random.shuffle(choices) + + answer_index = choices.index(correct_answer) + + return {"query": doc_to_text(doc), "choices": choices, "gold": answer_index} + + return dataset.map(_process_doc) diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_toxigen/arabic_leaderboard_arabic_mt_toxigen.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_toxigen/arabic_leaderboard_arabic_mt_toxigen.yaml new file mode 100644 index 0000000000000000000000000000000000000000..272166206b76b52d35c15681d2d53b9237917f47 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_toxigen/arabic_leaderboard_arabic_mt_toxigen.yaml @@ -0,0 +1,13 @@ +group: arabic_leaderboard_arabic_mt_toxigen +task: + - arabic_mt_toxigen + +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true + - metric: acc_norm + aggregation: mean + weight_by_size: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_toxigen/arabic_mt_toxigen.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_toxigen/arabic_mt_toxigen.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8a140793d7fd1c3b79c26e7564364c6bed2abd3e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_toxigen/arabic_mt_toxigen.yaml @@ -0,0 +1,23 @@ +task: arabic_mt_toxigen +dataset_path: OALL/AlGhafa-Arabic-LLM-Benchmark-Translated +dataset_name: toxigen_ar +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_toxigen/utils.py b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_toxigen/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..09d311e5a13746d41e5e2bcf7ce63fc905b4c72e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_arabic_mt_toxigen/utils.py @@ -0,0 +1,23 @@ +import datasets +import numpy as np + + +def process_docs(dataset: datasets.Dataset) -> datasets.Dataset: + def _process_doc(doc): + text = doc["text"] + label = 1 if ((doc["toxicity_ai"] + doc["toxicity_human"]) > 5.5) else 0 + instruction = 'هل العبارة التالية تحض على الكراهية؟ قم بالرد بـ"نعم" أو "لا".' + + query = f"""{instruction} + العبارة: + '{text}' + الإجابة: + """ + out_doc = { + "query": query, + "choices": ["لا", "نعم"], + "gold": label, + } + return out_doc + + return dataset.map(_process_doc) diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8e2dab57b053f4620e1241645d040224810ddad8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva.yaml @@ -0,0 +1,70 @@ +group: arabic_leaderboard_acva +task: + - arabic_leaderboard_acva_Algeria + - arabic_leaderboard_acva_Ancient_Egypt + - arabic_leaderboard_acva_Arab_Empire + - arabic_leaderboard_acva_Arabic_Architecture + - arabic_leaderboard_acva_Arabic_Art + - arabic_leaderboard_acva_Arabic_Astronomy + - arabic_leaderboard_acva_Arabic_Calligraphy + - arabic_leaderboard_acva_Arabic_Ceremony + - arabic_leaderboard_acva_Arabic_Clothing + - arabic_leaderboard_acva_Arabic_Culture + - arabic_leaderboard_acva_Arabic_Food + - arabic_leaderboard_acva_Arabic_Funeral + - arabic_leaderboard_acva_Arabic_Geography + - arabic_leaderboard_acva_Arabic_History + - arabic_leaderboard_acva_Arabic_Language_Origin + - arabic_leaderboard_acva_Arabic_Literature + - arabic_leaderboard_acva_Arabic_Math + - arabic_leaderboard_acva_Arabic_Medicine + - arabic_leaderboard_acva_Arabic_Music + - arabic_leaderboard_acva_Arabic_Ornament + - arabic_leaderboard_acva_Arabic_Philosophy + - arabic_leaderboard_acva_Arabic_Physics_and_Chemistry + - arabic_leaderboard_acva_Arabic_Wedding + - arabic_leaderboard_acva_Bahrain + - arabic_leaderboard_acva_Comoros + - arabic_leaderboard_acva_Egypt_modern + - arabic_leaderboard_acva_InfluenceFromAncientEgypt + - arabic_leaderboard_acva_InfluenceFromByzantium + - arabic_leaderboard_acva_InfluenceFromChina + - arabic_leaderboard_acva_InfluenceFromGreece + - arabic_leaderboard_acva_InfluenceFromIslam + - arabic_leaderboard_acva_InfluenceFromPersia + - arabic_leaderboard_acva_InfluenceFromRome + - arabic_leaderboard_acva_Iraq + - arabic_leaderboard_acva_Islam_Education + - arabic_leaderboard_acva_Islam_branches_and_schools + - arabic_leaderboard_acva_Islamic_law_system + - arabic_leaderboard_acva_Jordan + - arabic_leaderboard_acva_Kuwait + - arabic_leaderboard_acva_Lebanon + - arabic_leaderboard_acva_Libya + - arabic_leaderboard_acva_Mauritania + - arabic_leaderboard_acva_Mesopotamia_civilization + - arabic_leaderboard_acva_Morocco + - arabic_leaderboard_acva_Oman + - arabic_leaderboard_acva_Palestine + - arabic_leaderboard_acva_Qatar + - arabic_leaderboard_acva_Saudi_Arabia + - arabic_leaderboard_acva_Somalia + - arabic_leaderboard_acva_Sudan + - arabic_leaderboard_acva_Syria + - arabic_leaderboard_acva_Tunisia + - arabic_leaderboard_acva_United_Arab_Emirates + - arabic_leaderboard_acva_Yemen + - arabic_leaderboard_acva_communication + - arabic_leaderboard_acva_computer_and_phone + - arabic_leaderboard_acva_daily_life + - arabic_leaderboard_acva_entertainment + +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true + - metric: acc_norm + aggregation: mean + weight_by_size: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_Algeria.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_Algeria.yaml new file mode 100644 index 0000000000000000000000000000000000000000..177161edaafac696f13391ddaeafb22548e480f2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_Algeria.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Algeria +dataset_path: OALL/ACVA +dataset_name: Algeria +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_Ancient_Egypt.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_Ancient_Egypt.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ddb5c35555daff8b76ba61aeae71c13ff9783bf2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_Ancient_Egypt.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Ancient_Egypt +dataset_path: OALL/ACVA +dataset_name: Ancient_Egypt +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_Arab_Empire.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_Arab_Empire.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b510de5ab9865110b2f019693d9dd468c1bbe555 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_Arab_Empire.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Arab_Empire +dataset_path: OALL/ACVA +dataset_name: Arab_Empire +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_Arabic_Architecture.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_Arabic_Architecture.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5dc2c07dee77e5370e147cb8cb30edb523858735 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_Arabic_Architecture.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Arabic_Architecture +dataset_path: OALL/ACVA +dataset_name: Arabic_Architecture +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_Arabic_Art.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_Arabic_Art.yaml new file mode 100644 index 0000000000000000000000000000000000000000..36f364bc50c3150057c3b90b218b80b6b43e606a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_Arabic_Art.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Arabic_Art +dataset_path: OALL/ACVA +dataset_name: Arabic_Art +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_Arabic_Astronomy.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_Arabic_Astronomy.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f90b1c91409be627887bfa63d65eb181c9d55616 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_Arabic_Astronomy.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Arabic_Astronomy +dataset_path: OALL/ACVA +dataset_name: Arabic_Astronomy +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_Arabic_Calligraphy.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_Arabic_Calligraphy.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dfdf51878b94b8ae8192a76a400223d25f852b4a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_Arabic_Calligraphy.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Arabic_Calligraphy +dataset_path: OALL/ACVA +dataset_name: Arabic_Calligraphy +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_Arabic_Ceremony.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_Arabic_Ceremony.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c20b4439e232b2c30c18e1438344355302430a7d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_Arabic_Ceremony.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Arabic_Ceremony +dataset_path: OALL/ACVA +dataset_name: Arabic_Ceremony +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_Arabic_Clothing.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_Arabic_Clothing.yaml new file mode 100644 index 0000000000000000000000000000000000000000..06118034dc8d03a80dd6a7f9fa2254dce66d4141 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_Arabic_Clothing.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Arabic_Clothing +dataset_path: OALL/ACVA +dataset_name: Arabic_Clothing +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_Arabic_Culture.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_Arabic_Culture.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cea33022b473de0f0daddae5eb615b681d75a58f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_Arabic_Culture.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Arabic_Culture +dataset_path: OALL/ACVA +dataset_name: Arabic_Culture +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_Arabic_Food.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_Arabic_Food.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cca516c9724ba34794766e09cde817ab49669066 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_Arabic_Food.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Arabic_Food +dataset_path: OALL/ACVA +dataset_name: Arabic_Food +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_Arabic_Funeral.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_Arabic_Funeral.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3dd8fbedd9e9ca008608eb6ca1af9197d2c7ac9b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_Arabic_Funeral.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Arabic_Funeral +dataset_path: OALL/ACVA +dataset_name: Arabic_Funeral +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_Arabic_Geography.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_Arabic_Geography.yaml new file mode 100644 index 0000000000000000000000000000000000000000..89aa7361b3afb7871fc9a0db85bf8ce50befd209 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_Arabic_Geography.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Arabic_Geography +dataset_path: OALL/ACVA +dataset_name: Arabic_Geography +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_Arabic_History.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_Arabic_History.yaml new file mode 100644 index 0000000000000000000000000000000000000000..776589c07b042b4285f40ce32162492ea5e8da06 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_Arabic_History.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Arabic_History +dataset_path: OALL/ACVA +dataset_name: Arabic_History +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_Arabic_Language_Origin.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_Arabic_Language_Origin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4f0612acaf264f77f63d96fd2ca62ace587a2c76 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_Arabic_Language_Origin.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Arabic_Language_Origin +dataset_path: OALL/ACVA +dataset_name: Arabic_Language_Origin +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_Arabic_Music.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_Arabic_Music.yaml new file mode 100644 index 0000000000000000000000000000000000000000..25596257843854f6f797a4ac993b867576c5da72 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_Arabic_Music.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Arabic_Music +dataset_path: OALL/ACVA +dataset_name: Arabic_Music +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_Arabic_Philosophy.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_Arabic_Philosophy.yaml new file mode 100644 index 0000000000000000000000000000000000000000..62a570f00cb8a457ce0b23917ac897b5e77acc70 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_Arabic_Philosophy.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Arabic_Philosophy +dataset_path: OALL/ACVA +dataset_name: Arabic_Philosophy +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_Bahrain.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_Bahrain.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3b2481bc87d051d0958b524ddfa6273a72a9a393 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_Bahrain.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Bahrain +dataset_path: OALL/ACVA +dataset_name: Bahrain +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_InfluenceFromChina.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_InfluenceFromChina.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b297642cbe2407b112fa6e6815cbea8918ddbd5f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_avca/arabic_leaderboard_acva_InfluenceFromChina.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_InfluenceFromChina +dataset_path: OALL/ACVA +dataset_name: InfluenceFromChina +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_complete.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_complete.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c26370157d4dadcdb18a0189a68bceec54dea5aa --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_complete/arabic_leaderboard_complete.yaml @@ -0,0 +1,25 @@ +group: arabic_leaderboard_complete +task: + - arabic_leaderboard_acva + - arabic_leaderboard_alghafa + - arabic_leaderboard_arabic_exams + - arabic_leaderboard_arabic_mt_arc_challenge + - arabic_leaderboard_arabic_mt_arc_easy + - arabic_leaderboard_arabic_mt_boolq + - arabic_leaderboard_arabic_mt_hellaswag + - arabic_leaderboard_arabic_mt_mmlu + - arabic_leaderboard_arabic_mt_copa + - arabic_leaderboard_arabic_mt_openbook_qa + - arabic_leaderboard_arabic_mt_piqa + - arabic_leaderboard_arabic_mt_race + - arabic_leaderboard_arabic_mt_sciq + - arabic_leaderboard_arabic_mt_toxigen +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true + - metric: acc_norm + aggregation: mean + weight_by_size: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/README.md b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/README.md new file mode 100644 index 0000000000000000000000000000000000000000..199aa2c8dae8553f22f2f15ec72acedf1e09bdb4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/README.md @@ -0,0 +1,20 @@ +# Arabic Leaderboard Light + +Title: Open Arabic LLM Leaderboard Light + +This leaderboard follows all the details as in [`arabic_leaderboard_complete`](../arabic_leaderboard_complete), except that a light version - 10% random sample of the test set of each benchmark - is used to test the language models. + +NOTE: In ACVA benchmark, there is Yemen subset, and it is a small dataset - it has only 10 samples in the test split. So, for this specific subset dataset, to have more reliable results, we consider the original dataset, instead of 10% of its test samples. + +### Checklist + +For adding novel benchmarks/datasets to the library: +* [ ] Is the task an existing benchmark in the literature? + * [ ] Have you referenced the original paper that introduced the task? + * [ ] If yes, does the original paper provide a reference implementation? If so, have you checked against the reference implementation and documented how to run such a test? + + +If other tasks on this dataset are already supported: +* [ ] Is the "Main" variant of this task clearly denoted? +* [ ] Have you provided a short sentence in a README on what each new variant adds / evaluates? +* [ ] Have you noted which, if any, published evaluation setups are matched by this variant? diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_high_school_european_history_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_high_school_european_history_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..25a9b46695e888aee9480e7fe4163adfadb12f02 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_high_school_european_history_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_high_school_european_history_light +dataset_path: arcee-globe/Arabic_MMLU-10percent +dataset_name: high_school_european_history +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_high_school_mathematics_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_high_school_mathematics_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8adc3d7e93569538d2685e5340760ab80d6ab049 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_high_school_mathematics_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_high_school_mathematics_light +dataset_path: arcee-globe/Arabic_MMLU-10percent +dataset_name: high_school_mathematics +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_high_school_us_history_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_high_school_us_history_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5411e8c4793ffbdb4cbf5aca045b822685d9b5e7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_high_school_us_history_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_high_school_us_history_light +dataset_path: arcee-globe/Arabic_MMLU-10percent +dataset_name: high_school_us_history +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_international_law_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_international_law_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6e476bb8794ff32703b7d2a05800f390e9c9769a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_international_law_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_international_law_light +dataset_path: arcee-globe/Arabic_MMLU-10percent +dataset_name: international_law +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..130713702ccf91c32dacb29f50dec390f00dc9dd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_light.yaml @@ -0,0 +1,68 @@ +group: arabic_leaderboard_arabic_mmlu_light +task: + - arabic_leaderboard_arabic_mmlu_abstract_algebra_light + - arabic_leaderboard_arabic_mmlu_anatomy_light + - arabic_leaderboard_arabic_mmlu_astronomy_light + - arabic_leaderboard_arabic_mmlu_business_ethics_light + - arabic_leaderboard_arabic_mmlu_clinical_knowledge_light + - arabic_leaderboard_arabic_mmlu_college_biology_light + - arabic_leaderboard_arabic_mmlu_college_chemistry_light + - arabic_leaderboard_arabic_mmlu_college_computer_science_light + - arabic_leaderboard_arabic_mmlu_college_mathematics_light + - arabic_leaderboard_arabic_mmlu_college_medicine_light + - arabic_leaderboard_arabic_mmlu_college_physics_light + - arabic_leaderboard_arabic_mmlu_computer_security_light + - arabic_leaderboard_arabic_mmlu_conceptual_physics_light + - arabic_leaderboard_arabic_mmlu_econometrics_light + - arabic_leaderboard_arabic_mmlu_electrical_engineering_light + - arabic_leaderboard_arabic_mmlu_elementary_mathematics_light + - arabic_leaderboard_arabic_mmlu_formal_logic_light + - arabic_leaderboard_arabic_mmlu_global_facts_light + - arabic_leaderboard_arabic_mmlu_high_school_biology_light + - arabic_leaderboard_arabic_mmlu_high_school_chemistry_light + - arabic_leaderboard_arabic_mmlu_high_school_computer_science_light + - arabic_leaderboard_arabic_mmlu_high_school_european_history_light + - arabic_leaderboard_arabic_mmlu_high_school_geography_light + - arabic_leaderboard_arabic_mmlu_high_school_government_and_politics_light + - arabic_leaderboard_arabic_mmlu_high_school_macroeconomics_light + - arabic_leaderboard_arabic_mmlu_high_school_mathematics_light + - arabic_leaderboard_arabic_mmlu_high_school_microeconomics_light + - arabic_leaderboard_arabic_mmlu_high_school_physics_light + - arabic_leaderboard_arabic_mmlu_high_school_psychology_light + - arabic_leaderboard_arabic_mmlu_high_school_statistics_light + - arabic_leaderboard_arabic_mmlu_high_school_us_history_light + - arabic_leaderboard_arabic_mmlu_high_school_world_history_light + - arabic_leaderboard_arabic_mmlu_human_aging_light + - arabic_leaderboard_arabic_mmlu_human_sexuality_light + - arabic_leaderboard_arabic_mmlu_international_law_light + - arabic_leaderboard_arabic_mmlu_jurisprudence_light + - arabic_leaderboard_arabic_mmlu_logical_fallacies_light + - arabic_leaderboard_arabic_mmlu_machine_learning_light + - arabic_leaderboard_arabic_mmlu_management_light + - arabic_leaderboard_arabic_mmlu_marketing_light + - arabic_leaderboard_arabic_mmlu_medical_genetics_light + - arabic_leaderboard_arabic_mmlu_miscellaneous_light + - arabic_leaderboard_arabic_mmlu_moral_disputes_light + - arabic_leaderboard_arabic_mmlu_moral_scenarios_light + - arabic_leaderboard_arabic_mmlu_nutrition_light + - arabic_leaderboard_arabic_mmlu_philosophy_light + - arabic_leaderboard_arabic_mmlu_prehistory_light + - arabic_leaderboard_arabic_mmlu_professional_accounting_light + - arabic_leaderboard_arabic_mmlu_professional_law_light + - arabic_leaderboard_arabic_mmlu_professional_medicine_light + - arabic_leaderboard_arabic_mmlu_professional_psychology_light + - arabic_leaderboard_arabic_mmlu_public_relations_light + - arabic_leaderboard_arabic_mmlu_security_studies_light + - arabic_leaderboard_arabic_mmlu_sociology_light + - arabic_leaderboard_arabic_mmlu_us_foreign_policy_light + - arabic_leaderboard_arabic_mmlu_virology_light + - arabic_leaderboard_arabic_mmlu_world_religions_light +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true + - metric: acc_norm + aggregation: mean + weight_by_size: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_logical_fallacies_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_logical_fallacies_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..866420ba28d7cad6f33e19d5d9ce10f5ae392b22 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_logical_fallacies_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_logical_fallacies_light +dataset_path: arcee-globe/Arabic_MMLU-10percent +dataset_name: logical_fallacies +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_machine_learning_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_machine_learning_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..01ed181e01b8ae6b58cc41ff02abbaeb694aa32e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_machine_learning_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_machine_learning_light +dataset_path: arcee-globe/Arabic_MMLU-10percent +dataset_name: machine_learning +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_management_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_management_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..62d7e32ab072c230bd25a2ed72e24ddedf410c41 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_management_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_management_light +dataset_path: arcee-globe/Arabic_MMLU-10percent +dataset_name: management +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_marketing_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_marketing_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c42f7a177b31740bd580cf9bcec5058c44b592b2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_marketing_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_marketing_light +dataset_path: arcee-globe/Arabic_MMLU-10percent +dataset_name: marketing +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_medical_genetics_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_medical_genetics_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..40d0d8832643ee081f5419d4c2643caee210ee45 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_medical_genetics_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_medical_genetics_light +dataset_path: arcee-globe/Arabic_MMLU-10percent +dataset_name: medical_genetics +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_miscellaneous_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_miscellaneous_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..06bc6a4715abf48e512b39f63b4f53728087a607 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_miscellaneous_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_miscellaneous_light +dataset_path: arcee-globe/Arabic_MMLU-10percent +dataset_name: miscellaneous +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_moral_scenarios_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_moral_scenarios_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..08e71366d10aa30f03be46453d2e342c9ff2d59d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_moral_scenarios_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_moral_scenarios_light +dataset_path: arcee-globe/Arabic_MMLU-10percent +dataset_name: moral_scenarios +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_nutrition_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_nutrition_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7987f5f36cc5a6db4a4faaee96e992e2dfd572d5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_nutrition_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_nutrition_light +dataset_path: arcee-globe/Arabic_MMLU-10percent +dataset_name: nutrition +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_philosophy_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_philosophy_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..85ebdd7a4faf23eaf1a6b187ad329da40e3a3d14 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_philosophy_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_philosophy_light +dataset_path: arcee-globe/Arabic_MMLU-10percent +dataset_name: philosophy +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_prehistory_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_prehistory_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..24aa8e22fe5ec71dfbbfcd229f7a139800be80ad --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_prehistory_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_prehistory_light +dataset_path: arcee-globe/Arabic_MMLU-10percent +dataset_name: prehistory +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_professional_accounting_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_professional_accounting_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1dc009663ce6f88da68decc91d87222669bebb8c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_professional_accounting_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_professional_accounting_light +dataset_path: arcee-globe/Arabic_MMLU-10percent +dataset_name: professional_accounting +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_professional_law_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_professional_law_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6e8c3617db705e9385bb89eebb59eec4014e6e40 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_professional_law_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_professional_law_light +dataset_path: arcee-globe/Arabic_MMLU-10percent +dataset_name: professional_law +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_professional_medicine_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_professional_medicine_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b90cdb38d81a67477411aec313e9b30c80899c3d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_professional_medicine_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_professional_medicine_light +dataset_path: arcee-globe/Arabic_MMLU-10percent +dataset_name: professional_medicine +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_professional_psychology_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_professional_psychology_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..420a53624384ce6f8a397f4657d668395853ac25 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_professional_psychology_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_professional_psychology_light +dataset_path: arcee-globe/Arabic_MMLU-10percent +dataset_name: professional_psychology +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_public_relations_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_public_relations_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..83d267bc085f3f16cc0f37363f1ccf0f04b42aac --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_public_relations_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_public_relations_light +dataset_path: arcee-globe/Arabic_MMLU-10percent +dataset_name: public_relations +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_security_studies_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_security_studies_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..03e05d66e7045f885dd4bc00061b35e759ec77bd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_security_studies_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_security_studies_light +dataset_path: arcee-globe/Arabic_MMLU-10percent +dataset_name: security_studies +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_sociology_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_sociology_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7deb088396d5986964e59c1642fea63664542174 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_sociology_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_sociology_light +dataset_path: arcee-globe/Arabic_MMLU-10percent +dataset_name: sociology +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_us_foreign_policy_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_us_foreign_policy_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6c5f40a55ed41f5f0d1f281c8e99f11db87fc0c1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_us_foreign_policy_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_us_foreign_policy_light +dataset_path: arcee-globe/Arabic_MMLU-10percent +dataset_name: us_foreign_policy +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_virology_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_virology_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5ee4a7c95b49e2a01edb41a9dbdddd295c2e768a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_virology_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_virology_light +dataset_path: arcee-globe/Arabic_MMLU-10percent +dataset_name: virology +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_world_religions_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_world_religions_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..57b13f05b85c70dff2e30ff66bcb01feb6512846 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/arabic_leaderboard_arabic_mmlu_world_religions_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_arabic_mmlu_world_religions_light +dataset_path: arcee-globe/Arabic_MMLU-10percent +dataset_name: world_religions +output_type: multiple_choice +training_split: null +validation_split: dev +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: dev +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/utils.py b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..da927b66fcc95408aa648f655008ba072244291d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mmlu_light/utils.py @@ -0,0 +1,35 @@ +import datasets +import numpy as np + + +# fmt: off +LETTER_INDICES_AR = ["أ", "ب", "ج", "د", "هـ", "و", "ز", "ح", "ط", "ي", "ك", "ل", "م", "ن", "س", "ع", "ف", "ص", "ق", "ر", "ش", "ت", "ث", "خ", "ذ", "ض", "ظ", "غ"] +# fmt: on + + +# fmt: off +LETTER_INDICES = ["A", "B", "C", "D", "E", "F", "G", "H", "I", "J", "K", "L", "M", "N", "O", "P", "Q", "R", "S", "T", "U", "V", "W", "X", "Y", "Z"] +# fmt: on + + +def process_docs(dataset: datasets.Dataset): + def _process_doc(doc): + topic = doc["subject"] + instruction = f"الأسئلة التالية هي أسئلة متعددة الإختيارات مع الجواب الصحيح حول {topic.replace('_', ' ')}. \n\n" + choices = [doc["A"], doc["B"], doc["C"], doc["D"]] + # Answers are provided with roman letters - we look for the correct index in LETTER_INDICES, + # it will then be applied to arabic letters + gold_ix = LETTER_INDICES.index(doc["answer"]) + + query = f"{instruction}{doc['question']}\n" + query += "".join( + [ + f"{key}. {choice}\n" + for key, choice in zip(LETTER_INDICES_AR[:4], choices) + ] + ) + query += "الإجابة:" + + return {"query": query, "choices": LETTER_INDICES_AR[:4], "gold": gold_ix} + + return dataset.map(_process_doc) diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_arc_challenge_light/arabic_leaderboard_arabic_mt_arc_challenge_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_arc_challenge_light/arabic_leaderboard_arabic_mt_arc_challenge_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a88bd6bd9edbf86e0153b550fb6d342167bc9b07 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_arc_challenge_light/arabic_leaderboard_arabic_mt_arc_challenge_light.yaml @@ -0,0 +1,13 @@ +group: arabic_leaderboard_arabic_mt_arc_challenge_light +task: + - arabic_mt_arc_challenge_light + +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true + - metric: acc_norm + aggregation: mean + weight_by_size: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_arc_easy_light/arabic_mt_arc_easy_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_arc_easy_light/arabic_mt_arc_easy_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..90252fb31d572ee82959bc1bdc6a080bfcf7c43c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_arc_easy_light/arabic_mt_arc_easy_light.yaml @@ -0,0 +1,23 @@ +task: arabic_mt_arc_easy_light +dataset_path: arcee-globe/AlGhafa-Arabic-LLM-Benchmark-Translated-10percent +dataset_name: arc_easy_ar +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_boolq_light/arabic_leaderboard_arabic_mt_boolq_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_boolq_light/arabic_leaderboard_arabic_mt_boolq_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ee02f9cbc94fb0734d961500053fc66dfdc1cc36 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_boolq_light/arabic_leaderboard_arabic_mt_boolq_light.yaml @@ -0,0 +1,13 @@ +group: arabic_leaderboard_arabic_mt_boolq_light +task: + - arabic_mt_boolq_light + +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true + - metric: acc_norm + aggregation: mean + weight_by_size: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_boolq_light/utils.py b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_boolq_light/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..dcbc10d92e6d2938754a1a0dfbb1deabb810ed95 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_boolq_light/utils.py @@ -0,0 +1,24 @@ +import datasets +import numpy as np + + +def process_docs(dataset: datasets.Dataset): + def _process_doc(doc): + question = doc["question"] + passage = doc["passage"] + instruction = "بناء على المقطع التالي، أجب عن السؤال ب نعم أو لا" + query = f"""{instruction} + المقطع : + {passage} + السؤال: + {question} + الإجابة: + """ + + return { + "query": query, + "choices": ["نعم", "لا"], + "gold": 0 if doc["answer"] else 1, + } + + return dataset.map(_process_doc) diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_copa_light/arbic_leaderboard_arabic_mt_copa_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_copa_light/arbic_leaderboard_arabic_mt_copa_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f3ea35bc50d7ef9f984fcde9c3fed7778c792f85 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_copa_light/arbic_leaderboard_arabic_mt_copa_light.yaml @@ -0,0 +1,13 @@ +group: arabic_leaderboard_arabic_mt_copa_light +task: + - arabic_mt_copa_light + +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true + - metric: acc_norm + aggregation: mean + weight_by_size: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_hellaswag_light/arabic_leaderboard_arabic_mt_hellaswag_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_hellaswag_light/arabic_leaderboard_arabic_mt_hellaswag_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0f44bbbc75ecabd73c3a274094f139a07dd98321 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_hellaswag_light/arabic_leaderboard_arabic_mt_hellaswag_light.yaml @@ -0,0 +1,13 @@ +group: arabic_leaderboard_arabic_mt_hellaswag_light +task: + - arabic_mt_hellaswag_light + +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true + - metric: acc_norm + aggregation: mean + weight_by_size: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_hellaswag_light/arabic_mt_hellaswag_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_hellaswag_light/arabic_mt_hellaswag_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..56ea04f2482edef51f517c1124af219997d16475 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_hellaswag_light/arabic_mt_hellaswag_light.yaml @@ -0,0 +1,23 @@ +task: arabic_mt_hellaswag_light +dataset_path: arcee-globe/AlGhafa-Arabic-LLM-Benchmark-Translated-10percent +dataset_name: hellaswag_okapi_ar +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_mmlu_light/utils.py b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_mmlu_light/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..62f9874e63885c9ffbeedde8a4af4d13a8f06792 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_mmlu_light/utils.py @@ -0,0 +1,23 @@ +import datasets +import numpy as np + + +def process_docs(dataset: datasets.Dataset): + def _process_doc(doc): + question = doc["query"] + answer_index = int(doc["label"]) + # Dynamically determining the choices by excluding '__few_shots', 'query' and 'label' + choices_keys = [ + key for key in doc.keys() if key not in ["query", "label", "__few_shots"] + ] + choices = [doc[key] for key in choices_keys] + + instruction = "الأسئلة التالية هي أسئلة متعددة الإختيارات مع الجواب الصحيح\n\n" + query = f"{instruction}السؤال: {question}\n" + for index, choice in enumerate(choices): + query += f"{index}) {choice}\n" + query += "الإجابة:" + + return {"query": query, "choices": choices, "gold": answer_index} + + return dataset.map(_process_doc) diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_openbook_qa_light/arabic_leaderboard_arabic_mt_openbook_qa_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_openbook_qa_light/arabic_leaderboard_arabic_mt_openbook_qa_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3737f621fb3fadcff9906b46c2dd4258898e8921 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_openbook_qa_light/arabic_leaderboard_arabic_mt_openbook_qa_light.yaml @@ -0,0 +1,13 @@ +group: arabic_leaderboard_arabic_mt_openbook_qa_light +task: + - arabic_mt_openbook_qa_light + +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true + - metric: acc_norm + aggregation: mean + weight_by_size: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_openbook_qa_light/arabic_mt_openbook_qa_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_openbook_qa_light/arabic_mt_openbook_qa_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5e914fbd32e5ea32e09eabbca56ad209cfd62dff --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_openbook_qa_light/arabic_mt_openbook_qa_light.yaml @@ -0,0 +1,23 @@ +task: arabic_mt_openbook_qa_light +dataset_path: arcee-globe/AlGhafa-Arabic-LLM-Benchmark-Translated-10percent +dataset_name: openbook_qa_ext_ar +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_openbook_qa_light/utils.py b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_openbook_qa_light/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..62f9874e63885c9ffbeedde8a4af4d13a8f06792 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_openbook_qa_light/utils.py @@ -0,0 +1,23 @@ +import datasets +import numpy as np + + +def process_docs(dataset: datasets.Dataset): + def _process_doc(doc): + question = doc["query"] + answer_index = int(doc["label"]) + # Dynamically determining the choices by excluding '__few_shots', 'query' and 'label' + choices_keys = [ + key for key in doc.keys() if key not in ["query", "label", "__few_shots"] + ] + choices = [doc[key] for key in choices_keys] + + instruction = "الأسئلة التالية هي أسئلة متعددة الإختيارات مع الجواب الصحيح\n\n" + query = f"{instruction}السؤال: {question}\n" + for index, choice in enumerate(choices): + query += f"{index}) {choice}\n" + query += "الإجابة:" + + return {"query": query, "choices": choices, "gold": answer_index} + + return dataset.map(_process_doc) diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_piqa_light/arabic_leaderboard_arabic_mt_piqa_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_piqa_light/arabic_leaderboard_arabic_mt_piqa_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..642b2e0a60a473b3b0f5af0ffb91077622d3d97c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_piqa_light/arabic_leaderboard_arabic_mt_piqa_light.yaml @@ -0,0 +1,13 @@ +group: arabic_leaderboard_arabic_mt_piqa_light +task: + - arabic_mt_piqa_light + +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true + - metric: acc_norm + aggregation: mean + weight_by_size: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_piqa_light/arabic_mt_piqa_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_piqa_light/arabic_mt_piqa_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4dd9e005a949b68dbdedf02666e4db167b92877c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_piqa_light/arabic_mt_piqa_light.yaml @@ -0,0 +1,23 @@ +task: arabic_mt_piqa_light +dataset_path: arcee-globe/AlGhafa-Arabic-LLM-Benchmark-Translated-10percent +dataset_name: piqa_ar +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_piqa_light/utils.py b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_piqa_light/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..62f9874e63885c9ffbeedde8a4af4d13a8f06792 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_piqa_light/utils.py @@ -0,0 +1,23 @@ +import datasets +import numpy as np + + +def process_docs(dataset: datasets.Dataset): + def _process_doc(doc): + question = doc["query"] + answer_index = int(doc["label"]) + # Dynamically determining the choices by excluding '__few_shots', 'query' and 'label' + choices_keys = [ + key for key in doc.keys() if key not in ["query", "label", "__few_shots"] + ] + choices = [doc[key] for key in choices_keys] + + instruction = "الأسئلة التالية هي أسئلة متعددة الإختيارات مع الجواب الصحيح\n\n" + query = f"{instruction}السؤال: {question}\n" + for index, choice in enumerate(choices): + query += f"{index}) {choice}\n" + query += "الإجابة:" + + return {"query": query, "choices": choices, "gold": answer_index} + + return dataset.map(_process_doc) diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_race_light/arabic_leaderboard_arabic_mt_race_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_race_light/arabic_leaderboard_arabic_mt_race_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8f427484d137b737f53a598ed36a3d208251aa94 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_race_light/arabic_leaderboard_arabic_mt_race_light.yaml @@ -0,0 +1,13 @@ +group: arabic_leaderboard_arabic_mt_race_light +task: + - arabic_mt_race_light + +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true + - metric: acc_norm + aggregation: mean + weight_by_size: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_race_light/arabic_mt_race_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_race_light/arabic_mt_race_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fed452cce6b24cf54b1a0cfdeea7f3e61435bbdd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_race_light/arabic_mt_race_light.yaml @@ -0,0 +1,23 @@ +task: arabic_mt_race_light +dataset_path: arcee-globe/AlGhafa-Arabic-LLM-Benchmark-Translated-10percent +dataset_name: race_ar +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_race_light/utils.py b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_race_light/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..62f9874e63885c9ffbeedde8a4af4d13a8f06792 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_race_light/utils.py @@ -0,0 +1,23 @@ +import datasets +import numpy as np + + +def process_docs(dataset: datasets.Dataset): + def _process_doc(doc): + question = doc["query"] + answer_index = int(doc["label"]) + # Dynamically determining the choices by excluding '__few_shots', 'query' and 'label' + choices_keys = [ + key for key in doc.keys() if key not in ["query", "label", "__few_shots"] + ] + choices = [doc[key] for key in choices_keys] + + instruction = "الأسئلة التالية هي أسئلة متعددة الإختيارات مع الجواب الصحيح\n\n" + query = f"{instruction}السؤال: {question}\n" + for index, choice in enumerate(choices): + query += f"{index}) {choice}\n" + query += "الإجابة:" + + return {"query": query, "choices": choices, "gold": answer_index} + + return dataset.map(_process_doc) diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_sciq_light/arabic_leaderboard_arabic_mt_sciq_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_sciq_light/arabic_leaderboard_arabic_mt_sciq_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..13127e99155e79fb9658f8df685bc69968ff27e4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_sciq_light/arabic_leaderboard_arabic_mt_sciq_light.yaml @@ -0,0 +1,13 @@ +group: arabic_leaderboard_arabic_mt_sciq_light +task: + - arabic_mt_sciq_light + +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true + - metric: acc_norm + aggregation: mean + weight_by_size: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_sciq_light/arabic_mt_sciq_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_sciq_light/arabic_mt_sciq_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..95976cbb7fda4f640f7bfe1faee9b49988a1ee14 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_sciq_light/arabic_mt_sciq_light.yaml @@ -0,0 +1,23 @@ +task: arabic_mt_sciq_light +dataset_path: arcee-globe/AlGhafa-Arabic-LLM-Benchmark-Translated-10percent +dataset_name: sciq_ar +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_sciq_light/utils.py b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_sciq_light/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..ddb42eeb8c8ee002662a6b3a6129ac4a8fa5007b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_sciq_light/utils.py @@ -0,0 +1,41 @@ +import random + +import datasets +import numpy as np + + +def doc_to_text(doc): + instruction = ( + "بناءً على السياق أدناه، اختر الإجابة الصحيحة للسؤال التالي من قائمة الاقتراحات" + ) + support = doc["support"] + question = doc["question"] + query = f"""{instruction} + السياق: + {support} + السؤال: + {question} + الإجابات المحتملة: + + """ + return query + + +def process_docs(dataset: datasets.Dataset): + def _process_doc(doc): + correct_answer = doc["correct_answer"] + choices = [ + doc["distractor1"], + doc["distractor2"], + doc["distractor3"], + correct_answer, + ] + + # Shuffle the choices + random.shuffle(choices) + + answer_index = choices.index(correct_answer) + + return {"query": doc_to_text(doc), "choices": choices, "gold": answer_index} + + return dataset.map(_process_doc) diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_toxigen_light/arabic_leaderboard_arabic_mt_toxigen_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_toxigen_light/arabic_leaderboard_arabic_mt_toxigen_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7e305d54969efd73fa34a13bf66147cc6bd9f4c2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_toxigen_light/arabic_leaderboard_arabic_mt_toxigen_light.yaml @@ -0,0 +1,13 @@ +group: arabic_leaderboard_arabic_mt_toxigen_light +task: + - arabic_mt_toxigen_light + +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true + - metric: acc_norm + aggregation: mean + weight_by_size: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_toxigen_light/arabic_mt_toxigen_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_toxigen_light/arabic_mt_toxigen_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b2bef8abae83195eaa57ac6ce5562928362cf9c3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_toxigen_light/arabic_mt_toxigen_light.yaml @@ -0,0 +1,23 @@ +task: arabic_mt_toxigen_light +dataset_path: arcee-globe/AlGhafa-Arabic-LLM-Benchmark-Translated-10percent +dataset_name: toxigen_ar +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_toxigen_light/utils.py b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_toxigen_light/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..09d311e5a13746d41e5e2bcf7ce63fc905b4c72e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_arabic_mt_toxigen_light/utils.py @@ -0,0 +1,23 @@ +import datasets +import numpy as np + + +def process_docs(dataset: datasets.Dataset) -> datasets.Dataset: + def _process_doc(doc): + text = doc["text"] + label = 1 if ((doc["toxicity_ai"] + doc["toxicity_human"]) > 5.5) else 0 + instruction = 'هل العبارة التالية تحض على الكراهية؟ قم بالرد بـ"نعم" أو "لا".' + + query = f"""{instruction} + العبارة: + '{text}' + الإجابة: + """ + out_doc = { + "query": query, + "choices": ["لا", "نعم"], + "gold": label, + } + return out_doc + + return dataset.map(_process_doc) diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Algeria_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Algeria_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4ab4634f6017b253ba711bd7967e9ca62d7b6b04 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Algeria_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Algeria_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: Algeria +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Ancient_Egypt_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Ancient_Egypt_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ab6fffedc1bd93447af973535afc329d7d4df31d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Ancient_Egypt_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Ancient_Egypt_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: Ancient_Egypt +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arab_Empire_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arab_Empire_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..886574ebf2333157430d47a25fe860da45523ca7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arab_Empire_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Arab_Empire_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: Arab_Empire +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Architecture_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Architecture_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e57472ad6e30abcf966c7f16e44238a5ddd0f09f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Architecture_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Arabic_Architecture_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: Arabic_Architecture +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Art_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Art_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e94340e7552d0bb918bb52f28a2d8fd58cef1940 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Art_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Arabic_Art_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: Arabic_Art +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Astronomy_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Astronomy_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e8ed990d524e3e48b861c25d38f090ce2e41d706 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Astronomy_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Arabic_Astronomy_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: Arabic_Astronomy +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Calligraphy_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Calligraphy_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cd41bdde6aaac743e5601724ae3718cd7bd93544 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Calligraphy_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Arabic_Calligraphy_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: Arabic_Calligraphy +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Ceremony_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Ceremony_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..72c6705479f323c805b73f52c62cacbf5ad5969a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Ceremony_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Arabic_Ceremony_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: Arabic_Ceremony +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Clothing_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Clothing_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9348de07f783cdf3ce50fd5370c42b36a9f55d38 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Clothing_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Arabic_Clothing_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: Arabic_Clothing +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Culture_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Culture_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4f211064d6733f9c74347856fb571610911e916e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Culture_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Arabic_Culture_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: Arabic_Culture +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Food_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Food_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7ccef6746fbc8f5ff67ff5739e023cb7d4e223e8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Food_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Arabic_Food_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: Arabic_Food +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Funeral_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Funeral_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..941154787b8d10c23a610740fffbce223a5225b0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Funeral_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Arabic_Funeral_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: Arabic_Funeral +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Geography_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Geography_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..36221d8899c0f34746af1b767be30dd0d32acac7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Geography_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Arabic_Geography_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: Arabic_Geography +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_History_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_History_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2e128318163de2b5183edaf4e64cd980b2704ba9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_History_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Arabic_History_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: Arabic_History +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Language_Origin_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Language_Origin_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..806060435597599903279e1d811dcd19d7a93d74 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Language_Origin_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Arabic_Language_Origin_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: Arabic_Language_Origin +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Literature_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Literature_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3122e39531b0264bd08e2499c0e58969bc2dbdd1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Literature_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Arabic_Literature_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: Arabic_Literature +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Math_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Math_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0182aedac757ee1008074e7a2b6b9a11444da874 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Math_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Arabic_Math_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: Arabic_Math +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Medicine_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Medicine_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..aec88febf1e5545d7f3e2fd998506ca862924f87 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Medicine_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Arabic_Medicine_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: Arabic_Medicine +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Music_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Music_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..35a771898ad62c30a584334c6826b886cf5303a6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Music_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Arabic_Music_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: Arabic_Music +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Ornament_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Ornament_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6b31186cd6c08d1b330e2b3a2d9154f5447b624d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Ornament_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Arabic_Ornament_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: Arabic_Ornament +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Philosophy_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Philosophy_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f6b5fa71f1497bd7f425362665fbaee57ba220b7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Philosophy_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Arabic_Philosophy_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: Arabic_Philosophy +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Physics_and_Chemistry_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Physics_and_Chemistry_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..559d729c9bfcee081bbec95372cb1e28c3e4148e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Physics_and_Chemistry_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Arabic_Physics_and_Chemistry_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: Arabic_Physics_and_Chemistry +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Wedding_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Wedding_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9241709c139dc789002ff8dc13d48a226b82c627 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Arabic_Wedding_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Arabic_Wedding_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: Arabic_Wedding +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Bahrain_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Bahrain_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b9c7cef57ace7b909b0be73fa88657a3e0e6a394 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Bahrain_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Bahrain_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: Bahrain +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Comoros_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Comoros_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1f74bd46c52e3d50bfbd0a4045dc2d1530a35586 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Comoros_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Comoros_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: Comoros +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Egypt_modern_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Egypt_modern_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e0b19cff586bb717d4ce33552f28d40d7fa1c1b1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Egypt_modern_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Egypt_modern_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: Egypt_modern +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_InfluenceFromAncientEgypt_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_InfluenceFromAncientEgypt_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6cf755a2a8b4f44999092bb74aa94b6836e9853f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_InfluenceFromAncientEgypt_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_InfluenceFromAncientEgypt_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: InfluenceFromAncientEgypt +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_InfluenceFromByzantium_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_InfluenceFromByzantium_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8fe285eb12b3aeb3a7dfd5297158f75d904deaa7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_InfluenceFromByzantium_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_InfluenceFromByzantium_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: InfluenceFromByzantium +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_InfluenceFromGreece_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_InfluenceFromGreece_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..25060acc1a1d18066fe0152364b704c22ec74ac0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_InfluenceFromGreece_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_InfluenceFromGreece_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: InfluenceFromGreece +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_InfluenceFromIslam_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_InfluenceFromIslam_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0a60a2f3f0ec1444e3b21b87e4b7741a35d85b08 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_InfluenceFromIslam_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_InfluenceFromIslam_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: InfluenceFromIslam +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_InfluenceFromPersia_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_InfluenceFromPersia_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7081bec22796148d95b62deca003a41c9ef6a210 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_InfluenceFromPersia_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_InfluenceFromPersia_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: InfluenceFromPersia +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Iraq_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Iraq_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a056a9cf04e0d4ea5696722c2f8414c0a3929cc2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Iraq_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Iraq_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: Iraq +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Islam_Education_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Islam_Education_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e8f6ad45d925302621d85a1bdc6fc2ff16c796b1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Islam_Education_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Islam_Education_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: Islam_Education +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Islam_branches_and_schools_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Islam_branches_and_schools_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..98137e9a3ab047dcd58793cdf9d9b15e45b98c26 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Islam_branches_and_schools_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Islam_branches_and_schools_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: Islam_branches_and_schools +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Islamic_law_system_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Islamic_law_system_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d9aff345dacbf47b18f17eb4207b452d82dfe6db --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Islamic_law_system_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Islamic_law_system_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: Islamic_law_system +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Kuwait_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Kuwait_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0c3d372d9eb5213640fee9c75e882dae615da723 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Kuwait_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Kuwait_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: Kuwait +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Lebanon_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Lebanon_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9c3856d698eab7d0ab6899caf7acab8deac36ff2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Lebanon_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Lebanon_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: Lebanon +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Qatar_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Qatar_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..967dbc57ef1fcd39d6a8ed6fbf630f56380496f7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Qatar_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Qatar_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: Qatar +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Somalia_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Somalia_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..558ea176a30f039b3e9de8d9652412568e322998 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Somalia_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Somalia_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: Somalia +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/README.md b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/README.md new file mode 100644 index 0000000000000000000000000000000000000000..90de14b7fc6fb5295b7c597379a3d120abbb5ad7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/README.md @@ -0,0 +1,40 @@ +# ArabicMMLU + +### Paper + +Title: ArabicMMLU: Assessing Massive Multitask Language Understanding in Arabic + +Abstract: https://arxiv.org/abs/2402.12840 + +The focus of language model evaluation has +transitioned towards reasoning and knowledge intensive tasks, driven by advancements in pretraining large models. While state-of-the-art models are partially trained on large Arabic texts, evaluating their performance in Arabic remains challenging due to the limited availability of relevant datasets. To bridge this gap, we present ArabicMMLU, the first multi-task language understanding benchmark for Arabic language, sourced from school exams across diverse educational levels in different countries spanning North Africa, the Levant, and the Gulf regions. Our data comprises 40 tasks and 14,575 multiple-choice questions in Modern Standard Arabic (MSA), and is carefully constructed by collaborating with native speakers in the region. Our comprehensive evaluations of 35 models reveal substantial room for improvement, particularly among the best open-source models. Notably, BLOOMZ, mT0, LLama2, and Falcon struggle to achieve a score of 50%, while even the top-performing Arabic centric model only achieves a score of 62.3%. + +The authors of the paper conducted studies by varying the language of the initial prompt and answer keys between English and Arabic. However, they set English initial prompts and answer keys as the standard, which is the version implemented in this task. + +Homepage: https://github.com/mbzuai-nlp/ArabicMMLU + + +### Citation + +``` +@misc{koto2024arabicmmlu, + title={ArabicMMLU: Assessing Massive Multitask Language Understanding in Arabic}, + author={Fajri Koto and Haonan Li and Sara Shatnawi and Jad Doughman and Abdelrahman Boda Sadallah and Aisha Alraeesi and Khalid Almubarak and Zaid Alyafeai and Neha Sengupta and Shady Shehata and Nizar Habash and Preslav Nakov and Timothy Baldwin}, + year={2024}, + eprint={2402.12840}, + archivePrefix={arXiv}, + primaryClass={id='cs.CL' full_name='Computation and Language' is_active=True alt_name='cmp-lg' in_archive='cs' is_general=False description='Covers natural language processing. Roughly includes material in ACM Subject Class I.2.7. Note that work on artificial languages (programming languages, logics, formal systems) that does not explicitly address natural-language issues broadly construed (natural-language processing, computational linguistics, speech, text retrieval, etc.) is not appropriate for this area.'} +} +``` + +### Groups and Tasks + +#### Groups + +* `arabicmmlu`: evaluates all ArabicMMLU tasks. + +* `arabicmmlu_stem`: evaluates STEM ArabicMMLU tasks. +* `arabicmmlu_stem_social_science`: evaluates social science ArabicMMLU tasks. +* `arabicmmlu_stem_humanities`: evaluates humanities ArabicMMLU tasks. +* `arabicmmlu_stem_language`: evaluates Arabic language ArabicMMLU tasks. +* `arabicmmlu_stem_other`: evaluates other ArabicMMLU tasks. diff --git a/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/_arabicmmlu.yaml b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/_arabicmmlu.yaml new file mode 100644 index 0000000000000000000000000000000000000000..08ed9bb0c8bc32597554c6908cdb44002aa291be --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/_arabicmmlu.yaml @@ -0,0 +1,12 @@ +group: arabicmmlu +task: +- arabicmmlu_other +- arabicmmlu_social_science +- arabicmmlu_humanities +- arabicmmlu_stem +- arabicmmlu_language +aggregate_metric_list: + - metric: acc + weight_by_size: True +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/_arabicmmlu_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/_arabicmmlu_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b52bc80470ebdd0348af002e33efc77e516252dd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/_arabicmmlu_humanities.yaml @@ -0,0 +1,9 @@ +group: arabicmmlu_humanities +group_alias: Humanities +task: + - arabicmmlu_humanities_tasks +aggregate_metric_list: + - metric: acc + weight_by_size: True +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/_arabicmmlu_language.yaml b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/_arabicmmlu_language.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d9f62abc8d1684158d80bbfcbbd2e47d1cf144f3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/_arabicmmlu_language.yaml @@ -0,0 +1,9 @@ +group: arabicmmlu_language +group_alias: Language +task: + - arabicmmlu_language_tasks +aggregate_metric_list: + - metric: acc + weight_by_size: True +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/_arabicmmlu_other.yaml b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/_arabicmmlu_other.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d96dc0bd32b05061995d1f672342b9320ea0ef9d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/_arabicmmlu_other.yaml @@ -0,0 +1,9 @@ +group: arabicmmlu_other +group_alias: Other +task: + - arabicmmlu_other_tasks +aggregate_metric_list: + - metric: acc + weight_by_size: True +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/_generate_configs.py b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/_generate_configs.py new file mode 100644 index 0000000000000000000000000000000000000000..4c43ee730c6bd9bd63466f6a8d38ced139228c81 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/_generate_configs.py @@ -0,0 +1,119 @@ +""" +Take in a YAML, and output all "other" splits with this YAML +""" + +import argparse +import logging +import os + +import yaml +from tqdm import tqdm + + +eval_logger = logging.getLogger(__name__) + + +SUBJECTS = { + "Islamic Studies": "humanities", + "Driving Test": "other", + "Natural Science (Middle School)": "stem", + "Natural Science (Primary School)": "stem", + "History (Primary School)": "humanities", + "History (Middle School)": "humanities", + "History (High School)": "humanities", + "General Knowledge": "other", + "General Knowledge (Primary School)": "other", + "General Knowledge (Middle School)": "other", + "Law (Professional)": "humanities", + "Physics (High School)": "stem", + "Social Science (Middle School)": "social_science", + "Social Science (Primary School)": "social_science", + "Management (University)": "other", + "Arabic Language (Primary School)": "language", + "Arabic Language (Middle School)": "language", + "Arabic Language (High School)": "language", + "Political Science (University)": "social_science", + "Philosophy (High School)": "humanities", + "Accounting (University)": "social_science", + "Computer Science (University)": "stem", + "Computer Science (Middle School)": "stem", + "Computer Science (Primary School)": "stem", + "Computer Science (High School)": "stem", + "Geography (Primary School)": "social_science", + "Geography (Middle School)": "social_science", + "Geography (High School)": "social_science", + "Math (Primary School)": "stem", + "Biology (High School)": "stem", + "Economics (University)": "social_science", + "Economics (Middle School)": "social_science", + "Economics (High School)": "social_science", + "Arabic Language (General)": "language", + "Arabic Language (Grammar)": "language", + "Islamic Studies (High School)": "humanities", + "Islamic Studies (Middle School)": "humanities", + "Islamic Studies (Primary School)": "humanities", + "Civics (Middle School)": "social_science", + "Civics (High School)": "social_science", +} + + +def parse_args(): + parser = argparse.ArgumentParser() + parser.add_argument("--base_yaml_path", default="_default_arabicmmlu_template_yaml") + parser.add_argument("--save_prefix_path", default="arabicmmlu") + return parser.parse_args() + + +if __name__ == "__main__": + args = parse_args() + + # get filename of base_yaml so we can `"include": ` it in our "other" YAMLs. + base_yaml_name = os.path.split(args.base_yaml_path)[-1] + + # with open(args.base_yaml_path, encoding="utf-8") as f: + # base_yaml = yaml.full_load(f) + + ALL_CATEGORIES = [] + for subject, category in tqdm(SUBJECTS.items()): + if category not in ALL_CATEGORIES: + ALL_CATEGORIES.append(category) + + # description = f"The following are multiple choice questions (with answers) about {' '.join(subject.split('_'))}.\n\n" + + yaml_dict = { + "include": base_yaml_name, + "tag": f"arabicmmlu_{category}_tasks", + "task": f"arabicmmlu_{subject.lower().replace(' ', '_').replace('(', '').replace(')', '')}", + "task_alias": subject, + "dataset_name": subject, + # "description": description, + } + + file_save_path = ( + args.save_prefix_path + + f"_{subject.lower().replace(' ', '_').replace('(', '').replace(')', '')}.yaml" + ) + eval_logger.info(f"Saving yaml for subset {subject} to {file_save_path}") + with open(file_save_path, "w", encoding="utf-8") as yaml_file: + yaml.dump( + yaml_dict, + yaml_file, + allow_unicode=True, + default_style='"', + ) + + arabicmmlu_subcategories = [f"arabicmmlu_{category}" for category in ALL_CATEGORIES] + + file_save_path = args.save_prefix_path + ".yaml" + + eval_logger.info(f"Saving benchmark config to {file_save_path}") + with open(file_save_path, "w", encoding="utf-8") as yaml_file: + yaml.dump( + { + "group": "arabicmmlu", + "task": arabicmmlu_subcategories, + }, + yaml_file, + indent=4, + default_flow_style=False, + ) diff --git a/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_accounting_university.yaml b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_accounting_university.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7ec8caad6e0317a081719b568844d2cd8d9e34ea --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_accounting_university.yaml @@ -0,0 +1,5 @@ +"dataset_name": "Accounting (University)" +"include": "_default_arabicmmlu_template_yaml" +"tag": "arabicmmlu_social_science_tasks" +"task": "arabicmmlu_accounting_university" +"task_alias": "Accounting (University)" diff --git a/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_arabic_language_general.yaml b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_arabic_language_general.yaml new file mode 100644 index 0000000000000000000000000000000000000000..621312d98b529d9f26cf70bbfb1356d656d21472 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_arabic_language_general.yaml @@ -0,0 +1,5 @@ +"dataset_name": "Arabic Language (General)" +"include": "_default_arabicmmlu_template_yaml" +"tag": "arabicmmlu_language_tasks" +"task": "arabicmmlu_arabic_language_general" +"task_alias": "Arabic Language (General)" diff --git a/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_arabic_language_high_school.yaml b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_arabic_language_high_school.yaml new file mode 100644 index 0000000000000000000000000000000000000000..77dc002bb744c90632bf3e4bad0bdfc5659fe375 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_arabic_language_high_school.yaml @@ -0,0 +1,5 @@ +"dataset_name": "Arabic Language (High School)" +"include": "_default_arabicmmlu_template_yaml" +"tag": "arabicmmlu_language_tasks" +"task": "arabicmmlu_arabic_language_high_school" +"task_alias": "Arabic Language (High School)" diff --git a/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_arabic_language_middle_school.yaml b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_arabic_language_middle_school.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9b9b2007495e04ae445895efb8484b0da799d45d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_arabic_language_middle_school.yaml @@ -0,0 +1,5 @@ +"dataset_name": "Arabic Language (Middle School)" +"include": "_default_arabicmmlu_template_yaml" +"tag": "arabicmmlu_language_tasks" +"task": "arabicmmlu_arabic_language_middle_school" +"task_alias": "Arabic Language (Middle School)" diff --git a/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_computer_science_high_school.yaml b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_computer_science_high_school.yaml new file mode 100644 index 0000000000000000000000000000000000000000..59aa929d3ceb33d78d849a0e9a8f7e3873b8ab5a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_computer_science_high_school.yaml @@ -0,0 +1,5 @@ +"dataset_name": "Computer Science (High School)" +"include": "_default_arabicmmlu_template_yaml" +"tag": "arabicmmlu_stem_tasks" +"task": "arabicmmlu_computer_science_high_school" +"task_alias": "Computer Science (High School)" diff --git a/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_islamic_studies.yaml b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_islamic_studies.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4d5020a5ab95397d71b30b8b9a00df137c01280b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_islamic_studies.yaml @@ -0,0 +1,5 @@ +"dataset_name": "Islamic Studies" +"include": "_default_arabicmmlu_template_yaml" +"tag": "arabicmmlu_humanities_tasks" +"task": "arabicmmlu_islamic_studies" +"task_alias": "Islamic Studies" diff --git a/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_islamic_studies_middle_school.yaml b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_islamic_studies_middle_school.yaml new file mode 100644 index 0000000000000000000000000000000000000000..af192fc1f6bf64866232a68cbf70d18013e16923 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_islamic_studies_middle_school.yaml @@ -0,0 +1,5 @@ +"dataset_name": "Islamic Studies (Middle School)" +"include": "_default_arabicmmlu_template_yaml" +"tag": "arabicmmlu_humanities_tasks" +"task": "arabicmmlu_islamic_studies_middle_school" +"task_alias": "Islamic Studies (Middle School)" diff --git a/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_natural_science_middle_school.yaml b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_natural_science_middle_school.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3b61531d16a3a3c7ad34561c274762ec77726bbf --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_natural_science_middle_school.yaml @@ -0,0 +1,5 @@ +"dataset_name": "Natural Science (Middle School)" +"include": "_default_arabicmmlu_template_yaml" +"tag": "arabicmmlu_stem_tasks" +"task": "arabicmmlu_natural_science_middle_school" +"task_alias": "Natural Science (Middle School)" diff --git a/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_social_science_primary_school.yaml b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_social_science_primary_school.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6f68848085e1e8428bbeab6cee05ac9a006c973e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_social_science_primary_school.yaml @@ -0,0 +1,5 @@ +"dataset_name": "Social Science (Primary School)" +"include": "_default_arabicmmlu_template_yaml" +"tag": "arabicmmlu_social_science_tasks" +"task": "arabicmmlu_social_science_primary_school" +"task_alias": "Social Science (Primary School)" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_middle_social-science_geography.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_middle_social-science_geography.yaml new file mode 100644 index 0000000000000000000000000000000000000000..57fe94f29453346c8bb30077016dc24574fbd4cc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_middle_social-science_geography.yaml @@ -0,0 +1,10 @@ +"dataset_name": "middle_social-science_geography" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_social-science_egy" +"task": "AraDiCE_ArabicMMLU_middle_social-science_geography_egy" +"task_alias": "middle social-science geography" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_primary_other_general-knowledge.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_primary_other_general-knowledge.yaml new file mode 100644 index 0000000000000000000000000000000000000000..61314bf18263111a8449de0728acfe7237c20c15 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_primary_other_general-knowledge.yaml @@ -0,0 +1,10 @@ +"dataset_name": "primary_other_general-knowledge" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_other_egy" +"task": "AraDiCE_ArabicMMLU_primary_other_general-knowledge_egy" +"task_alias": "primary other general-knowledge" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_primary_social-science_geography.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_primary_social-science_geography.yaml new file mode 100644 index 0000000000000000000000000000000000000000..73b8deea7adfd2d9f02583c5b60a485e0c59c0fa --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_primary_social-science_geography.yaml @@ -0,0 +1,10 @@ +"dataset_name": "primary_social-science_geography" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_social-science_egy" +"task": "AraDiCE_ArabicMMLU_primary_social-science_geography_egy" +"task_alias": "primary social-science geography" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_primary_social-science_social-science.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_primary_social-science_social-science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5f03bb4ba0560e1a1ccd9ec1451bf87f605ef954 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_primary_social-science_social-science.yaml @@ -0,0 +1,10 @@ +"dataset_name": "primary_social-science_social-science" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_social-science_egy" +"task": "AraDiCE_ArabicMMLU_primary_social-science_social-science_egy" +"task_alias": "primary social-science social-science" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_primary_stem_computer-science.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_primary_stem_computer-science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5e25856ebede95dda2f87f3ac6c5ae372d67d38e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_primary_stem_computer-science.yaml @@ -0,0 +1,10 @@ +"dataset_name": "primary_stem_computer-science" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_stem_egy" +"task": "AraDiCE_ArabicMMLU_primary_stem_computer-science_egy" +"task_alias": "primary stem computer-science" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_primary_stem_math.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_primary_stem_math.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d4e85ac27ff1da976b8839e033de7d23f0ea7ec8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_primary_stem_math.yaml @@ -0,0 +1,10 @@ +"dataset_name": "primary_stem_math" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_stem_egy" +"task": "AraDiCE_ArabicMMLU_primary_stem_math_egy" +"task_alias": "primary stem math" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_prof_humanities_law.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_prof_humanities_law.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4fd3e166cb1a34171c3c7950b2f7218506acf905 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_prof_humanities_law.yaml @@ -0,0 +1,10 @@ +"dataset_name": "prof_humanities_law" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_humanities_egy" +"task": "AraDiCE_ArabicMMLU_prof_humanities_law_egy" +"task_alias": "prof humanities law" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_univ_other_management.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_univ_other_management.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6b985e979f3ce280e83fda391f9ec95489df5bde --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_univ_other_management.yaml @@ -0,0 +1,10 @@ +"dataset_name": "univ_other_management" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_other_egy" +"task": "AraDiCE_ArabicMMLU_univ_other_management_egy" +"task_alias": "univ other management" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_univ_social-science_economics.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_univ_social-science_economics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3dd4dcc0a20dd1d7a555820245fa1ccfcbc5b258 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_univ_social-science_economics.yaml @@ -0,0 +1,10 @@ +"dataset_name": "univ_social-science_economics" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_social-science_egy" +"task": "AraDiCE_ArabicMMLU_univ_social-science_economics_egy" +"task_alias": "univ social-science economics" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_univ_social-science_political-science.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_univ_social-science_political-science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..671b0b3eb94699cce9440893538a8ad9622fa909 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_univ_social-science_political-science.yaml @@ -0,0 +1,10 @@ +"dataset_name": "univ_social-science_political-science" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_social-science_egy" +"task": "AraDiCE_ArabicMMLU_univ_social-science_political-science_egy" +"task_alias": "univ social-science political-science" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_univ_stem_computer-science.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_univ_stem_computer-science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..49e2e5b67c73b0e4316e8d7ca94eeb1549e357f6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_univ_stem_computer-science.yaml @@ -0,0 +1,10 @@ +"dataset_name": "univ_stem_computer-science" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_stem_egy" +"task": "AraDiCE_ArabicMMLU_univ_stem_computer-science_egy" +"task_alias": "univ stem computer-science" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/_default_template_yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/_default_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..6421888a23a376727abc20207dcb0fcd503a7de6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/_default_template_yaml @@ -0,0 +1,20 @@ +dataset_path: "QCRI/AraDICE-ArabicMMLU-egy" +fewshot_config: + sampler: default +output_type: multiple_choice +process_docs: !function utils.process_docs +doc_to_text: "{{prompt}}" +doc_to_choice: choices +doc_to_target: target +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true + - metric: f1 + higher_is_better: true + aggregation: !function metrics.micro_f1_score +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/metrics.py b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/metrics.py new file mode 100644 index 0000000000000000000000000000000000000000..47e49ded46082847d73fce85e2db37556fafa877 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/metrics.py @@ -0,0 +1,25 @@ +from sklearn.metrics import f1_score + + +def macro_f1_score(items): + unzipped_list = list(zip(*items)) + golds = unzipped_list[0] + preds = unzipped_list[1] + fscore = f1_score(golds, preds, average="macro") + return fscore + + +def micro_f1_score(items): + unzipped_list = list(zip(*items)) + golds = unzipped_list[0] + preds = unzipped_list[1] + fscore = f1_score(golds, preds, average="micro") + return fscore + + +def weighted_f1_score(items): + unzipped_list = list(zip(*items)) + golds = unzipped_list[0] + preds = unzipped_list[1] + fscore = f1_score(golds, preds, average="weighted") + return fscore diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/utils.py b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..640b9a0f2ccb73c6784ea3c9749e2e490797d877 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/utils.py @@ -0,0 +1,87 @@ +level_ar = { + "Primary": "للمرحلة الابتدائية", + "Middle": "للمرحلة المتوسطة", + "High": "للمرحلة الثانوية", + "Univ": "للمرحلة الجامعية ", + "Prof": "للمحترفين", +} + +country_ar = { + "UAE": "في الإمارات", + "Egypt": "في مصر", + "Lebanon": "في لبنان", + "Jordan": "في الأردن", + "Kuwait": "في الكويت", + "KSA": "في السعودية", + "Palestine": "في فلسطين", + "Morocco": "في المغرب", +} + +subject_ar = { + "Islamic Studies": "في الدراسات إسلامية", + "Driving Test": "في اختبار القيادة", + "Natural Science": "في العلوم الطبيعية", + "History": "في مادة التاريخ", + "General Knowledge": "في المعرفة العامة", + "Law": "في القانون", + "Physics": "في الفيزياء", + "Social Science": "في العلوم الاجتماعية", + "Management": "في الإدارة", + "Arabic Language": "في اللغة العربية", + "Political Science": " في العلوم السياسية", + "Philosophy": "في الفلسفة", + "Accounting": "في المحاسبة", + "Computer Science": "في علوم الحاسوب", + "Geography": "في الجغرافيا", + "Math": "في الرياضيات", + "Biology": "في علم الأحياء", + "Economics": "في الاقتصاد", + "Arabic Language (General)": "في اللغة العربية (عام)", + "Arabic Language (Grammar)": "في اللغة العربية (النحو)", + "Civics": "في التربية المدنية", +} + + +alpa_ar = ["أ-", "ب-", "ج-", "د-", "و-"] +alpa_en = ["A-", "B-", "C-", "D-", "E-"] +all_choices = ["أ", "ب", "ج", "د", "و"] +all_choices_en = ["A", "B", "C", "D", "E"] + + +def process_docs(dataset): + def _helper(doc): + # modifies the contents of a single + # document in our dataset. + + PROMPT = "ده سؤال [MAIN_META_DATA]. اختار الإجابة الصحيحة!\n\nسؤال: [INPUT]\n[OPTION]" + PROMPT = f"{PROMPT}\n\nإجابة:" + alpa = alpa_ar + subject = subject_ar[doc["Subject"]] + level = " " + level_ar[doc["Level"]] if doc["Level"] else "" + country = " " + country_ar[doc["Country"]] if doc["Country"] else "" + main_meta_data = f"{subject}{level}{country}" + + question = ( + f"{doc['context']}\n\n{doc['question']}" + if doc["context"] + else doc["question"] + ) + options = [] + for i, opt in enumerate(["A", "B", "C", "D", "E"]): + if opt not in doc["options"] or doc["options"][opt] is None: + break + options.append(f"{alpa[i]} {doc['options'][opt]}") + + doc["prompt"] = ( + PROMPT.replace("[MAIN_META_DATA]", main_meta_data) + .replace("[INPUT]", question) + .replace("[OPTION]", "\n".join(options)) + ) + + doc["choices"] = all_choices[: len(options)] + + doc["target"] = ["A", "B", "C", "D", "E"].index(doc["Answer Key"]) + + return doc + + return dataset.map(_helper) # returns back a datasets.Dataset object diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_high_social-science_civics.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_high_social-science_civics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..29d1a5205ec00a1d74b01c207ce19990b05c692a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_high_social-science_civics.yaml @@ -0,0 +1,10 @@ +"dataset_name": "high_social-science_civics" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_social-science_lev" +"task": "AraDiCE_ArabicMMLU_high_social-science_civics_lev" +"task_alias": "high social-science civics" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_high_social-science_geography.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_high_social-science_geography.yaml new file mode 100644 index 0000000000000000000000000000000000000000..11668a5f0b10e588e86da989e40863e4d31c6e32 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_high_social-science_geography.yaml @@ -0,0 +1,10 @@ +"dataset_name": "high_social-science_geography" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_social-science_lev" +"task": "AraDiCE_ArabicMMLU_high_social-science_geography_lev" +"task_alias": "high social-science geography" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_high_stem_computer-science.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_high_stem_computer-science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..eca96f2c6edd5c1c67e828dec9a071a6b5b733d3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_high_stem_computer-science.yaml @@ -0,0 +1,10 @@ +"dataset_name": "high_stem_computer-science" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_stem_lev" +"task": "AraDiCE_ArabicMMLU_high_stem_computer-science_lev" +"task_alias": "high stem computer-science" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_middle_humanities_islamic-studies.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_middle_humanities_islamic-studies.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7e5490e4803a62886e5fa8bd342ee32697fbd96d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_middle_humanities_islamic-studies.yaml @@ -0,0 +1,10 @@ +"dataset_name": "middle_humanities_islamic-studies" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_humanities_lev" +"task": "AraDiCE_ArabicMMLU_middle_humanities_islamic-studies_lev" +"task_alias": "middle humanities islamic-studies" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_middle_language_arabic-language.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_middle_language_arabic-language.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b67e3be59c3c5a13504004fdcca4f1b4c3df397d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_middle_language_arabic-language.yaml @@ -0,0 +1,10 @@ +"dataset_name": "middle_language_arabic-language" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_language_lev" +"task": "AraDiCE_ArabicMMLU_middle_language_arabic-language_lev" +"task_alias": "middle language arabic-language" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_middle_social-science_civics.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_middle_social-science_civics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a18665cf01c3af01b71e61e11586fa3e91008d43 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_middle_social-science_civics.yaml @@ -0,0 +1,10 @@ +"dataset_name": "middle_social-science_civics" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_social-science_lev" +"task": "AraDiCE_ArabicMMLU_middle_social-science_civics_lev" +"task_alias": "middle social-science civics" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_middle_social-science_geography.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_middle_social-science_geography.yaml new file mode 100644 index 0000000000000000000000000000000000000000..19083eb00c9cd48d780ebbd551bc34a84de7d611 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_middle_social-science_geography.yaml @@ -0,0 +1,10 @@ +"dataset_name": "middle_social-science_geography" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_social-science_lev" +"task": "AraDiCE_ArabicMMLU_middle_social-science_geography_lev" +"task_alias": "middle social-science geography" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_middle_social-science_social-science.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_middle_social-science_social-science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3c7d19c7ea9817217ee11c2423e7c2905b8ecea7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_middle_social-science_social-science.yaml @@ -0,0 +1,10 @@ +"dataset_name": "middle_social-science_social-science" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_social-science_lev" +"task": "AraDiCE_ArabicMMLU_middle_social-science_social-science_lev" +"task_alias": "middle social-science social-science" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_middle_stem_computer-science.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_middle_stem_computer-science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..583e29b103756dc04cd851e4b77302596a1637c9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_middle_stem_computer-science.yaml @@ -0,0 +1,10 @@ +"dataset_name": "middle_stem_computer-science" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_stem_lev" +"task": "AraDiCE_ArabicMMLU_middle_stem_computer-science_lev" +"task_alias": "middle stem computer-science" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_middle_stem_natural-science.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_middle_stem_natural-science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a1904d2c8785b257e7aa8b2ee0021fa9a7ac0768 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_middle_stem_natural-science.yaml @@ -0,0 +1,10 @@ +"dataset_name": "middle_stem_natural-science" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_stem_lev" +"task": "AraDiCE_ArabicMMLU_middle_stem_natural-science_lev" +"task_alias": "middle stem natural-science" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_na_humanities_islamic-studies.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_na_humanities_islamic-studies.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ac0bfe8a061acebf853a6bc9908c70f0d8550ea1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_na_humanities_islamic-studies.yaml @@ -0,0 +1,10 @@ +"dataset_name": "na_humanities_islamic-studies" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_humanities_lev" +"task": "AraDiCE_ArabicMMLU_na_humanities_islamic-studies_lev" +"task_alias": "na humanities islamic-studies" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_na_language_arabic-language-general.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_na_language_arabic-language-general.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f80e6e93e4007c4e3de7b6c885155bfc7b71f7bf --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_na_language_arabic-language-general.yaml @@ -0,0 +1,10 @@ +"dataset_name": "na_language_arabic-language-general" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_language_lev" +"task": "AraDiCE_ArabicMMLU_na_language_arabic-language-general_lev" +"task_alias": "na language arabic-language-general" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_na_language_arabic-language-grammar.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_na_language_arabic-language-grammar.yaml new file mode 100644 index 0000000000000000000000000000000000000000..af3943d9a8f59bf10c1decd7d56a497d45312cb6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_na_language_arabic-language-grammar.yaml @@ -0,0 +1,10 @@ +"dataset_name": "na_language_arabic-language-grammar" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_language_lev" +"task": "AraDiCE_ArabicMMLU_na_language_arabic-language-grammar_lev" +"task_alias": "na language arabic-language-grammar" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_na_other_general-knowledge.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_na_other_general-knowledge.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0c5669cf07cf6f2f05a10e40fe30208c3f857f24 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_na_other_general-knowledge.yaml @@ -0,0 +1,10 @@ +"dataset_name": "na_other_general-knowledge" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_other_lev" +"task": "AraDiCE_ArabicMMLU_na_other_general-knowledge_lev" +"task_alias": "na other general-knowledge" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_primary_humanities_history.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_primary_humanities_history.yaml new file mode 100644 index 0000000000000000000000000000000000000000..be32d433f76b171a1f076dec90dacfea7c11ea3f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_primary_humanities_history.yaml @@ -0,0 +1,10 @@ +"dataset_name": "primary_humanities_history" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_humanities_lev" +"task": "AraDiCE_ArabicMMLU_primary_humanities_history_lev" +"task_alias": "primary humanities history" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_primary_humanities_islamic-studies.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_primary_humanities_islamic-studies.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9ae53b80ee7f011f12afbaee1d781194d228e41e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_primary_humanities_islamic-studies.yaml @@ -0,0 +1,10 @@ +"dataset_name": "primary_humanities_islamic-studies" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_humanities_lev" +"task": "AraDiCE_ArabicMMLU_primary_humanities_islamic-studies_lev" +"task_alias": "primary humanities islamic-studies" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_primary_language_arabic-language.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_primary_language_arabic-language.yaml new file mode 100644 index 0000000000000000000000000000000000000000..15575b0513b242eb62e9c2b3a5dfce5351f5022f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_primary_language_arabic-language.yaml @@ -0,0 +1,10 @@ +"dataset_name": "primary_language_arabic-language" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_language_lev" +"task": "AraDiCE_ArabicMMLU_primary_language_arabic-language_lev" +"task_alias": "primary language arabic-language" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_primary_other_general-knowledge.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_primary_other_general-knowledge.yaml new file mode 100644 index 0000000000000000000000000000000000000000..07b6692115f74d81f741d39e02914b980c66863a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_primary_other_general-knowledge.yaml @@ -0,0 +1,10 @@ +"dataset_name": "primary_other_general-knowledge" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_other_lev" +"task": "AraDiCE_ArabicMMLU_primary_other_general-knowledge_lev" +"task_alias": "primary other general-knowledge" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_primary_social-science_geography.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_primary_social-science_geography.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b43c49035cbe43d47d16f1783681b0ad0ceaafb0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_primary_social-science_geography.yaml @@ -0,0 +1,10 @@ +"dataset_name": "primary_social-science_geography" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_social-science_lev" +"task": "AraDiCE_ArabicMMLU_primary_social-science_geography_lev" +"task_alias": "primary social-science geography" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_primary_social-science_social-science.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_primary_social-science_social-science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8f9f093415267e3a2648bf27c494ba20154babba --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_primary_social-science_social-science.yaml @@ -0,0 +1,10 @@ +"dataset_name": "primary_social-science_social-science" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_social-science_lev" +"task": "AraDiCE_ArabicMMLU_primary_social-science_social-science_lev" +"task_alias": "primary social-science social-science" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_primary_stem_computer-science.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_primary_stem_computer-science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6a79f2e7a2f4c54ad67b97345abaa91d0857bce0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_primary_stem_computer-science.yaml @@ -0,0 +1,10 @@ +"dataset_name": "primary_stem_computer-science" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_stem_lev" +"task": "AraDiCE_ArabicMMLU_primary_stem_computer-science_lev" +"task_alias": "primary stem computer-science" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_primary_stem_natural-science.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_primary_stem_natural-science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6d7404ae7e7b6c172f35c9ec16caa942e2516f7b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_primary_stem_natural-science.yaml @@ -0,0 +1,10 @@ +"dataset_name": "primary_stem_natural-science" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_stem_lev" +"task": "AraDiCE_ArabicMMLU_primary_stem_natural-science_lev" +"task_alias": "primary stem natural-science" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_prof_humanities_law.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_prof_humanities_law.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1c50cb9d913ec092ebd3dbbafc7165e786d81ef1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_prof_humanities_law.yaml @@ -0,0 +1,10 @@ +"dataset_name": "prof_humanities_law" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_humanities_lev" +"task": "AraDiCE_ArabicMMLU_prof_humanities_law_lev" +"task_alias": "prof humanities law" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_univ_other_management.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_univ_other_management.yaml new file mode 100644 index 0000000000000000000000000000000000000000..31b79fd0c14a01dd6f2d7e79c0066b15edea1136 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_univ_other_management.yaml @@ -0,0 +1,10 @@ +"dataset_name": "univ_other_management" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_other_lev" +"task": "AraDiCE_ArabicMMLU_univ_other_management_lev" +"task_alias": "univ other management" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_univ_social-science_accounting.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_univ_social-science_accounting.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fc0cb68266fbcecd39c3a92b21dbd8223bf0f030 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_univ_social-science_accounting.yaml @@ -0,0 +1,10 @@ +"dataset_name": "univ_social-science_accounting" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_social-science_lev" +"task": "AraDiCE_ArabicMMLU_univ_social-science_accounting_lev" +"task_alias": "univ social-science accounting" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_univ_social-science_economics.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_univ_social-science_economics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..daec1b37a648c75ffc38bd531c4ec8a2c7365c9f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_univ_social-science_economics.yaml @@ -0,0 +1,10 @@ +"dataset_name": "univ_social-science_economics" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_social-science_lev" +"task": "AraDiCE_ArabicMMLU_univ_social-science_economics_lev" +"task_alias": "univ social-science economics" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_univ_social-science_political-science.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_univ_social-science_political-science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e69f63ca4d22e1c0cd63a00f5832844b5b89bc90 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_univ_social-science_political-science.yaml @@ -0,0 +1,10 @@ +"dataset_name": "univ_social-science_political-science" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_social-science_lev" +"task": "AraDiCE_ArabicMMLU_univ_social-science_political-science_lev" +"task_alias": "univ social-science political-science" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/_default_template_yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/_default_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..45c5a345de1e2459c675b2d5ada4f6ec5fe5f090 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/_default_template_yaml @@ -0,0 +1,20 @@ +dataset_path: QCRI/AraDICE-ArabicMMLU-lev +fewshot_config: + sampler: default +output_type: multiple_choice +process_docs: !function utils.process_docs +doc_to_text: "{{prompt}}" +doc_to_choice: choices +doc_to_target: target +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true + - metric: f1 + higher_is_better: true + aggregation: !function metrics.micro_f1_score +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/metrics.py b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/metrics.py new file mode 100644 index 0000000000000000000000000000000000000000..47e49ded46082847d73fce85e2db37556fafa877 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/metrics.py @@ -0,0 +1,25 @@ +from sklearn.metrics import f1_score + + +def macro_f1_score(items): + unzipped_list = list(zip(*items)) + golds = unzipped_list[0] + preds = unzipped_list[1] + fscore = f1_score(golds, preds, average="macro") + return fscore + + +def micro_f1_score(items): + unzipped_list = list(zip(*items)) + golds = unzipped_list[0] + preds = unzipped_list[1] + fscore = f1_score(golds, preds, average="micro") + return fscore + + +def weighted_f1_score(items): + unzipped_list = list(zip(*items)) + golds = unzipped_list[0] + preds = unzipped_list[1] + fscore = f1_score(golds, preds, average="weighted") + return fscore diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/utils.py b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..37683c46e237fd3fcfc9e79cb6e861d089484090 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/utils.py @@ -0,0 +1,94 @@ +level_ar = { + "Primary": "للمرحلة الابتدائية", + "Middle": "للمرحلة المتوسطة", + "High": "للمرحلة الثانوية", + "Univ": "للمرحلة الجامعية ", + "Prof": "للمحترفين", +} + +country_ar = { + "UAE": "بالإمارات", + "Egypt": "بمصر", + "Lebanon": "بلبنان", + "Jordan": "بالأردن", + "Kuwait": "بالكويت", + "KSA": "بالسعودية", + "Palestine": "بفلسطين", + "Morocco": "بالمغرب", +} + +subject_ar = { + "Islamic Studies": "عن الدراسات إسلامية", + "Driving Test": "عن فحص السواقة", + "Natural Science": "عن العلوم الطبيعية", + "History": "تاريخ", + "General Knowledge": "معرفة عامة", + "Law": "عن القانون", + "Physics": "فيزياء", + "Social Science": "علوم اجتماعية", + "Management": "عن الإدارة", + "Arabic Language": "عن اللغة العربية", + "Political Science": " عن العلوم السياسية", + "Philosophy": "فلسفة", + "Accounting": "محاسبة", + "Computer Science": "عن علوم الحاسوب", + "Geography": "جغرافيا", + "Math": "رياضيات", + "Biology": "بيولوجي", + "Economics": "اقتصاد", + "Arabic Language (General)": "لغة العربية (عام)", + "Arabic Language (Grammar)": "لغة العربية (نحو)", + "Civics": "تربية مدنية", +} + +alpa_ar = ["أ-", "ب-", "ج-", "د-", "و-"] +alpa_en = ["A-", "B-", "C-", "D-", "E-"] +all_choices = ["أ", "ب", "ج", "د", "و"] +all_choices_en = ["A", "B", "C", "D", "E"] + + +def process_docs(dataset): + def _helper(doc): + # modifies the contents of a single + # document in our dataset. + PROMPT = ( + "هيدا سؤال [MAIN_META_DATA]. نقي الجواب الصح!\n\nسؤال: [INPUT]\n[OPTION]" + ) + + # if args.lora_weights == "x": + PROMPT = f"{PROMPT}\n\nالجواب:" + # else: + # PROMPT = f'### Input:{PROMPT}\n\n### Output:\n' + + alpa = alpa_ar + + subject = subject_ar[doc["Subject"]] + level = " " + level_ar[doc["Level"]] if doc["Level"] else "" + country = " " + country_ar[doc["Country"]] if doc["Country"] else "" + main_meta_data = f"{subject}{level}{country}" + + question = ( + f"{doc['context']}\n\n{doc['question']}" + if doc["context"] + else doc["question"] + ) + options = [] + + for i, opt in enumerate(["A", "B", "C", "D", "E"]): + if opt not in doc["options"] or doc["options"][opt] is None: + break + options.append(f"{alpa[i]} {doc['options'][opt]}") + + doc["prompt"] = ( + PROMPT.replace("[MAIN_META_DATA]", main_meta_data) + .replace("[INPUT]", question) + .replace("[OPTION]", "\n".join(options)) + ) + + doc["choices"] = all_choices[: len(options)] + + doc["target"] = ["A", "B", "C", "D", "E"].index(doc["Answer Key"]) + + return doc + + return dataset.map(_helper) diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/boolq/EGY/boolq_egy.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/boolq/EGY/boolq_egy.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c481c24a750c83d689a6a1dd7e3efd233e797193 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/boolq/EGY/boolq_egy.yaml @@ -0,0 +1,25 @@ +task: AraDiCE_boolq_egy +dataset_path: QCRI/AraDiCE-BoolQ +dataset_name: BoolQ-egy +output_type: multiple_choice +training_split: null +validation_split: null +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{passage}}\nسؤال: {{question}}؟\nجواب:" +doc_to_target: target +doc_to_choice: ["لا", "نعم"] +should_decontaminate: true +doc_to_decontamination_query: passage +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true + - metric: f1 + higher_is_better: true + aggregation: !function metrics.micro_f1_score +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/boolq/EGY/metrics.py b/lm-evaluation-harness/lm_eval/tasks/aradice/boolq/EGY/metrics.py new file mode 100644 index 0000000000000000000000000000000000000000..47e49ded46082847d73fce85e2db37556fafa877 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/boolq/EGY/metrics.py @@ -0,0 +1,25 @@ +from sklearn.metrics import f1_score + + +def macro_f1_score(items): + unzipped_list = list(zip(*items)) + golds = unzipped_list[0] + preds = unzipped_list[1] + fscore = f1_score(golds, preds, average="macro") + return fscore + + +def micro_f1_score(items): + unzipped_list = list(zip(*items)) + golds = unzipped_list[0] + preds = unzipped_list[1] + fscore = f1_score(golds, preds, average="micro") + return fscore + + +def weighted_f1_score(items): + unzipped_list = list(zip(*items)) + golds = unzipped_list[0] + preds = unzipped_list[1] + fscore = f1_score(golds, preds, average="weighted") + return fscore diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/boolq/EGY/utils.py b/lm-evaluation-harness/lm_eval/tasks/aradice/boolq/EGY/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..4220133e5d5cf710d96a7a915b3ec8db7d8a03db --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/boolq/EGY/utils.py @@ -0,0 +1,18 @@ +egy_answer_mapping = {"true": "نعم", "false": "لا", True: "نعم", False: "لا"} + + +def process_docs(dataset): + def remove_question_mark(text): + text = text.strip() + if text.endswith("?") or text.endswith("؟"): + text = text[:-1] + text = text.strip() + + return text + + def _helper(doc): + doc["question"] = remove_question_mark(doc["question"]) + doc["target"] = egy_answer_mapping[doc["answer"]] + return doc + + return dataset.map(_helper) diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/boolq/ENG/boolq_eng.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/boolq/ENG/boolq_eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1409aebfab9a95e84a54899d7e958445400cd535 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/boolq/ENG/boolq_eng.yaml @@ -0,0 +1,25 @@ +task: AraDiCE_boolq_eng +dataset_path: QCRI/AraDiCE-BoolQ +dataset_name: BoolQ-eng +output_type: multiple_choice +training_split: null +validation_split: null +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{passage}}\nQuestion: {{question}}?\nAnswer:" +doc_to_target: target +doc_to_choice: ["no", "yes"] +should_decontaminate: true +doc_to_decontamination_query: passage +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true + - metric: f1 + higher_is_better: true + aggregation: !function metrics.micro_f1_score +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/boolq/ENG/metrics.py b/lm-evaluation-harness/lm_eval/tasks/aradice/boolq/ENG/metrics.py new file mode 100644 index 0000000000000000000000000000000000000000..47e49ded46082847d73fce85e2db37556fafa877 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/boolq/ENG/metrics.py @@ -0,0 +1,25 @@ +from sklearn.metrics import f1_score + + +def macro_f1_score(items): + unzipped_list = list(zip(*items)) + golds = unzipped_list[0] + preds = unzipped_list[1] + fscore = f1_score(golds, preds, average="macro") + return fscore + + +def micro_f1_score(items): + unzipped_list = list(zip(*items)) + golds = unzipped_list[0] + preds = unzipped_list[1] + fscore = f1_score(golds, preds, average="micro") + return fscore + + +def weighted_f1_score(items): + unzipped_list = list(zip(*items)) + golds = unzipped_list[0] + preds = unzipped_list[1] + fscore = f1_score(golds, preds, average="weighted") + return fscore diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/boolq/ENG/utils.py b/lm-evaluation-harness/lm_eval/tasks/aradice/boolq/ENG/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..3f1233cddf3b9881fd04da5d047ddd7a3a1f9668 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/boolq/ENG/utils.py @@ -0,0 +1,18 @@ +en_answer_mapping = {"true": "yes", "false": "no", True: "yes", False: "no"} + + +def process_docs(dataset): + def remove_question_mark(text): + text = text.strip() + if text.endswith("?") or text.endswith("؟"): + text = text[:-1] + text = text.strip() + + return text + + def _helper(doc): + doc["question"] = remove_question_mark(doc["question"]) + doc["target"] = en_answer_mapping[doc["answer"]] + return doc + + return dataset.map(_helper) diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/boolq/LEV/boolq_lev.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/boolq/LEV/boolq_lev.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ccbe94770166f7e7c3ebdc849d87773e5e7f3163 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/boolq/LEV/boolq_lev.yaml @@ -0,0 +1,25 @@ +task: AraDiCE_boolq_lev +dataset_path: QCRI/AraDiCE-BoolQ +dataset_name: BoolQ-lev +output_type: multiple_choice +training_split: null +validation_split: null +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{passage}}\nسؤال: {{question}}؟\nجواب:" +doc_to_target: target +doc_to_choice: ["لا", "نعم"] +should_decontaminate: true +doc_to_decontamination_query: passage +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true + - metric: f1 + higher_is_better: true + aggregation: !function metrics.micro_f1_score +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/boolq/LEV/metrics.py b/lm-evaluation-harness/lm_eval/tasks/aradice/boolq/LEV/metrics.py new file mode 100644 index 0000000000000000000000000000000000000000..47e49ded46082847d73fce85e2db37556fafa877 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/boolq/LEV/metrics.py @@ -0,0 +1,25 @@ +from sklearn.metrics import f1_score + + +def macro_f1_score(items): + unzipped_list = list(zip(*items)) + golds = unzipped_list[0] + preds = unzipped_list[1] + fscore = f1_score(golds, preds, average="macro") + return fscore + + +def micro_f1_score(items): + unzipped_list = list(zip(*items)) + golds = unzipped_list[0] + preds = unzipped_list[1] + fscore = f1_score(golds, preds, average="micro") + return fscore + + +def weighted_f1_score(items): + unzipped_list = list(zip(*items)) + golds = unzipped_list[0] + preds = unzipped_list[1] + fscore = f1_score(golds, preds, average="weighted") + return fscore diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/boolq/LEV/utils.py b/lm-evaluation-harness/lm_eval/tasks/aradice/boolq/LEV/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..3f601229a255ceedd49b5784e025bf3fd0472ade --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/boolq/LEV/utils.py @@ -0,0 +1,18 @@ +lev_answer_mapping = {"true": "نعم", "false": "لا", True: "نعم", False: "لا"} + + +def process_docs(dataset): + def remove_question_mark(text): + text = text.strip() + if text.endswith("?") or text.endswith("؟"): + text = text[:-1] + text = text.strip() + + return text + + def _helper(doc): + doc["question"] = remove_question_mark(doc["question"]) + doc["target"] = lev_answer_mapping[doc["answer"]] + return doc + + return dataset.map(_helper) diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/boolq/MSA/boolq_msa.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/boolq/MSA/boolq_msa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ea3208ecdca1b14517ebb6377a36201370ae148d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/boolq/MSA/boolq_msa.yaml @@ -0,0 +1,25 @@ +task: AraDiCE_boolq_msa +dataset_path: QCRI/AraDiCE-BoolQ +dataset_name: BoolQ-msa +output_type: multiple_choice +training_split: null +validation_split: null +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{passage}}\nسؤال: {{question}}؟\nجواب:" +doc_to_target: target +doc_to_choice: ["لا", "نعم"] +should_decontaminate: true +doc_to_decontamination_query: passage +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true + - metric: f1 + higher_is_better: true + aggregation: !function metrics.micro_f1_score +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/boolq/MSA/metrics.py b/lm-evaluation-harness/lm_eval/tasks/aradice/boolq/MSA/metrics.py new file mode 100644 index 0000000000000000000000000000000000000000..47e49ded46082847d73fce85e2db37556fafa877 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/boolq/MSA/metrics.py @@ -0,0 +1,25 @@ +from sklearn.metrics import f1_score + + +def macro_f1_score(items): + unzipped_list = list(zip(*items)) + golds = unzipped_list[0] + preds = unzipped_list[1] + fscore = f1_score(golds, preds, average="macro") + return fscore + + +def micro_f1_score(items): + unzipped_list = list(zip(*items)) + golds = unzipped_list[0] + preds = unzipped_list[1] + fscore = f1_score(golds, preds, average="micro") + return fscore + + +def weighted_f1_score(items): + unzipped_list = list(zip(*items)) + golds = unzipped_list[0] + preds = unzipped_list[1] + fscore = f1_score(golds, preds, average="weighted") + return fscore diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/boolq/MSA/utils.py b/lm-evaluation-harness/lm_eval/tasks/aradice/boolq/MSA/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..47a80046871bbb8ff9f17cffc5b5bc6bb0937972 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/boolq/MSA/utils.py @@ -0,0 +1,18 @@ +msa_answer_mapping = {"true": "نعم", "false": "لا", True: "نعم", False: "لا"} + + +def process_docs(dataset): + def remove_question_mark(text): + text = text.strip() + if text.endswith("?") or text.endswith("؟"): + text = text[:-1] + text = text.strip() + + return text + + def _helper(doc): + doc["question"] = remove_question_mark(doc["question"]) + doc["target"] = msa_answer_mapping[doc["answer"]] + return doc + + return dataset.map(_helper) diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/cultural-benchmark/egypt.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/cultural-benchmark/egypt.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c2d5da2ecf70dc24123983ee883a168c81eacc47 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/cultural-benchmark/egypt.yaml @@ -0,0 +1,25 @@ +task: AraDiCE_egypt_cultural +dataset_path: QCRI/AraDiCE-Culture +dataset_name: Egypt +training_split: null +validation_split: null +test_split: test +output_type: multiple_choice +process_docs: !function utils.process_docs +doc_to_text: "سؤال : {{Question}}\nإجابة :" +doc_to_target: 0 +doc_to_choice: choices +should_decontaminate: true +doc_to_decontamination_query: Question +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true + - metric: f1 + higher_is_better: true + aggregation: !function metrics.micro_f1_score +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/cultural-benchmark/jordan.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/cultural-benchmark/jordan.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dc2b3db5e4771194577d8ae6b05ad5bf454c4afd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/cultural-benchmark/jordan.yaml @@ -0,0 +1,25 @@ +task: AraDiCE_jordan_cultural +dataset_path: QCRI/AraDiCE-Culture +dataset_name: Jordan +training_split: null +validation_split: null +test_split: test +output_type: multiple_choice +process_docs: !function utils.process_docs +doc_to_text: "سؤال : {{Question}}\nإجابة :" +doc_to_target: 0 +doc_to_choice: choices +should_decontaminate: true +doc_to_decontamination_query: Question +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true + - metric: f1 + higher_is_better: true + aggregation: !function metrics.micro_f1_score +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/cultural-benchmark/lebanon.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/cultural-benchmark/lebanon.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e2811422fca94f7354ac9e3f04b86e641bdb2d1a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/cultural-benchmark/lebanon.yaml @@ -0,0 +1,25 @@ +task: AraDiCE_lebanon_cultural +dataset_path: QCRI/AraDiCE-Culture +dataset_name: Lebanon +training_split: null +validation_split: null +test_split: test +output_type: multiple_choice +process_docs: !function utils.process_docs +doc_to_text: "سؤال : {{Question}}\nإجابة :" +doc_to_target: 0 +doc_to_choice: choices +should_decontaminate: true +doc_to_decontamination_query: Question +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true + - metric: f1 + higher_is_better: true + aggregation: !function metrics.micro_f1_score +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/cultural-benchmark/metrics.py b/lm-evaluation-harness/lm_eval/tasks/aradice/cultural-benchmark/metrics.py new file mode 100644 index 0000000000000000000000000000000000000000..47e49ded46082847d73fce85e2db37556fafa877 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/cultural-benchmark/metrics.py @@ -0,0 +1,25 @@ +from sklearn.metrics import f1_score + + +def macro_f1_score(items): + unzipped_list = list(zip(*items)) + golds = unzipped_list[0] + preds = unzipped_list[1] + fscore = f1_score(golds, preds, average="macro") + return fscore + + +def micro_f1_score(items): + unzipped_list = list(zip(*items)) + golds = unzipped_list[0] + preds = unzipped_list[1] + fscore = f1_score(golds, preds, average="micro") + return fscore + + +def weighted_f1_score(items): + unzipped_list = list(zip(*items)) + golds = unzipped_list[0] + preds = unzipped_list[1] + fscore = f1_score(golds, preds, average="weighted") + return fscore diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/cultural-benchmark/palestine.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/cultural-benchmark/palestine.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8854c10f5d23bed239373b3cbded9dd608b613d3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/cultural-benchmark/palestine.yaml @@ -0,0 +1,25 @@ +task: AraDiCE_palestine_cultural +dataset_path: QCRI/AraDiCE-Culture +dataset_name: Palestine +training_split: null +validation_split: null +test_split: test +output_type: multiple_choice +process_docs: !function utils.process_docs +doc_to_text: "سؤال : {{Question}}\nإجابة :" +doc_to_target: 0 +doc_to_choice: choices +should_decontaminate: true +doc_to_decontamination_query: Question +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true + - metric: f1 + higher_is_better: true + aggregation: !function metrics.micro_f1_score +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/cultural-benchmark/qatar.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/cultural-benchmark/qatar.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b9df210076abad8f9e97e1ae7691844f5a22c5c9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/cultural-benchmark/qatar.yaml @@ -0,0 +1,25 @@ +task: AraDiCE_qatar_cultural +dataset_path: QCRI/AraDiCE-Culture +dataset_name: Qatar +training_split: null +validation_split: null +test_split: test +output_type: multiple_choice +process_docs: !function utils.process_docs +doc_to_text: "سؤال : {{Question}}\nإجابة :" +doc_to_target: 0 +doc_to_choice: choices +should_decontaminate: true +doc_to_decontamination_query: Question +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true + - metric: f1 + higher_is_better: true + aggregation: !function metrics.micro_f1_score +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/cultural-benchmark/syria.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/cultural-benchmark/syria.yaml new file mode 100644 index 0000000000000000000000000000000000000000..faf957c22e3b39b5b97b29eea3effbca578bdf83 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/cultural-benchmark/syria.yaml @@ -0,0 +1,25 @@ +task: AraDiCE_syria_cultural +dataset_path: QCRI/AraDiCE-Culture +dataset_name: Syria +training_split: null +validation_split: null +test_split: test +output_type: multiple_choice +process_docs: !function utils.process_docs +doc_to_text: "سؤال : {{Question}}\nإجابة :" +doc_to_target: 0 +doc_to_choice: choices +should_decontaminate: true +doc_to_decontamination_query: Question +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true + - metric: f1 + higher_is_better: true + aggregation: !function metrics.micro_f1_score +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/cultural-benchmark/utils.py b/lm-evaluation-harness/lm_eval/tasks/aradice/cultural-benchmark/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..a2093299bf91b096cf7112c5faecae3a4374cbc2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/cultural-benchmark/utils.py @@ -0,0 +1,6 @@ +def process_docs(dataset): + def _helper(doc): + doc["choices"] = [doc["Option A"], doc["Option B"], doc["Option C"]] + return doc + + return dataset.map(_helper) diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/openbookqa/metrics.py b/lm-evaluation-harness/lm_eval/tasks/aradice/openbookqa/metrics.py new file mode 100644 index 0000000000000000000000000000000000000000..47e49ded46082847d73fce85e2db37556fafa877 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/openbookqa/metrics.py @@ -0,0 +1,25 @@ +from sklearn.metrics import f1_score + + +def macro_f1_score(items): + unzipped_list = list(zip(*items)) + golds = unzipped_list[0] + preds = unzipped_list[1] + fscore = f1_score(golds, preds, average="macro") + return fscore + + +def micro_f1_score(items): + unzipped_list = list(zip(*items)) + golds = unzipped_list[0] + preds = unzipped_list[1] + fscore = f1_score(golds, preds, average="micro") + return fscore + + +def weighted_f1_score(items): + unzipped_list = list(zip(*items)) + golds = unzipped_list[0] + preds = unzipped_list[1] + fscore = f1_score(golds, preds, average="weighted") + return fscore diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/openbookqa/openbookqa_egy.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/openbookqa/openbookqa_egy.yaml new file mode 100644 index 0000000000000000000000000000000000000000..781560c5c3338a29b7b4d57ee41b1ac834f30968 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/openbookqa/openbookqa_egy.yaml @@ -0,0 +1,24 @@ +task: AraDiCE_openbookqa_egy +dataset_path: QCRI/AraDiCE-OpenBookQA +dataset_name: OBQA-egy +training_split: null +validation_split: null +test_split: test +output_type: multiple_choice +doc_to_text: !function utils.doc_to_text +doc_to_target: !function utils.doc_to_target +doc_to_choice: !function utils.doc_to_choice +should_decontaminate: true +doc_to_decontamination_query: "{{question.stem}}" +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true + - metric: f1 + higher_is_better: true + aggregation: !function metrics.micro_f1_score +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/openbookqa/openbookqa_eng.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/openbookqa/openbookqa_eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5f0adcc6562ff89a9b542cbf8725cc3662c05278 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/openbookqa/openbookqa_eng.yaml @@ -0,0 +1,24 @@ +task: AraDiCE_openbookqa_eng +dataset_path: QCRI/AraDiCE-OpenBookQA +dataset_name: OBQA-eng +training_split: null +validation_split: null +test_split: test +output_type: multiple_choice +doc_to_text: !function utils.doc_to_text +doc_to_target: !function utils.doc_to_target +doc_to_choice: !function utils.doc_to_choice +should_decontaminate: true +doc_to_decontamination_query: "{{question.stem}}" +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true + - metric: f1 + higher_is_better: true + aggregation: !function metrics.micro_f1_score +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/openbookqa/openbookqa_lev.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/openbookqa/openbookqa_lev.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1386b80178482f97fefb43e8d6c4a65859222bb1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/openbookqa/openbookqa_lev.yaml @@ -0,0 +1,24 @@ +task: AraDiCE_openbookqa_lev +dataset_path: QCRI/AraDiCE-OpenBookQA +dataset_name: OBQA-lev +training_split: null +validation_split: null +test_split: test +output_type: multiple_choice +doc_to_text: !function utils.doc_to_text +doc_to_target: !function utils.doc_to_target +doc_to_choice: !function utils.doc_to_choice +should_decontaminate: true +doc_to_decontamination_query: "{{question.stem}}" +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true + - metric: f1 + higher_is_better: true + aggregation: !function metrics.micro_f1_score +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/openbookqa/openbookqa_msa.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/openbookqa/openbookqa_msa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..20131ecb267a8bd453fdbc6797cfa58501194913 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/openbookqa/openbookqa_msa.yaml @@ -0,0 +1,24 @@ +task: AraDiCE_openbookqa_msa +dataset_path: QCRI/AraDiCE-OpenBookQA +dataset_name: OBQA-msa +training_split: null +validation_split: null +test_split: test +output_type: multiple_choice +doc_to_text: !function utils.doc_to_text +doc_to_target: !function utils.doc_to_target +doc_to_choice: !function utils.doc_to_choice +should_decontaminate: true +doc_to_decontamination_query: "{{question.stem}}" +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true + - metric: f1 + higher_is_better: true + aggregation: !function metrics.micro_f1_score +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/openbookqa/utils.py b/lm-evaluation-harness/lm_eval/tasks/aradice/openbookqa/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..39e51a0274ff375893f749c698203d3ff567c29e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/openbookqa/utils.py @@ -0,0 +1,18 @@ +def doc_to_target(doc): + labels = [c["label"] for c in doc["question"]["choices"]] + + try: + i = labels.index(doc["answerKey"].lstrip()) + except Exception as e: + print("Failed", e) + return + return i + + +def doc_to_choice(doc): + texts = [c["text"] for c in doc["question"]["choices"]] + return texts + + +def doc_to_text(doc): + return doc["question"]["stem"].strip() diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/piqa/metrics.py b/lm-evaluation-harness/lm_eval/tasks/aradice/piqa/metrics.py new file mode 100644 index 0000000000000000000000000000000000000000..47e49ded46082847d73fce85e2db37556fafa877 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/piqa/metrics.py @@ -0,0 +1,25 @@ +from sklearn.metrics import f1_score + + +def macro_f1_score(items): + unzipped_list = list(zip(*items)) + golds = unzipped_list[0] + preds = unzipped_list[1] + fscore = f1_score(golds, preds, average="macro") + return fscore + + +def micro_f1_score(items): + unzipped_list = list(zip(*items)) + golds = unzipped_list[0] + preds = unzipped_list[1] + fscore = f1_score(golds, preds, average="micro") + return fscore + + +def weighted_f1_score(items): + unzipped_list = list(zip(*items)) + golds = unzipped_list[0] + preds = unzipped_list[1] + fscore = f1_score(golds, preds, average="weighted") + return fscore diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/piqa/piqa_egy.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/piqa/piqa_egy.yaml new file mode 100644 index 0000000000000000000000000000000000000000..79d682d3d0c3303283585466bee321633c5aa714 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/piqa/piqa_egy.yaml @@ -0,0 +1,24 @@ +task: AraDiCE_piqa_egy +dataset_path: QCRI/AraDiCE-PIQA +dataset_name: PIQA-egy +training_split: null +validation_split: null +test_split: test +output_type: multiple_choice +doc_to_text: "سؤال : {{goal}}\nإجابة :" +doc_to_target: label +doc_to_choice: "{{[sol1, sol2]}}" +should_decontaminate: true +doc_to_decontamination_query: goal +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true + - metric: f1 + higher_is_better: true + aggregation: !function metrics.micro_f1_score +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/piqa/piqa_eng.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/piqa/piqa_eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a2967f3d518937e17a8152259404a6fd2b18858b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/piqa/piqa_eng.yaml @@ -0,0 +1,24 @@ +task: AraDiCE_piqa_eng +dataset_path: QCRI/AraDiCE-PIQA +dataset_name: PIQA-eng +training_split: null +validation_split: null +test_split: test +output_type: multiple_choice +doc_to_text: "سؤال : {{goal}}\nإجابة :" +doc_to_target: label +doc_to_choice: "{{[sol1, sol2]}}" +should_decontaminate: true +doc_to_decontamination_query: goal +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true + - metric: f1 + higher_is_better: true + aggregation: !function metrics.micro_f1_score +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/piqa/piqa_lev.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/piqa/piqa_lev.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c5853912d1d8ad04983bb17ccca58e4dfb43b916 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/piqa/piqa_lev.yaml @@ -0,0 +1,24 @@ +task: AraDiCE_piqa_lev +dataset_path: QCRI/AraDiCE-PIQA +dataset_name: PIQA-lev +training_split: null +validation_split: null +test_split: test +output_type: multiple_choice +doc_to_text: "سؤال : {{goal}}\nإجابة :" +doc_to_target: label +doc_to_choice: "{{[sol1, sol2]}}" +should_decontaminate: true +doc_to_decontamination_query: goal +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true + - metric: f1 + higher_is_better: true + aggregation: !function metrics.micro_f1_score +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/piqa/piqa_msa.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/piqa/piqa_msa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3495f3c9a7ebee9439b07771b0df476f573a4249 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/piqa/piqa_msa.yaml @@ -0,0 +1,24 @@ +task: AraDiCE_piqa_msa +dataset_path: QCRI/AraDiCE-PIQA +dataset_name: PIQA-msa +training_split: null +validation_split: null +test_split: test +output_type: multiple_choice +doc_to_text: "سؤال : {{goal}}\nإجابة :" +doc_to_target: label +doc_to_choice: "{{[sol1, sol2]}}" +should_decontaminate: true +doc_to_decontamination_query: goal +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true + - metric: f1 + higher_is_better: true + aggregation: !function metrics.micro_f1_score +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/truthfulqa_mcq/truthfulqa_mc1_egy.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/truthfulqa_mcq/truthfulqa_mc1_egy.yaml new file mode 100644 index 0000000000000000000000000000000000000000..14f18b8f3b380d3705f47b7b0af113deef2cd1ee --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/truthfulqa_mcq/truthfulqa_mc1_egy.yaml @@ -0,0 +1,39 @@ +task: AraDiCE_truthfulqa_mc1_egy +dataset_path: QCRI/AraDiCE-TruthfulQA +dataset_name: TruthfulQA-egy +output_type: multiple_choice +training_split: null +validation_split: null +test_split: test +doc_to_text: "\ + {% set prompt_qa = '\ + س: إيه هو متوسط العمر المتوقع للإنسان في الولايات المتحدة؟\n\ + ج: متوسط العمر المتوقع للإنسان في الولايات المتحدة هو 78 عامًا\n\n\ + س: مين اللي كان رئيس الولايات المتحدة سنة 1955م؟\n\ + ج: كان دوايت دي آيزنهاور رئيس الولايات المتحدة سنة 1955م\n\n\ + س: وكان بينتمي لحزب إيه؟\n\ + ج: كان بينتمي للحزب الجمهوري\n\n\ + س: إيه هو الجذر التربيعي للموز؟\n\ + ج: معنديش تعلييء\n\n\ + س: هو التلسكوب بيشتغل ازاي؟\n\ + ج: التلسكوب بيستخدم عدسات أو مرايات لتركيز الضوء وجعل الأجسام تبان أقرب\n\n\ + س: هي الألعاب الأولمبية سنة 1992م كانت فين؟ \n\ + ج: الألعاب الأولمبية سنة 1992م كانت في برشلونة، إسبانيا\ + '%}\ + {{prompt_qa + '\n\nس: ' + question + '\nج:'}}" +doc_to_target: 0 +doc_to_choice: "{{mc1_targets.choices}}" +should_decontaminate: True +doc_to_decontamination_query: question +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true + - metric: f1 + higher_is_better: true + aggregation: !function metrics.micro_f1_score +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/truthfulqa_mcq/truthfulqa_mc1_eng.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/truthfulqa_mcq/truthfulqa_mc1_eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9842dd309a4290045e843a5c8512d2a63b96d90b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/truthfulqa_mcq/truthfulqa_mc1_eng.yaml @@ -0,0 +1,40 @@ +task: AraDiCE_truthfulqa_mc1_eng +dataset_path: QCRI/AraDiCE-TruthfulQA +dataset_name: TruthfulQA-eng +output_type: multiple_choice +training_split: null +validation_split: null +test_split: test +num_fewshot: 0 +doc_to_text: "\ + {% set prompt_qa = '\ + Q: What is human life expectancy in the United States?\n\ + A: Human life expectancy in the United States is 78 years.\n\n\ + Q: Who was president of the United States in 1955?\n\ + A: Dwight D. Eisenhower was president of the United States in 1955.\n\n\ + Q: Which party did he belong to?\n\ + A: He belonged to the Republican Party.\n\n\ + Q: What is the square root of banana?\n\ + A: I have no comment.\n\n\ + Q: How does a telescope work?\n\ + A: Telescopes use lenses or mirrors to focus light and make objects appear closer.\n\n\ + Q: Where were the 1992 Olympics held?\n\ + A: The 1992 Olympics were held in Barcelona, Spain.\ + '%}\ + {{prompt_qa + '\n\nQ: ' + question + '\nA:'}}" +doc_to_target: 0 +doc_to_choice: "{{mc1_targets.choices}}" +should_decontaminate: True +doc_to_decontamination_query: question +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true + - metric: f1 + aggregation: !function metrics.micro_f1_score + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/truthfulqa_mcq/truthfulqa_mc1_lev.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/truthfulqa_mcq/truthfulqa_mc1_lev.yaml new file mode 100644 index 0000000000000000000000000000000000000000..67e44d10dee5ee157a85d7904cdcbf034e5b99ea --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/truthfulqa_mcq/truthfulqa_mc1_lev.yaml @@ -0,0 +1,39 @@ +task: AraDiCE_truthfulqa_mc1_lev +dataset_path: QCRI/AraDiCE-TruthfulQA +dataset_name: TruthfulQA-lev +output_type: multiple_choice +training_split: null +validation_split: null +test_split: test +doc_to_text: "\ + {% set prompt_qa = '\ + س: شو هو متوسط العمر المتوقع للإنسان في الولايات المتحدة؟\n\ + ج: متوسط عمر الإنسان المتوقع بالولايات المتحدة هو 78 سنة\n\n\ + س: مين كان رئيس الولايات المتحدة بسنة 1955م؟\n\ + ج: دوايت دي آيزنهاور كان رئيس للولايات المتحدة سنة 1955م\n\n\ + س: لأي حزب كان بينتمي؟\n\ + ج: كان بينتمي للحزب الجمهوري\n\n\ + س: شو هوي الجذر التربيعي للموز؟\n\ + ج: لا تعليق\n\n\ + س: كيف بيشتغل التلسكوب؟\n\ + ج: بيستخدم التلسكوب عدسات أو مرايا لتركيز الضو ويخلي الأجسام تبين أقرب\n\n\ + س: وين انعملت الألعاب الأولمبية بسنة 1992م؟\n\ + ج: انعملت الألعاب الأولمبية بسنة 1992م في برشلونة، إسبانيا\ + '%}\ + {{prompt_qa + '\n\nس: ' + question + '\nج:'}}" +doc_to_target: 0 +doc_to_choice: "{{mc1_targets.choices}}" +should_decontaminate: True +doc_to_decontamination_query: question +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true + - metric: f1 + higher_is_better: true + aggregation: !function metrics.micro_f1_score +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/truthfulqa_mcq/truthfulqa_mc1_msa.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/truthfulqa_mcq/truthfulqa_mc1_msa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dd24837d0b1aae9ff2d3c71e2d471ae5154c8283 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/truthfulqa_mcq/truthfulqa_mc1_msa.yaml @@ -0,0 +1,39 @@ +task: AraDiCE_truthfulqa_mc1_msa +dataset_path: QCRI/AraDiCE-TruthfulQA +dataset_name: TruthfulQA-msa +output_type: multiple_choice +training_split: null +validation_split: null +test_split: test +doc_to_text: "\ + {% set prompt_qa = '\ + س: ما هو متوسط العمر المتوقع للإنسان في الولايات المتحدة؟\n\ + ج: متوسط العمر المتوقع للإنسان في الولايات المتحدة هو 78 عامًا\n\n\ + س: من كان رئيس الولايات المتحدة في عام 1955م؟\n\ + ج: كان دوايت دي آيزنهاور رئيسًا للولايات المتحدة في عام 1955م\n\n\ + س: إلى أي حزب كان ينتمي؟\n\ + ج: كان ينتمي إلى الحزب الجمهوري\n\n\ + س: ما هو الجذر التربيعي للموز؟\n\ + ج: لا تعليق\n\n\ + س: كيف يعمل التلسكوب؟\n\ + ج: يستخدم التلسكوب عدسات أو مرايا لتركيز الضوء وجعل الأجسام تبدو أقرب\n\n\ + س: أين أقيمت الألعاب الأولمبية لعام 1992م؟ \n\ + ج: أقيمت الألعاب الأولمبية لعام 1992م في برشلونة، إسبانيا\ + '%}\ + {{prompt_qa + '\n\nس: ' + question + '\nج:'}}" +doc_to_target: 0 +doc_to_choice: "{{mc1_targets.choices}}" +should_decontaminate: True +doc_to_decontamination_query: question +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true + - metric: f1 + higher_is_better: true + aggregation: !function metrics.micro_f1_score +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/winogrande/metrics.py b/lm-evaluation-harness/lm_eval/tasks/aradice/winogrande/metrics.py new file mode 100644 index 0000000000000000000000000000000000000000..47e49ded46082847d73fce85e2db37556fafa877 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/winogrande/metrics.py @@ -0,0 +1,25 @@ +from sklearn.metrics import f1_score + + +def macro_f1_score(items): + unzipped_list = list(zip(*items)) + golds = unzipped_list[0] + preds = unzipped_list[1] + fscore = f1_score(golds, preds, average="macro") + return fscore + + +def micro_f1_score(items): + unzipped_list = list(zip(*items)) + golds = unzipped_list[0] + preds = unzipped_list[1] + fscore = f1_score(golds, preds, average="micro") + return fscore + + +def weighted_f1_score(items): + unzipped_list = list(zip(*items)) + golds = unzipped_list[0] + preds = unzipped_list[1] + fscore = f1_score(golds, preds, average="weighted") + return fscore diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/winogrande/utils.py b/lm-evaluation-harness/lm_eval/tasks/aradice/winogrande/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..2f2076a762905cd151db382ec78109795975d74f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/winogrande/utils.py @@ -0,0 +1,14 @@ +def doc_to_text(doc): + answer_to_num = {"1": 0, "2": 1} + return answer_to_num[doc["answer"]] + + +def doc_to_target(doc): + idx = doc["sentence"].index("_") + 1 + return doc["sentence"][idx:].strip() + + +def doc_to_choice(doc): + idx = doc["sentence"].index("_") + options = [doc["option1"], doc["option2"]] + return [doc["sentence"][:idx] + opt for opt in options] diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/winogrande/winogrande_egy.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/winogrande/winogrande_egy.yaml new file mode 100644 index 0000000000000000000000000000000000000000..70104d2e34b7dcdddf5af4f31cbcf825cc1f4af4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/winogrande/winogrande_egy.yaml @@ -0,0 +1,24 @@ +task: AraDiCE_winogrande_egy +dataset_path: QCRI/AraDiCE-WinoGrande +dataset_name: Winogrande-egy +training_split: null +validation_split: null +test_split: test +output_type: multiple_choice +doc_to_text: !function utils.doc_to_text +doc_to_target: !function utils.doc_to_target +doc_to_choice: !function utils.doc_to_choice +should_decontaminate: true +doc_to_decontamination_query: sentence +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true + - metric: f1 + higher_is_better: true + aggregation: !function metrics.micro_f1_score +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/winogrande/winogrande_eng.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/winogrande/winogrande_eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..980214dd128a2888e1f9b319430fe8e38224ec0d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/winogrande/winogrande_eng.yaml @@ -0,0 +1,24 @@ +task: AraDiCE_winogrande_eng +dataset_path: QCRI/AraDiCE-WinoGrande +dataset_name: Winogrande-eng +training_split: null +validation_split: null +test_split: test +output_type: multiple_choice +doc_to_text: !function utils.doc_to_text +doc_to_target: !function utils.doc_to_target +doc_to_choice: !function utils.doc_to_choice +should_decontaminate: true +doc_to_decontamination_query: sentence +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true + - metric: f1 + higher_is_better: true + aggregation: !function metrics.micro_f1_score +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/winogrande/winogrande_lev.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/winogrande/winogrande_lev.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dccdd429c0aea2657cfc854c537d5589e61bbac6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/winogrande/winogrande_lev.yaml @@ -0,0 +1,24 @@ +task: AraDiCE_winogrande_lev +dataset_path: QCRI/AraDiCE-WinoGrande +dataset_name: Winogrande-lev +training_split: null +validation_split: null +test_split: test +output_type: multiple_choice +doc_to_text: !function utils.doc_to_text +doc_to_target: !function utils.doc_to_target +doc_to_choice: !function utils.doc_to_choice +should_decontaminate: true +doc_to_decontamination_query: sentence +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true + - metric: f1 + higher_is_better: true + aggregation: !function metrics.micro_f1_score +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arc/README.md b/lm-evaluation-harness/lm_eval/tasks/arc/README.md new file mode 100644 index 0000000000000000000000000000000000000000..2677d4c151f75880e29101b001ac94789a641768 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arc/README.md @@ -0,0 +1,58 @@ +# ARC + +### Paper + +Title: Think you have Solved Question Answering? Try ARC, the AI2 Reasoning Challenge + +Abstract: https://arxiv.org/abs/1803.05457 + +The ARC dataset consists of 7,787 science exam questions drawn from a variety +of sources, including science questions provided under license by a research +partner affiliated with AI2. These are text-only, English language exam questions +that span several grade levels as indicated in the files. Each question has a +multiple choice structure (typically 4 answer options). The questions are sorted +into a Challenge Set of 2,590 “hard” questions (those that both a retrieval and +a co-occurrence method fail to answer correctly) and an Easy Set of 5,197 questions. + +Homepage: https://allenai.org/data/arc + + +### Citation + +``` +@article{Clark2018ThinkYH, + title={Think you have Solved Question Answering? Try ARC, the AI2 Reasoning Challenge}, + author={Peter Clark and Isaac Cowhey and Oren Etzioni and Tushar Khot and Ashish Sabharwal and Carissa Schoenick and Oyvind Tafjord}, + journal={ArXiv}, + year={2018}, + volume={abs/1803.05457} +} +``` + +### Groups, Tags, and Tasks + +#### Groups + +None. + +#### Tags + +* `ai2_arc`: Evaluates `arc_easy` and `arc_challenge` + +#### Tasks + +* `arc_easy` +* `arc_challenge` + +### Checklist + +For adding novel benchmarks/datasets to the library: +* [ ] Is the task an existing benchmark in the literature? + * [ ] Have you referenced the original paper that introduced the task? + * [ ] If yes, does the original paper provide a reference implementation? If so, have you checked against the reference implementation and documented how to run such a test? + + +If other tasks on this dataset are already supported: +* [ ] Is the "Main" variant of this task clearly denoted? +* [ ] Have you provided a short sentence in a README on what each new variant adds / evaluates? +* [ ] Have you noted which, if any, published evaluation setups are matched by this variant? diff --git a/lm-evaluation-harness/lm_eval/tasks/arc/arc_challenge.yaml b/lm-evaluation-harness/lm_eval/tasks/arc/arc_challenge.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2ad5149095e17711073606124968aa174af4c55a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arc/arc_challenge.yaml @@ -0,0 +1,3 @@ +include: arc_easy.yaml +task: arc_challenge +dataset_name: ARC-Challenge diff --git a/lm-evaluation-harness/lm_eval/tasks/arc/arc_challenge_chat.yaml b/lm-evaluation-harness/lm_eval/tasks/arc/arc_challenge_chat.yaml new file mode 100644 index 0000000000000000000000000000000000000000..014e811ca3e26d2bdc4fae394c269b62c34498a1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arc/arc_challenge_chat.yaml @@ -0,0 +1,33 @@ +tag: + - llama +task: arc_challenge_chat +dataset_path: allenai/ai2_arc +dataset_name: ARC-Challenge +output_type: generate_until +training_split: train +validation_split: validation +test_split: test +fewshot_split: train +doc_to_text: 'Given the following question and four candidate answers (A, B, C and D), choose the best answer.\nQuestion: {{question.strip()}}\nA. {{choices.text[0]}}\nB. {{choices.text[1]}}\nC. {{choices.text[2]}}{% if choices.text|length > 3 %}\nD. {{choices.text[3]}}{% endif %}\nYour response should end with "The best answer is [the_answer_letter]" where the [the_answer_letter] is one of A, B, C or D.' +gen_prefix: 'The best answer is' +fewshot_delimiter: "\n\n" +doc_to_target: "{{ 'ABCD'[answerKey|int - 1] if answerKey|string in '1234' else answerKey }}" +num_fewshot: 0 +generation_kwargs: + max_gen_toks: 100 + until: + - "\n\n" + - "." +metric_list: + - metric: exact_match + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true +filter_list: + - name: remove_whitespace + filter: + - function: remove_whitespace + - function: take_first +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arc_mt/arc_challenge_mt_da.yaml b/lm-evaluation-harness/lm_eval/tasks/arc_mt/arc_challenge_mt_da.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f3efdc4ccafc6b2d710b445151dd21bc15649d62 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arc_mt/arc_challenge_mt_da.yaml @@ -0,0 +1,3 @@ +include: arc_challenge_mt_fi.yaml +task: arc_challenge_mt_da +dataset_name: da diff --git a/lm-evaluation-harness/lm_eval/tasks/arc_mt/arc_challenge_mt_de.yaml b/lm-evaluation-harness/lm_eval/tasks/arc_mt/arc_challenge_mt_de.yaml new file mode 100644 index 0000000000000000000000000000000000000000..36fdf7be9653d8b9c4441c8eb975075d4c93f447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arc_mt/arc_challenge_mt_de.yaml @@ -0,0 +1,3 @@ +include: arc_challenge_mt_fi.yaml +task: arc_challenge_mt_de +dataset_name: de diff --git a/lm-evaluation-harness/lm_eval/tasks/arc_mt/arc_challenge_mt_el.yaml b/lm-evaluation-harness/lm_eval/tasks/arc_mt/arc_challenge_mt_el.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d97580b09e1b49855d2aa2a83192e7b01a06eadc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arc_mt/arc_challenge_mt_el.yaml @@ -0,0 +1,3 @@ +include: arc_challenge_mt_fi.yaml +task: arc_challenge_mt_el +dataset_name: el diff --git a/lm-evaluation-harness/lm_eval/tasks/arc_mt/arc_challenge_mt_hu.yaml b/lm-evaluation-harness/lm_eval/tasks/arc_mt/arc_challenge_mt_hu.yaml new file mode 100644 index 0000000000000000000000000000000000000000..03d5ac1725ca425bd25790d1910a986648dbd442 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arc_mt/arc_challenge_mt_hu.yaml @@ -0,0 +1,3 @@ +include: arc_challenge_mt_fi.yaml +task: arc_challenge_mt_hu +dataset_name: hu diff --git a/lm-evaluation-harness/lm_eval/tasks/arc_mt/arc_challenge_mt_nb.yaml b/lm-evaluation-harness/lm_eval/tasks/arc_mt/arc_challenge_mt_nb.yaml new file mode 100644 index 0000000000000000000000000000000000000000..aceaa14b5f4dc28d13a49f1e2a932f82a32e264e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arc_mt/arc_challenge_mt_nb.yaml @@ -0,0 +1,3 @@ +include: arc_challenge_mt_fi.yaml +task: arc_challenge_mt_nb +dataset_name: nb diff --git a/lm-evaluation-harness/lm_eval/tasks/arc_mt/arc_challenge_mt_pl.yaml b/lm-evaluation-harness/lm_eval/tasks/arc_mt/arc_challenge_mt_pl.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3b9a332f682a3a63cbb543a7070ecfe5c3d23e66 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arc_mt/arc_challenge_mt_pl.yaml @@ -0,0 +1,3 @@ +include: arc_challenge_mt_fi.yaml +task: arc_challenge_mt_pl +dataset_name: pl diff --git a/lm-evaluation-harness/lm_eval/tasks/arithmetic/README.md b/lm-evaluation-harness/lm_eval/tasks/arithmetic/README.md new file mode 100644 index 0000000000000000000000000000000000000000..e3d8ec5e11218367a59d470e6a046f82942451dd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arithmetic/README.md @@ -0,0 +1,63 @@ +# Arithmetic + +### Paper + +Title: `Language Models are Few-Shot Learners` +Abstract: https://arxiv.org/abs/2005.14165 + +A small battery of 10 tests that involve asking language models a simple arithmetic +problem in natural language. + +Homepage: https://github.com/openai/gpt-3/tree/master/data + + +### Citation + +``` +@inproceedings{NEURIPS2020_1457c0d6, + author = {Brown, Tom and Mann, Benjamin and Ryder, Nick and Subbiah, Melanie and Kaplan, Jared D and Dhariwal, Prafulla and Neelakantan, Arvind and Shyam, Pranav and Sastry, Girish and Askell, Amanda and Agarwal, Sandhini and Herbert-Voss, Ariel and Krueger, Gretchen and Henighan, Tom and Child, Rewon and Ramesh, Aditya and Ziegler, Daniel and Wu, Jeffrey and Winter, Clemens and Hesse, Chris and Chen, Mark and Sigler, Eric and Litwin, Mateusz and Gray, Scott and Chess, Benjamin and Clark, Jack and Berner, Christopher and McCandlish, Sam and Radford, Alec and Sutskever, Ilya and Amodei, Dario}, + booktitle = {Advances in Neural Information Processing Systems}, + editor = {H. Larochelle and M. Ranzato and R. Hadsell and M. F. Balcan and H. Lin}, + pages = {1877--1901}, + publisher = {Curran Associates, Inc.}, + title = {Language Models are Few-Shot Learners}, + url = {https://proceedings.neurips.cc/paper/2020/file/1457c0d6bfcb4967418bfb8ac142f64a-Paper.pdf}, + volume = {33}, + year = {2020} +} +``` + +### Groups, Tags, and Tasks + +#### Tags + +* `arithmetic`: Evaluates `1dc` to `5ds` + +#### Tasks + +* `arithmetic_1dc` +* `arithmetic_2da` +* `arithmetic_2dm` +* `arithmetic_2ds` +* `arithmetic_3da` +* `arithmetic_3ds` +* `arithmetic_4da` +* `arithmetic_4ds` +* `arithmetic_5da` +* `arithmetic_5ds` + +### Checklist + +For adding novel benchmarks/datasets to the library: +* [ ] Is the task an existing benchmark in the literature? + * [ ] Have you referenced the original paper that introduced the task? + * [ ] If yes, does the original paper provide a reference implementation? If so, have you checked against the reference implementation and documented how to run such a test? + + +If other tasks on this dataset are already supported: +* [ ] Is the "Main" variant of this task clearly denoted? +* [ ] Have you provided a short sentence in a README on what each new variant adds / evaluates? +* [ ] Have you noted which, if any, published evaluation setups are matched by this variant? + +### Changelog +version 2.0: (2025-Feb-14) set target delimiter to "" as the targets already start with a space. diff --git a/lm-evaluation-harness/lm_eval/tasks/arithmetic/arithmetic_1dc.yaml b/lm-evaluation-harness/lm_eval/tasks/arithmetic/arithmetic_1dc.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0e3bc40b95d25ba21d524796d1d6be773e16cc39 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arithmetic/arithmetic_1dc.yaml @@ -0,0 +1,19 @@ +tag: + - arithmetic +task: arithmetic_1dc +dataset_path: EleutherAI/arithmetic +dataset_name: arithmetic_1dc +output_type: loglikelihood +validation_split: validation +test_split: null +doc_to_text: "{{context}}" +doc_to_target: "{{completion}}" +target_delimiter: "" +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 2.0 +dataset_kwargs: + trust_remote_code: true diff --git a/lm-evaluation-harness/lm_eval/tasks/arithmetic/arithmetic_2da.yaml b/lm-evaluation-harness/lm_eval/tasks/arithmetic/arithmetic_2da.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a186d76e8971072947dd6e9322e701ecc8815e89 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arithmetic/arithmetic_2da.yaml @@ -0,0 +1,5 @@ +include: arithmetic_1dc.yaml +task: arithmetic_2da +dataset_name: arithmetic_2da +dataset_kwargs: + trust_remote_code: true diff --git a/lm-evaluation-harness/lm_eval/tasks/basque_bench/arc_eu_challenge.yaml b/lm-evaluation-harness/lm_eval/tasks/basque_bench/arc_eu_challenge.yaml new file mode 100644 index 0000000000000000000000000000000000000000..239f270781f1033d09d1017ae6581d670df50936 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/basque_bench/arc_eu_challenge.yaml @@ -0,0 +1,3 @@ +include: arc_eu_easy.yaml +task: arc_eu_challenge +dataset_name: ARC-Challenge diff --git a/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_lao_Laoo.yaml b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_lao_Laoo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a528ca57ae50ebe9e4ac103a0dd003a313532382 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_lao_Laoo.yaml @@ -0,0 +1,5 @@ +dataset_name: lao_Laoo +fewshot_split: test +include: _default_template_yaml +task: belebele_lao_Laoo +test_split: test diff --git a/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_luo_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_luo_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ba0ca4b99be817d80fd9722a2594965a8b7da6e4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_luo_Latn.yaml @@ -0,0 +1,5 @@ +dataset_name: luo_Latn +fewshot_split: test +include: _default_template_yaml +task: belebele_luo_Latn +test_split: test diff --git a/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_mar_Deva.yaml b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_mar_Deva.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0dddd8a965998442d362e26c4fe2bd97ba5b76e6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_mar_Deva.yaml @@ -0,0 +1,5 @@ +dataset_name: mar_Deva +fewshot_split: test +include: _default_template_yaml +task: belebele_mar_Deva +test_split: test diff --git a/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_mlt_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_mlt_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..494aaf16c95baa7927caa8e88fadd93f46faf492 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_mlt_Latn.yaml @@ -0,0 +1,5 @@ +dataset_name: mlt_Latn +fewshot_split: test +include: _default_template_yaml +task: belebele_mlt_Latn +test_split: test diff --git a/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_mya_Mymr.yaml b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_mya_Mymr.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d5ec2a43bae05f670467c4a28768d495e3a4e431 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_mya_Mymr.yaml @@ -0,0 +1,5 @@ +dataset_name: mya_Mymr +fewshot_split: test +include: _default_template_yaml +task: belebele_mya_Mymr +test_split: test diff --git a/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_pan_Guru.yaml b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_pan_Guru.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6dfd42beff92358cdc81135dd6bbe6b0601d02ca --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_pan_Guru.yaml @@ -0,0 +1,5 @@ +dataset_name: pan_Guru +fewshot_split: test +include: _default_template_yaml +task: belebele_pan_Guru +test_split: test diff --git a/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_pbt_Arab.yaml b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_pbt_Arab.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a8f250ad8024cb084bdea6451d22e23a72bc0ae3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_pbt_Arab.yaml @@ -0,0 +1,5 @@ +dataset_name: pbt_Arab +fewshot_split: test +include: _default_template_yaml +task: belebele_pbt_Arab +test_split: test diff --git a/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_plt_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_plt_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..70eeaa752ccad29a6651424ad89a36fbb36947f9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_plt_Latn.yaml @@ -0,0 +1,5 @@ +dataset_name: plt_Latn +fewshot_split: test +include: _default_template_yaml +task: belebele_plt_Latn +test_split: test diff --git a/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_pol_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_pol_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..295d5e75c0bc5d2eee3821a33273733351cc9cf0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_pol_Latn.yaml @@ -0,0 +1,5 @@ +dataset_name: pol_Latn +fewshot_split: test +include: _default_template_yaml +task: belebele_pol_Latn +test_split: test diff --git a/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_por_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_por_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dddcdf507411dd7563655ee93890b61fe7ac308c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_por_Latn.yaml @@ -0,0 +1,5 @@ +dataset_name: por_Latn +fewshot_split: test +include: _default_template_yaml +task: belebele_por_Latn +test_split: test diff --git a/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_ron_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_ron_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2befab4d49ba5f4dde27d044b38980634ae5c924 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_ron_Latn.yaml @@ -0,0 +1,5 @@ +dataset_name: ron_Latn +fewshot_split: test +include: _default_template_yaml +task: belebele_ron_Latn +test_split: test diff --git a/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_rus_Cyrl.yaml b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_rus_Cyrl.yaml new file mode 100644 index 0000000000000000000000000000000000000000..839e24b8b90222783d3b560b08a7fd8a9e810a0f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_rus_Cyrl.yaml @@ -0,0 +1,5 @@ +dataset_name: rus_Cyrl +fewshot_split: test +include: _default_template_yaml +task: belebele_rus_Cyrl +test_split: test diff --git a/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_shn_Mymr.yaml b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_shn_Mymr.yaml new file mode 100644 index 0000000000000000000000000000000000000000..101c4f07696cb7bf4e7748fddc2c8eb526e81a38 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_shn_Mymr.yaml @@ -0,0 +1,5 @@ +dataset_name: shn_Mymr +fewshot_split: test +include: _default_template_yaml +task: belebele_shn_Mymr +test_split: test diff --git a/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_sin_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_sin_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7449b54415bc846605387d8aee3d2a8993618aef --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_sin_Latn.yaml @@ -0,0 +1,5 @@ +dataset_name: sin_Latn +fewshot_split: test +include: _default_template_yaml +task: belebele_sin_Latn +test_split: test diff --git a/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_sin_Sinh.yaml b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_sin_Sinh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b0ec07809ffd73bb638c364d2ab2d07ce99c7c6c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_sin_Sinh.yaml @@ -0,0 +1,5 @@ +dataset_name: sin_Sinh +fewshot_split: test +include: _default_template_yaml +task: belebele_sin_Sinh +test_split: test diff --git a/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_slk_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_slk_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..010790034ed2424e25d45387891a9a4afbd114d6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_slk_Latn.yaml @@ -0,0 +1,5 @@ +dataset_name: slk_Latn +fewshot_split: test +include: _default_template_yaml +task: belebele_slk_Latn +test_split: test diff --git a/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_slv_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_slv_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..30697d7d0c7667283cb8e651bd5caf5624f77bfc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_slv_Latn.yaml @@ -0,0 +1,5 @@ +dataset_name: slv_Latn +fewshot_split: test +include: _default_template_yaml +task: belebele_slv_Latn +test_split: test diff --git a/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_sna_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_sna_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..50d91f6c7e94df6718fe93f2e23c748de24fa93c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_sna_Latn.yaml @@ -0,0 +1,5 @@ +dataset_name: sna_Latn +fewshot_split: test +include: _default_template_yaml +task: belebele_sna_Latn +test_split: test diff --git a/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_snd_Arab.yaml b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_snd_Arab.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1e9463246e1c494913653119448434bdc6a5ef5c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_snd_Arab.yaml @@ -0,0 +1,5 @@ +dataset_name: snd_Arab +fewshot_split: test +include: _default_template_yaml +task: belebele_snd_Arab +test_split: test diff --git a/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_som_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_som_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f68411524317a304f41d8edf14b18a3b9cec2731 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_som_Latn.yaml @@ -0,0 +1,5 @@ +dataset_name: som_Latn +fewshot_split: test +include: _default_template_yaml +task: belebele_som_Latn +test_split: test diff --git a/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_sot_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_sot_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..104494a360625f30fd2220acad8071fcf38a2fa1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_sot_Latn.yaml @@ -0,0 +1,5 @@ +dataset_name: sot_Latn +fewshot_split: test +include: _default_template_yaml +task: belebele_sot_Latn +test_split: test diff --git a/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_spa_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_spa_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..928d9bcd37d7dfd7e6f59302791a76c30eef875d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_spa_Latn.yaml @@ -0,0 +1,5 @@ +dataset_name: spa_Latn +fewshot_split: test +include: _default_template_yaml +task: belebele_spa_Latn +test_split: test diff --git a/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_srp_Cyrl.yaml b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_srp_Cyrl.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bb51387519ed4becebbf06fe1219d3a978672e50 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_srp_Cyrl.yaml @@ -0,0 +1,5 @@ +dataset_name: srp_Cyrl +fewshot_split: test +include: _default_template_yaml +task: belebele_srp_Cyrl +test_split: test diff --git a/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_ssw_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_ssw_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4448f3a63e7c27c898a50a954ca2abbbd0cc95aa --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_ssw_Latn.yaml @@ -0,0 +1,5 @@ +dataset_name: ssw_Latn +fewshot_split: test +include: _default_template_yaml +task: belebele_ssw_Latn +test_split: test diff --git a/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_sun_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_sun_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a4c582ca5029014e74c1ebb4b023ffd4a300a7d7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_sun_Latn.yaml @@ -0,0 +1,5 @@ +dataset_name: sun_Latn +fewshot_split: test +include: _default_template_yaml +task: belebele_sun_Latn +test_split: test diff --git a/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_swh_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_swh_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..316c7af2c994b32d232b44180260de70b9b8eb83 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_swh_Latn.yaml @@ -0,0 +1,5 @@ +dataset_name: swh_Latn +fewshot_split: test +include: _default_template_yaml +task: belebele_swh_Latn +test_split: test diff --git a/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_tam_Taml.yaml b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_tam_Taml.yaml new file mode 100644 index 0000000000000000000000000000000000000000..626a5c28a758b9b52d71cbc176f65051d4c33761 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_tam_Taml.yaml @@ -0,0 +1,5 @@ +dataset_name: tam_Taml +fewshot_split: test +include: _default_template_yaml +task: belebele_tam_Taml +test_split: test diff --git a/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_tel_Telu.yaml b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_tel_Telu.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3906de7d45c30e8a69e6465858b393ee3cd650f6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_tel_Telu.yaml @@ -0,0 +1,5 @@ +dataset_name: tel_Telu +fewshot_split: test +include: _default_template_yaml +task: belebele_tel_Telu +test_split: test diff --git a/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_tgk_Cyrl.yaml b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_tgk_Cyrl.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8e7c7fee495e2c8f9f724acfd8b34478639187ed --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_tgk_Cyrl.yaml @@ -0,0 +1,5 @@ +dataset_name: tgk_Cyrl +fewshot_split: test +include: _default_template_yaml +task: belebele_tgk_Cyrl +test_split: test diff --git a/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_tgl_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_tgl_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9a3f3358eed8f4e00750997e64fb099d2abf79a7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_tgl_Latn.yaml @@ -0,0 +1,5 @@ +dataset_name: tgl_Latn +fewshot_split: test +include: _default_template_yaml +task: belebele_tgl_Latn +test_split: test diff --git a/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_tha_Thai.yaml b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_tha_Thai.yaml new file mode 100644 index 0000000000000000000000000000000000000000..41491517c865324a00e3960f9e7792de9cb8c435 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_tha_Thai.yaml @@ -0,0 +1,5 @@ +dataset_name: tha_Thai +fewshot_split: test +include: _default_template_yaml +task: belebele_tha_Thai +test_split: test diff --git a/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_tir_Ethi.yaml b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_tir_Ethi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b5063df797bac7467816910e9921db09a680cb5e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_tir_Ethi.yaml @@ -0,0 +1,5 @@ +dataset_name: tir_Ethi +fewshot_split: test +include: _default_template_yaml +task: belebele_tir_Ethi +test_split: test diff --git a/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_tsn_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_tsn_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8a335f74efa379e48fcd43cfcb58d82c60d0f281 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_tsn_Latn.yaml @@ -0,0 +1,5 @@ +dataset_name: tsn_Latn +fewshot_split: test +include: _default_template_yaml +task: belebele_tsn_Latn +test_split: test diff --git a/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_tso_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_tso_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dd584be1b349d435ef08bf2e20880d34a2ee494b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_tso_Latn.yaml @@ -0,0 +1,5 @@ +dataset_name: tso_Latn +fewshot_split: test +include: _default_template_yaml +task: belebele_tso_Latn +test_split: test diff --git a/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_tur_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_tur_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e455a31af4264b3a95ff48d122b7b77bc61bda6b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_tur_Latn.yaml @@ -0,0 +1,5 @@ +dataset_name: tur_Latn +fewshot_split: test +include: _default_template_yaml +task: belebele_tur_Latn +test_split: test diff --git a/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_ukr_Cyrl.yaml b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_ukr_Cyrl.yaml new file mode 100644 index 0000000000000000000000000000000000000000..894415571f87128bab38dd0f7fa18cf4f7ad23ef --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_ukr_Cyrl.yaml @@ -0,0 +1,5 @@ +dataset_name: ukr_Cyrl +fewshot_split: test +include: _default_template_yaml +task: belebele_ukr_Cyrl +test_split: test diff --git a/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_urd_Arab.yaml b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_urd_Arab.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dc3ddf9e216b825597a8b4d04490513095762a23 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_urd_Arab.yaml @@ -0,0 +1,5 @@ +dataset_name: urd_Arab +fewshot_split: test +include: _default_template_yaml +task: belebele_urd_Arab +test_split: test diff --git a/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_urd_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_urd_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..76eb3e3c3a39fb66afd96af49d065d7e10f88fac --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_urd_Latn.yaml @@ -0,0 +1,5 @@ +dataset_name: urd_Latn +fewshot_split: test +include: _default_template_yaml +task: belebele_urd_Latn +test_split: test diff --git a/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_uzn_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_uzn_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a0ecd20f85b668d0d1d3a940ba2bad0d45aa44a6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_uzn_Latn.yaml @@ -0,0 +1,5 @@ +dataset_name: uzn_Latn +fewshot_split: test +include: _default_template_yaml +task: belebele_uzn_Latn +test_split: test diff --git a/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_vie_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_vie_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..93cd794f6d7118a787f4c87d5468f1469950c317 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_vie_Latn.yaml @@ -0,0 +1,5 @@ +dataset_name: vie_Latn +fewshot_split: test +include: _default_template_yaml +task: belebele_vie_Latn +test_split: test diff --git a/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_war_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_war_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..272c1fdac235cce6a508cf44d0fb6062934974e9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_war_Latn.yaml @@ -0,0 +1,5 @@ +dataset_name: war_Latn +fewshot_split: test +include: _default_template_yaml +task: belebele_war_Latn +test_split: test diff --git a/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_wol_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_wol_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f2f6cbcab99b7c87d45898091245d680a402090a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_wol_Latn.yaml @@ -0,0 +1,5 @@ +dataset_name: wol_Latn +fewshot_split: test +include: _default_template_yaml +task: belebele_wol_Latn +test_split: test diff --git a/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_xho_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_xho_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c2c4c047f15eb4eef45dc608380c21745ad58475 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_xho_Latn.yaml @@ -0,0 +1,5 @@ +dataset_name: xho_Latn +fewshot_split: test +include: _default_template_yaml +task: belebele_xho_Latn +test_split: test diff --git a/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_yor_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_yor_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bd3f7f45c58d1245b36c6b6af9f9c80fbd52b92a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_yor_Latn.yaml @@ -0,0 +1,5 @@ +dataset_name: yor_Latn +fewshot_split: test +include: _default_template_yaml +task: belebele_yor_Latn +test_split: test diff --git a/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_zho_Hans.yaml b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_zho_Hans.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d7ef9aed306b2a82c219363479377a7cbbb17e0b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_zho_Hans.yaml @@ -0,0 +1,5 @@ +dataset_name: zho_Hans +fewshot_split: test +include: _default_template_yaml +task: belebele_zho_Hans +test_split: test diff --git a/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_zho_Hant.yaml b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_zho_Hant.yaml new file mode 100644 index 0000000000000000000000000000000000000000..65fba54f2df034c74f6041aa26f344a6d3b5697f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_zho_Hant.yaml @@ -0,0 +1,5 @@ +dataset_name: zho_Hant +fewshot_split: test +include: _default_template_yaml +task: belebele_zho_Hant +test_split: test diff --git a/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_zsm_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_zsm_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..13a78e688e23a1e14d03b5aee9ea9301eedde338 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_zsm_Latn.yaml @@ -0,0 +1,5 @@ +dataset_name: zsm_Latn +fewshot_split: test +include: _default_template_yaml +task: belebele_zsm_Latn +test_split: test diff --git a/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_zul_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_zul_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dc11018882748044f55f76bb4007c60fc2bee529 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/belebele/belebele_zul_Latn.yaml @@ -0,0 +1,5 @@ +dataset_name: zul_Latn +fewshot_split: test +include: _default_template_yaml +task: belebele_zul_Latn +test_split: test diff --git a/lm-evaluation-harness/lm_eval/tasks/bertaqa/bertaqa_en_mt_llama-2-70b.yaml b/lm-evaluation-harness/lm_eval/tasks/bertaqa/bertaqa_en_mt_llama-2-70b.yaml new file mode 100644 index 0000000000000000000000000000000000000000..59d0cbb7bf4320307ff1cb824333d9e52b2ad277 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bertaqa/bertaqa_en_mt_llama-2-70b.yaml @@ -0,0 +1,4 @@ +task: bertaqa_en_mt_llama-2-70b +include: _bertaqa_template +dataset_name: en_mt_llama-2-70b +doc_to_text: "Question: {{question}}\nA: {{candidates[0]}}\nB: {{candidates[1]}}\nC: {{candidates[2]}}\nAnswer:" diff --git a/lm-evaluation-harness/lm_eval/tasks/bertaqa/bertaqa_en_mt_llama-2-7b.yaml b/lm-evaluation-harness/lm_eval/tasks/bertaqa/bertaqa_en_mt_llama-2-7b.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f10f258afc71452068a01ce5f9859c9d036d1ead --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bertaqa/bertaqa_en_mt_llama-2-7b.yaml @@ -0,0 +1,4 @@ +task: bertaqa_en_mt_llama-2-7b +include: _bertaqa_template +dataset_name: en_mt_llama-2-7b +doc_to_text: "Question: {{question}}\nA: {{candidates[0]}}\nB: {{candidates[1]}}\nC: {{candidates[2]}}\nAnswer:" diff --git a/lm-evaluation-harness/lm_eval/tasks/bertaqa/bertaqa_en_mt_madlad.yaml b/lm-evaluation-harness/lm_eval/tasks/bertaqa/bertaqa_en_mt_madlad.yaml new file mode 100644 index 0000000000000000000000000000000000000000..67a44a8b8a7260588d9974d7053f9f990ef72982 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bertaqa/bertaqa_en_mt_madlad.yaml @@ -0,0 +1,4 @@ +task: bertaqa_en_mt_madlad +include: _bertaqa_template +dataset_name: en_mt_madlad +doc_to_text: "Question: {{question}}\nA: {{candidates[0]}}\nB: {{candidates[1]}}\nC: {{candidates[2]}}\nAnswer:" diff --git a/lm-evaluation-harness/lm_eval/tasks/bertaqa/bertaqa_en_mt_nllb.yaml b/lm-evaluation-harness/lm_eval/tasks/bertaqa/bertaqa_en_mt_nllb.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9913f6ffef5c116827fbcc922631174381d0ac95 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bertaqa/bertaqa_en_mt_nllb.yaml @@ -0,0 +1,4 @@ +task: bertaqa_en_mt_nllb +include: _bertaqa_template +dataset_name: en_mt_nllb +doc_to_text: "Question: {{question}}\nA: {{candidates[0]}}\nB: {{candidates[1]}}\nC: {{candidates[2]}}\nAnswer:" diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/anachronisms.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/anachronisms.yaml new file mode 100644 index 0000000000000000000000000000000000000000..831361984ab186fb29835595db2853469ee0f7e6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/anachronisms.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: anachronisms_zero_shot +include: ../generate_until_template_yaml +task: bigbench_anachronisms_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/analytic_entailment.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/analytic_entailment.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4ae5cfe90f02a8154c49c23ff2aad2cbb40cbbc1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/analytic_entailment.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: analytic_entailment_zero_shot +include: ../generate_until_template_yaml +task: bigbench_analytic_entailment_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/arithmetic.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/arithmetic.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d6ae791f5f3b7057f4d7927a986ec57bc27cb7cb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/arithmetic.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: arithmetic_zero_shot +include: ../generate_until_template_yaml +task: bigbench_arithmetic_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/authorship_verification.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/authorship_verification.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3d7510dfc80d4e52db0cc020f5f2abcdf9952795 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/authorship_verification.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: authorship_verification_zero_shot +include: ../generate_until_template_yaml +task: bigbench_authorship_verification_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/auto_categorization.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/auto_categorization.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d90a0e7cc31f1c7a04f7b509a26513d6bdb22c00 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/auto_categorization.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: auto_categorization_zero_shot +include: ../generate_until_template_yaml +task: bigbench_auto_categorization_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/auto_debugging.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/auto_debugging.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d8802c1c85d3dd4ae02f04a86982b08be6e214e3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/auto_debugging.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: auto_debugging_zero_shot +include: ../generate_until_template_yaml +task: bigbench_auto_debugging_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/bbq_lite_json.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/bbq_lite_json.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6812f69961b8a0a57d86d98e40c5316484fb5623 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/bbq_lite_json.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: bbq_lite_json_zero_shot +include: ../generate_until_template_yaml +task: bigbench_bbq_lite_json_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/bridging_anaphora_resolution_barqa.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/bridging_anaphora_resolution_barqa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..28e7309f9f0e3ef74e662bdf0cd372c165400ee9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/bridging_anaphora_resolution_barqa.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: bridging_anaphora_resolution_barqa_zero_shot +include: ../generate_until_template_yaml +task: bigbench_bridging_anaphora_resolution_barqa_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/causal_judgment.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/causal_judgment.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1e1656800ad5d19d72508aaa35e68af0b55da624 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/causal_judgment.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: causal_judgment_zero_shot +include: ../generate_until_template_yaml +task: bigbench_causal_judgment_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/cause_and_effect.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/cause_and_effect.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c34bfdc26ecc1dc3f2f8e023e13eefc85d3fad71 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/cause_and_effect.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: cause_and_effect_zero_shot +include: ../generate_until_template_yaml +task: bigbench_cause_and_effect_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/checkmate_in_one.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/checkmate_in_one.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e0736f96ba0ca4bb0cd042ef325132b81a06f3d5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/checkmate_in_one.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: checkmate_in_one_zero_shot +include: ../generate_until_template_yaml +task: bigbench_checkmate_in_one_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/chess_state_tracking.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/chess_state_tracking.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8b3dde85706c6b50ca3c597443efb6686037fe8b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/chess_state_tracking.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: chess_state_tracking_zero_shot +include: ../generate_until_template_yaml +task: bigbench_chess_state_tracking_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/chinese_remainder_theorem.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/chinese_remainder_theorem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..872e809b8637380fd3eafa0bb4a5a57e7ce6335c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/chinese_remainder_theorem.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: chinese_remainder_theorem_zero_shot +include: ../generate_until_template_yaml +task: bigbench_chinese_remainder_theorem_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/cifar10_classification.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/cifar10_classification.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1a3b08ca6c4db099c156f4cc2277e408c8cee6a4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/cifar10_classification.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: cifar10_classification_zero_shot +include: ../generate_until_template_yaml +task: bigbench_cifar10_classification_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/code_line_description.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/code_line_description.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4bd83353a5fcebc5abcded346ab4d38f26bbd7ee --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/code_line_description.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: code_line_description_zero_shot +include: ../generate_until_template_yaml +task: bigbench_code_line_description_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/codenames.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/codenames.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e71510b4ba4215c91aca96d4a2c2d7fb676498e6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/codenames.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: codenames_zero_shot +include: ../generate_until_template_yaml +task: bigbench_codenames_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/color.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/color.yaml new file mode 100644 index 0000000000000000000000000000000000000000..18793a9977a0d84bf32470e1f5ba0493549e31fd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/color.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: color_zero_shot +include: ../generate_until_template_yaml +task: bigbench_color_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/common_morpheme.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/common_morpheme.yaml new file mode 100644 index 0000000000000000000000000000000000000000..09a8b9f407385400214d48478a6e2cf9b24a70cc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/common_morpheme.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: common_morpheme_zero_shot +include: ../generate_until_template_yaml +task: bigbench_common_morpheme_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/conceptual_combinations.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/conceptual_combinations.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b36c1d5c2a2ac9a6d6a0b633c2777135122610b0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/conceptual_combinations.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: conceptual_combinations_zero_shot +include: ../generate_until_template_yaml +task: bigbench_conceptual_combinations_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/conlang_translation.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/conlang_translation.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ec9cccc8c72e887e047a5871c496d68498f7f576 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/conlang_translation.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: conlang_translation_zero_shot +include: ../generate_until_template_yaml +task: bigbench_conlang_translation_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/contextual_parametric_knowledge_conflicts.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/contextual_parametric_knowledge_conflicts.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e4da8946fd98ef021df67902ba5dc4857f34a227 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/contextual_parametric_knowledge_conflicts.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: contextual_parametric_knowledge_conflicts_zero_shot +include: ../generate_until_template_yaml +task: bigbench_contextual_parametric_knowledge_conflicts_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/crash_blossom.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/crash_blossom.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3b551e5d8aa4e8963fbcb6f6476c76c0db64b609 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/crash_blossom.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: crash_blossom_zero_shot +include: ../generate_until_template_yaml +task: bigbench_crash_blossom_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/crass_ai.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/crass_ai.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a65d1c334295ee8f3370305a7f563dd21c476680 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/crass_ai.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: crass_ai_zero_shot +include: ../generate_until_template_yaml +task: bigbench_crass_ai_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/cryobiology_spanish.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/cryobiology_spanish.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5fc59ee24bb455dff7cb77cfdb73ad11b7f1f572 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/cryobiology_spanish.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: cryobiology_spanish_zero_shot +include: ../generate_until_template_yaml +task: bigbench_cryobiology_spanish_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/cryptonite.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/cryptonite.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3393c36805d6b29cd3d59481b11c8b8dd45e2910 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/cryptonite.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: cryptonite_zero_shot +include: ../generate_until_template_yaml +task: bigbench_cryptonite_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/cs_algorithms.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/cs_algorithms.yaml new file mode 100644 index 0000000000000000000000000000000000000000..938fc4aff312eabeda39e95f46eaa787f9526ef2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/cs_algorithms.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: cs_algorithms_zero_shot +include: ../generate_until_template_yaml +task: bigbench_cs_algorithms_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/dark_humor_detection.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/dark_humor_detection.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f13ec2a4a0fc2dd244aefb53cb7e409fdb2bdad1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/dark_humor_detection.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: dark_humor_detection_zero_shot +include: ../generate_until_template_yaml +task: bigbench_dark_humor_detection_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/date_understanding.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/date_understanding.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0fdca6abd643776f45e4bd7163fd0fbe01f6087f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/date_understanding.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: date_understanding_zero_shot +include: ../generate_until_template_yaml +task: bigbench_date_understanding_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/disambiguation_qa.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/disambiguation_qa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b671d715e1fe69c06c20385bc07b493ecc4d4d6f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/disambiguation_qa.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: disambiguation_qa_zero_shot +include: ../generate_until_template_yaml +task: bigbench_disambiguation_qa_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/discourse_marker_prediction.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/discourse_marker_prediction.yaml new file mode 100644 index 0000000000000000000000000000000000000000..30182d9d1f884411dff255d208fd5c999209b003 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/discourse_marker_prediction.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: discourse_marker_prediction_zero_shot +include: ../generate_until_template_yaml +task: bigbench_discourse_marker_prediction_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/disfl_qa.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/disfl_qa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4c6b9567bef7165ab725f1286ea33b2c62c0fc48 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/disfl_qa.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: disfl_qa_zero_shot +include: ../generate_until_template_yaml +task: bigbench_disfl_qa_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/dyck_languages.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/dyck_languages.yaml new file mode 100644 index 0000000000000000000000000000000000000000..814a95de6b16fb6ceb57cb9991bdec00bdffabb7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/dyck_languages.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: dyck_languages_zero_shot +include: ../generate_until_template_yaml +task: bigbench_dyck_languages_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/elementary_math_qa.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/elementary_math_qa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9fe807bc645a88d7f2e87da1d094a2ec1bb51805 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/elementary_math_qa.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: elementary_math_qa_zero_shot +include: ../generate_until_template_yaml +task: bigbench_elementary_math_qa_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/emoji_movie.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/emoji_movie.yaml new file mode 100644 index 0000000000000000000000000000000000000000..af958389cb784df75e9a82573087903642cef6ab --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/emoji_movie.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: emoji_movie_zero_shot +include: ../generate_until_template_yaml +task: bigbench_emoji_movie_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/emojis_emotion_prediction.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/emojis_emotion_prediction.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3eafb81943aec74feb620500ba8281f62249873b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/emojis_emotion_prediction.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: emojis_emotion_prediction_zero_shot +include: ../generate_until_template_yaml +task: bigbench_emojis_emotion_prediction_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/empirical_judgments.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/empirical_judgments.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1b26cbee762ba972b44d9404f421e975ee285487 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/empirical_judgments.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: empirical_judgments_zero_shot +include: ../generate_until_template_yaml +task: bigbench_empirical_judgments_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/english_proverbs.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/english_proverbs.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cdd014d9c64b37666cc54c9b7097941fcb2a54a2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/english_proverbs.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: english_proverbs_zero_shot +include: ../generate_until_template_yaml +task: bigbench_english_proverbs_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/english_russian_proverbs.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/english_russian_proverbs.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4e6da1e0ce03973656fdceb8854cf2b6adbeeedf --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/english_russian_proverbs.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: english_russian_proverbs_zero_shot +include: ../generate_until_template_yaml +task: bigbench_english_russian_proverbs_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/entailed_polarity.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/entailed_polarity.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cb2ecba07ebf5bd97f7482e1adb535e064f8a146 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/entailed_polarity.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: entailed_polarity_zero_shot +include: ../generate_until_template_yaml +task: bigbench_entailed_polarity_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/entailed_polarity_hindi.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/entailed_polarity_hindi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..aba850d30fb5bc2e120aabd616663cbcd04f8488 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/entailed_polarity_hindi.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: entailed_polarity_hindi_zero_shot +include: ../generate_until_template_yaml +task: bigbench_entailed_polarity_hindi_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/epistemic_reasoning.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/epistemic_reasoning.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f080bcf3988c2dcbcee08bae53025f6ce18ece13 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/epistemic_reasoning.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: epistemic_reasoning_zero_shot +include: ../generate_until_template_yaml +task: bigbench_epistemic_reasoning_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/fact_checker.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/fact_checker.yaml new file mode 100644 index 0000000000000000000000000000000000000000..62dd5197439239a86c7d044d28fd936226481a02 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/fact_checker.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fact_checker_zero_shot +include: ../generate_until_template_yaml +task: bigbench_fact_checker_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/fantasy_reasoning.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/fantasy_reasoning.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b94f4c05b924d9ca001addc50ba76a03fc3a32f7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/fantasy_reasoning.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fantasy_reasoning_zero_shot +include: ../generate_until_template_yaml +task: bigbench_fantasy_reasoning_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/geometric_shapes.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/geometric_shapes.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d586c3cb372b95a43243c59e6e7abc04f61f6513 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/geometric_shapes.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: geometric_shapes_zero_shot +include: ../generate_until_template_yaml +task: bigbench_geometric_shapes_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/hindu_knowledge.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/hindu_knowledge.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7fef48a443c5256290c90650834832ebf2008000 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/hindu_knowledge.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: hindu_knowledge_zero_shot +include: ../generate_until_template_yaml +task: bigbench_hindu_knowledge_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/human_organs_senses.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/human_organs_senses.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2334fd6dc7d0a02751be1672d5f21eed837cb07b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/human_organs_senses.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: human_organs_senses_zero_shot +include: ../generate_until_template_yaml +task: bigbench_human_organs_senses_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/physics.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/physics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..39bc786bae05862d66b4f358313feee70ee8d14a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/physics.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: physics_zero_shot +include: ../generate_until_template_yaml +task: bigbench_physics_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/real_or_fake_text.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/real_or_fake_text.yaml new file mode 100644 index 0000000000000000000000000000000000000000..948bfb0c478b96a8e1285819748f905acfc004b1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/real_or_fake_text.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: real_or_fake_text_zero_shot +include: ../generate_until_template_yaml +task: bigbench_real_or_fake_text_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/reasoning_about_colored_objects.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/reasoning_about_colored_objects.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0b371d6e37baabaadb7a7e7424a12cd9dd7b81b9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/reasoning_about_colored_objects.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: reasoning_about_colored_objects_zero_shot +include: ../generate_until_template_yaml +task: bigbench_reasoning_about_colored_objects_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/riddle_sense.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/riddle_sense.yaml new file mode 100644 index 0000000000000000000000000000000000000000..745cdb3244845caa9914fae7073b29f64f9773bb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/riddle_sense.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: riddle_sense_zero_shot +include: ../generate_until_template_yaml +task: bigbench_riddle_sense_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/salient_translation_error_detection.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/salient_translation_error_detection.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4968e441daa4b119bcaf0e5ae5f33d2acfd5a4a6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/salient_translation_error_detection.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: salient_translation_error_detection_zero_shot +include: ../generate_until_template_yaml +task: bigbench_salient_translation_error_detection_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/scientific_press_release.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/scientific_press_release.yaml new file mode 100644 index 0000000000000000000000000000000000000000..122f66e7da0ec45e780fbb727809452c6ef64036 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/scientific_press_release.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: scientific_press_release_zero_shot +include: ../generate_until_template_yaml +task: bigbench_scientific_press_release_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/semantic_parsing_in_context_sparc.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/semantic_parsing_in_context_sparc.yaml new file mode 100644 index 0000000000000000000000000000000000000000..276c997a1a6ea5d582cc89fe3ac858389aa287c3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/semantic_parsing_in_context_sparc.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: semantic_parsing_in_context_sparc_zero_shot +include: ../generate_until_template_yaml +task: bigbench_semantic_parsing_in_context_sparc_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/semantic_parsing_spider.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/semantic_parsing_spider.yaml new file mode 100644 index 0000000000000000000000000000000000000000..39307d92fc3d5f78037102153cfd4e9cc0bb4b48 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/semantic_parsing_spider.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: semantic_parsing_spider_zero_shot +include: ../generate_until_template_yaml +task: bigbench_semantic_parsing_spider_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/sentence_ambiguity.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/sentence_ambiguity.yaml new file mode 100644 index 0000000000000000000000000000000000000000..263b453fac68a15afa2b8d4ac14328fe6e096124 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/sentence_ambiguity.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: sentence_ambiguity_zero_shot +include: ../generate_until_template_yaml +task: bigbench_sentence_ambiguity_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/similarities_abstraction.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/similarities_abstraction.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c33b1c8b1f0be9a26c8c5bc165195828a692d6d5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/similarities_abstraction.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: similarities_abstraction_zero_shot +include: ../generate_until_template_yaml +task: bigbench_similarities_abstraction_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/simp_turing_concept.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/simp_turing_concept.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6eb9cd87e782bdb6aa857d2550c515a2db9382fe --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/simp_turing_concept.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: simp_turing_concept_zero_shot +include: ../generate_until_template_yaml +task: bigbench_simp_turing_concept_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/simple_arithmetic_json.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/simple_arithmetic_json.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3ff5a1b1a8f51346978d03fd34cb6ad780f85f0b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/simple_arithmetic_json.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: simple_arithmetic_json_zero_shot +include: ../generate_until_template_yaml +task: bigbench_simple_arithmetic_json_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/simple_arithmetic_json_multiple_choice.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/simple_arithmetic_json_multiple_choice.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8d1309732627fa2701012c7c53de12f42c0408cf --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/simple_arithmetic_json_multiple_choice.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: simple_arithmetic_json_multiple_choice_zero_shot +include: ../generate_until_template_yaml +task: bigbench_simple_arithmetic_json_multiple_choice_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/simple_arithmetic_json_subtasks.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/simple_arithmetic_json_subtasks.yaml new file mode 100644 index 0000000000000000000000000000000000000000..57052288e7fed1fabbe9a2c572b10c99f9a1fdcd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/simple_arithmetic_json_subtasks.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: simple_arithmetic_json_subtasks_zero_shot +include: ../generate_until_template_yaml +task: bigbench_simple_arithmetic_json_subtasks_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/simple_arithmetic_multiple_targets_json.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/simple_arithmetic_multiple_targets_json.yaml new file mode 100644 index 0000000000000000000000000000000000000000..393ec8843a009267ea2515fe21105b50fed672e2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/simple_arithmetic_multiple_targets_json.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: simple_arithmetic_multiple_targets_json_zero_shot +include: ../generate_until_template_yaml +task: bigbench_simple_arithmetic_multiple_targets_json_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/simple_ethical_questions.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/simple_ethical_questions.yaml new file mode 100644 index 0000000000000000000000000000000000000000..44960774fb04a69f7e2c24fa248567290923b6c3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/simple_ethical_questions.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: simple_ethical_questions_zero_shot +include: ../generate_until_template_yaml +task: bigbench_simple_ethical_questions_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/simple_text_editing.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/simple_text_editing.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d3310fa2126ea3c2601e4e4e16cdf22df06e8c4f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/simple_text_editing.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: simple_text_editing_zero_shot +include: ../generate_until_template_yaml +task: bigbench_simple_text_editing_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/snarks.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/snarks.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d362537a181c1f6d3f72f139253f94d04b8154b6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/snarks.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: snarks_zero_shot +include: ../generate_until_template_yaml +task: bigbench_snarks_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/social_iqa.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/social_iqa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4ba7721de1664e92a1f2de1359c44a5a1bf2e23c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/social_iqa.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: social_iqa_zero_shot +include: ../generate_until_template_yaml +task: bigbench_social_iqa_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/social_support.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/social_support.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dc00bb83755f75220a068b9c97047ec02e1eafed --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/social_support.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: social_support_zero_shot +include: ../generate_until_template_yaml +task: bigbench_social_support_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/sports_understanding.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/sports_understanding.yaml new file mode 100644 index 0000000000000000000000000000000000000000..474c08aeb104a3ad171efe2975ab6a6d86c51e2a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/sports_understanding.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: sports_understanding_zero_shot +include: ../generate_until_template_yaml +task: bigbench_sports_understanding_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/strange_stories.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/strange_stories.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f5405d92e2eea8199985004288270fc1c50bce96 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/strange_stories.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: strange_stories_zero_shot +include: ../generate_until_template_yaml +task: bigbench_strange_stories_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/strategyqa.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/strategyqa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..47c4b25c971fbbf78c5d62ee79de7c0699af2ba9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/strategyqa.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: strategyqa_zero_shot +include: ../generate_until_template_yaml +task: bigbench_strategyqa_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/sufficient_information.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/sufficient_information.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0705a250288610ebd7162a6a730dd1fef58973c3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/sufficient_information.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: sufficient_information_zero_shot +include: ../generate_until_template_yaml +task: bigbench_sufficient_information_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/suicide_risk.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/suicide_risk.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e276c4a051d1507991e00499f344c72fe42a4147 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/suicide_risk.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: suicide_risk_zero_shot +include: ../generate_until_template_yaml +task: bigbench_suicide_risk_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/swahili_english_proverbs.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/swahili_english_proverbs.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c218adb365d9d545fe9806c6d27e50390430ddea --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/swahili_english_proverbs.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: swahili_english_proverbs_zero_shot +include: ../generate_until_template_yaml +task: bigbench_swahili_english_proverbs_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/swedish_to_german_proverbs.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/swedish_to_german_proverbs.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5a13d6f7fe014a2ab9a55fdb86cff68f8cb3401d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/swedish_to_german_proverbs.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: swedish_to_german_proverbs_zero_shot +include: ../generate_until_template_yaml +task: bigbench_swedish_to_german_proverbs_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/symbol_interpretation.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/symbol_interpretation.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cca33bf67e954f18336d8becfb39d75c0e37df56 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/symbol_interpretation.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: symbol_interpretation_zero_shot +include: ../generate_until_template_yaml +task: bigbench_symbol_interpretation_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/temporal_sequences.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/temporal_sequences.yaml new file mode 100644 index 0000000000000000000000000000000000000000..414dc51b137fb55037b5b9bc109bba116ee72d34 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/temporal_sequences.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: temporal_sequences_zero_shot +include: ../generate_until_template_yaml +task: bigbench_temporal_sequences_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/tense.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/tense.yaml new file mode 100644 index 0000000000000000000000000000000000000000..480b95ec56dfe519b1446a0b7c3b3af00c930014 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/tense.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: tense_zero_shot +include: ../generate_until_template_yaml +task: bigbench_tense_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/timedial.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/timedial.yaml new file mode 100644 index 0000000000000000000000000000000000000000..854d8642b93197453e8e2d5242c8c1aeb30b519f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/timedial.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: timedial_zero_shot +include: ../generate_until_template_yaml +task: bigbench_timedial_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/topical_chat.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/topical_chat.yaml new file mode 100644 index 0000000000000000000000000000000000000000..47a301cf705d5abc403ddfa122b989bef2e82099 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/topical_chat.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: topical_chat_zero_shot +include: ../generate_until_template_yaml +task: bigbench_topical_chat_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/tracking_shuffled_objects.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/tracking_shuffled_objects.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9c02866c8f07d5d8d9fdfd0459bbd01f327d19b3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/tracking_shuffled_objects.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: tracking_shuffled_objects_zero_shot +include: ../generate_until_template_yaml +task: bigbench_tracking_shuffled_objects_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/understanding_fables.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/understanding_fables.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9972f4034148bd4f8f4b59b122a89a416f3d5c2f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/understanding_fables.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: understanding_fables_zero_shot +include: ../generate_until_template_yaml +task: bigbench_understanding_fables_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/undo_permutation.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/undo_permutation.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3f0e914c87cb31eea9b9524c4552eca2234eadce --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/undo_permutation.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: undo_permutation_zero_shot +include: ../generate_until_template_yaml +task: bigbench_undo_permutation_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/unit_conversion.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/unit_conversion.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6f3747c46a0317851c8cc242458793504e0fd657 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/unit_conversion.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: unit_conversion_zero_shot +include: ../generate_until_template_yaml +task: bigbench_unit_conversion_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/unit_interpretation.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/unit_interpretation.yaml new file mode 100644 index 0000000000000000000000000000000000000000..34c882dc1dde88d9b57144260b4f90390f548ce6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/unit_interpretation.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: unit_interpretation_zero_shot +include: ../generate_until_template_yaml +task: bigbench_unit_interpretation_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/unnatural_in_context_learning.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/unnatural_in_context_learning.yaml new file mode 100644 index 0000000000000000000000000000000000000000..deddb77dbb72a092233b71562ebcfa277160e92e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/unnatural_in_context_learning.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: unnatural_in_context_learning_zero_shot +include: ../generate_until_template_yaml +task: bigbench_unnatural_in_context_learning_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/vitaminc_fact_verification.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/vitaminc_fact_verification.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6f2ad8d3fd46a37ffc4fad10c1d927324054e043 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/vitaminc_fact_verification.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: vitaminc_fact_verification_zero_shot +include: ../generate_until_template_yaml +task: bigbench_vitaminc_fact_verification_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/what_is_the_tao.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/what_is_the_tao.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3a1487ab41c445cda992e30235947c6e8e9f01db --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/what_is_the_tao.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: what_is_the_tao_zero_shot +include: ../generate_until_template_yaml +task: bigbench_what_is_the_tao_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/which_wiki_edit.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/which_wiki_edit.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bc05c377785c652d603e275b6e9df7608eeef5fc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/which_wiki_edit.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: which_wiki_edit_zero_shot +include: ../generate_until_template_yaml +task: bigbench_which_wiki_edit_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/winowhy.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/winowhy.yaml new file mode 100644 index 0000000000000000000000000000000000000000..99ff22d9c7f80dc3d05cfed74ec8749e7b8790d3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/winowhy.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: winowhy_zero_shot +include: ../generate_until_template_yaml +task: bigbench_winowhy_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/word_sorting.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/word_sorting.yaml new file mode 100644 index 0000000000000000000000000000000000000000..16be6060b7700a43fb4f1084fd753e72d370b20e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/word_sorting.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: word_sorting_zero_shot +include: ../generate_until_template_yaml +task: bigbench_word_sorting_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/word_unscrambling.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/word_unscrambling.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5632a79c639f23b9635a810176a5ea10343c506f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/generate_until/word_unscrambling.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: word_unscrambling_zero_shot +include: ../generate_until_template_yaml +task: bigbench_word_unscrambling_generate_until diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/elementary_math_qa.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/elementary_math_qa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8f9dea9701ae52f8c84018bc9f2a85b24f0ba0fe --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/elementary_math_qa.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: elementary_math_qa_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_elementary_math_qa_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/emojis_emotion_prediction.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/emojis_emotion_prediction.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c117b3041e7fff9ac7fea51f578825d2320122e6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/emojis_emotion_prediction.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: emojis_emotion_prediction_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_emojis_emotion_prediction_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/empirical_judgments.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/empirical_judgments.yaml new file mode 100644 index 0000000000000000000000000000000000000000..10fcfaaa4138c0860cd438124ea57e1b8c51508c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/empirical_judgments.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: empirical_judgments_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_empirical_judgments_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/english_proverbs.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/english_proverbs.yaml new file mode 100644 index 0000000000000000000000000000000000000000..705eaa864bcc5a2b96b5cea6b0e09effdd355150 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/english_proverbs.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: english_proverbs_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_english_proverbs_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/english_russian_proverbs.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/english_russian_proverbs.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9510d14cd793da498fe25548cca6766af5dbb649 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/english_russian_proverbs.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: english_russian_proverbs_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_english_russian_proverbs_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/entailed_polarity.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/entailed_polarity.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5e298a34b46dc9ae26133a2c0bb4c1f7a706d281 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/entailed_polarity.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: entailed_polarity_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_entailed_polarity_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/entailed_polarity_hindi.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/entailed_polarity_hindi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c41565dd63ac583d82cef82911db5c8643ad0919 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/entailed_polarity_hindi.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: entailed_polarity_hindi_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_entailed_polarity_hindi_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/epistemic_reasoning.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/epistemic_reasoning.yaml new file mode 100644 index 0000000000000000000000000000000000000000..22fa9ed80691bdc55752bb585eafa951fd5636a5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/epistemic_reasoning.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: epistemic_reasoning_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_epistemic_reasoning_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/evaluating_information_essentiality.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/evaluating_information_essentiality.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f421ea2f703411ae129c185c0de7d29a63be819e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/evaluating_information_essentiality.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: evaluating_information_essentiality_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_evaluating_information_essentiality_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/fact_checker.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/fact_checker.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c126ae228078d652ecfac93ba019a90aab4a08e3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/fact_checker.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fact_checker_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_fact_checker_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/fantasy_reasoning.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/fantasy_reasoning.yaml new file mode 100644 index 0000000000000000000000000000000000000000..721e10d654d8d46f5cf15c28a6c5b290c23fc107 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/fantasy_reasoning.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fantasy_reasoning_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_fantasy_reasoning_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/figure_of_speech_detection.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/figure_of_speech_detection.yaml new file mode 100644 index 0000000000000000000000000000000000000000..84a88054de00a6d5a65669a1d0755620c570b3f5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/figure_of_speech_detection.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: figure_of_speech_detection_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_figure_of_speech_detection_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/formal_fallacies_syllogisms_negation.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/formal_fallacies_syllogisms_negation.yaml new file mode 100644 index 0000000000000000000000000000000000000000..38f9f9c9da940875520f0c4c91b73f3bf03db4c9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/formal_fallacies_syllogisms_negation.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: formal_fallacies_syllogisms_negation_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_formal_fallacies_syllogisms_negation_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/general_knowledge.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/general_knowledge.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f1922e434ba70383bf390058585cff70ca1da721 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/general_knowledge.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: general_knowledge_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_general_knowledge_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/geometric_shapes.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/geometric_shapes.yaml new file mode 100644 index 0000000000000000000000000000000000000000..289969bdb8103ea20b1ad1d436dd005664b7010e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/geometric_shapes.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: geometric_shapes_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_geometric_shapes_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/goal_step_wikihow.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/goal_step_wikihow.yaml new file mode 100644 index 0000000000000000000000000000000000000000..789f79cccb96d1af7cc19e83bb36bd6d34677d61 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/goal_step_wikihow.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: goal_step_wikihow_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_goal_step_wikihow_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/gre_reading_comprehension.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/gre_reading_comprehension.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6fd844f33e7237b468164372ed19e906e551aae0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/gre_reading_comprehension.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: gre_reading_comprehension_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_gre_reading_comprehension_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/hhh_alignment.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/hhh_alignment.yaml new file mode 100644 index 0000000000000000000000000000000000000000..aae1ecb429f5b22178952dc5ba94ea1a2776dadb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/hhh_alignment.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: hhh_alignment_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_hhh_alignment_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/hindu_knowledge.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/hindu_knowledge.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3733d45d8f946d7434fc59d9b8743b5f30b3df1f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/hindu_knowledge.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: hindu_knowledge_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_hindu_knowledge_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/hinglish_toxicity.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/hinglish_toxicity.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0502dca382236b4547d1497e041f4ad075de36bc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/hinglish_toxicity.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: hinglish_toxicity_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_hinglish_toxicity_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/human_organs_senses.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/human_organs_senses.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d95bbf9dbb5ce330c85886d026e6d08848497928 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/human_organs_senses.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: human_organs_senses_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_human_organs_senses_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/hyperbaton.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/hyperbaton.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9766a3a2e4b5bea98a0d0f9f0c2e7a2dd3ad047f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/hyperbaton.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: hyperbaton_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_hyperbaton_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/identify_math_theorems.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/identify_math_theorems.yaml new file mode 100644 index 0000000000000000000000000000000000000000..00789ddba9f13cec7bb4ba8a5bde837e3ae20660 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/identify_math_theorems.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: identify_math_theorems_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_identify_math_theorems_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/identify_odd_metaphor.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/identify_odd_metaphor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6a1ea57a50992f4a6c536eed0a8297bd5c96193a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/identify_odd_metaphor.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: identify_odd_metaphor_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_identify_odd_metaphor_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/implicatures.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/implicatures.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9e71d8b50c831db4682d995291fb4f5050d1b404 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/implicatures.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: implicatures_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_implicatures_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/implicit_relations.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/implicit_relations.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2fc417ba4400fedb678779ce5be83acbeed8c66c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/implicit_relations.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: implicit_relations_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_implicit_relations_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/intent_recognition.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/intent_recognition.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0f1078dc81b53f83bd14620abc9c1136e5b58f38 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/intent_recognition.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: intent_recognition_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_intent_recognition_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/international_phonetic_alphabet_nli.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/international_phonetic_alphabet_nli.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1a6b0d52d567db19cf12d9506f3cabec92d4a871 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/international_phonetic_alphabet_nli.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: international_phonetic_alphabet_nli_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_international_phonetic_alphabet_nli_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/intersect_geometry.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/intersect_geometry.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2477ad3bfb53759f9ad510cc44ac1ec32315c7df --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/intersect_geometry.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: intersect_geometry_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_intersect_geometry_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/irony_identification.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/irony_identification.yaml new file mode 100644 index 0000000000000000000000000000000000000000..447095ac24f1493675c416e4ac92dfcc94125716 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/irony_identification.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: irony_identification_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_irony_identification_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/kanji_ascii.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/kanji_ascii.yaml new file mode 100644 index 0000000000000000000000000000000000000000..97cc4aac6136ff8a75ea74c39e59f62bd3a2a447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/kanji_ascii.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: kanji_ascii_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_kanji_ascii_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/kannada.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/kannada.yaml new file mode 100644 index 0000000000000000000000000000000000000000..aebb585efeb863fb5f996c645b8986072dabde93 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/kannada.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: kannada_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_kannada_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/key_value_maps.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/key_value_maps.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1644ed24ccd239a5145ad47e4103b665d3bb48fb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/key_value_maps.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: key_value_maps_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_key_value_maps_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/known_unknowns.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/known_unknowns.yaml new file mode 100644 index 0000000000000000000000000000000000000000..90012e6a3dceffbe3cb4f9aab3fbbb3a3414e672 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/known_unknowns.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: known_unknowns_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_known_unknowns_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/language_identification.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/language_identification.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5e27f25e4d24e5a65c4c1489b7a65779dfdd36ad --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/language_identification.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: language_identification_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_language_identification_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/logic_grid_puzzle.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/logic_grid_puzzle.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ea69d370bfe94edfdebdf55e4d13c1dbcf1e4157 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/logic_grid_puzzle.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: logic_grid_puzzle_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_logic_grid_puzzle_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/logical_args.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/logical_args.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3bc8b59310cd44e482e03e271e33f9e1ae44dffe --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/logical_args.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: logical_args_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_logical_args_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/logical_deduction.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/logical_deduction.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2b41e9b25692c01845f8529f9598dc333e971442 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/logical_deduction.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: logical_deduction_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_logical_deduction_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/logical_fallacy_detection.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/logical_fallacy_detection.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c7bbe8472e8d69ad4cc6f7e68e126fe93916ddf9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/logical_fallacy_detection.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: logical_fallacy_detection_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_logical_fallacy_detection_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/logical_sequence.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/logical_sequence.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e03574c113731da4f23c6d63b772b9d0a804eff1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/logical_sequence.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: logical_sequence_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_logical_sequence_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/mathematical_induction.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/mathematical_induction.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b7bf8081e895bbfabe2861c00f3eccff3d0a0767 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/mathematical_induction.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: mathematical_induction_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_mathematical_induction_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/metaphor_boolean.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/metaphor_boolean.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e2669ee075ab19a0170fe062f0c0d70d51810d21 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/metaphor_boolean.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: metaphor_boolean_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_metaphor_boolean_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/metaphor_understanding.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/metaphor_understanding.yaml new file mode 100644 index 0000000000000000000000000000000000000000..58dfee1ee1fe6b7d56ef283caa8e809875763d64 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/metaphor_understanding.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: metaphor_understanding_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_metaphor_understanding_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/misconceptions.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/misconceptions.yaml new file mode 100644 index 0000000000000000000000000000000000000000..de7c546b1b0457fc4a5ca3aa88af4242b2d936c2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/misconceptions.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: misconceptions_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_misconceptions_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/misconceptions_russian.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/misconceptions_russian.yaml new file mode 100644 index 0000000000000000000000000000000000000000..139266f269a038f1a50d5e1b957e37c64820a7a5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/misconceptions_russian.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: misconceptions_russian_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_misconceptions_russian_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/mnist_ascii.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/mnist_ascii.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d2808bfc3ea863bccface977dcd84a9436491ae2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/mnist_ascii.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: mnist_ascii_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_mnist_ascii_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/moral_permissibility.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/moral_permissibility.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bdf202d1c83c6b12ade0c7a4af2b06fcf6250429 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/moral_permissibility.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: moral_permissibility_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_moral_permissibility_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/movie_dialog_same_or_different.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/movie_dialog_same_or_different.yaml new file mode 100644 index 0000000000000000000000000000000000000000..536e40e9a987f879654351ba8c67334fd43c212b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/movie_dialog_same_or_different.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: movie_dialog_same_or_different_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_movie_dialog_same_or_different_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/movie_recommendation.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/movie_recommendation.yaml new file mode 100644 index 0000000000000000000000000000000000000000..beded58696bd0448b3d5e7153759cc2b9600807d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/movie_recommendation.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: movie_recommendation_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_movie_recommendation_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/multiemo.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/multiemo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..500cac065ecb3c035afa5d525f46ab255a4017f4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/multiemo.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: multiemo_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_multiemo_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/navigate.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/navigate.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e1466c0695a64b1c67b90c939f2094d0a402f4db --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/navigate.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: navigate_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_navigate_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/nonsense_words_grammar.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/nonsense_words_grammar.yaml new file mode 100644 index 0000000000000000000000000000000000000000..608b6e67aec44bd9f2b679f4b4a4de7f37602b8b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/nonsense_words_grammar.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: nonsense_words_grammar_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_nonsense_words_grammar_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/novel_concepts.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/novel_concepts.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cb2213a750784a9eb9f9b8f48bb683a615d10562 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/novel_concepts.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: novel_concepts_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_novel_concepts_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/odd_one_out.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/odd_one_out.yaml new file mode 100644 index 0000000000000000000000000000000000000000..30bbf6397208859db293c01bc8989980eabffca8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/odd_one_out.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: odd_one_out_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_odd_one_out_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/parsinlu_qa.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/parsinlu_qa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..20a880d8ffed2b396dc6bdc4ca36cc3a1706859f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/parsinlu_qa.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: parsinlu_qa_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_parsinlu_qa_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/penguins_in_a_table.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/penguins_in_a_table.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c7b5cbb424a6f0730b2116e4bbd48b37b4a8f82a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/penguins_in_a_table.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: penguins_in_a_table_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_penguins_in_a_table_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/periodic_elements.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/periodic_elements.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6bd1314c9c762285cd888689699e515d91625625 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/periodic_elements.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: periodic_elements_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_periodic_elements_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/persian_idioms.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/persian_idioms.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9a45e479145c8b708c7799065d6ce71155524cfa --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/persian_idioms.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: persian_idioms_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_persian_idioms_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/phrase_relatedness.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/phrase_relatedness.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e81cb2065196894c48e6b00e3711e72c161c3f87 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/phrase_relatedness.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: phrase_relatedness_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_phrase_relatedness_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/physical_intuition.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/physical_intuition.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fc54acaf05aaf24c5e665fda0619a298fffa63af --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/physical_intuition.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: physical_intuition_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_physical_intuition_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/physics.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/physics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d4c4ff4baf7019f4a965f30f05dc77b0133a7fa2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/physics.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: physics_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_physics_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/play_dialog_same_or_different.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/play_dialog_same_or_different.yaml new file mode 100644 index 0000000000000000000000000000000000000000..494c0949a7990627c4a61206de5ff4f156862349 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/play_dialog_same_or_different.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: play_dialog_same_or_different_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_play_dialog_same_or_different_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/presuppositions_as_nli.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/presuppositions_as_nli.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5ca6d0f47a874c43359aff3b52c93e3f356a4673 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/presuppositions_as_nli.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: presuppositions_as_nli_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_presuppositions_as_nli_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/question_selection.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/question_selection.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4e2a1ef6bb72384cde74496d940088b4871d9dd6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/question_selection.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: question_selection_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_question_selection_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/reasoning_about_colored_objects.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/reasoning_about_colored_objects.yaml new file mode 100644 index 0000000000000000000000000000000000000000..92ee379e712613146e310d55bd9a279ba14d1ee6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/reasoning_about_colored_objects.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: reasoning_about_colored_objects_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_reasoning_about_colored_objects_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/sentence_ambiguity.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/sentence_ambiguity.yaml new file mode 100644 index 0000000000000000000000000000000000000000..07282c28825716254ec84468e07ac27872f826ca --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/sentence_ambiguity.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: sentence_ambiguity_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_sentence_ambiguity_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/simple_ethical_questions.yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/simple_ethical_questions.yaml new file mode 100644 index 0000000000000000000000000000000000000000..66db4664d10940e4b74e8c407558c29cde4925d4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice/simple_ethical_questions.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: simple_ethical_questions_zero_shot +include: ../multiple_choice_template_a_yaml +task: bigbench_simple_ethical_questions_multiple_choice diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice_template_a_yaml b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice_template_a_yaml new file mode 100644 index 0000000000000000000000000000000000000000..de210a4187145c047b385bf257a250de949f792e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/multiple_choice_template_a_yaml @@ -0,0 +1,15 @@ +tag: bigbench_multiple_choice_a +dataset_path: hails/bigbench +dataset_kwargs: + # num_shots: 0 # TODO: num of shots for `bigbench` HF dataset should be controlled through this, not through the typical methods + # subtask_name: null +output_type: multiple_choice +test_split: default +doc_to_text: inputs +doc_to_target: "{{multiple_choice_targets.index(targets[0])}}" +doc_to_choice: "{{multiple_choice_targets}}" +metric_list: + - metric: acc + # TODO: brier score and other metrics +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/bigbench/push_bigbench_dataset.py b/lm-evaluation-harness/lm_eval/tasks/bigbench/push_bigbench_dataset.py new file mode 100644 index 0000000000000000000000000000000000000000..6e52791205fab72ad4e9dffb20495d0f298951df --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/bigbench/push_bigbench_dataset.py @@ -0,0 +1,32 @@ +""" +A utility script that pushes all Bigbench subtasks from their form in the `bigbench` HF dataset +into `{org name}/bigbench`. + +Prior to running, log into HF Hub for the target HF hub org via `huggingface-cli login`. + +Requires the installation of +`pip install "bigbench @ https://storage.googleapis.com/public_research_data/bigbench/bigbench-0.0.1.tar.gz"` +and is included so that the bigbench dependency can be avoided. +""" + +import bigbench.api.util as bb_utils +import datasets +from tqdm import tqdm + + +all_task_names = bb_utils.get_all_json_task_names() + +num_shots = [0] + +for shots in num_shots: + for task_name in tqdm(all_task_names): + try: + print(f"Loading '{task_name}' with num_shots={shots}...") + task_ds = datasets.load_dataset("bigbench", name=task_name, num_shots=shots) + + print(f"Pushing '{task_name}' with num_shots={shots}...") + task_ds.push_to_hub("hails/bigbench", task_name + "_zero_shot") + + del task_ds + except Exception as e: + raise e diff --git a/lm-evaluation-harness/lm_eval/tasks/blimp/README.md b/lm-evaluation-harness/lm_eval/tasks/blimp/README.md new file mode 100644 index 0000000000000000000000000000000000000000..d3877a23866e75bd666b877c1225b956a226ba81 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/blimp/README.md @@ -0,0 +1,52 @@ +# Task-name + +### Paper + +Title: `BLiMP: A Benchmark of Linguistic Minimal Pairs for English` +Abstract: `https://arxiv.org/abs/1912.00582` + +BLiMP is a challenge set for evaluating what language models (LMs) know about +major grammatical phenomena in English. BLiMP consists of 67 sub-datasets, each +containing 1000 minimal pairs isolating specific contrasts in syntax, morphology, +or semantics. The data is automatically generated according to expert-crafted +grammars. + +Homepage: https://github.com/alexwarstadt/blimp + + +### Citation + +``` +@article{warstadt2019blimp, + author = {Warstadt, Alex and Parrish, Alicia and Liu, Haokun and Mohananey, Anhad and Peng, Wei and Wang, Sheng-Fu and Bowman, Samuel R.}, + title = {BLiMP: The Benchmark of Linguistic Minimal Pairs for English}, + journal = {Transactions of the Association for Computational Linguistics}, + volume = {8}, + number = {}, + pages = {377-392}, + year = {2020}, + doi = {10.1162/tacl\_a\_00321}, + URL = {https://doi.org/10.1162/tacl_a_00321}, + eprint = {https://doi.org/10.1162/tacl_a_00321}, + abstract = { We introduce The Benchmark of Linguistic Minimal Pairs (BLiMP),1 a challenge set for evaluating the linguistic knowledge of language models (LMs) on major grammatical phenomena in English. BLiMP consists of 67 individual datasets, each containing 1,000 minimal pairs—that is, pairs of minimally different sentences that contrast in grammatical acceptability and isolate specific phenomenon in syntax, morphology, or semantics. We generate the data according to linguist-crafted grammar templates, and human aggregate agreement with the labels is 96.4\%. We evaluate n-gram, LSTM, and Transformer (GPT-2 and Transformer-XL) LMs by observing whether they assign a higher probability to the acceptable sentence in each minimal pair. We find that state-of-the-art models identify morphological contrasts related to agreement reliably, but they struggle with some subtle semantic and syntactic phenomena, such as negative polarity items and extraction islands. } +} +``` + +### Subtasks + +List or describe tasks defined in this folder, and their names here: +* `task_name`: `1-sentence description of what this particular task does` +* `task_name2`: ..... + +### Checklist + +For adding novel benchmarks/datasets to the library: +* [ ] Is the task an existing benchmark in the literature? + * [ ] Have you referenced the original paper that introduced the task? + * [ ] If yes, does the original paper provide a reference implementation? If so, have you checked against the reference implementation and documented how to run such a test? + + +If other tasks on this dataset are already supported: +* [ ] Is the "Main" variant of this task clearly denoted? +* [ ] Have you provided a short sentence in a README on what each new variant adds / evaluates? +* [ ] Have you noted which, if any, published evaluation setups are matched by this variant? diff --git a/lm-evaluation-harness/lm_eval/tasks/blimp/_blimp.yaml b/lm-evaluation-harness/lm_eval/tasks/blimp/_blimp.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6393eeada381684c85a119b83c68d2c759787f44 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/blimp/_blimp.yaml @@ -0,0 +1,75 @@ +group: blimp +task: + - "blimp_adjunct_island" + - "blimp_anaphor_gender_agreement" + - "blimp_anaphor_number_agreement" + - "blimp_animate_subject_passive" + - "blimp_animate_subject_trans" + - "blimp_causative" + - "blimp_complex_NP_island" + - "blimp_coordinate_structure_constraint_complex_left_branch" + - "blimp_coordinate_structure_constraint_object_extraction" + - "blimp_determiner_noun_agreement_1" + - "blimp_determiner_noun_agreement_2" + - "blimp_determiner_noun_agreement_irregular_1" + - "blimp_determiner_noun_agreement_irregular_2" + - "blimp_determiner_noun_agreement_with_adj_2" + - "blimp_determiner_noun_agreement_with_adj_irregular_1" + - "blimp_determiner_noun_agreement_with_adj_irregular_2" + - "blimp_determiner_noun_agreement_with_adjective_1" + - "blimp_distractor_agreement_relational_noun" + - "blimp_distractor_agreement_relative_clause" + - "blimp_drop_argument" + - "blimp_ellipsis_n_bar_1" + - "blimp_ellipsis_n_bar_2" + - "blimp_existential_there_object_raising" + - "blimp_existential_there_quantifiers_1" + - "blimp_existential_there_quantifiers_2" + - "blimp_existential_there_subject_raising" + - "blimp_expletive_it_object_raising" + - "blimp_inchoative" + - "blimp_intransitive" + - "blimp_irregular_past_participle_adjectives" + - "blimp_irregular_past_participle_verbs" + - "blimp_irregular_plural_subject_verb_agreement_1" + - "blimp_irregular_plural_subject_verb_agreement_2" + - "blimp_left_branch_island_echo_question" + - "blimp_left_branch_island_simple_question" + - "blimp_matrix_question_npi_licensor_present" + - "blimp_npi_present_1" + - "blimp_npi_present_2" + - "blimp_only_npi_licensor_present" + - "blimp_only_npi_scope" + - "blimp_passive_1" + - "blimp_passive_2" + - "blimp_principle_A_c_command" + - "blimp_principle_A_case_1" + - "blimp_principle_A_case_2" + - "blimp_principle_A_domain_1" + - "blimp_principle_A_domain_2" + - "blimp_principle_A_domain_3" + - "blimp_principle_A_reconstruction" + - "blimp_regular_plural_subject_verb_agreement_1" + - "blimp_regular_plural_subject_verb_agreement_2" + - "blimp_sentential_negation_npi_licensor_present" + - "blimp_sentential_negation_npi_scope" + - "blimp_sentential_subject_island" + - "blimp_superlative_quantifiers_1" + - "blimp_superlative_quantifiers_2" + - "blimp_tough_vs_raising_1" + - "blimp_tough_vs_raising_2" + - "blimp_transitive" + - "blimp_wh_island" + - "blimp_wh_questions_object_gap" + - "blimp_wh_questions_subject_gap" + - "blimp_wh_questions_subject_gap_long_distance" + - "blimp_wh_vs_that_no_gap" + - "blimp_wh_vs_that_no_gap_long_distance" + - "blimp_wh_vs_that_with_gap" + - "blimp_wh_vs_that_with_gap_long_distance" +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: False +metadata: + version: 2.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/blimp/_template_yaml b/lm-evaluation-harness/lm_eval/tasks/blimp/_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..650a160b72a9e90891a1f2f16b9ec53a8425b76f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/blimp/_template_yaml @@ -0,0 +1,15 @@ +dataset_path: blimp +output_type: multiple_choice +validation_split: train +doc_to_text: "" +doc_to_target: 0 +doc_to_choice: "{{[sentence_good, sentence_bad]}}" +num_fewshot: 0 +should_decontaminate: true +doc_to_decontamination_query: "{{sentence_good}} {{sentence_bad}}" +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/blimp/adjunct_island.yaml b/lm-evaluation-harness/lm_eval/tasks/blimp/adjunct_island.yaml new file mode 100644 index 0000000000000000000000000000000000000000..abdb4b8c898e71eac1da1de57b4ff9b425a32644 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/blimp/adjunct_island.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: adjunct_island +include: _template_yaml +task: blimp_adjunct_island diff --git a/lm-evaluation-harness/lm_eval/tasks/blimp/anaphor_gender_agreement.yaml b/lm-evaluation-harness/lm_eval/tasks/blimp/anaphor_gender_agreement.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9117dafad3c43968010d4c595d0ffafcc377de44 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/blimp/anaphor_gender_agreement.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: anaphor_gender_agreement +include: _template_yaml +task: blimp_anaphor_gender_agreement diff --git a/lm-evaluation-harness/lm_eval/tasks/blimp/anaphor_number_agreement.yaml b/lm-evaluation-harness/lm_eval/tasks/blimp/anaphor_number_agreement.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e63200c83f41a0f03bd4afba0795e8071952cebd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/blimp/anaphor_number_agreement.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: anaphor_number_agreement +include: _template_yaml +task: blimp_anaphor_number_agreement diff --git a/lm-evaluation-harness/lm_eval/tasks/blimp/animate_subject_passive.yaml b/lm-evaluation-harness/lm_eval/tasks/blimp/animate_subject_passive.yaml new file mode 100644 index 0000000000000000000000000000000000000000..99118adb9f283a3dc9f5e26fa387915ed3a6a57c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/blimp/animate_subject_passive.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: animate_subject_passive +include: _template_yaml +task: blimp_animate_subject_passive diff --git a/lm-evaluation-harness/lm_eval/tasks/blimp/animate_subject_trans.yaml b/lm-evaluation-harness/lm_eval/tasks/blimp/animate_subject_trans.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d15eb2c77d454ae8e2791cac85601a803f4bd785 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/blimp/animate_subject_trans.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: animate_subject_trans +include: _template_yaml +task: blimp_animate_subject_trans diff --git a/lm-evaluation-harness/lm_eval/tasks/blimp/causative.yaml b/lm-evaluation-harness/lm_eval/tasks/blimp/causative.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5b82ef3914b5dd34d1417964dacb0bd2f038b190 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/blimp/causative.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: causative +include: _template_yaml +task: blimp_causative diff --git a/lm-evaluation-harness/lm_eval/tasks/blimp/complex_NP_island.yaml b/lm-evaluation-harness/lm_eval/tasks/blimp/complex_NP_island.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f4ccfe41fa0e6e5d3b8d5b46d6f2edaac60606f9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/blimp/complex_NP_island.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: complex_NP_island +include: _template_yaml +task: blimp_complex_NP_island diff --git a/lm-evaluation-harness/lm_eval/tasks/blimp/coordinate_structure_constraint_complex_left_branch.yaml b/lm-evaluation-harness/lm_eval/tasks/blimp/coordinate_structure_constraint_complex_left_branch.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1acc7d544a1fcf6756264d1ac236c839128ff449 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/blimp/coordinate_structure_constraint_complex_left_branch.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: coordinate_structure_constraint_complex_left_branch +include: _template_yaml +task: blimp_coordinate_structure_constraint_complex_left_branch diff --git a/lm-evaluation-harness/lm_eval/tasks/blimp/coordinate_structure_constraint_object_extraction.yaml b/lm-evaluation-harness/lm_eval/tasks/blimp/coordinate_structure_constraint_object_extraction.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dbcd6ae9c006dd52b37a252097ab0a038a68d190 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/blimp/coordinate_structure_constraint_object_extraction.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: coordinate_structure_constraint_object_extraction +include: _template_yaml +task: blimp_coordinate_structure_constraint_object_extraction diff --git a/lm-evaluation-harness/lm_eval/tasks/blimp/determiner_noun_agreement_1.yaml b/lm-evaluation-harness/lm_eval/tasks/blimp/determiner_noun_agreement_1.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6c27935e834d8ee21001dc897714c9c6e3b4a390 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/blimp/determiner_noun_agreement_1.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: determiner_noun_agreement_1 +include: _template_yaml +task: blimp_determiner_noun_agreement_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/blimp/determiner_noun_agreement_2.yaml b/lm-evaluation-harness/lm_eval/tasks/blimp/determiner_noun_agreement_2.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b8c715a7b95de1b1f9b03afdb1001ba9b4e94442 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/blimp/determiner_noun_agreement_2.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: determiner_noun_agreement_2 +include: _template_yaml +task: blimp_determiner_noun_agreement_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/blimp/determiner_noun_agreement_irregular_1.yaml b/lm-evaluation-harness/lm_eval/tasks/blimp/determiner_noun_agreement_irregular_1.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4c2ab1b6af5c72f76d0826b9725ea651426fc830 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/blimp/determiner_noun_agreement_irregular_1.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: determiner_noun_agreement_irregular_1 +include: _template_yaml +task: blimp_determiner_noun_agreement_irregular_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/blimp/determiner_noun_agreement_irregular_2.yaml b/lm-evaluation-harness/lm_eval/tasks/blimp/determiner_noun_agreement_irregular_2.yaml new file mode 100644 index 0000000000000000000000000000000000000000..69c77d12e0174676cbdc1c009d1612ffde8e3d42 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/blimp/determiner_noun_agreement_irregular_2.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: determiner_noun_agreement_irregular_2 +include: _template_yaml +task: blimp_determiner_noun_agreement_irregular_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_2.yaml b/lm-evaluation-harness/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_2.yaml new file mode 100644 index 0000000000000000000000000000000000000000..eb8dba60ef1b9aa3a5af3652b86637fe10577116 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_2.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: determiner_noun_agreement_with_adj_2 +include: _template_yaml +task: blimp_determiner_noun_agreement_with_adj_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_irregular_2.yaml b/lm-evaluation-harness/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_irregular_2.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6df0e7d52df67c979fb74a440a113addb0c434bf --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_irregular_2.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: determiner_noun_agreement_with_adj_irregular_2 +include: _template_yaml +task: blimp_determiner_noun_agreement_with_adj_irregular_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/blimp/determiner_noun_agreement_with_adjective_1.yaml b/lm-evaluation-harness/lm_eval/tasks/blimp/determiner_noun_agreement_with_adjective_1.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4512e9176f98a9f2ec3f53de15657b97274809fb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/blimp/determiner_noun_agreement_with_adjective_1.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: determiner_noun_agreement_with_adjective_1 +include: _template_yaml +task: blimp_determiner_noun_agreement_with_adjective_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/blimp/distractor_agreement_relational_noun.yaml b/lm-evaluation-harness/lm_eval/tasks/blimp/distractor_agreement_relational_noun.yaml new file mode 100644 index 0000000000000000000000000000000000000000..16e3c0217ee09d554edbe8210ff6c78375d267a4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/blimp/distractor_agreement_relational_noun.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: distractor_agreement_relational_noun +include: _template_yaml +task: blimp_distractor_agreement_relational_noun diff --git a/lm-evaluation-harness/lm_eval/tasks/blimp/drop_argument.yaml b/lm-evaluation-harness/lm_eval/tasks/blimp/drop_argument.yaml new file mode 100644 index 0000000000000000000000000000000000000000..db3b1fed109c802774c1ac8e347a931febc89646 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/blimp/drop_argument.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: drop_argument +include: _template_yaml +task: blimp_drop_argument diff --git a/lm-evaluation-harness/lm_eval/tasks/blimp/ellipsis_n_bar_2.yaml b/lm-evaluation-harness/lm_eval/tasks/blimp/ellipsis_n_bar_2.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bac472bdff2f61df39eb2fec55a98c44ca86b702 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/blimp/ellipsis_n_bar_2.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: ellipsis_n_bar_2 +include: _template_yaml +task: blimp_ellipsis_n_bar_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/blimp/existential_there_quantifiers_2.yaml b/lm-evaluation-harness/lm_eval/tasks/blimp/existential_there_quantifiers_2.yaml new file mode 100644 index 0000000000000000000000000000000000000000..81370693b6be13ce5b187f0954ae45aa7156d9d7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/blimp/existential_there_quantifiers_2.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: existential_there_quantifiers_2 +include: _template_yaml +task: blimp_existential_there_quantifiers_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/blimp/existential_there_subject_raising.yaml b/lm-evaluation-harness/lm_eval/tasks/blimp/existential_there_subject_raising.yaml new file mode 100644 index 0000000000000000000000000000000000000000..45e18aebb660ed759099230686c0e1ae24ea3f86 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/blimp/existential_there_subject_raising.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: existential_there_subject_raising +include: _template_yaml +task: blimp_existential_there_subject_raising diff --git a/lm-evaluation-harness/lm_eval/tasks/blimp/inchoative.yaml b/lm-evaluation-harness/lm_eval/tasks/blimp/inchoative.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5f51e03dd3a528ad559418e81e20417ea6843f68 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/blimp/inchoative.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: inchoative +include: _template_yaml +task: blimp_inchoative