diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_InfluenceFromChina_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_InfluenceFromChina_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bb028b0892f077165a3fc1736bf9121065ceb9ad --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_InfluenceFromChina_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_InfluenceFromChina_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: InfluenceFromChina +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_InfluenceFromRome_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_InfluenceFromRome_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8c64cf3bbe636c0a43e1387a8bda4febcd2efd14 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_InfluenceFromRome_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_InfluenceFromRome_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: InfluenceFromRome +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Jordan_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Jordan_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..674a998e0168231c8494d3ffa34ceb00e465ea5d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Jordan_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Jordan_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: Jordan +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Libya_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Libya_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6070ccbfb884bbb77fff7b12f96ec443be654c09 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Libya_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Libya_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: Libya +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Mauritania_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Mauritania_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0b1deda61449b19ee76e41c77a405d2b876ca203 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Mauritania_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Mauritania_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: Mauritania +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Mesopotamia_civilization_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Mesopotamia_civilization_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..65474b724bc85755cc281fdf4815eb5f0b72438f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Mesopotamia_civilization_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Mesopotamia_civilization_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: Mesopotamia_civilization +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Morocco_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Morocco_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d752434a5a40a5520fdd741e98cfc5d24169b785 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Morocco_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Morocco_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: Morocco +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Oman_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Oman_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..448498f4a115ca19cab4ebe3332f208f4c717a53 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Oman_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Oman_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: Oman +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Palestine_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Palestine_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a619c460a1a353978c329e2f1756a23cdd77a2e2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Palestine_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Palestine_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: Palestine +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Saudi_Arabia_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Saudi_Arabia_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d45558b9ff3eabc0965c1c6b9f6af9d0685db2f9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Saudi_Arabia_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Saudi_Arabia_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: Saudi_Arabia +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Sudan_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Sudan_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ce599733063f8eb767b4fbde58d8a8d2253311dd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Sudan_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Sudan_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: Sudan +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Syria_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Syria_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8b0bd7aebcd361b025f903e1fb9c603197b5bf97 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Syria_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Syria_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: Syria +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Tunisia_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Tunisia_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a53c5e0bf9017efd6e8077e609b20f3ca0cce7dd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Tunisia_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Tunisia_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: Tunisia +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_United_Arab_Emirates_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_United_Arab_Emirates_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1ce5993a674c98003fefd017a320b1f72d9c9df0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_United_Arab_Emirates_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_United_Arab_Emirates_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: United_Arab_Emirates +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Yemen_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Yemen_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e480b19d605a0f28df819cc6e5eeef12eb9961ad --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Yemen_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_Yemen_light +dataset_path: OALL/ACVA +dataset_name: Yemen +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_communication_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_communication_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2814278ace335717ce3ccedff63804d48fc1b722 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_communication_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_communication_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: communication +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_computer_and_phone_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_computer_and_phone_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ddd07e3f50c2f8eecf70ae7a8c3e6b4e7330e6b2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_computer_and_phone_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_computer_and_phone_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: computer_and_phone +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_daily_life_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_daily_life_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2d975e4e85cef1b266b803d2916bcae5447ceee5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_daily_life_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_daily_life_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: daily_life +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_entertainment_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_entertainment_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..721e6cdd3b366e21e7561584e27257c4a62bc505 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_entertainment_light.yaml @@ -0,0 +1,23 @@ +task: arabic_leaderboard_acva_entertainment_light +dataset_path: arcee-globe/ACVA-10percent +dataset_name: entertainment +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{gold}}" +doc_to_choice: "choices" +fewshot_split: validation +fewshot_config: + sampler: first_n +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_light.yaml b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_light.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ea4a89771f1414322b14eed184eef7d14a7c267a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_light.yaml @@ -0,0 +1,70 @@ +group: arabic_leaderboard_acva_light +task: + - arabic_leaderboard_acva_Algeria_light + - arabic_leaderboard_acva_Ancient_Egypt_light + - arabic_leaderboard_acva_Arab_Empire_light + - arabic_leaderboard_acva_Arabic_Architecture_light + - arabic_leaderboard_acva_Arabic_Art_light + - arabic_leaderboard_acva_Arabic_Astronomy_light + - arabic_leaderboard_acva_Arabic_Calligraphy_light + - arabic_leaderboard_acva_Arabic_Ceremony_light + - arabic_leaderboard_acva_Arabic_Clothing_light + - arabic_leaderboard_acva_Arabic_Culture_light + - arabic_leaderboard_acva_Arabic_Food_light + - arabic_leaderboard_acva_Arabic_Funeral_light + - arabic_leaderboard_acva_Arabic_Geography_light + - arabic_leaderboard_acva_Arabic_History_light + - arabic_leaderboard_acva_Arabic_Language_Origin_light + - arabic_leaderboard_acva_Arabic_Literature_light + - arabic_leaderboard_acva_Arabic_Math_light + - arabic_leaderboard_acva_Arabic_Medicine_light + - arabic_leaderboard_acva_Arabic_Music_light + - arabic_leaderboard_acva_Arabic_Ornament_light + - arabic_leaderboard_acva_Arabic_Philosophy_light + - arabic_leaderboard_acva_Arabic_Physics_and_Chemistry_light + - arabic_leaderboard_acva_Arabic_Wedding_light + - arabic_leaderboard_acva_Bahrain_light + - arabic_leaderboard_acva_Comoros_light + - arabic_leaderboard_acva_Egypt_modern_light + - arabic_leaderboard_acva_InfluenceFromAncientEgypt_light + - arabic_leaderboard_acva_InfluenceFromByzantium_light + - arabic_leaderboard_acva_InfluenceFromChina_light + - arabic_leaderboard_acva_InfluenceFromGreece_light + - arabic_leaderboard_acva_InfluenceFromIslam_light + - arabic_leaderboard_acva_InfluenceFromPersia_light + - arabic_leaderboard_acva_InfluenceFromRome_light + - arabic_leaderboard_acva_Iraq_light + - arabic_leaderboard_acva_Islam_Education_light + - arabic_leaderboard_acva_Islam_branches_and_schools_light + - arabic_leaderboard_acva_Islamic_law_system_light + - arabic_leaderboard_acva_Jordan_light + - arabic_leaderboard_acva_Kuwait_light + - arabic_leaderboard_acva_Lebanon_light + - arabic_leaderboard_acva_Libya_light + - arabic_leaderboard_acva_Mauritania_light + - arabic_leaderboard_acva_Mesopotamia_civilization_light + - arabic_leaderboard_acva_Morocco_light + - arabic_leaderboard_acva_Oman_light + - arabic_leaderboard_acva_Palestine_light + - arabic_leaderboard_acva_Qatar_light + - arabic_leaderboard_acva_Saudi_Arabia_light + - arabic_leaderboard_acva_Somalia_light + - arabic_leaderboard_acva_Sudan_light + - arabic_leaderboard_acva_Syria_light + - arabic_leaderboard_acva_Tunisia_light + - arabic_leaderboard_acva_United_Arab_Emirates_light + - arabic_leaderboard_acva_Yemen_light + - arabic_leaderboard_acva_communication_light + - arabic_leaderboard_acva_computer_and_phone_light + - arabic_leaderboard_acva_daily_life_light + - arabic_leaderboard_acva_entertainment_light + +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true + - metric: acc_norm + aggregation: mean + weight_by_size: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/utils.py b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..7e91496f59df5e940e4c206bceee69007c9f159c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/utils.py @@ -0,0 +1,16 @@ +import datasets +import numpy as np + + +def process_docs(dataset: datasets.Dataset): + def _process_doc(doc): + question = doc["question"] + answer = doc["answer"] + + return { + "query": f"السؤال: {question}\nالإجابة:", + "choices": ["صح", "خطأ"], + "gold": ["صح", "خطأ"].index(answer), + } + + return dataset.map(_process_doc) diff --git a/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/_arabicmmlu_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/_arabicmmlu_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b40e7c808981291c369b7364cc52a2b153c51e43 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/_arabicmmlu_social_science.yaml @@ -0,0 +1,9 @@ +group: arabicmmlu_social_science +group_alias: Social Science +task: + - arabicmmlu_social_science_tasks +aggregate_metric_list: + - metric: acc + weight_by_size: True +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/_arabicmmlu_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/_arabicmmlu_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5065d0bde9f54ce93f2f189a9c7e3484f3e1da50 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/_arabicmmlu_stem.yaml @@ -0,0 +1,9 @@ +group: arabicmmlu_stem +group_alias: STEM +task: + - arabicmmlu_stem_tasks +aggregate_metric_list: + - metric: acc + weight_by_size: True +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/_default_arabicmmlu_template_yaml b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/_default_arabicmmlu_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..471c0fc0b44ead783ea6b80d7029b13592703ff9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/_default_arabicmmlu_template_yaml @@ -0,0 +1,15 @@ +dataset_path: MBZUAI/ArabicMMLU +test_split: test +fewshot_split: dev +fewshot_config: + sampler: first_n +output_type: multiple_choice +doc_to_text: !function utils.doc_to_text +doc_to_choice: !function utils.doc_to_choice +doc_to_target: "Answer Key" +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_arabic_language_grammar.yaml b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_arabic_language_grammar.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0511b9d91de2bc92705fee06a42d7bf68f14b231 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_arabic_language_grammar.yaml @@ -0,0 +1,5 @@ +"dataset_name": "Arabic Language (Grammar)" +"include": "_default_arabicmmlu_template_yaml" +"tag": "arabicmmlu_language_tasks" +"task": "arabicmmlu_arabic_language_grammar" +"task_alias": "Arabic Language (Grammar)" diff --git a/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_arabic_language_primary_school.yaml b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_arabic_language_primary_school.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3c0f045d8825b6a6e778211f0aead037b628b907 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_arabic_language_primary_school.yaml @@ -0,0 +1,5 @@ +"dataset_name": "Arabic Language (Primary School)" +"include": "_default_arabicmmlu_template_yaml" +"tag": "arabicmmlu_language_tasks" +"task": "arabicmmlu_arabic_language_primary_school" +"task_alias": "Arabic Language (Primary School)" diff --git a/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_biology_high_school.yaml b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_biology_high_school.yaml new file mode 100644 index 0000000000000000000000000000000000000000..865a477dee67f0ba7a038151844702906a23ea5e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_biology_high_school.yaml @@ -0,0 +1,5 @@ +"dataset_name": "Biology (High School)" +"include": "_default_arabicmmlu_template_yaml" +"tag": "arabicmmlu_stem_tasks" +"task": "arabicmmlu_biology_high_school" +"task_alias": "Biology (High School)" diff --git a/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_civics_high_school.yaml b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_civics_high_school.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6f81e9220f3def1a6198a1feaf07f78145ab2520 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_civics_high_school.yaml @@ -0,0 +1,5 @@ +"dataset_name": "Civics (High School)" +"include": "_default_arabicmmlu_template_yaml" +"tag": "arabicmmlu_social_science_tasks" +"task": "arabicmmlu_civics_high_school" +"task_alias": "Civics (High School)" diff --git a/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_civics_middle_school.yaml b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_civics_middle_school.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3e82c777caa01b761af706bffcf834f13cb02b87 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_civics_middle_school.yaml @@ -0,0 +1,5 @@ +"dataset_name": "Civics (Middle School)" +"include": "_default_arabicmmlu_template_yaml" +"tag": "arabicmmlu_social_science_tasks" +"task": "arabicmmlu_civics_middle_school" +"task_alias": "Civics (Middle School)" diff --git a/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_computer_science_middle_school.yaml b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_computer_science_middle_school.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3ecdc10616fa01a32f0ce9924f6140effc7271ef --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_computer_science_middle_school.yaml @@ -0,0 +1,5 @@ +"dataset_name": "Computer Science (Middle School)" +"include": "_default_arabicmmlu_template_yaml" +"tag": "arabicmmlu_stem_tasks" +"task": "arabicmmlu_computer_science_middle_school" +"task_alias": "Computer Science (Middle School)" diff --git a/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_computer_science_primary_school.yaml b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_computer_science_primary_school.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8feec4aaadf6bc87576f70d042aee1e2cf7461f5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_computer_science_primary_school.yaml @@ -0,0 +1,5 @@ +"dataset_name": "Computer Science (Primary School)" +"include": "_default_arabicmmlu_template_yaml" +"tag": "arabicmmlu_stem_tasks" +"task": "arabicmmlu_computer_science_primary_school" +"task_alias": "Computer Science (Primary School)" diff --git a/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_computer_science_university.yaml b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_computer_science_university.yaml new file mode 100644 index 0000000000000000000000000000000000000000..327cfab645fdd32688f3c337f1e8877a9d94e0ad --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_computer_science_university.yaml @@ -0,0 +1,5 @@ +"dataset_name": "Computer Science (University)" +"include": "_default_arabicmmlu_template_yaml" +"tag": "arabicmmlu_stem_tasks" +"task": "arabicmmlu_computer_science_university" +"task_alias": "Computer Science (University)" diff --git a/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_driving_test.yaml b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_driving_test.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ab951dfc878fa8f98a7ada1e13d3ed9e7d856494 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_driving_test.yaml @@ -0,0 +1,5 @@ +"dataset_name": "Driving Test" +"include": "_default_arabicmmlu_template_yaml" +"tag": "arabicmmlu_other_tasks" +"task": "arabicmmlu_driving_test" +"task_alias": "Driving Test" diff --git a/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_economics_high_school.yaml b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_economics_high_school.yaml new file mode 100644 index 0000000000000000000000000000000000000000..78cba021270843b6056ba1845a90556f98ade506 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_economics_high_school.yaml @@ -0,0 +1,5 @@ +"dataset_name": "Economics (High School)" +"include": "_default_arabicmmlu_template_yaml" +"tag": "arabicmmlu_social_science_tasks" +"task": "arabicmmlu_economics_high_school" +"task_alias": "Economics (High School)" diff --git a/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_economics_middle_school.yaml b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_economics_middle_school.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ed004b34a774212a6fb3f69ff05fa809d87f3fa7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_economics_middle_school.yaml @@ -0,0 +1,5 @@ +"dataset_name": "Economics (Middle School)" +"include": "_default_arabicmmlu_template_yaml" +"tag": "arabicmmlu_social_science_tasks" +"task": "arabicmmlu_economics_middle_school" +"task_alias": "Economics (Middle School)" diff --git a/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_economics_university.yaml b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_economics_university.yaml new file mode 100644 index 0000000000000000000000000000000000000000..76bfe4f1c53af0e17fef375e6676db490f68c82e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_economics_university.yaml @@ -0,0 +1,5 @@ +"dataset_name": "Economics (University)" +"include": "_default_arabicmmlu_template_yaml" +"tag": "arabicmmlu_social_science_tasks" +"task": "arabicmmlu_economics_university" +"task_alias": "Economics (University)" diff --git a/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_general_knowledge.yaml b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_general_knowledge.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8ac6e71066262860a28a47ec00b887b2b8d7329d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_general_knowledge.yaml @@ -0,0 +1,5 @@ +"dataset_name": "General Knowledge" +"include": "_default_arabicmmlu_template_yaml" +"tag": "arabicmmlu_other_tasks" +"task": "arabicmmlu_general_knowledge" +"task_alias": "General Knowledge" diff --git a/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_general_knowledge_middle_school.yaml b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_general_knowledge_middle_school.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a6e4b7c97c93a885f3ff9c40392517af8e4bad3d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_general_knowledge_middle_school.yaml @@ -0,0 +1,5 @@ +"dataset_name": "General Knowledge (Middle School)" +"include": "_default_arabicmmlu_template_yaml" +"tag": "arabicmmlu_other_tasks" +"task": "arabicmmlu_general_knowledge_middle_school" +"task_alias": "General Knowledge (Middle School)" diff --git a/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_general_knowledge_primary_school.yaml b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_general_knowledge_primary_school.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0735829975cde6ad6ddcd3bf7a12ae81f1f8852e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_general_knowledge_primary_school.yaml @@ -0,0 +1,5 @@ +"dataset_name": "General Knowledge (Primary School)" +"include": "_default_arabicmmlu_template_yaml" +"tag": "arabicmmlu_other_tasks" +"task": "arabicmmlu_general_knowledge_primary_school" +"task_alias": "General Knowledge (Primary School)" diff --git a/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_geography_high_school.yaml b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_geography_high_school.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b6264fc45801cb0af139e08bd742693058986020 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_geography_high_school.yaml @@ -0,0 +1,5 @@ +"dataset_name": "Geography (High School)" +"include": "_default_arabicmmlu_template_yaml" +"tag": "arabicmmlu_social_science_tasks" +"task": "arabicmmlu_geography_high_school" +"task_alias": "Geography (High School)" diff --git a/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_geography_middle_school.yaml b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_geography_middle_school.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6483749f897a1d353c19020186e5f43dbd5ba8ab --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_geography_middle_school.yaml @@ -0,0 +1,5 @@ +"dataset_name": "Geography (Middle School)" +"include": "_default_arabicmmlu_template_yaml" +"tag": "arabicmmlu_social_science_tasks" +"task": "arabicmmlu_geography_middle_school" +"task_alias": "Geography (Middle School)" diff --git a/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_geography_primary_school.yaml b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_geography_primary_school.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1465fb05a5e39106479b9532f99fd0b0d7349fe4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_geography_primary_school.yaml @@ -0,0 +1,5 @@ +"dataset_name": "Geography (Primary School)" +"include": "_default_arabicmmlu_template_yaml" +"tag": "arabicmmlu_social_science_tasks" +"task": "arabicmmlu_geography_primary_school" +"task_alias": "Geography (Primary School)" diff --git a/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_history_high_school.yaml b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_history_high_school.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b97a081a71bce1ddc84e2207d53993f182a8270b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_history_high_school.yaml @@ -0,0 +1,5 @@ +"dataset_name": "History (High School)" +"include": "_default_arabicmmlu_template_yaml" +"tag": "arabicmmlu_humanities_tasks" +"task": "arabicmmlu_history_high_school" +"task_alias": "History (High School)" diff --git a/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_history_middle_school.yaml b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_history_middle_school.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3435604a4159ffb2fa425813997bf38343a3f27b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_history_middle_school.yaml @@ -0,0 +1,5 @@ +"dataset_name": "History (Middle School)" +"include": "_default_arabicmmlu_template_yaml" +"tag": "arabicmmlu_humanities_tasks" +"task": "arabicmmlu_history_middle_school" +"task_alias": "History (Middle School)" diff --git a/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_history_primary_school.yaml b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_history_primary_school.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c156ff521a7af619adc9ae3ba30f08358e6751d0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_history_primary_school.yaml @@ -0,0 +1,5 @@ +"dataset_name": "History (Primary School)" +"include": "_default_arabicmmlu_template_yaml" +"tag": "arabicmmlu_humanities_tasks" +"task": "arabicmmlu_history_primary_school" +"task_alias": "History (Primary School)" diff --git a/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_islamic_studies_high_school.yaml b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_islamic_studies_high_school.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5bae042f3143a82c29c131320049a921a9bb98a4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_islamic_studies_high_school.yaml @@ -0,0 +1,5 @@ +"dataset_name": "Islamic Studies (High School)" +"include": "_default_arabicmmlu_template_yaml" +"tag": "arabicmmlu_humanities_tasks" +"task": "arabicmmlu_islamic_studies_high_school" +"task_alias": "Islamic Studies (High School)" diff --git a/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_islamic_studies_primary_school.yaml b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_islamic_studies_primary_school.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c4e5d3543dda46309d310b5eef5edebde575bb22 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_islamic_studies_primary_school.yaml @@ -0,0 +1,5 @@ +"dataset_name": "Islamic Studies (Primary School)" +"include": "_default_arabicmmlu_template_yaml" +"tag": "arabicmmlu_humanities_tasks" +"task": "arabicmmlu_islamic_studies_primary_school" +"task_alias": "Islamic Studies (Primary School)" diff --git a/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_law_professional.yaml b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_law_professional.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5e2b6a4a42c720bfadfa9a505265b88ffbcc9660 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_law_professional.yaml @@ -0,0 +1,5 @@ +"dataset_name": "Law (Professional)" +"include": "_default_arabicmmlu_template_yaml" +"tag": "arabicmmlu_humanities_tasks" +"task": "arabicmmlu_law_professional" +"task_alias": "Law (Professional)" diff --git a/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_management_university.yaml b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_management_university.yaml new file mode 100644 index 0000000000000000000000000000000000000000..386c8e6b7623a5e51c0a557fb4f8958a7604ddf4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_management_university.yaml @@ -0,0 +1,5 @@ +"dataset_name": "Management (University)" +"include": "_default_arabicmmlu_template_yaml" +"tag": "arabicmmlu_other_tasks" +"task": "arabicmmlu_management_university" +"task_alias": "Management (University)" diff --git a/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_math_primary_school.yaml b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_math_primary_school.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1df99b8a0f46d57147bfcb7c8ae16a7b6bca1f9b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_math_primary_school.yaml @@ -0,0 +1,5 @@ +"dataset_name": "Math (Primary School)" +"include": "_default_arabicmmlu_template_yaml" +"tag": "arabicmmlu_stem_tasks" +"task": "arabicmmlu_math_primary_school" +"task_alias": "Math (Primary School)" diff --git a/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_natural_science_primary_school.yaml b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_natural_science_primary_school.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1efd6c9bdf9c63d48c298639090e187588c68c7d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_natural_science_primary_school.yaml @@ -0,0 +1,5 @@ +"dataset_name": "Natural Science (Primary School)" +"include": "_default_arabicmmlu_template_yaml" +"tag": "arabicmmlu_stem_tasks" +"task": "arabicmmlu_natural_science_primary_school" +"task_alias": "Natural Science (Primary School)" diff --git a/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_philosophy_high_school.yaml b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_philosophy_high_school.yaml new file mode 100644 index 0000000000000000000000000000000000000000..66715bb054d1e907694ef86c8077ef9de16938c9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_philosophy_high_school.yaml @@ -0,0 +1,5 @@ +"dataset_name": "Philosophy (High School)" +"include": "_default_arabicmmlu_template_yaml" +"tag": "arabicmmlu_humanities_tasks" +"task": "arabicmmlu_philosophy_high_school" +"task_alias": "Philosophy (High School)" diff --git a/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_physics_high_school.yaml b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_physics_high_school.yaml new file mode 100644 index 0000000000000000000000000000000000000000..00ecf8ad181aed62a1e987d0abfc6d5c21ea15de --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_physics_high_school.yaml @@ -0,0 +1,5 @@ +"dataset_name": "Physics (High School)" +"include": "_default_arabicmmlu_template_yaml" +"tag": "arabicmmlu_stem_tasks" +"task": "arabicmmlu_physics_high_school" +"task_alias": "Physics (High School)" diff --git a/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_political_science_university.yaml b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_political_science_university.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1f64125fefbff9e2a1196dc2d281fc7496115ceb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_political_science_university.yaml @@ -0,0 +1,5 @@ +"dataset_name": "Political Science (University)" +"include": "_default_arabicmmlu_template_yaml" +"tag": "arabicmmlu_social_science_tasks" +"task": "arabicmmlu_political_science_university" +"task_alias": "Political Science (University)" diff --git a/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_social_science_middle_school.yaml b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_social_science_middle_school.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b876649f9e8b0ade91766baa95a318aaf727ef5d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_social_science_middle_school.yaml @@ -0,0 +1,5 @@ +"dataset_name": "Social Science (Middle School)" +"include": "_default_arabicmmlu_template_yaml" +"tag": "arabicmmlu_social_science_tasks" +"task": "arabicmmlu_social_science_middle_school" +"task_alias": "Social Science (Middle School)" diff --git a/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/utils.py b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..a572489e118564601243e6a6bf813b77cbe95220 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/arabicmmlu/utils.py @@ -0,0 +1,44 @@ +PROMPT = "This is a {}. Select the correct answer!\n\nQuestion: {}\n{}\n\nAnswer:" + +level_en = { + "Primary": "primary school", + "Middle": "middle school", + "High": "high school", + "Univ": "university", + "Prof": "professional", +} + +alpa = ["A.", "B.", "C.", "D.", "E."] + + +def doc_to_text(doc): + """ + Refactoring `prepare_data_en` to fit with the lm harness framework. + https://github.com/mbzuai-nlp/ArabicMMLU/blob/main/util_prompt.py + """ + + level = "" if not doc["Level"] else " for " + level_en[doc["Level"]] + country = "" if not doc["Country"] else " in " + doc["Country"] + main_meta_data = f"{doc['Subject']} question{level}{country}" + + question = ( + doc["Question"] + if not doc["Context"] + else f"{doc['Context']}\n\n{doc['Question']}" + ) + + options = [] + for i, opt in enumerate( + ["Option 1", "Option 2", "Option 3", "Option 4", "Option 5"] + ): + if not doc[opt]: + break + options.append(f"{alpa[i]} {doc[opt]}") + + doc_text = PROMPT.format(main_meta_data, question, "\n".join(options)) + + return doc_text + + +def doc_to_choice(doc): + return [alpa[i][0] for i in range(5) if doc[f"Option {i + 1}"]] diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU.yaml new file mode 100644 index 0000000000000000000000000000000000000000..77cbf95ace833b0c513034e240513bae3259caa4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU.yaml @@ -0,0 +1,12 @@ +group: AraDiCE_ArabicMMLU_egy +task: +- AraDiCE_ArabicMMLU_humanities_egy +- AraDiCE_ArabicMMLU_language_egy +- AraDiCE_ArabicMMLU_social-science_egy +- AraDiCE_ArabicMMLU_stem_egy +- AraDiCE_ArabicMMLU_other_egy +aggregate_metric_list: + - metric: acc + weight_by_size: True + - metric: acc_norm + weight_by_size: True diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_high_humanities_history.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_high_humanities_history.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5a03177d137ae08ff327788992100fb62588f139 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_high_humanities_history.yaml @@ -0,0 +1,10 @@ +"dataset_name": "high_humanities_history" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_humanities_egy" +"task": "AraDiCE_ArabicMMLU_high_humanities_history_egy" +"task_alias": "high humanities history" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_high_humanities_islamic-studies.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_high_humanities_islamic-studies.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ee65adc6dbf36ef7632ec9638ee990f8a73360d8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_high_humanities_islamic-studies.yaml @@ -0,0 +1,10 @@ +"dataset_name": "high_humanities_islamic-studies" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_humanities_egy" +"task": "AraDiCE_ArabicMMLU_high_humanities_islamic-studies_egy" +"task_alias": "high humanities islamic-studies" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_high_humanities_philosophy.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_high_humanities_philosophy.yaml new file mode 100644 index 0000000000000000000000000000000000000000..123f696f30977f71872c926325fc924e12f0dccc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_high_humanities_philosophy.yaml @@ -0,0 +1,10 @@ +"dataset_name": "high_humanities_philosophy" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_humanities_egy" +"task": "AraDiCE_ArabicMMLU_high_humanities_philosophy_egy" +"task_alias": "high humanities philosophy" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_high_language_arabic-language.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_high_language_arabic-language.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1df05181daeebb89e12dc8ca66d24becb950ab72 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_high_language_arabic-language.yaml @@ -0,0 +1,10 @@ +"dataset_name": "high_language_arabic-language" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_language_egy" +"task": "AraDiCE_ArabicMMLU_high_language_arabic-language_egy" +"task_alias": "high language arabic-language" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_high_social-science_civics.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_high_social-science_civics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7b42490b066b7919d83c0cbad44398dec147fac5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_high_social-science_civics.yaml @@ -0,0 +1,10 @@ +"dataset_name": "high_social-science_civics" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_social-science_egy" +"task": "AraDiCE_ArabicMMLU_high_social-science_civics_egy" +"task_alias": "high social-science civics" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_high_social-science_economics.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_high_social-science_economics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5518b2cda31c2f2482e56fffc4cfa7ca44bc1bb5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_high_social-science_economics.yaml @@ -0,0 +1,10 @@ +"dataset_name": "high_social-science_economics" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_social-science_egy" +"task": "AraDiCE_ArabicMMLU_high_social-science_economics_egy" +"task_alias": "high social-science economics" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_high_social-science_geography.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_high_social-science_geography.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d9a2d5b332976d20a362baf53c8633ec1132e62f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_high_social-science_geography.yaml @@ -0,0 +1,10 @@ +"dataset_name": "high_social-science_geography" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_social-science_egy" +"task": "AraDiCE_ArabicMMLU_high_social-science_geography_egy" +"task_alias": "high social-science geography" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_high_stem_biology.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_high_stem_biology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3f1ab8a7b8768712e46cb9950113c014af205dc8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_high_stem_biology.yaml @@ -0,0 +1,10 @@ +"dataset_name": "high_stem_biology" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_stem_egy" +"task": "AraDiCE_ArabicMMLU_high_stem_biology_egy" +"task_alias": "high stem biology" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_high_stem_computer-science.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_high_stem_computer-science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c27f5be3185b1140ba07e56c6335c275fa9e0b1e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_high_stem_computer-science.yaml @@ -0,0 +1,10 @@ +"dataset_name": "high_stem_computer-science" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_stem_egy" +"task": "AraDiCE_ArabicMMLU_high_stem_computer-science_egy" +"task_alias": "high stem computer-science" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_high_stem_physics.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_high_stem_physics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4e24a2f4fbbd0a7ae25abc501bae68b3909b7259 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_high_stem_physics.yaml @@ -0,0 +1,10 @@ +"dataset_name": "high_stem_physics" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_stem_egy" +"task": "AraDiCE_ArabicMMLU_high_stem_physics_egy" +"task_alias": "high stem physics" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_middle_humanities_history.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_middle_humanities_history.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9f2c3770406823a48930876f3d761a7e0bfe8e28 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_middle_humanities_history.yaml @@ -0,0 +1,10 @@ +"dataset_name": "middle_humanities_history" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_humanities_egy" +"task": "AraDiCE_ArabicMMLU_middle_humanities_history_egy" +"task_alias": "middle humanities history" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_middle_humanities_islamic-studies.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_middle_humanities_islamic-studies.yaml new file mode 100644 index 0000000000000000000000000000000000000000..41995c4aa3122b88018270586d0622b4e4c839c1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_middle_humanities_islamic-studies.yaml @@ -0,0 +1,10 @@ +"dataset_name": "middle_humanities_islamic-studies" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_humanities_egy" +"task": "AraDiCE_ArabicMMLU_middle_humanities_islamic-studies_egy" +"task_alias": "middle humanities islamic-studies" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_middle_language_arabic-language.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_middle_language_arabic-language.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e33bf590a19b7a8c42f1b52c529b5c7df4dec731 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_middle_language_arabic-language.yaml @@ -0,0 +1,10 @@ +"dataset_name": "middle_language_arabic-language" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_language_egy" +"task": "AraDiCE_ArabicMMLU_middle_language_arabic-language_egy" +"task_alias": "middle language arabic-language" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_middle_other_general-knowledge.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_middle_other_general-knowledge.yaml new file mode 100644 index 0000000000000000000000000000000000000000..73fc902702363d8f1793f59814357b6356ba2d61 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_middle_other_general-knowledge.yaml @@ -0,0 +1,10 @@ +"dataset_name": "middle_other_general-knowledge" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_other_egy" +"task": "AraDiCE_ArabicMMLU_middle_other_general-knowledge_egy" +"task_alias": "middle other general-knowledge" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_middle_social-science_civics.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_middle_social-science_civics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8407f36e7f356f75d1df8f7ce51f65734a7700bb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_middle_social-science_civics.yaml @@ -0,0 +1,10 @@ +"dataset_name": "middle_social-science_civics" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_social-science_egy" +"task": "AraDiCE_ArabicMMLU_middle_social-science_civics_egy" +"task_alias": "middle social-science civics" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_middle_social-science_economics.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_middle_social-science_economics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fbcb040d27ea95bded3a8043d984ad67ebe9eb19 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_middle_social-science_economics.yaml @@ -0,0 +1,10 @@ +"dataset_name": "middle_social-science_economics" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_social-science_egy" +"task": "AraDiCE_ArabicMMLU_middle_social-science_economics_egy" +"task_alias": "middle social-science economics" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_middle_social-science_social-science.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_middle_social-science_social-science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..115170b8cc57e5365b0e4a57660c2bf70b3b3de9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_middle_social-science_social-science.yaml @@ -0,0 +1,10 @@ +"dataset_name": "middle_social-science_social-science" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_social-science_egy" +"task": "AraDiCE_ArabicMMLU_middle_social-science_social-science_egy" +"task_alias": "middle social-science social-science" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_middle_stem_computer-science.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_middle_stem_computer-science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1d8787e3c065b0d6ac941624a5e8273b3f195fbf --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_middle_stem_computer-science.yaml @@ -0,0 +1,10 @@ +"dataset_name": "middle_stem_computer-science" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_stem_egy" +"task": "AraDiCE_ArabicMMLU_middle_stem_computer-science_egy" +"task_alias": "middle stem computer-science" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_middle_stem_natural-science.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_middle_stem_natural-science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ee09058ce4b90040f387fae7ac836f5e81d1177e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_middle_stem_natural-science.yaml @@ -0,0 +1,10 @@ +"dataset_name": "middle_stem_natural-science" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_stem_egy" +"task": "AraDiCE_ArabicMMLU_middle_stem_natural-science_egy" +"task_alias": "middle stem natural-science" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_na_humanities_islamic-studies.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_na_humanities_islamic-studies.yaml new file mode 100644 index 0000000000000000000000000000000000000000..995aa28c2f55ba68c915c1feb69abab239dcb61d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_na_humanities_islamic-studies.yaml @@ -0,0 +1,10 @@ +"dataset_name": "na_humanities_islamic-studies" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_humanities_egy" +"task": "AraDiCE_ArabicMMLU_na_humanities_islamic-studies_egy" +"task_alias": "na humanities islamic-studies" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_na_language_arabic-language-general.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_na_language_arabic-language-general.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8691250702eccbb58651aec19f77c0ec9cf9b419 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_na_language_arabic-language-general.yaml @@ -0,0 +1,10 @@ +"dataset_name": "na_language_arabic-language-general" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_language_egy" +"task": "AraDiCE_ArabicMMLU_na_language_arabic-language-general_egy" +"task_alias": "na language arabic-language-general" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_na_language_arabic-language-grammar.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_na_language_arabic-language-grammar.yaml new file mode 100644 index 0000000000000000000000000000000000000000..453e41435dc1a85a2b7860bafebaf91a188bd307 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_na_language_arabic-language-grammar.yaml @@ -0,0 +1,10 @@ +"dataset_name": "na_language_arabic-language-grammar" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_language_egy" +"task": "AraDiCE_ArabicMMLU_na_language_arabic-language-grammar_egy" +"task_alias": "na language arabic-language-grammar" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_na_other_driving-test.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_na_other_driving-test.yaml new file mode 100644 index 0000000000000000000000000000000000000000..abc097210fdb0e90aaab2c344c40648daf9c4ba0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_na_other_driving-test.yaml @@ -0,0 +1,10 @@ +"dataset_name": "na_other_driving-test" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_other_egy" +"task": "AraDiCE_ArabicMMLU_na_other_driving-test_egy" +"task_alias": "na other driving-test" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_na_other_general-knowledge.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_na_other_general-knowledge.yaml new file mode 100644 index 0000000000000000000000000000000000000000..72af8e7f5310fd895fca58e74fdd5e28119084c9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_na_other_general-knowledge.yaml @@ -0,0 +1,10 @@ +"dataset_name": "na_other_general-knowledge" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_other_egy" +"task": "AraDiCE_ArabicMMLU_na_other_general-knowledge_egy" +"task_alias": "na other general-knowledge" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_primary_humanities_history.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_primary_humanities_history.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5e640faa54b05b6f7234896207926a28000c0dfc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_primary_humanities_history.yaml @@ -0,0 +1,10 @@ +"dataset_name": "primary_humanities_history" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_humanities_egy" +"task": "AraDiCE_ArabicMMLU_primary_humanities_history_egy" +"task_alias": "primary humanities history" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_primary_humanities_islamic-studies.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_primary_humanities_islamic-studies.yaml new file mode 100644 index 0000000000000000000000000000000000000000..120dfa14350e6025f94636be53828ef14ee5fafe --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_primary_humanities_islamic-studies.yaml @@ -0,0 +1,10 @@ +"dataset_name": "primary_humanities_islamic-studies" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_humanities_egy" +"task": "AraDiCE_ArabicMMLU_primary_humanities_islamic-studies_egy" +"task_alias": "primary humanities islamic-studies" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_primary_language_arabic-language.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_primary_language_arabic-language.yaml new file mode 100644 index 0000000000000000000000000000000000000000..57c460a01b329f9a21605bf7121ef10162d597b1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_primary_language_arabic-language.yaml @@ -0,0 +1,10 @@ +"dataset_name": "primary_language_arabic-language" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_language_egy" +"task": "AraDiCE_ArabicMMLU_primary_language_arabic-language_egy" +"task_alias": "primary language arabic-language" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_primary_stem_natural-science.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_primary_stem_natural-science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..04591fcd81028726441b2bfb66d7314919c51e17 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_primary_stem_natural-science.yaml @@ -0,0 +1,10 @@ +"dataset_name": "primary_stem_natural-science" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_stem_egy" +"task": "AraDiCE_ArabicMMLU_primary_stem_natural-science_egy" +"task_alias": "primary stem natural-science" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_univ_social-science_accounting.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_univ_social-science_accounting.yaml new file mode 100644 index 0000000000000000000000000000000000000000..48ec0e75d852057e4080f3dc4ea84c417beafaf7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/EGY/AraDiCE_ArabicMMLU_univ_social-science_accounting.yaml @@ -0,0 +1,10 @@ +"dataset_name": "univ_social-science_accounting" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_social-science_egy" +"task": "AraDiCE_ArabicMMLU_univ_social-science_accounting_egy" +"task_alias": "univ social-science accounting" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU.yaml new file mode 100644 index 0000000000000000000000000000000000000000..df64389d8ece88b80a4029845f09f810131da7fe --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU.yaml @@ -0,0 +1,12 @@ +group: AraDiCE_ArabicMMLU_lev +task: +- AraDiCE_ArabicMMLU_humanities_lev +- AraDiCE_ArabicMMLU_language_lev +- AraDiCE_ArabicMMLU_social-science_lev +- AraDiCE_ArabicMMLU_stem_lev +- AraDiCE_ArabicMMLU_other_lev +aggregate_metric_list: + - metric: acc + weight_by_size: True + - metric: acc_norm + weight_by_size: True diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_high_humanities_history.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_high_humanities_history.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fbe1838c0f9ad4c741b58b457771f01c3e109fad --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_high_humanities_history.yaml @@ -0,0 +1,10 @@ +"dataset_name": "high_humanities_history" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_humanities_lev" +"task": "AraDiCE_ArabicMMLU_high_humanities_history_lev" +"task_alias": "high humanities history" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_high_humanities_islamic-studies.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_high_humanities_islamic-studies.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2e1d874eaf0ea69031c06aa947bac25839286b69 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_high_humanities_islamic-studies.yaml @@ -0,0 +1,10 @@ +"dataset_name": "high_humanities_islamic-studies" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_humanities_lev" +"task": "AraDiCE_ArabicMMLU_high_humanities_islamic-studies_lev" +"task_alias": "high humanities islamic-studies" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_high_humanities_philosophy.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_high_humanities_philosophy.yaml new file mode 100644 index 0000000000000000000000000000000000000000..752a95f3db174d00f2de8c13fafe5512aede2467 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_high_humanities_philosophy.yaml @@ -0,0 +1,10 @@ +"dataset_name": "high_humanities_philosophy" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_humanities_lev" +"task": "AraDiCE_ArabicMMLU_high_humanities_philosophy_lev" +"task_alias": "high humanities philosophy" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_high_language_arabic-language.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_high_language_arabic-language.yaml new file mode 100644 index 0000000000000000000000000000000000000000..27d14f96d16d01469dfe2b9b054c9e3223f2e421 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_high_language_arabic-language.yaml @@ -0,0 +1,10 @@ +"dataset_name": "high_language_arabic-language" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_language_lev" +"task": "AraDiCE_ArabicMMLU_high_language_arabic-language_lev" +"task_alias": "high language arabic-language" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_high_social-science_economics.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_high_social-science_economics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..378587a8feba7b2fb6745425025a5748c6cd634c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_high_social-science_economics.yaml @@ -0,0 +1,10 @@ +"dataset_name": "high_social-science_economics" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_social-science_lev" +"task": "AraDiCE_ArabicMMLU_high_social-science_economics_lev" +"task_alias": "high social-science economics" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_high_stem_biology.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_high_stem_biology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..80900b2f52c50ca7a3c0b0657a22cb81be951fbd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_high_stem_biology.yaml @@ -0,0 +1,10 @@ +"dataset_name": "high_stem_biology" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_stem_lev" +"task": "AraDiCE_ArabicMMLU_high_stem_biology_lev" +"task_alias": "high stem biology" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_high_stem_physics.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_high_stem_physics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1d21bcc69ffa6a500d64d87b424ae720f0177e26 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_high_stem_physics.yaml @@ -0,0 +1,10 @@ +"dataset_name": "high_stem_physics" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_stem_lev" +"task": "AraDiCE_ArabicMMLU_high_stem_physics_lev" +"task_alias": "high stem physics" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_middle_humanities_history.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_middle_humanities_history.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8dd3cfb9e1a3db0af98b158759618fa994792437 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_middle_humanities_history.yaml @@ -0,0 +1,10 @@ +"dataset_name": "middle_humanities_history" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_humanities_lev" +"task": "AraDiCE_ArabicMMLU_middle_humanities_history_lev" +"task_alias": "middle humanities history" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_middle_other_general-knowledge.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_middle_other_general-knowledge.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bd43ebe3ddabeb27f725fc41a8ea42b3a8a562a1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_middle_other_general-knowledge.yaml @@ -0,0 +1,10 @@ +"dataset_name": "middle_other_general-knowledge" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_other_lev" +"task": "AraDiCE_ArabicMMLU_middle_other_general-knowledge_lev" +"task_alias": "middle other general-knowledge" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_middle_social-science_economics.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_middle_social-science_economics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e1de265b6b9f530aaea9d7766a1ea72e8fbbc6d8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_middle_social-science_economics.yaml @@ -0,0 +1,10 @@ +"dataset_name": "middle_social-science_economics" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_social-science_lev" +"task": "AraDiCE_ArabicMMLU_middle_social-science_economics_lev" +"task_alias": "middle social-science economics" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_na_other_driving-test.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_na_other_driving-test.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0af542f0d6ab8de6be76a898022ad4adb242520d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/ArabicMMLU/LEV/AraDiCE_ArabicMMLU_na_other_driving-test.yaml @@ -0,0 +1,10 @@ +"dataset_name": "na_other_driving-test" +"description": "" +"fewshot_split": !!null "null" +"include": "_default_template_yaml" +"tag": "AraDiCE_ArabicMMLU_other_lev" +"task": "AraDiCE_ArabicMMLU_na_other_driving-test_lev" +"task_alias": "na other driving-test" +"test_split": "test" +"training_split": !!null "null" +"validation_split": !!null "null" diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/README.md b/lm-evaluation-harness/lm_eval/tasks/aradice/README.md new file mode 100644 index 0000000000000000000000000000000000000000..c0f1043df5e2048af610bf101fd3b4d390611533 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/README.md @@ -0,0 +1,49 @@ +# AraDiCE + +### Paper + +**Title:** AraDiCE: Benchmarks for Dialectal and Cultural Capabilities in LLMs + +**Abstract:** Arabic, with its rich diversity of dialects, remains significantly underrepresented in Large Language Models, particularly in dialectal variations. We address this gap by introducing seven synthetic datasets in dialects alongside Modern Standard Arabic (MSA), created using Machine Translation (MT) combined with human post-editing. We present AraDiCE, a benchmark for Arabic Dialect and Cultural Evaluation. We evaluate LLMs on dialect comprehension and generation, focusing specifically on low-resource Arabic dialects. Additionally, we introduce the first-ever fine-grained benchmark designed to evaluate cultural awareness across the Gulf, Egypt, and Levant regions, providing a novel dimension to LLM evaluation. Our findings demonstrate that while Arabic-specific models like Jais and AceGPT outperform multilingual models on dialectal tasks, significant challenges persist in dialect identification, generation, and translation. This work contributes ~45K post-edited samples, a cultural benchmark, and highlights the importance of tailored training to improve LLM performance in capturing the nuances of diverse Arabic dialects and cultural contexts. We will release the dialectal translation models and benchmarks curated in this study. + +**Homepage:** +https://huggingface.co/datasets/QCRI/AraDiCE + + + +### Citation + +``` +@article{mousi2024aradicebenchmarksdialectalcultural, + title={{AraDiCE}: Benchmarks for Dialectal and Cultural Capabilities in LLMs}, + author={Basel Mousi and Nadir Durrani and Fatema Ahmad and Md. Arid Hasan and Maram Hasanain and Tameem Kabbani and Fahim Dalvi and Shammur Absar Chowdhury and Firoj Alam}, + year={2024}, + publisher={arXiv:2409.11404}, + url={https://arxiv.org/abs/2409.11404}, +} +``` + +### Groups, Tags, and Tasks + +#### Groups + +* `AraDiCE`: Overall results for all tasks associated with different datasets. + + +#### Tasks + +* `aradice`: Overall results for all tasks associated with different datasets. +* `arabicmmlu`: TODO + + +### Checklist + +* [x] Is the task an existing benchmark in the literature? + * [x] Have you referenced the original paper that introduced the task? + * [x] If yes, does the original paper provide a reference implementation? If so, have you checked against the reference implementation and documented how to run such a test? + + +If other tasks on this dataset are already supported: +* [x] Is the "Main" variant of this task clearly denoted? +* [x] Have you provided a short sentence in a README on what each new variant adds / evaluates? +* [x] Have you noted which, if any, published evaluation setups are matched by this variant? diff --git a/lm-evaluation-harness/lm_eval/tasks/aradice/aradice.yaml b/lm-evaluation-harness/lm_eval/tasks/aradice/aradice.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8c7759f2c38a88289050771d2b044ebc6a1abf2f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aradice/aradice.yaml @@ -0,0 +1,30 @@ +group: AraDiCE +task: +- AraDiCE_ArabicMMLU_lev +- AraDiCE_ArabicMMLU_egy +- AraDiCE_boolq_egy +- AraDiCE_boolq_eng +- AraDiCE_boolq_lev +- AraDiCE_boolq_msa +- AraDiCE_egypt_cultural +- AraDiCE_jordan_cultural +- AraDiCE_lebanon_cultural +- AraDiCE_palestine_cultural +- AraDiCE_qatar_cultural +- AraDiCE_syria_cultural +- AraDiCE_openbookqa_egy +- AraDiCE_openbookqa_eng +- AraDiCE_openbookqa_lev +- AraDiCE_openbookqa_msa +- AraDiCE_piqa_egy +- AraDiCE_piqa_eng +- AraDiCE_piqa_lev +- AraDiCE_piqa_msa +- AraDiCE_truthfulqa_mc1_egy +- AraDiCE_truthfulqa_mc1_eng +- AraDiCE_truthfulqa_mc1_lev +- AraDiCE_truthfulqa_mc1_msa +- AraDiCE_winogrande_egy +- AraDiCE_winogrande_eng +- AraDiCE_winogrande_lev +- AraDiCE_winogrande_msa diff --git a/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_default_nutrition.yaml b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_default_nutrition.yaml new file mode 100644 index 0000000000000000000000000000000000000000..23d52c45e07134b2ff4f7c1a8e55ba19acfbcfd9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_default_nutrition.yaml @@ -0,0 +1,4 @@ +"dataset_name": "nutrition" +"description": "以下是关于营养学的单项选择题,请直接给出正确答案的选项。\n\n" +"include": "_default_template_yaml" +"task": "cmmlu_nutrition" diff --git a/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_default_sociology.yaml b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_default_sociology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..54f4367d010fb4ae5b0fe6b8a120f139b39cb0ab --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_default_sociology.yaml @@ -0,0 +1,4 @@ +"dataset_name": "sociology" +"description": "以下是关于社会学的单项选择题,请直接给出正确答案的选项。\n\n" +"include": "_default_template_yaml" +"task": "cmmlu_sociology" diff --git a/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_default_sports_science.yaml b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_default_sports_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..35e5bb9cc4c40abcf271955f068788f85e44794a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_default_sports_science.yaml @@ -0,0 +1,4 @@ +"dataset_name": "sports_science" +"description": "以下是关于体育学的单项选择题,请直接给出正确答案的选项。\n\n" +"include": "_default_template_yaml" +"task": "cmmlu_sports_science" diff --git a/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_default_virology.yaml b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_default_virology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1560b84f682493ef53a9c26ae1d36ac520ff46c7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_default_virology.yaml @@ -0,0 +1,4 @@ +"dataset_name": "virology" +"description": "以下是关于病毒学的单项选择题,请直接给出正确答案的选项。\n\n" +"include": "_default_template_yaml" +"task": "cmmlu_virology" diff --git a/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_default_world_history.yaml b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_default_world_history.yaml new file mode 100644 index 0000000000000000000000000000000000000000..993ce0ab6e390a81286df213e5d3ddd9fe3908bd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_default_world_history.yaml @@ -0,0 +1,4 @@ +"dataset_name": "world_history" +"description": "以下是关于世界历史的单项选择题,请直接给出正确答案的选项。\n\n" +"include": "_default_template_yaml" +"task": "cmmlu_world_history" diff --git a/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_default_world_religions.yaml b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_default_world_religions.yaml new file mode 100644 index 0000000000000000000000000000000000000000..13229fc95d7d85408cc8d3649208404e9a8476d2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_default_world_religions.yaml @@ -0,0 +1,4 @@ +"dataset_name": "world_religions" +"description": "以下是关于世界宗教的单项选择题,请直接给出正确答案的选项。\n\n" +"include": "_default_template_yaml" +"task": "cmmlu_world_religions" diff --git a/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_economics.yaml b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_economics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4602efb430d49e3a876b7243c4cfffe506094b34 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_economics.yaml @@ -0,0 +1,4 @@ +"dataset_name": "economics" +"description": "以下是关于经济学的单项选择题,请直接给出正确答案的选项。\n\n" +"include": "_default_template_yaml" +"task": "cmmlu_economics" diff --git a/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_education.yaml b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_education.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f1dc8a8a4fbc9664da04e2288cf782a9cc1e1877 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_education.yaml @@ -0,0 +1,4 @@ +"dataset_name": "education" +"description": "以下是关于教育学的单项选择题,请直接给出正确答案的选项。\n\n" +"include": "_default_template_yaml" +"task": "cmmlu_education" diff --git a/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_electrical_engineering.yaml b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_electrical_engineering.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2bb920b53ab8856d717fea8e07e87077ec3b3f71 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_electrical_engineering.yaml @@ -0,0 +1,4 @@ +"dataset_name": "electrical_engineering" +"description": "以下是关于电气工程的单项选择题,请直接给出正确答案的选项。\n\n" +"include": "_default_template_yaml" +"task": "cmmlu_electrical_engineering" diff --git a/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_elementary_chinese.yaml b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_elementary_chinese.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6f67be3fc40f5c038b455edcc6076675a4451261 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_elementary_chinese.yaml @@ -0,0 +1,4 @@ +"dataset_name": "elementary_chinese" +"description": "以下是关于小学语文的单项选择题,请直接给出正确答案的选项。\n\n" +"include": "_default_template_yaml" +"task": "cmmlu_elementary_chinese" diff --git a/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_elementary_commonsense.yaml b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_elementary_commonsense.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3017edd999a0ee04de4a5dd8c7dc4b1b6218f5e3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_elementary_commonsense.yaml @@ -0,0 +1,4 @@ +"dataset_name": "elementary_commonsense" +"description": "以下是关于小学常识的单项选择题,请直接给出正确答案的选项。\n\n" +"include": "_default_template_yaml" +"task": "cmmlu_elementary_commonsense" diff --git a/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_elementary_information_and_technology.yaml b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_elementary_information_and_technology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..98c7d3c8f2d85f3c52a3314253d2d2151f7116ae --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_elementary_information_and_technology.yaml @@ -0,0 +1,4 @@ +"dataset_name": "elementary_information_and_technology" +"description": "以下是关于小学信息技术的单项选择题,请直接给出正确答案的选项。\n\n" +"include": "_default_template_yaml" +"task": "cmmlu_elementary_information_and_technology" diff --git a/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_elementary_mathematics.yaml b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_elementary_mathematics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f702312ca07c2b882d17c88d30dbe87a837ce5c6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_elementary_mathematics.yaml @@ -0,0 +1,4 @@ +"dataset_name": "elementary_mathematics" +"description": "以下是关于初等数学的单项选择题,请直接给出正确答案的选项。\n\n" +"include": "_default_template_yaml" +"task": "cmmlu_elementary_mathematics" diff --git a/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_ethnology.yaml b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_ethnology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..88a653a9ee5e5978113626a35acbe50bd2ea5437 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_ethnology.yaml @@ -0,0 +1,4 @@ +"dataset_name": "ethnology" +"description": "以下是关于民族学的单项选择题,请直接给出正确答案的选项。\n\n" +"include": "_default_template_yaml" +"task": "cmmlu_ethnology" diff --git a/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_food_science.yaml b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_food_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9be450ca2ea2190c6dd3b0639ad9fbd12d968443 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_food_science.yaml @@ -0,0 +1,4 @@ +"dataset_name": "food_science" +"description": "以下是关于食品科学的单项选择题,请直接给出正确答案的选项。\n\n" +"include": "_default_template_yaml" +"task": "cmmlu_food_science" diff --git a/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_genetics.yaml b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_genetics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..be57628b6f0d3dd2bc6719e08f9aaddb45ac7fa2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_genetics.yaml @@ -0,0 +1,4 @@ +"dataset_name": "genetics" +"description": "以下是关于遗传学的单项选择题,请直接给出正确答案的选项。\n\n" +"include": "_default_template_yaml" +"task": "cmmlu_genetics" diff --git a/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_global_facts.yaml b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_global_facts.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6982be9468bebc3d99a53baf120a11eae52704bb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_global_facts.yaml @@ -0,0 +1,4 @@ +"dataset_name": "global_facts" +"description": "以下是关于全球事实的单项选择题,请直接给出正确答案的选项。\n\n" +"include": "_default_template_yaml" +"task": "cmmlu_global_facts" diff --git a/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_high_school_biology.yaml b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_high_school_biology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a56e797420f80bba1814e2bffc3aaa7f009d74f4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_high_school_biology.yaml @@ -0,0 +1,4 @@ +"dataset_name": "high_school_biology" +"description": "以下是关于高中生物的单项选择题,请直接给出正确答案的选项。\n\n" +"include": "_default_template_yaml" +"task": "cmmlu_high_school_biology" diff --git a/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_high_school_chemistry.yaml b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_high_school_chemistry.yaml new file mode 100644 index 0000000000000000000000000000000000000000..34e99ea0f47b7017206bd6e9078ca7a5c2b25f0e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_high_school_chemistry.yaml @@ -0,0 +1,4 @@ +"dataset_name": "high_school_chemistry" +"description": "以下是关于高中化学的单项选择题,请直接给出正确答案的选项。\n\n" +"include": "_default_template_yaml" +"task": "cmmlu_high_school_chemistry" diff --git a/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_high_school_geography.yaml b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_high_school_geography.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c73ebe9171df9e9f0fbdf2fecddb251e56884702 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_high_school_geography.yaml @@ -0,0 +1,4 @@ +"dataset_name": "high_school_geography" +"description": "以下是关于高中地理的单项选择题,请直接给出正确答案的选项。\n\n" +"include": "_default_template_yaml" +"task": "cmmlu_high_school_geography" diff --git a/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_high_school_mathematics.yaml b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_high_school_mathematics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3598501c1763d5f1c19444e1b18bb242149fdd34 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_high_school_mathematics.yaml @@ -0,0 +1,4 @@ +"dataset_name": "high_school_mathematics" +"description": "以下是关于高中数学的单项选择题,请直接给出正确答案的选项。\n\n" +"include": "_default_template_yaml" +"task": "cmmlu_high_school_mathematics" diff --git a/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_high_school_physics.yaml b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_high_school_physics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..676fca166369b2f2b18a077ab2ec61b74a777d5c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_high_school_physics.yaml @@ -0,0 +1,4 @@ +"dataset_name": "high_school_physics" +"description": "以下是关于高中物理学的单项选择题,请直接给出正确答案的选项。\n\n" +"include": "_default_template_yaml" +"task": "cmmlu_high_school_physics" diff --git a/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_high_school_politics.yaml b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_high_school_politics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5f689dff61a4ea55628b04f9bed5202e48c6eb70 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_high_school_politics.yaml @@ -0,0 +1,4 @@ +"dataset_name": "high_school_politics" +"description": "以下是关于高中政治的单项选择题,请直接给出正确答案的选项。\n\n" +"include": "_default_template_yaml" +"task": "cmmlu_high_school_politics" diff --git a/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_human_sexuality.yaml b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_human_sexuality.yaml new file mode 100644 index 0000000000000000000000000000000000000000..39ff32e728dd228dd675f708dc6e2680c96f0900 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_human_sexuality.yaml @@ -0,0 +1,4 @@ +"dataset_name": "human_sexuality" +"description": "以下是关于人类性行为的单项选择题,请直接给出正确答案的选项。\n\n" +"include": "_default_template_yaml" +"task": "cmmlu_human_sexuality" diff --git a/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_international_law.yaml b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_international_law.yaml new file mode 100644 index 0000000000000000000000000000000000000000..32112d3c8b6ee26ee786439053c2d1f1da5b04c2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_international_law.yaml @@ -0,0 +1,4 @@ +"dataset_name": "international_law" +"description": "以下是关于国际法学的单项选择题,请直接给出正确答案的选项。\n\n" +"include": "_default_template_yaml" +"task": "cmmlu_international_law" diff --git a/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_journalism.yaml b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_journalism.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9f283816f59760033f375d8aba352fdd860b3338 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_journalism.yaml @@ -0,0 +1,4 @@ +"dataset_name": "journalism" +"description": "以下是关于新闻学的单项选择题,请直接给出正确答案的选项。\n\n" +"include": "_default_template_yaml" +"task": "cmmlu_journalism" diff --git a/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_jurisprudence.yaml b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_jurisprudence.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ab40da40bafeb56459ae462b795be8c8584fb02a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_jurisprudence.yaml @@ -0,0 +1,4 @@ +"dataset_name": "jurisprudence" +"description": "以下是关于法理学的单项选择题,请直接给出正确答案的选项。\n\n" +"include": "_default_template_yaml" +"task": "cmmlu_jurisprudence" diff --git a/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_legal_and_moral_basis.yaml b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_legal_and_moral_basis.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a5e3ee13b6e9670f33068bc731acebf7489737ec --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_legal_and_moral_basis.yaml @@ -0,0 +1,4 @@ +"dataset_name": "legal_and_moral_basis" +"description": "以下是关于法律与道德基础的单项选择题,请直接给出正确答案的选项。\n\n" +"include": "_default_template_yaml" +"task": "cmmlu_legal_and_moral_basis" diff --git a/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_logical.yaml b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_logical.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4c4ac2e12abb2fa29dd2e194f5f1b9417f61142b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_logical.yaml @@ -0,0 +1,4 @@ +"dataset_name": "logical" +"description": "以下是关于逻辑学的单项选择题,请直接给出正确答案的选项。\n\n" +"include": "_default_template_yaml" +"task": "cmmlu_logical" diff --git a/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_machine_learning.yaml b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_machine_learning.yaml new file mode 100644 index 0000000000000000000000000000000000000000..062cd1cd73add5caf387f6b4717c5ed837e2c7f7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_machine_learning.yaml @@ -0,0 +1,4 @@ +"dataset_name": "machine_learning" +"description": "以下是关于机器学习的单项选择题,请直接给出正确答案的选项。\n\n" +"include": "_default_template_yaml" +"task": "cmmlu_machine_learning" diff --git a/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_management.yaml b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_management.yaml new file mode 100644 index 0000000000000000000000000000000000000000..aa5681babeb650cc451c15e3496ca4d0ed3a1e0f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_management.yaml @@ -0,0 +1,4 @@ +"dataset_name": "management" +"description": "以下是关于管理学的单项选择题,请直接给出正确答案的选项。\n\n" +"include": "_default_template_yaml" +"task": "cmmlu_management" diff --git a/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_marketing.yaml b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_marketing.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a729641f9059060ec9abadeac611cf3e74528165 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_marketing.yaml @@ -0,0 +1,4 @@ +"dataset_name": "marketing" +"description": "以下是关于市场营销的单项选择题,请直接给出正确答案的选项。\n\n" +"include": "_default_template_yaml" +"task": "cmmlu_marketing" diff --git a/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_marxist_theory.yaml b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_marxist_theory.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f99fa17514a10e8bf587b50ae9dd997b80c00225 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_marxist_theory.yaml @@ -0,0 +1,4 @@ +"dataset_name": "marxist_theory" +"description": "以下是关于马克思主义理论的单项选择题,请直接给出正确答案的选项。\n\n" +"include": "_default_template_yaml" +"task": "cmmlu_marxist_theory" diff --git a/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_modern_chinese.yaml b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_modern_chinese.yaml new file mode 100644 index 0000000000000000000000000000000000000000..13b2ccc4f939876616ceeda42d211e96347ce060 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_modern_chinese.yaml @@ -0,0 +1,4 @@ +"dataset_name": "modern_chinese" +"description": "以下是关于现代汉语的单项选择题,请直接给出正确答案的选项。\n\n" +"include": "_default_template_yaml" +"task": "cmmlu_modern_chinese" diff --git a/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_nutrition.yaml b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_nutrition.yaml new file mode 100644 index 0000000000000000000000000000000000000000..23d52c45e07134b2ff4f7c1a8e55ba19acfbcfd9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_nutrition.yaml @@ -0,0 +1,4 @@ +"dataset_name": "nutrition" +"description": "以下是关于营养学的单项选择题,请直接给出正确答案的选项。\n\n" +"include": "_default_template_yaml" +"task": "cmmlu_nutrition" diff --git a/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_philosophy.yaml b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_philosophy.yaml new file mode 100644 index 0000000000000000000000000000000000000000..17340fa490f0350e6e532b2c67f8c81fa63bfb3a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_philosophy.yaml @@ -0,0 +1,4 @@ +"dataset_name": "philosophy" +"description": "以下是关于哲学的单项选择题,请直接给出正确答案的选项。\n\n" +"include": "_default_template_yaml" +"task": "cmmlu_philosophy" diff --git a/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_professional_accounting.yaml b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_professional_accounting.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bed3485d787d921fb25bbbfbad7671118acfc42b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_professional_accounting.yaml @@ -0,0 +1,4 @@ +"dataset_name": "professional_accounting" +"description": "以下是关于专业会计的单项选择题,请直接给出正确答案的选项。\n\n" +"include": "_default_template_yaml" +"task": "cmmlu_professional_accounting" diff --git a/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_professional_law.yaml b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_professional_law.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dec4c6cf1d7b095fab8fb293b9cf7600765f24db --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_professional_law.yaml @@ -0,0 +1,4 @@ +"dataset_name": "professional_law" +"description": "以下是关于专业法学的单项选择题,请直接给出正确答案的选项。\n\n" +"include": "_default_template_yaml" +"task": "cmmlu_professional_law" diff --git a/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_professional_medicine.yaml b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_professional_medicine.yaml new file mode 100644 index 0000000000000000000000000000000000000000..92fed45e74f9b69b2c7b595a4bb682318fe0b81c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_professional_medicine.yaml @@ -0,0 +1,4 @@ +"dataset_name": "professional_medicine" +"description": "以下是关于专业医学的单项选择题,请直接给出正确答案的选项。\n\n" +"include": "_default_template_yaml" +"task": "cmmlu_professional_medicine" diff --git a/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_professional_psychology.yaml b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_professional_psychology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..83f0255591a17711d6ac99cf164a29ffe2a69866 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_professional_psychology.yaml @@ -0,0 +1,4 @@ +"dataset_name": "professional_psychology" +"description": "以下是关于专业心理学的单项选择题,请直接给出正确答案的选项。\n\n" +"include": "_default_template_yaml" +"task": "cmmlu_professional_psychology" diff --git a/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_public_relations.yaml b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_public_relations.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a1c3711ef7734df27852065cf894f9c9cff9d776 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_public_relations.yaml @@ -0,0 +1,4 @@ +"dataset_name": "public_relations" +"description": "以下是关于公共关系的单项选择题,请直接给出正确答案的选项。\n\n" +"include": "_default_template_yaml" +"task": "cmmlu_public_relations" diff --git a/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_security_study.yaml b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_security_study.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c9660f041fcb24ed83089c624f7ef6c6962c5d8b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_security_study.yaml @@ -0,0 +1,4 @@ +"dataset_name": "security_study" +"description": "以下是关于安全研究的单项选择题,请直接给出正确答案的选项。\n\n" +"include": "_default_template_yaml" +"task": "cmmlu_security_study" diff --git a/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_sociology.yaml b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_sociology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..54f4367d010fb4ae5b0fe6b8a120f139b39cb0ab --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_sociology.yaml @@ -0,0 +1,4 @@ +"dataset_name": "sociology" +"description": "以下是关于社会学的单项选择题,请直接给出正确答案的选项。\n\n" +"include": "_default_template_yaml" +"task": "cmmlu_sociology" diff --git a/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_sports_science.yaml b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_sports_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..35e5bb9cc4c40abcf271955f068788f85e44794a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_sports_science.yaml @@ -0,0 +1,4 @@ +"dataset_name": "sports_science" +"description": "以下是关于体育学的单项选择题,请直接给出正确答案的选项。\n\n" +"include": "_default_template_yaml" +"task": "cmmlu_sports_science" diff --git a/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_traditional_chinese_medicine.yaml b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_traditional_chinese_medicine.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ed4627deefd6a9a1737cc700604b940b31635cf8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_traditional_chinese_medicine.yaml @@ -0,0 +1,4 @@ +"dataset_name": "traditional_chinese_medicine" +"description": "以下是关于中医中药的单项选择题,请直接给出正确答案的选项。\n\n" +"include": "_default_template_yaml" +"task": "cmmlu_traditional_chinese_medicine" diff --git a/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_virology.yaml b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_virology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1560b84f682493ef53a9c26ae1d36ac520ff46c7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_virology.yaml @@ -0,0 +1,4 @@ +"dataset_name": "virology" +"description": "以下是关于病毒学的单项选择题,请直接给出正确答案的选项。\n\n" +"include": "_default_template_yaml" +"task": "cmmlu_virology" diff --git a/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_world_history.yaml b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_world_history.yaml new file mode 100644 index 0000000000000000000000000000000000000000..993ce0ab6e390a81286df213e5d3ddd9fe3908bd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_world_history.yaml @@ -0,0 +1,4 @@ +"dataset_name": "world_history" +"description": "以下是关于世界历史的单项选择题,请直接给出正确答案的选项。\n\n" +"include": "_default_template_yaml" +"task": "cmmlu_world_history" diff --git a/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_world_religions.yaml b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_world_religions.yaml new file mode 100644 index 0000000000000000000000000000000000000000..13229fc95d7d85408cc8d3649208404e9a8476d2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/cmmlu/cmmlu_world_religions.yaml @@ -0,0 +1,4 @@ +"dataset_name": "world_religions" +"description": "以下是关于世界宗教的单项选择题,请直接给出正确答案的选项。\n\n" +"include": "_default_template_yaml" +"task": "cmmlu_world_religions" diff --git a/lm-evaluation-harness/lm_eval/tasks/coqa/utils.py b/lm-evaluation-harness/lm_eval/tasks/coqa/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..29911cfec5cd345b41c631064a7e281b9d15000e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/coqa/utils.py @@ -0,0 +1,77 @@ +from itertools import zip_longest + +import transformers.data.metrics.squad_metrics as squad_metrics + + +def doc_to_text(doc): + # Given a passage p, the conversation history {q1, a1, . . . qi−1, ai−1} + # and a question qi, the task is to predict the answer ai + doc_text = doc["story"] + "\n\n" + for q, a in zip_longest( + doc["questions"]["input_text"], doc["answers"]["input_text"][:-1] + ): # omit target answer ai + question = f"Q: {q}\n\n" + answer = f"A: {a}\n\n" if a is not None else "A:" + doc_text += question + answer + return doc_text + + +def doc_to_target(doc): + turn_id = len(doc["questions"]["input_text"]) + # Returns unique answers and valid alternatives (Some questions in CoQA have multiple valid answers). + answers = [] + answer_forturn = doc["answers"]["input_text"][turn_id - 1] + answers.append(answer_forturn) + + additional_answers = doc.get("additional_answers") + if additional_answers: + for key in additional_answers: + additional_answer_for_turn = additional_answers[key]["input_text"][ + turn_id - 1 + ] + if additional_answer_for_turn.lower() not in map(str.lower, answers): + answers.append(additional_answer_for_turn) + return answers + + +def em(gold_list, pred): + # tests for exact match and on the normalised answer (compute_exact) + em_sum = 0.0 + if len(gold_list) > 1: + for i in range(len(gold_list)): + gold_answers = gold_list[0:i] + gold_list[i + 1 :] + # predictions compared against (n) golds and take maximum + em_sum += max(squad_metrics.compute_exact(a, pred) for a in gold_answers) + else: + em_sum += max(squad_metrics.compute_exact(a, pred) for a in gold_list) + + return em_sum / max(1, len(gold_list)) + + +def compute_scores(gold_list, pred): + # tests for exact match and on the normalised answer (compute_exact) + # test for overlap (compute_f1) + f1_sum = 0.0 + em_sum = 0.0 + if len(gold_list) > 1: + for i in range(len(gold_list)): + gold_answers = gold_list[0:i] + gold_list[i + 1 :] + # predictions compared against (n) golds and take maximum + em_sum += max(squad_metrics.compute_exact(a, pred) for a in gold_answers) + f1_sum += max(squad_metrics.compute_f1(a, pred) for a in gold_answers) + else: + em_sum += max(squad_metrics.compute_exact(a, pred) for a in gold_list) + f1_sum += max(squad_metrics.compute_f1(a, pred) for a in gold_list) + + return { + "em": em_sum / max(1, len(gold_list)), + "f1": f1_sum / max(1, len(gold_list)), + } + + +def process_results(doc, results): + gold_list = doc_to_target(doc) + pred = results[0].strip().split("\n")[0] + + scores = compute_scores(gold_list, pred) + return scores diff --git a/lm-evaluation-harness/lm_eval/tasks/crows_pairs/README.md b/lm-evaluation-harness/lm_eval/tasks/crows_pairs/README.md new file mode 100644 index 0000000000000000000000000000000000000000..9532179d8b0977573b6ee35e304c31f6c8867165 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/crows_pairs/README.md @@ -0,0 +1,101 @@ +# CrowS-Pairs + +### Paper + +CrowS-Pairs: A Challenge Dataset for Measuring Social Biases in Masked Language Models +https://aclanthology.org/2020.emnlp-main.154/ +French CrowS-Pairs: Extending a challenge dataset for measuring social bias in masked +language models to a language other than English +https://aclanthology.org/2022.acl-long.583/ + +CrowS-Pairs is a challenge set for evaluating what language models (LMs) on their tendency +to generate biased outputs. CrowS-Pairs comes in 2 languages and the English subset has +a newer version which fixes some of the issues with the original version. + +Homepage: https://github.com/nyu-mll/crows-pairs, https://gitlab.inria.fr/french-crows-pairs + +### Citation + +```bibtex +@inproceedings{nangia-etal-2020-crows, + title = "{C}row{S}-Pairs: A Challenge Dataset for Measuring Social Biases in Masked Language Models", + author = "Nangia, Nikita and + Vania, Clara and + Bhalerao, Rasika and + Bowman, Samuel R.", + booktitle = "Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing (EMNLP)", + month = nov, + year = "2020", + address = "Online", + publisher = "Association for Computational Linguistics", + url = "https://aclanthology.org/2020.emnlp-main.154", + doi = "10.18653/v1/2020.emnlp-main.154", + pages = "1953--1967", + abstract = "Pretrained language models, especially masked language models (MLMs) have seen success across many NLP tasks. However, there is ample evidence that they use the cultural biases that are undoubtedly present in the corpora they are trained on, implicitly creating harm with biased representations. To measure some forms of social bias in language models against protected demographic groups in the US, we introduce the Crowdsourced Stereotype Pairs benchmark (CrowS-Pairs). CrowS-Pairs has 1508 examples that cover stereotypes dealing with nine types of bias, like race, religion, and age. In CrowS-Pairs a model is presented with two sentences: one that is more stereotyping and another that is less stereotyping. The data focuses on stereotypes about historically disadvantaged groups and contrasts them with advantaged groups. We find that all three of the widely-used MLMs we evaluate substantially favor sentences that express stereotypes in every category in CrowS-Pairs. As work on building less biased models advances, this dataset can be used as a benchmark to evaluate progress.", +} + +@inproceedings{neveol-etal-2022-french, + title = "{F}rench {C}row{S}-Pairs: Extending a challenge dataset for measuring social bias in masked language models to a language other than {E}nglish", + author = {N{\'e}v{\'e}ol, Aur{\'e}lie and + Dupont, Yoann and + Bezan{\c{c}}on, Julien and + Fort, Kar{\"e}n}, + booktitle = "Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)", + month = may, + year = "2022", + address = "Dublin, Ireland", + publisher = "Association for Computational Linguistics", + url = "https://aclanthology.org/2022.acl-long.583", + doi = "10.18653/v1/2022.acl-long.583", + pages = "8521--8531", + abstract = "Warning: This paper contains explicit statements of offensive stereotypes which may be upsetting.Much work on biases in natural language processing has addressed biases linked to the social and cultural experience of English speaking individuals in the United States. We seek to widen the scope of bias studies by creating material to measure social bias in language models (LMs) against specific demographic groups in France. We build on the US-centered CrowS-pairs dataset to create a multilingual stereotypes dataset that allows for comparability across languages while also characterizing biases that are specific to each country and language. We introduce 1,679 sentence pairs in French that cover stereotypes in ten types of bias like gender and age. 1,467 sentence pairs are translated from CrowS-pairs and 212 are newly crowdsourced. The sentence pairs contrast stereotypes concerning underadvantaged groups with the same sentence concerning advantaged groups. We find that four widely used language models (three French, one multilingual) favor sentences that express stereotypes in most bias categories. We report on the translation process from English into French, which led to a characterization of stereotypes in CrowS-pairs including the identification of US-centric cultural traits. We offer guidelines to further extend the dataset to other languages and cultural environments.", +} +``` + +### Groups and Tasks + +#### Groups + +- `crows_pairs_english`: The entire English subset of the CrowS-Pairs dataset. +- `crows_pairs_french`: The entire French subset of the CrowS-Pairs dataset. + +#### Tasks + + +The following tasks evaluate sub-areas of bias in the English CrowS-Pairs dataset: +- `crows_pairs_english_age` +- `crows_pairs_english_autre` +- `crows_pairs_english_disability` +- `crows_pairs_english_gender` +- `crows_pairs_english_nationality` +- `crows_pairs_english_physical_appearance` +- `crows_pairs_english_race_color` +- `crows_pairs_english_religion` +- `crows_pairs_english_sexual_orientation` +- `crows_pairs_english_socioeconomic` + +The following tasks evaluate sub-areas of bias in the French CrowS-Pairs dataset: +- `crows_pairs_french_age` +- `crows_pairs_french_autre` +- `crows_pairs_french_disability` +- `crows_pairs_french_gender` +- `crows_pairs_french_nationality` +- `crows_pairs_french_physical_appearance` +- `crows_pairs_french_race_color` +- `crows_pairs_french_religion` +- `crows_pairs_french_sexual_orientation` +- `crows_pairs_french_socioeconomic` + +All tasks evaluate the percentage of more-stereotypical sentences that are rated as more likely by a model than the non-stereotypical sentences (`pct_stereotype`), as well as the average absolute difference of loglikelihoods between the sentences in the pairs. + +### Checklist + +* [x] Is the task an existing benchmark in the literature? + * [x] Have you referenced the original paper that introduced the task? + * [x] If yes, does the original paper provide a reference implementation? + * [x] The original paper does not for causal language models, so this is a novel formulation of the task for autoregressive LMs. + +If other tasks on this dataset are already supported: +* [x] Is the "Main" variant of this task clearly denoted? +* [x] Have you provided a short sentence in a README on what each new variant adds / evaluates? +* [x] Have you noted which, if any, published evaluation setups are matched by this variant? diff --git a/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_english.yaml b/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_english.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3daf5f75fc3fe8090336a43e3617fe79ceb22bdd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_english.yaml @@ -0,0 +1,21 @@ +tag: + - crows_pairs +task: crows_pairs_english +dataset_path: BigScienceBiasEval/crows_pairs_multilingual +dataset_name: english +test_split: test +output_type: multiple_choice +doc_to_text: "" +doc_to_target: 0 +doc_to_choice: !function utils.doc_to_choice +target_delimiter: "" +process_results: !function utils.process_results +metric_list: + - metric: likelihood_diff + aggregation: mean + higher_is_better: false + - metric: pct_stereotype + aggregation: mean + higher_is_better: false +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_english_age.yaml b/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_english_age.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ccf500fa37f3c4f55c31f85ac77d8152505b42b6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_english_age.yaml @@ -0,0 +1,3 @@ +include: crows_pairs_english.yaml +task: crows_pairs_english_age +process_docs: !function utils.filter_age diff --git a/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_english_autre.yaml b/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_english_autre.yaml new file mode 100644 index 0000000000000000000000000000000000000000..138873d7e839c0e51fdfc471cc4e223c971920ab --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_english_autre.yaml @@ -0,0 +1,3 @@ +include: crows_pairs_english.yaml +task: crows_pairs_english_autre +process_docs: !function utils.filter_autre diff --git a/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_english_disability.yaml b/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_english_disability.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a8cc280f1542ed84a42c69ce67cda3dc54335bca --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_english_disability.yaml @@ -0,0 +1,3 @@ +include: crows_pairs_english.yaml +task: crows_pairs_english_disability +process_docs: !function utils.filter_disability diff --git a/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_english_gender.yaml b/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_english_gender.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2dba35cfcc6dc0b8b039a47e6cbe9164bf773cb7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_english_gender.yaml @@ -0,0 +1,3 @@ +include: crows_pairs_english.yaml +task: crows_pairs_english_gender +process_docs: !function utils.filter_gender diff --git a/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_english_nationality.yaml b/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_english_nationality.yaml new file mode 100644 index 0000000000000000000000000000000000000000..13fcf889752df39ca5f7ad877d989105727bb5e5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_english_nationality.yaml @@ -0,0 +1,3 @@ +include: crows_pairs_english.yaml +task: crows_pairs_english_nationality +process_docs: !function utils.filter_nationality diff --git a/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_english_physical_appearance.yaml b/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_english_physical_appearance.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8b4dd4c228879770a4fa70ce50967baa464550b6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_english_physical_appearance.yaml @@ -0,0 +1,3 @@ +include: crows_pairs_english.yaml +task: crows_pairs_english_physical_appearance +process_docs: !function utils.filter_appearance diff --git a/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_english_race_color.yaml b/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_english_race_color.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3a214fe2ea5d87d87f3ded4a8bbf3491feee40ac --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_english_race_color.yaml @@ -0,0 +1,3 @@ +include: crows_pairs_english.yaml +task: crows_pairs_english_race_color +process_docs: !function utils.filter_race_color diff --git a/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_english_religion.yaml b/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_english_religion.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5ff24b1c1af572a718479d1cc4a7fb35902e8066 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_english_religion.yaml @@ -0,0 +1,3 @@ +include: crows_pairs_english.yaml +task: crows_pairs_english_religion +process_docs: !function utils.filter_religion diff --git a/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_english_sexual_orientation.yaml b/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_english_sexual_orientation.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e53df12924cfb4f12afe0b764323cb0e00b75f39 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_english_sexual_orientation.yaml @@ -0,0 +1,3 @@ +include: crows_pairs_english.yaml +task: crows_pairs_english_sexual_orientation +process_docs: !function utils.filter_orientation diff --git a/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_english_socioeconomic.yaml b/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_english_socioeconomic.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4fef30285d1fbe73b9f4a9555fc2586c72b0c46a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_english_socioeconomic.yaml @@ -0,0 +1,3 @@ +include: crows_pairs_english.yaml +task: crows_pairs_english_socioeconomic +process_docs: !function utils.filter_socio diff --git a/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_french.yaml b/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_french.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4eb7f0034149f08f30249758c2baff4a8f0164e9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_french.yaml @@ -0,0 +1,3 @@ +include: crows_pairs_english.yaml +task: crows_pairs_french +dataset_name: french diff --git a/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_french_age.yaml b/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_french_age.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3857686cff80be1791288d3f106e20b7394e8dc8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_french_age.yaml @@ -0,0 +1,3 @@ +include: crows_pairs_french.yaml +task: crows_pairs_french_age +process_docs: !function utils.filter_age diff --git a/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_french_autre.yaml b/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_french_autre.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b5198fe6758b9498a3f9168704bd210cf1e33c1e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_french_autre.yaml @@ -0,0 +1,3 @@ +include: crows_pairs_french.yaml +task: crows_pairs_french_autre +process_docs: !function utils.filter_autre diff --git a/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_french_disability.yaml b/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_french_disability.yaml new file mode 100644 index 0000000000000000000000000000000000000000..72c78f7bbc3a16128a13897dbb6d7d3b5fe1447d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_french_disability.yaml @@ -0,0 +1,3 @@ +include: crows_pairs_french.yaml +task: crows_pairs_french_disability +process_docs: !function utils.filter_disability diff --git a/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_french_gender.yaml b/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_french_gender.yaml new file mode 100644 index 0000000000000000000000000000000000000000..00eaa84691c7f70dd01c9172e0c353e0b09a4ab2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_french_gender.yaml @@ -0,0 +1,3 @@ +include: crows_pairs_french.yaml +task: crows_pairs_french_gender +process_docs: !function utils.filter_gender diff --git a/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_french_nationality.yaml b/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_french_nationality.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fffcad38d2804103f9c15d504ff519b399581e09 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_french_nationality.yaml @@ -0,0 +1,3 @@ +include: crows_pairs_french.yaml +task: crows_pairs_french_nationality +process_docs: !function utils.filter_nationality diff --git a/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_french_physical_appearance.yaml b/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_french_physical_appearance.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c5015169dc062ebf1c35572765ce627d42b3dfe4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_french_physical_appearance.yaml @@ -0,0 +1,3 @@ +include: crows_pairs_french.yaml +task: crows_pairs_french_physical_appearance +process_docs: !function utils.filter_appearance diff --git a/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_french_race_color.yaml b/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_french_race_color.yaml new file mode 100644 index 0000000000000000000000000000000000000000..87bd572ec9b8d7bdecc45d622a76cc7a3ed3499f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_french_race_color.yaml @@ -0,0 +1,3 @@ +include: crows_pairs_french.yaml +task: crows_pairs_french_race_color +process_docs: !function utils.filter_race_color diff --git a/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_french_religion.yaml b/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_french_religion.yaml new file mode 100644 index 0000000000000000000000000000000000000000..80b7add39b13d2dec822fd24099c35b65abbe0b1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_french_religion.yaml @@ -0,0 +1,3 @@ +include: crows_pairs_french.yaml +task: crows_pairs_french_religion +process_docs: !function utils.filter_religion diff --git a/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_french_sexual_orientation.yaml b/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_french_sexual_orientation.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7d13573fe2a5a56726c80372d5305378fac5dd25 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_french_sexual_orientation.yaml @@ -0,0 +1,3 @@ +include: crows_pairs_french.yaml +task: crows_pairs_french_sexual_orientation +process_docs: !function utils.filter_orientation diff --git a/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_french_socioeconomic.yaml b/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_french_socioeconomic.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9def52b93d660224fe87a09f07a5efc730753b2f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/crows_pairs/crows_pairs_french_socioeconomic.yaml @@ -0,0 +1,3 @@ +include: crows_pairs_french.yaml +task: crows_pairs_french_socioeconomic +process_docs: !function utils.filter_socio diff --git a/lm-evaluation-harness/lm_eval/tasks/crows_pairs/utils.py b/lm-evaluation-harness/lm_eval/tasks/crows_pairs/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..8cb38bdb4923f10f4fb237f7332bdfc785cd521f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/crows_pairs/utils.py @@ -0,0 +1,64 @@ +import datasets + + +def process_results(doc, results): + lls, _ = zip(*results) + + likelihood1, likelihood2 = lls + + # Calculate the absolute difference in loglikelihoods + diff = abs(likelihood1 - likelihood2) + + # if stereotypical sentence more likely (loglikelihood higher) + # then treat this as predicting stereotyped sentence + acc = 1.0 if likelihood1 > likelihood2 else 0.0 + + return {"likelihood_diff": diff, "pct_stereotype": acc} + + +def doc_to_choice(doc): + return [doc["sent_more"], doc["sent_less"]] + + +def filter_dataset(dataset: datasets.Dataset, bias_type: str) -> datasets.Dataset: + return dataset.filter(lambda example: example["bias_type"].startswith(bias_type)) + + +def filter_race_color(dataset: datasets.Dataset) -> datasets.Dataset: + return filter_dataset(dataset, "race-color") + + +def filter_socio(dataset: datasets.Dataset) -> datasets.Dataset: + return filter_dataset(dataset, "socioeconomic") + + +def filter_gender(dataset: datasets.Dataset) -> datasets.Dataset: + return filter_dataset(dataset, "gender") + + +def filter_age(dataset: datasets.Dataset) -> datasets.Dataset: + return filter_dataset(dataset, "age") + + +def filter_religion(dataset: datasets.Dataset) -> datasets.Dataset: + return filter_dataset(dataset, "religion") + + +def filter_disability(dataset: datasets.Dataset) -> datasets.Dataset: + return filter_dataset(dataset, "disability") + + +def filter_orientation(dataset: datasets.Dataset) -> datasets.Dataset: + return filter_dataset(dataset, "sexual-orientation") + + +def filter_nationality(dataset: datasets.Dataset) -> datasets.Dataset: + return filter_dataset(dataset, "nationality") + + +def filter_appearance(dataset: datasets.Dataset) -> datasets.Dataset: + return filter_dataset(dataset, "physical-appearance") + + +def filter_autre(dataset: datasets.Dataset) -> datasets.Dataset: + return filter_dataset(dataset, "autre") diff --git a/lm-evaluation-harness/lm_eval/tasks/csatqa/_csatqa.yaml b/lm-evaluation-harness/lm_eval/tasks/csatqa/_csatqa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8b9e30131097d9123f3dbbae3ab250fbcf4f6ad3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/csatqa/_csatqa.yaml @@ -0,0 +1,17 @@ +group: csatqa +task: + - csatqa_gr + - csatqa_li + - csatqa_rch + - csatqa_rcs + - csatqa_rcss + - csatqa_wr +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true + - metric: acc_norm + aggregation: mean + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/csatqa/_default_csatqa_yaml b/lm-evaluation-harness/lm_eval/tasks/csatqa/_default_csatqa_yaml new file mode 100644 index 0000000000000000000000000000000000000000..fdd6d148c5d7f9d522d7d0f9ad433f9b9ad4179e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/csatqa/_default_csatqa_yaml @@ -0,0 +1,16 @@ +dataset_path: HAERAE-HUB/csatqa +test_split: test +output_type: multiple_choice +process_docs: !function utils.process_docs +doc_to_text: "{{question}}" +doc_to_choice: "{{choices}}" +doc_to_target: "{{gold}}" +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/csatqa/_generate_configs.py b/lm-evaluation-harness/lm_eval/tasks/csatqa/_generate_configs.py new file mode 100644 index 0000000000000000000000000000000000000000..1ef34b8cf1e1a21ff511d6e1ef12d52da7781082 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/csatqa/_generate_configs.py @@ -0,0 +1,53 @@ +""" +Take in a YAML, and output all other splits with this YAML +""" + +import argparse +import logging +import os + +import yaml +from tqdm import tqdm + + +eval_logger = logging.getLogger(__name__) + + +SUBSETS = ["WR", "GR", "RCS", "RCSS", "RCH", "LI"] + + +def parse_args(): + parser = argparse.ArgumentParser() + parser.add_argument("--base_yaml_path", required=True) + parser.add_argument("--save_prefix_path", default="csatqa") + parser.add_argument("--task_prefix", default="") + return parser.parse_args() + + +if __name__ == "__main__": + args = parse_args() + + # get filename of base_yaml so we can `"include": ` it in our other YAMLs. + base_yaml_name = os.path.split(args.base_yaml_path)[-1] + with open(args.base_yaml_path, encoding="utf-8") as f: + base_yaml = yaml.full_load(f) + + for name in tqdm(SUBSETS): + yaml_dict = { + "include": base_yaml_name, + "task": f"csatqa_{args.task_prefix}_{name}" + if args.task_prefix != "" + else f"csatqa_{name.lower()}", + "dataset_name": name, + } + + file_save_path = args.save_prefix_path + f"_{name.lower()}.yaml" + eval_logger.info(f"Saving yaml for subset {name} to {file_save_path}") + with open(file_save_path, "w", encoding="utf-8") as yaml_file: + yaml.dump( + yaml_dict, + yaml_file, + width=float("inf"), + allow_unicode=True, + default_style='"', + ) diff --git a/lm-evaluation-harness/lm_eval/tasks/csatqa/csatqa_gr.yaml b/lm-evaluation-harness/lm_eval/tasks/csatqa/csatqa_gr.yaml new file mode 100644 index 0000000000000000000000000000000000000000..856ba617faa7953582bc8e7956f1a491b31cd2f2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/csatqa/csatqa_gr.yaml @@ -0,0 +1,3 @@ +"dataset_name": "GR" +"include": "_default_csatqa_yaml" +"task": "csatqa_gr" diff --git a/lm-evaluation-harness/lm_eval/tasks/csatqa/csatqa_li.yaml b/lm-evaluation-harness/lm_eval/tasks/csatqa/csatqa_li.yaml new file mode 100644 index 0000000000000000000000000000000000000000..eb7e178365b95d9562f47f8604a764f31817ea89 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/csatqa/csatqa_li.yaml @@ -0,0 +1,3 @@ +"dataset_name": "LI" +"include": "_default_csatqa_yaml" +"task": "csatqa_li" diff --git a/lm-evaluation-harness/lm_eval/tasks/csatqa/csatqa_rch.yaml b/lm-evaluation-harness/lm_eval/tasks/csatqa/csatqa_rch.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4eaffaa90c5d10fee0cbeb4c1d9b2b67361dfdc5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/csatqa/csatqa_rch.yaml @@ -0,0 +1,3 @@ +"dataset_name": "RCH" +"include": "_default_csatqa_yaml" +"task": "csatqa_rch" diff --git a/lm-evaluation-harness/lm_eval/tasks/csatqa/csatqa_rcs.yaml b/lm-evaluation-harness/lm_eval/tasks/csatqa/csatqa_rcs.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e2efcc1704195d51d1548c7b3527593c49be3be2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/csatqa/csatqa_rcs.yaml @@ -0,0 +1,3 @@ +"dataset_name": "RCS" +"include": "_default_csatqa_yaml" +"task": "csatqa_rcs" diff --git a/lm-evaluation-harness/lm_eval/tasks/csatqa/csatqa_rcss.yaml b/lm-evaluation-harness/lm_eval/tasks/csatqa/csatqa_rcss.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7b099fe2d56b06c3eda2c9624b5a27158c69c26e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/csatqa/csatqa_rcss.yaml @@ -0,0 +1,3 @@ +"dataset_name": "RCSS" +"include": "_default_csatqa_yaml" +"task": "csatqa_rcss" diff --git a/lm-evaluation-harness/lm_eval/tasks/csatqa/csatqa_wr.yaml b/lm-evaluation-harness/lm_eval/tasks/csatqa/csatqa_wr.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6276dfef9b0d53f96c9a7d5fb11fcf5e2d57e87f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/csatqa/csatqa_wr.yaml @@ -0,0 +1,3 @@ +"dataset_name": "WR" +"include": "_default_csatqa_yaml" +"task": "csatqa_wr" diff --git a/lm-evaluation-harness/lm_eval/tasks/csatqa/utils.py b/lm-evaluation-harness/lm_eval/tasks/csatqa/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..485a724cc0eb367aeb7567e2f75e78cad82bee50 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/csatqa/utils.py @@ -0,0 +1,20 @@ +import datasets + + +def process_docs(dataset: datasets.Dataset) -> datasets.Dataset: + def _process_doc(doc): + instruction = f"""다음을 읽고 정답으로 알맞은 것을 고르시요. +### Context: {doc["context"]} +### Question: {doc["question"]} +### Options: +(1) {doc["option#1"]}\n(2) {doc["option#2"]}\n(3) {doc["option#3"]}\n(4) {doc["option#4"]}\n(5) {doc["option#5"]} +### Answer: 주어진 문제의 정답은""" + + out_doc = { + "question": instruction, + "choices": ["(1)", "(2)", "(3)", "(4)", "(5)"], + "gold": int(doc["gold"]) - 1, + } + return out_doc + + return dataset.map(_process_doc) diff --git a/lm-evaluation-harness/lm_eval/tasks/darija_bench/README.md b/lm-evaluation-harness/lm_eval/tasks/darija_bench/README.md new file mode 100644 index 0000000000000000000000000000000000000000..5e68e101d53cb1f166ccb93824dba9332a09e49c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darija_bench/README.md @@ -0,0 +1,65 @@ +# DarijaBench + +### Paper + +Title: Atlas-Chat: Adapting Large Language Models for Low-Resource Moroccan Arabic Dialect + +Abstract: [https://arxiv.org/abs/2409.17912](https://arxiv.org/abs/2409.17912) + +DarijaBench, a comprehensive evaluation dataset tailored for Moroccan Darija. DarijaBench includes different datasets for core NLP tasks such as translation (based on four datasets, [DODa-10K](https://huggingface.co/datasets/MBZUAI-Paris/DODa-10K), [FLORES+](https://github.com/openlanguagedata/flores), [NLLB-Seed](https://github.com/openlanguagedata/seed) and [MADAR](https://sites.google.com/nyu.edu/madar/)), summarization (based on [MArSum](https://github.com/KamelGaanoun/MoroccanSummarization)) and, sentiment analysis (based on five datasets, [MAC](https://github.com/LeMGarouani/MAC), [MYC](https://github.com/MouadJb/MYC), [MSAC](https://hal.science/hal-03670346/document), [MSDA](https://cc.um6p.ma/cc_datasets) and, [ElectroMorocco2016](https://github.com/sentiprojects/ElecMorocco2016)), in addition to a new transliteration task to convert between Darija (written in Arabic letters) and Arabizi (written in Latin letters) it is based on [DODa-10K](https://huggingface.co/datasets/MBZUAI-Paris/DODa-10K) dataset. + + +Homepage: [https://huggingface.co/datasets/MBZUAI-Paris/DarijaBench](https://huggingface.co/datasets/MBZUAI-Paris/DarijaBench) + + +### Citation + +``` +@article{shang2024atlaschatadaptinglargelanguage, + title={Atlas-Chat: Adapting Large Language Models for Low-Resource Moroccan Arabic Dialect}, + author={Guokan Shang and Hadi Abdine and Yousef Khoubrane and Amr Mohamed and Yassine Abbahaddou and Sofiane Ennadir and Imane Momayiz and Xuguang Ren and Eric Moulines and Preslav Nakov and Michalis Vazirgiannis and Eric Xing}, + year={2024}, + eprint={2409.17912}, + archivePrefix={arXiv}, + primaryClass={cs.CL}, + url={https://arxiv.org/abs/2409.17912}, +} +``` + +### Groups and Tasks + +#### Groups + +* `darija_sentiment`: evaluates all Darija sentiment analysis tasks. +* `darija_summarization`: evaluates Darija summarization task. +* `darija_translation`: evaluates all Darija Translation tasks. +* `darija_transliteration`: evaluates Darija transliteration task. + +#### Tasks + +* `darija_sentiment_mac`: evaluates Darija translation task from [MAC](https://github.com/LeMGarouani/MAC) dataset. +* `darija_sentiment_myc`: evaluates Darija translation task from [MYC](https://github.com/MouadJb/MYC) dataset. +* `darija_sentiment_msac`: evaluates Darija translation task from [MSAC](https://hal.science/hal-03670346/document) dataset. +* `darija_sentiment_msda`: evaluates Darija translation task from [MSDA](https://cc.um6p.ma/cc_datasets) dataset. +* `darija_sentiment_electrom`: evaluates Darija translation task from [ElectroMorocco2016](https://github.com/sentiprojects/ElecMorocco2016) dataset. +* `darija_summarization_task`: evaluates Darija summarization task from [MArSum](https://github.com/KamelGaanoun/MoroccanSummarization) corpus. +* `darija_translation_doda`: evaluates Darija translation task from [DODa-10k](https://huggingface.co/datasets/MBZUAI-Paris/DODa-10K) corpus. +* `darija_translation_flores`: evaluates Darija translation task from [FLORES+](https://github.com/openlanguagedata/flores) dataset. +* `darija_translation_madar`: evaluates Darija translation task from [MADAR](https://sites.google.com/nyu.edu/madar/) dataset. +* `darija_translation_seed`: evaluates Darija translation task from [NLLB-Seed](https://github.com/openlanguagedata/seed) datasets. +* `darija_transliteration_task`: evaluates Darija transliteration task from [DODa-10K](https://huggingface.co/datasets/MBZUAI-Paris/DODa-10K) corpus. + +Note: depending on the model, padding and padding side could affect the results. The default padding side in this library is forced to left. Use batch size equal to 1 to avoid problems. + +### Checklist + +For adding novel benchmarks/datasets to the library: +* [x] Is the task an existing benchmark in the literature? + * [x] Have you referenced the original paper that introduced the task? + * [ ] If yes, does the original paper provide a reference implementation? If so, have you checked against the reference implementation and documented how to run such a test? + + +If other tasks on this dataset are already supported: +* [ ] Is the "Main" variant of this task clearly denoted? +* [ ] Have you provided a short sentence in a README on what each new variant adds / evaluates? +* [ ] Have you noted which, if any, published evaluation setups are matched by this variant? diff --git a/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_sentiment/README.md b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_sentiment/README.md new file mode 100644 index 0000000000000000000000000000000000000000..c49be3d89aba8c39b6a612c15757e0b5cdb66b8e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_sentiment/README.md @@ -0,0 +1,54 @@ +# DarijaBench: Sentiment Analysis + +### Paper + +Title: Atlas-Chat: Adapting Large Language Models for Low-Resource Moroccan Arabic Dialect + +Abstract: [https://arxiv.org/abs/2409.17912](https://arxiv.org/abs/2409.17912) + +DarijaBench, a comprehensive evaluation dataset tailored for Moroccan Darija. DarijaBench includes different datasets for core NLP tasks such as translation (based on four datasets, [DODa-10K](https://huggingface.co/datasets/MBZUAI-Paris/DODa-10K), [FLORES+](https://github.com/openlanguagedata/flores), [NLLB-Seed](https://github.com/openlanguagedata/seed) and [MADAR](https://sites.google.com/nyu.edu/madar/)), summarization (based on [MArSum](https://github.com/KamelGaanoun/MoroccanSummarization)) and, sentiment analysis (based on five datasets, [MAC](https://github.com/LeMGarouani/MAC), [MYC](https://github.com/MouadJb/MYC), [MSAC](https://hal.science/hal-03670346/document), [MSDA](https://cc.um6p.ma/cc_datasets) and, [ElectroMorocco2016](https://github.com/sentiprojects/ElecMorocco2016)), in addition to a new transliteration task to convert between Darija (written in Arabic letters) and Arabizi (written in Latin letters) it is based on [DODa-10K](https://huggingface.co/datasets/MBZUAI-Paris/DODa-10K) dataset. + + +Homepage: [https://huggingface.co/datasets/MBZUAI-Paris/DarijaBench](https://huggingface.co/datasets/MBZUAI-Paris/DarijaBench) + + +### Citation + +``` +@article{shang2024atlaschatadaptinglargelanguage, + title={Atlas-Chat: Adapting Large Language Models for Low-Resource Moroccan Arabic Dialect}, + author={Guokan Shang and Hadi Abdine and Yousef Khoubrane and Amr Mohamed and Yassine Abbahaddou and Sofiane Ennadir and Imane Momayiz and Xuguang Ren and Eric Moulines and Preslav Nakov and Michalis Vazirgiannis and Eric Xing}, + year={2024}, + eprint={2409.17912}, + archivePrefix={arXiv}, + primaryClass={cs.CL}, + url={https://arxiv.org/abs/2409.17912}, +} +``` + +### Groups and Tasks + +#### Groups + +* `darija_sentiment`: evaluates all Darija sentiment analysis tasks. + +#### Tasks + +* `darija_sentiment_mac`: evaluates Darija translation task from [MAC](https://github.com/LeMGarouani/MAC) dataset. +* `darija_sentiment_myc`: evaluates Darija translation task from [MYC](https://github.com/MouadJb/MYC) dataset. +* `darija_sentiment_msac`: evaluates Darija translation task from [MSAC](https://hal.science/hal-03670346/document) dataset. +* `darija_sentiment_msda`: evaluates Darija translation task from [MSDA](https://cc.um6p.ma/cc_datasets) dataset. +* `darija_sentiment_electrom`: evaluates Darija translation task from [ElectroMorocco2016](https://github.com/sentiprojects/ElecMorocco2016) dataset. + +### Checklist + +For adding novel benchmarks/datasets to the library: +* [x] Is the task an existing benchmark in the literature? + * [x] Have you referenced the original paper that introduced the task? + * [ ] If yes, does the original paper provide a reference implementation? If so, have you checked against the reference implementation and documented how to run such a test? + + +If other tasks on this dataset are already supported: +* [ ] Is the "Main" variant of this task clearly denoted? +* [ ] Have you provided a short sentence in a README on what each new variant adds / evaluates? +* [ ] Have you noted which, if any, published evaluation setups are matched by this variant? diff --git a/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_sentiment/darija_sentiment.yaml b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_sentiment/darija_sentiment.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b4033c4ccd5d9cf45a57fc9355a2e4d53b7a9e04 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_sentiment/darija_sentiment.yaml @@ -0,0 +1,9 @@ +group: darija_sentiment +group_alias: Sentiment_Analysis +task: + - darija_sentiment_tasks +aggregate_metric_list: + - metric: acc + weight_by_size: True +metadata: + version: 0 diff --git a/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_sentiment/darija_sentiment_electrom.yaml b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_sentiment/darija_sentiment_electrom.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ec1a9f97f11c60bb534db86fd0c4bc746371aae1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_sentiment/darija_sentiment_electrom.yaml @@ -0,0 +1,7 @@ +test_split: electro_maroc +"include": "default_darija_sentiment_template_yaml" +"tag": +- "darija_sentiment_tasks" +"task": "darija_sentiment_electrom" +"task_alias": "Electro Maroc" +doc_to_choice: !function utils.doc_to_choice_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_sentiment/darija_sentiment_mac.yaml b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_sentiment/darija_sentiment_mac.yaml new file mode 100644 index 0000000000000000000000000000000000000000..db75f43dd4af3ca6126651afd3bbbb6ca25ea8cc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_sentiment/darija_sentiment_mac.yaml @@ -0,0 +1,7 @@ +test_split: mac +"include": "default_darija_sentiment_template_yaml" +"tag": +- "darija_sentiment_tasks" +"task": "darija_sentiment_mac" +"task_alias": "MAC" +doc_to_choice: !function utils.doc_to_choice_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_sentiment/darija_sentiment_msac.yaml b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_sentiment/darija_sentiment_msac.yaml new file mode 100644 index 0000000000000000000000000000000000000000..654e7211301e71151a81d9435a2153f93995bac0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_sentiment/darija_sentiment_msac.yaml @@ -0,0 +1,7 @@ +test_split: msac +"include": "default_darija_sentiment_template_yaml" +"tag": +- "darija_sentiment_tasks" +"task": "darija_sentiment_msac" +"task_alias": "MSAC" +doc_to_choice: !function utils.doc_to_choice_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_sentiment/darija_sentiment_msda.yaml b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_sentiment/darija_sentiment_msda.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dc9773e2236c9b75cec2eb0d511ff2f09cdd9eb3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_sentiment/darija_sentiment_msda.yaml @@ -0,0 +1,7 @@ +test_split: msda +"include": "default_darija_sentiment_template_yaml" +"tag": +- "darija_sentiment_tasks" +"task": "darija_sentiment_msda" +"task_alias": "MSDA" +doc_to_choice: !function utils.doc_to_choice_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_sentiment/darija_sentiment_myc.yaml b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_sentiment/darija_sentiment_myc.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c520b0a1add7aa24fffb76eb0e137b7da89a468b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_sentiment/darija_sentiment_myc.yaml @@ -0,0 +1,7 @@ +test_split: myc +"include": "default_darija_sentiment_template_yaml" +"tag": +- "darija_sentiment_tasks" +"task": "darija_sentiment_myc" +"task_alias": "MYC" +doc_to_choice: !function utils.doc_to_choice_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_sentiment/default_darija_sentiment_template_yaml b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_sentiment/default_darija_sentiment_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..30d918c8a31c216bd313a4b28b4d8bca87aff123 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_sentiment/default_darija_sentiment_template_yaml @@ -0,0 +1,13 @@ +dataset_path: MBZUAI-Paris/DarijaBench +output_type: multiple_choice +doc_to_text: !function utils.doc_to_text +doc_to_choice: !function utils.doc_to_choice_3 +doc_to_target: !function utils.doc_to_target +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 +dataset_kwargs: + trust_remote_code: true diff --git a/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_sentiment/utils.py b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_sentiment/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..1e32d0db2819715f7e8b5c2179d1e78d8f59a940 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_sentiment/utils.py @@ -0,0 +1,30 @@ +from lm_eval.api.filter import Filter +from lm_eval.api.registry import register_filter + + +alpha = ["A", "B", "C"] +out_dic = {"ايجابي": 1, "سلبي": 0, "ماكينش إحساس": 2} + + +def doc_to_text(doc): + return ( + doc["messages"][0]["content"] + .replace("-سلبي", "A. سلبي") + .replace("-ايجابي", "B. ايجابي") + .replace( + "-ماكينش إحساس", + "C. ماكينش إحساس\nThe answer should be strictly one letter of the following: A, B, C.", + ) + ) # .replace('شنو هو الإحساس ديال هاد الجملة؟', 'شنو هو الإحساس ديال هاد الجملة؟') + + +def doc_to_choice_3(doc): + return alpha + + +def doc_to_choice_2(doc): + return alpha[:2] + + +def doc_to_target(doc): + return alpha[out_dic[doc["messages"][1]["content"]]] diff --git a/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_summarization/README.md b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_summarization/README.md new file mode 100644 index 0000000000000000000000000000000000000000..0b87cfb654ade42fcd1e014c46093f6a83992609 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_summarization/README.md @@ -0,0 +1,51 @@ +# DarijaBench: Summarization + +### Paper + +Title: Atlas-Chat: Adapting Large Language Models for Low-Resource Moroccan Arabic Dialect + +Abstract: [https://arxiv.org/abs/2409.17912](https://arxiv.org/abs/2409.17912) + +DarijaBench, a comprehensive evaluation dataset tailored for Moroccan Darija. DarijaBench includes different datasets for core NLP tasks such as translation (based on four datasets, [DODa-10K](https://huggingface.co/datasets/MBZUAI-Paris/DODa-10K), [FLORES+](https://github.com/openlanguagedata/flores), [NLLB-Seed](https://github.com/openlanguagedata/seed) and [MADAR](https://sites.google.com/nyu.edu/madar/)), summarization (based on [MArSum](https://github.com/KamelGaanoun/MoroccanSummarization)) and, sentiment analysis (based on five datasets, [MAC](https://github.com/LeMGarouani/MAC), [MYC](https://github.com/MouadJb/MYC), [MSAC](https://hal.science/hal-03670346/document), [MSDA](https://cc.um6p.ma/cc_datasets) and, [ElectroMorocco2016](https://github.com/sentiprojects/ElecMorocco2016)), in addition to a new transliteration task to convert between Darija (written in Arabic letters) and Arabizi (written in Latin letters) it is based on [DODa-10K](https://huggingface.co/datasets/MBZUAI-Paris/DODa-10K) dataset. + + +Homepage: [https://huggingface.co/datasets/MBZUAI-Paris/DarijaBench](https://huggingface.co/datasets/MBZUAI-Paris/DarijaBench) + + +### Citation + +``` +@article{shang2024atlaschatadaptinglargelanguage, + title={Atlas-Chat: Adapting Large Language Models for Low-Resource Moroccan Arabic Dialect}, + author={Guokan Shang and Hadi Abdine and Yousef Khoubrane and Amr Mohamed and Yassine Abbahaddou and Sofiane Ennadir and Imane Momayiz and Xuguang Ren and Eric Moulines and Preslav Nakov and Michalis Vazirgiannis and Eric Xing}, + year={2024}, + eprint={2409.17912}, + archivePrefix={arXiv}, + primaryClass={cs.CL}, + url={https://arxiv.org/abs/2409.17912}, +} +``` + +### Groups and Tasks + +#### Groups + +* `darija_summarization`: evaluates Darija summarization task. + +#### Tasks + +* `darija_summarization_task`: evaluates Darija summarization task from [MArSum](https://github.com/KamelGaanoun/MoroccanSummarization) corpus. + + +### Checklist + +For adding novel benchmarks/datasets to the library: +* [x] Is the task an existing benchmark in the literature? + * [x] Have you referenced the original paper that introduced the task? + * [ ] If yes, does the original paper provide a reference implementation? If so, have you checked against the reference implementation and documented how to run such a test? + + +If other tasks on this dataset are already supported: +* [ ] Is the "Main" variant of this task clearly denoted? +* [ ] Have you provided a short sentence in a README on what each new variant adds / evaluates? +* [ ] Have you noted which, if any, published evaluation setups are matched by this variant? diff --git a/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_summarization/summarization.yaml b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_summarization/summarization.yaml new file mode 100644 index 0000000000000000000000000000000000000000..58d000531af7ff6ae86aecd0e22d8dc7247ea801 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_summarization/summarization.yaml @@ -0,0 +1,2 @@ +"include": "summarization_common_yaml" +"task": "darija_summarization_task" diff --git a/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_summarization/summarization_common_yaml b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_summarization/summarization_common_yaml new file mode 100644 index 0000000000000000000000000000000000000000..e244eb65a49200f9c0a59438855fefeb8a1b7421 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_summarization/summarization_common_yaml @@ -0,0 +1,31 @@ +output_type: generate_until +dataset_path: MBZUAI-Paris/DarijaBench +test_split: marsum +doc_to_text: !function utils.doc_to_text +doc_to_target: !function utils.doc_to_target +metric_list: + - metric: !function utils.rouge1 + - metric: !function utils.rouge2 + - metric: !function utils.rougeL + - metric: !function utils.rougeLsum + - metric: !function utils.bert + - metric: chrf +generation_kwargs: + until: + - "" + - "" + - "" + - "<|end_of_text|>" + - "<|eot_id|>" + - "<|endoftext|>" + do_sample: false + temperature: 0.0 + max_new_tokens: 128 +repeats: 1 +filter_list: + - name: "STRIP_ANSWER" + filter: + - function: "custom" + filter_fn: !function utils.strip +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_summarization/summarization_darija.yaml b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_summarization/summarization_darija.yaml new file mode 100644 index 0000000000000000000000000000000000000000..16e1390c1cf6f0bd4d33dcddf362e5a324cd74fd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_summarization/summarization_darija.yaml @@ -0,0 +1,24 @@ +group: darija_summarization +task: + - darija_summarization_task +metric_list: + - metric: !function utils.rouge1 + aggregation: !function utils.agg_rouge1 + higher_is_better: true + - metric: !function utils.rouge2 + aggregation: !function utils.agg_rouge2 + higher_is_better: true + - metric: !function utils.rougeL + aggregation: !function utils.agg_rougel + higher_is_better: true + - metric: !function utils.rougeLsum + aggregation: !function utils.agg_rougelsum + higher_is_better: true + - metric: !function utils.bert + aggregation: !function utils.darijabert + higher_is_better: true + - metric: chrf + aggregation: chrf + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_summarization/utils.py b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_summarization/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..b69a6a7b293f64fbdb65374626f8b5c0a0b8df6e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_summarization/utils.py @@ -0,0 +1,81 @@ +import datasets +import evaluate + + +def strip(resps, docs): + """ + Assuming each entry of `resps` is a list of model responses, we discard all but the first response. + """ + return map(lambda r: r[0].strip(), resps) + + +def doc_to_text(doc): + doc_text = doc["messages"][0]["content"].replace( + "لخص هاد المقطع", "لخص هاد المقطع في ٣٠ كلمة" + ) + return doc_text + + +def doc_to_target(doc): + return doc["messages"][1]["content"] + + +def bert(items): + return items + + +def Average(lst): + return sum(lst) / len(lst) + + +def darijabert(items): + bert_model = "SI2M-Lab/DarijaBERT" + bert_score = evaluate.load("bertscore") + predictions, references = zip(*items) + bert = bert_score.compute( + predictions=predictions, + references=references, + model_type=bert_model, + num_layers=12, + ) + return Average(bert["f1"]) + + +def rouge1(items): + return items + + +def rougeL(items): + return items + + +def rouge2(items): + return items + + +def rougeLsum(items): + return items + + +def agg_rougelsum(items): + rouge = evaluate.load("rouge") + predictions, references = zip(*items) + return rouge.compute(predictions=predictions, references=references)["rougeLsum"] + + +def agg_rouge1(items): + rouge = evaluate.load("rouge") + predictions, references = zip(*items) + return rouge.compute(predictions=predictions, references=references)["rouge1"] + + +def agg_rouge2(items): + rouge = evaluate.load("rouge") + predictions, references = zip(*items) + return rouge.compute(predictions=predictions, references=references)["rouge2"] + + +def agg_rougel(items): + rouge = evaluate.load("rouge") + predictions, references = zip(*items) + return rouge.compute(predictions=predictions, references=references)["rougeL"] diff --git a/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/README.md b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/README.md new file mode 100644 index 0000000000000000000000000000000000000000..6dbe9454c8bb1197e999e040fcb4e045e3b3e30a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/README.md @@ -0,0 +1,53 @@ +# DarijaBench: Translation + +### Paper + +Title: Atlas-Chat: Adapting Large Language Models for Low-Resource Moroccan Arabic Dialect + +Abstract: [https://arxiv.org/abs/2409.17912](https://arxiv.org/abs/2409.17912) + +DarijaBench, a comprehensive evaluation dataset tailored for Moroccan Darija. DarijaBench includes different datasets for core NLP tasks such as translation (based on four datasets, [DODa-10K](https://huggingface.co/datasets/MBZUAI-Paris/DODa-10K), [FLORES+](https://github.com/openlanguagedata/flores), [NLLB-Seed](https://github.com/openlanguagedata/seed) and [MADAR](https://sites.google.com/nyu.edu/madar/)), summarization (based on [MArSum](https://github.com/KamelGaanoun/MoroccanSummarization)) and, sentiment analysis (based on five datasets, [MAC](https://github.com/LeMGarouani/MAC), [MYC](https://github.com/MouadJb/MYC), [MSAC](https://hal.science/hal-03670346/document), [MSDA](https://cc.um6p.ma/cc_datasets) and, [ElectroMorocco2016](https://github.com/sentiprojects/ElecMorocco2016)), in addition to a new transliteration task to convert between Darija (written in Arabic letters) and Arabizi (written in Latin letters) it is based on [DODa-10K](https://huggingface.co/datasets/MBZUAI-Paris/DODa-10K) dataset. + + +Homepage: [https://huggingface.co/datasets/MBZUAI-Paris/DarijaBench](https://huggingface.co/datasets/MBZUAI-Paris/DarijaBench) + + +### Citation + +``` +@article{shang2024atlaschatadaptinglargelanguage, + title={Atlas-Chat: Adapting Large Language Models for Low-Resource Moroccan Arabic Dialect}, + author={Guokan Shang and Hadi Abdine and Yousef Khoubrane and Amr Mohamed and Yassine Abbahaddou and Sofiane Ennadir and Imane Momayiz and Xuguang Ren and Eric Moulines and Preslav Nakov and Michalis Vazirgiannis and Eric Xing}, + year={2024}, + eprint={2409.17912}, + archivePrefix={arXiv}, + primaryClass={cs.CL}, + url={https://arxiv.org/abs/2409.17912}, +} +``` + +### Groups and Tasks + +#### Groups + +* `darija_translation`: evaluates all Darija Translation tasks. + +#### Tasks + +* `darija_translation_doda`: evaluates Darija translation task from [DODa-10k](https://huggingface.co/datasets/MBZUAI-Paris/DODa-10K) corpus. +* `darija_translation_flores`: evaluates Darija translation task from [FLORES+](https://github.com/openlanguagedata/flores) dataset. +* `darija_translation_madar`: evaluates Darija translation task from [MADAR](https://sites.google.com/nyu.edu/madar/) dataset. +* `darija_translation_seed`: evaluates Darija translation task from [NLLB-Seed](https://github.com/openlanguagedata/seed) datasets. + +### Checklist + +For adding novel benchmarks/datasets to the library: +* [x] Is the task an existing benchmark in the literature? + * [x] Have you referenced the original paper that introduced the task? + * [ ] If yes, does the original paper provide a reference implementation? If so, have you checked against the reference implementation and documented how to run such a test? + + +If other tasks on this dataset are already supported: +* [ ] Is the "Main" variant of this task clearly denoted? +* [ ] Have you provided a short sentence in a README on what each new variant adds / evaluates? +* [ ] Have you noted which, if any, published evaluation setups are matched by this variant? diff --git a/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/doda_common_yaml b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/doda_common_yaml new file mode 100644 index 0000000000000000000000000000000000000000..8b3ed3d75d28127f07371e541b6ef221bb599787 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/doda_common_yaml @@ -0,0 +1 @@ +test_split: doda diff --git a/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/doda_translation_all.yaml b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/doda_translation_all.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6301c69e5f323d5f29de00aeec461a75805731ba --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/doda_translation_all.yaml @@ -0,0 +1,11 @@ +include: + - translation_common_yaml + - doda_common_yaml +"tag": +- "darija_translation_tasks_doda" +"task": "trasnlation_all_doda" +"task_alias": "all_doda" +metric_list: + - metric: !function utils.bert + aggregation: !function utils.mbert + higher_is_better: true diff --git a/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/doda_translation_darija.yaml b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/doda_translation_darija.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ed73c1ed755112704909bf24bb3ba41c8b200f9d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/doda_translation_darija.yaml @@ -0,0 +1,19 @@ +group: darija_translation_doda +group_alias: translation_doda +task: + - darija_translation_tasks_doda +metric_list: + - metric: bleu + aggregation: bleu + higher_is_better: true + - metric: chrf + aggregation: chrf + higher_is_better: true + - metric: ter + aggregation: ter + higher_is_better: false + - metric: !function utils.bert + aggregation: !function utils.mbert + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/doda_translation_dr_en.yaml b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/doda_translation_dr_en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5990ab90f20b2028d52b9690fa79ccda0bd49b8e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/doda_translation_dr_en.yaml @@ -0,0 +1,12 @@ +"process_docs": !function utils.dr_en +include: + - translation_common_yaml + - doda_common_yaml +"tag": +- "darija_translation_tasks_doda" +"task": "trasnlation_dr_en_doda" +"task_alias": "dr_en_doda" +metric_list: + - metric: !function utils.bert + aggregation: !function utils.bertbase + higher_is_better: true diff --git a/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/doda_translation_dr_fr.yaml b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/doda_translation_dr_fr.yaml new file mode 100644 index 0000000000000000000000000000000000000000..08bc4b0dbcc6b8d8a5ba96f2e6e439d06e3e97c7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/doda_translation_dr_fr.yaml @@ -0,0 +1,12 @@ +"process_docs": !function utils.dr_fr +include: + - translation_common_yaml + - doda_common_yaml +"tag": +- "darija_translation_tasks_doda" +"task": "trasnlation_dr_fr_doda" +"task_alias": "dr_fr_doda" +metric_list: + - metric: !function utils.bert + aggregation: !function utils.camembert + higher_is_better: true diff --git a/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/doda_translation_dr_msa.yaml b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/doda_translation_dr_msa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9dc348600153b3a2e0eb44566a9d5770ff37e3ef --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/doda_translation_dr_msa.yaml @@ -0,0 +1,12 @@ +"process_docs": !function utils.dr_msa +include: + - translation_common_yaml + - doda_common_yaml +"tag": +- "darija_translation_tasks_doda" +"task": "trasnlation_dr_msa_doda" +"task_alias": "dr_msa_doda" +metric_list: + - metric: !function utils.bert + aggregation: !function utils.arabert + higher_is_better: true diff --git a/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/doda_translation_en_dr.yaml b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/doda_translation_en_dr.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d4edfcfa1941b97503342499a02a9e99c281c4d5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/doda_translation_en_dr.yaml @@ -0,0 +1,12 @@ +"process_docs": !function utils.en_dr +include: + - translation_common_yaml + - doda_common_yaml +"tag": + - "darija_translation_tasks_doda" +"task": "trasnlation_en_dr_doda" +"task_alias": "en_dr_doda" +metric_list: + - metric: !function utils.bert + aggregation: !function utils.darijabert + higher_is_better: true diff --git a/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/doda_translation_fr_dr.yaml b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/doda_translation_fr_dr.yaml new file mode 100644 index 0000000000000000000000000000000000000000..51bd38f8d95c47f9c2411c96c0a2f63a063e6b59 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/doda_translation_fr_dr.yaml @@ -0,0 +1,12 @@ +"process_docs": !function utils.fr_dr +include: + - translation_common_yaml + - doda_common_yaml +"tag": +- "darija_translation_tasks_doda" +"task": "trasnlation_fr_dr_doda" +"task_alias": "fr_dr_doda" +metric_list: + - metric: !function utils.bert + aggregation: !function utils.darijabert + higher_is_better: true diff --git a/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/doda_translation_msa_dr.yaml b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/doda_translation_msa_dr.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6e27383957b1b02958ec41497eb66aa4514004e6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/doda_translation_msa_dr.yaml @@ -0,0 +1,12 @@ +"process_docs": !function utils.msa_dr +include: + - translation_common_yaml + - doda_common_yaml +"tag": +- "darija_translation_tasks_doda" +"task": "trasnlation_msa_dr_doda" +"task_alias": "msa_dr_doda" +metric_list: + - metric: !function utils.bert + aggregation: !function utils.darijabert + higher_is_better: true diff --git a/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/flores_common_yaml b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/flores_common_yaml new file mode 100644 index 0000000000000000000000000000000000000000..fa54a293ef8e456a922fe55d1c55bc1e0d4e5959 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/flores_common_yaml @@ -0,0 +1 @@ +test_split: flores_plus diff --git a/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/flores_translation_all.yaml b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/flores_translation_all.yaml new file mode 100644 index 0000000000000000000000000000000000000000..56742634110a77e94a73fb859b41ab95f716c0be --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/flores_translation_all.yaml @@ -0,0 +1,11 @@ +include: + - translation_common_yaml + - flores_common_yaml +"tag": +- "darija_translation_tasks_flores" +"task": "trasnlation_all_flores" +"task_alias": "all_flores" +metric_list: + - metric: !function utils.bert + aggregation: !function utils.mbert + higher_is_better: true diff --git a/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/flores_translation_darija.yaml b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/flores_translation_darija.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1add52995ac40f5185f6d1509dab2bb121044103 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/flores_translation_darija.yaml @@ -0,0 +1,19 @@ +group: darija_translation_flores +group_alias: translation_flores +task: + - darija_translation_tasks_flores +metric_list: + - metric: bleu + aggregation: bleu + higher_is_better: true + - metric: chrf + aggregation: chrf + higher_is_better: true + - metric: ter + aggregation: ter + higher_is_better: false + - metric: !function utils.bert + aggregation: !function utils.mbert + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/flores_translation_dr_en.yaml b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/flores_translation_dr_en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..edc52a0fbf2500d7d8f825cabd67b15b77197b92 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/flores_translation_dr_en.yaml @@ -0,0 +1,12 @@ +"process_docs": !function utils.dr_en +include: + - translation_common_yaml + - flores_common_yaml +"tag": +- "darija_translation_tasks_flores" +"task": "trasnlation_dr_en_flores" +"task_alias": "dr_en_flores" +metric_list: + - metric: !function utils.bert + aggregation: !function utils.bertbase + higher_is_better: true diff --git a/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/flores_translation_dr_fr.yaml b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/flores_translation_dr_fr.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0ec99775ea83469675a0dbfc19477b72aa46ec43 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/flores_translation_dr_fr.yaml @@ -0,0 +1,12 @@ +"process_docs": !function utils.dr_fr +include: + - translation_common_yaml + - flores_common_yaml +"tag": +- "darija_translation_tasks_flores" +"task": "trasnlation_dr_fr_flores" +"task_alias": "dr_fr_flores" +metric_list: + - metric: !function utils.bert + aggregation: !function utils.camembert + higher_is_better: true diff --git a/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/flores_translation_dr_msa.yaml b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/flores_translation_dr_msa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e02c5da6cb6fd8d714ed3bf72e62c3c5172c9ed1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/flores_translation_dr_msa.yaml @@ -0,0 +1,12 @@ +"process_docs": !function utils.dr_msa +include: + - translation_common_yaml + - flores_common_yaml +"tag": +- "darija_translation_tasks_flores" +"task": "trasnlation_dr_msa_flores" +"task_alias": "dr_msa_flores" +metric_list: + - metric: !function utils.bert + aggregation: !function utils.arabert + higher_is_better: true diff --git a/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/flores_translation_en_dr.yaml b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/flores_translation_en_dr.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e5923feb65b5e243c9605d69292afb8f3f4d656a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/flores_translation_en_dr.yaml @@ -0,0 +1,12 @@ +"process_docs": !function utils.en_dr +include: + - translation_common_yaml + - flores_common_yaml +"tag": + - "darija_translation_tasks_flores" +"task": "trasnlation_en_dr_flores" +"task_alias": "en_dr_flores" +metric_list: + - metric: !function utils.bert + aggregation: !function utils.darijabert + higher_is_better: true diff --git a/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/flores_translation_fr_dr.yaml b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/flores_translation_fr_dr.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8fbe45aa0f272eeb3dc04acc9172d566946cf4ac --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/flores_translation_fr_dr.yaml @@ -0,0 +1,12 @@ +"process_docs": !function utils.fr_dr +include: + - translation_common_yaml + - flores_common_yaml +"tag": +- "darija_translation_tasks_flores" +"task": "trasnlation_fr_dr_flores" +"task_alias": "fr_dr_flores" +metric_list: + - metric: !function utils.bert + aggregation: !function utils.darijabert + higher_is_better: true diff --git a/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/flores_translation_msa_dr.yaml b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/flores_translation_msa_dr.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2252ac80f9df178073eb878b509a3278d3db63e7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/flores_translation_msa_dr.yaml @@ -0,0 +1,12 @@ +"process_docs": !function utils.msa_dr +include: + - translation_common_yaml + - flores_common_yaml +"tag": +- "darija_translation_tasks_flores" +"task": "trasnlation_msa_dr_flores" +"task_alias": "msa_dr_flores" +metric_list: + - metric: !function utils.bert + aggregation: !function utils.darijabert + higher_is_better: true diff --git a/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/madar_common_yaml b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/madar_common_yaml new file mode 100644 index 0000000000000000000000000000000000000000..cc33710eceb920d867fa357b3aa3534c32d1c87c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/madar_common_yaml @@ -0,0 +1 @@ +test_split: madar diff --git a/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/madar_translation_all.yaml b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/madar_translation_all.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ee921df81685fbfc22b57e4d650930293a99b4a5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/madar_translation_all.yaml @@ -0,0 +1,11 @@ +include: + - translation_common_yaml + - madar_common_yaml +"tag": +- "darija_translation_tasks_madar" +"task": "trasnlation_all_madar" +"task_alias": "all_madar" +metric_list: + - metric: !function utils.bert + aggregation: !function utils.mbert + higher_is_better: true diff --git a/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/madar_translation_darija.yaml b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/madar_translation_darija.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cff64b075c409bb8733d4695f159e0afe057d1c4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/madar_translation_darija.yaml @@ -0,0 +1,19 @@ +group: darija_translation_madar +group_alias: translation_madar +task: + - darija_translation_tasks_madar +metric_list: + - metric: bleu + aggregation: bleu + higher_is_better: true + - metric: chrf + aggregation: chrf + higher_is_better: true + - metric: ter + aggregation: ter + higher_is_better: false + - metric: !function utils.bert + aggregation: !function utils.mbert + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/madar_translation_dr_msa.yaml b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/madar_translation_dr_msa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bbca5556d07503cac26b21932d478610d57916c8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/madar_translation_dr_msa.yaml @@ -0,0 +1,12 @@ +"process_docs": !function utils.dr_msa +include: + - translation_common_yaml + - madar_common_yaml +"tag": +- "darija_translation_tasks_madar" +"task": "trasnlation_dr_msa_madar" +"task_alias": "dr_msa_madar" +metric_list: + - metric: !function utils.bert + aggregation: !function utils.arabert + higher_is_better: true diff --git a/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/madar_translation_msa_dr.yaml b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/madar_translation_msa_dr.yaml new file mode 100644 index 0000000000000000000000000000000000000000..580ebe209ba8936bd1cb3042c52f8783c1cac9e8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/madar_translation_msa_dr.yaml @@ -0,0 +1,12 @@ +"process_docs": !function utils.msa_dr +include: + - translation_common_yaml + - madar_common_yaml +"tag": +- "darija_translation_tasks_madar" +"task": "trasnlation_msa_dr_madar" +"task_alias": "msa_dr_madar" +metric_list: + - metric: !function utils.bert + aggregation: !function utils.darijabert + higher_is_better: true diff --git a/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/seed_common_yaml b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/seed_common_yaml new file mode 100644 index 0000000000000000000000000000000000000000..9937615b1e26255aed28c91ebc80b1c35a1ecb14 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/seed_common_yaml @@ -0,0 +1 @@ +test_split: seed diff --git a/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/seed_translation_all.yaml b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/seed_translation_all.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3e0c5591d6e89cbe5a93f843c7785a50aa08b239 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/seed_translation_all.yaml @@ -0,0 +1,11 @@ +include: + - translation_common_yaml + - seed_common_yaml +"tag": +- "darija_translation_tasks_seed" +"task": "trasnlation_all_seed" +"task_alias": "all_seed" +metric_list: + - metric: !function utils.bert + aggregation: !function utils.mbert + higher_is_better: true diff --git a/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/seed_translation_darija.yaml b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/seed_translation_darija.yaml new file mode 100644 index 0000000000000000000000000000000000000000..10ee141f2b2bc83fb15a4354d176d06776fa4f8e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/seed_translation_darija.yaml @@ -0,0 +1,19 @@ +group: darija_translation_seed +group_alias: translation_seed +task: + - darija_translation_tasks_seed +metric_list: + - metric: bleu + aggregation: bleu + higher_is_better: true + - metric: chrf + aggregation: chrf + higher_is_better: true + - metric: ter + aggregation: ter + higher_is_better: false + - metric: !function utils.bert + aggregation: !function utils.mbert + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/seed_translation_dr_en.yaml b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/seed_translation_dr_en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f286742db9cc205a87a0eff921773fb98fe2a280 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/seed_translation_dr_en.yaml @@ -0,0 +1,12 @@ +"process_docs": !function utils.dr_en +include: + - translation_common_yaml + - seed_common_yaml +"tag": +- "darija_translation_tasks_seed" +"task": "trasnlation_dr_en_seed" +"task_alias": "dr_en_seed" +metric_list: + - metric: !function utils.bert + aggregation: !function utils.bertbase + higher_is_better: true diff --git a/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/seed_translation_en_dr.yaml b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/seed_translation_en_dr.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2b707b3d818ecf6b11b5297ec3ef199718e4b13e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/seed_translation_en_dr.yaml @@ -0,0 +1,12 @@ +"process_docs": !function utils.en_dr +include: + - translation_common_yaml + - seed_common_yaml +"tag": + - "darija_translation_tasks_seed" +"task": "trasnlation_en_dr_seed" +"task_alias": "en_dr_seed" +metric_list: + - metric: !function utils.bert + aggregation: !function utils.darijabert + higher_is_better: true diff --git a/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/translation_common_yaml b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/translation_common_yaml new file mode 100644 index 0000000000000000000000000000000000000000..a8be981a81712d1d4c3546e6e9b9dff939781262 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/translation_common_yaml @@ -0,0 +1,27 @@ +output_type: generate_until +dataset_path: MBZUAI-Paris/DarijaBench +doc_to_text: !function utils.doc_to_text +doc_to_target: !function utils.doc_to_target +metric_list: + - metric: bleu + - metric: ter + - metric: chrf + - metric: !function utils.bert +generation_kwargs: + until: + - "" + - "" + - "" + - "<|end_of_text|>" + - "<|eot_id|>" + - "<|endoftext|>" + do_sample: false + temperature: 0.0 +filter_list: + - name: "STRIP_ANSWER" + filter: + - function: "custom" + filter_fn: !function utils.strip +repeats: 1 +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/translation_darija.yaml b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/translation_darija.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d343c287fbe1153b1318462d82b47794982a6b64 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/translation_darija.yaml @@ -0,0 +1,22 @@ +group: darija_translation +group_alias: translation +task: + - darija_translation_doda + - darija_translation_flores + - darija_translation_madar + - darija_translation_seed +metric_list: + - metric: bleu + aggregation: bleu + higher_is_better: true + - metric: chrf + aggregation: chrf + higher_is_better: true + - metric: ter + aggregation: ter + higher_is_better: false + - metric: !function utils.bert + aggregation: !function utils.mbert + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/utils.py b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..89d1ec2d046e9e579f4df337110f561f314665ba --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_translation/utils.py @@ -0,0 +1,125 @@ +import datasets +import evaluate + + +def strip(resps, docs): + """ + Assuming each entry of `resps` is a list of model responses, we discard all but the first response. + """ + return map(lambda r: r[0].strip(), resps) + + +def dr_fr(dataset: datasets.Dataset): + return dataset.filter(lambda x: x["direction"] == "dr_fr") + + +def dr_en(dataset: datasets.Dataset): + return dataset.filter(lambda x: x["direction"] == "dr_en") + + +def dr_msa(dataset: datasets.Dataset): + return dataset.filter(lambda x: x["direction"] == "dr_msa") + + +def fr_dr(dataset: datasets.Dataset): + return dataset.filter(lambda x: x["direction"] == "fr_dr") + + +def en_dr(dataset: datasets.Dataset): + return dataset.filter(lambda x: x["direction"] == "en_dr") + + +def msa_dr(dataset: datasets.Dataset): + return dataset.filter(lambda x: x["direction"] == "msa_dr") + + +prompt_templates = { + "fr_dr": "ترجم من الفرنساوية للدارجة:\n{0}", + "dr_fr": "ترجم من الدارجة للفرنساوية:\n{0}", + "en_dr": "ترجم من الإنجليزية للدارجة:\n{0}", + "dr_en": "ترجم من الدارجة للإنجليزية:\n{0}", + "msa_dr": "ترجم من الفصحى للدارجة:\n{0}", + "dr_msa": "ترجم من الدارجة للفصحى:\n{0}", +} + + +def doc_to_text(doc): + doc_text = doc["messages"][0]["content"] + return doc_text + + +def doc_to_target(doc): + return doc["messages"][1]["content"] + + +def bert(items): + return items + + +def Average(lst): + return sum(lst) / len(lst) + + +def camembert(items): + bert_model = "almanach/camembert-base" + bert_score = evaluate.load("bertscore") + predictions, references = zip(*items) + bert = bert_score.compute( + predictions=predictions, + references=references, + model_type=bert_model, + num_layers=12, + ) + return Average(bert["f1"]) + + +def darijabert(items): + bert_model = "SI2M-Lab/DarijaBERT" + bert_score = evaluate.load("bertscore") + predictions, references = zip(*items) + bert = bert_score.compute( + predictions=predictions, + references=references, + model_type=bert_model, + num_layers=12, + ) + return Average(bert["f1"]) + + +def arabert(items): + bert_model = "aubmindlab/bert-base-arabert" + bert_score = evaluate.load("bertscore") + predictions, references = zip(*items) + bert = bert_score.compute( + predictions=predictions, + references=references, + model_type=bert_model, + num_layers=12, + ) + return Average(bert["f1"]) + + +def bertbase(items): + bert_model = "google-bert/bert-base-uncased" + bert_score = evaluate.load("bertscore") + predictions, references = zip(*items) + bert = bert_score.compute( + predictions=predictions, + references=references, + model_type=bert_model, + num_layers=12, + ) + return Average(bert["f1"]) + + +def mbert(items): + bert_model = "google-bert/bert-base-multilingual-cased" + bert_score = evaluate.load("bertscore") + predictions, references = zip(*items) + bert = bert_score.compute( + predictions=predictions, + references=references, + model_type=bert_model, + num_layers=12, + ) + return Average(bert["f1"]) diff --git a/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_transliteration/README.md b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_transliteration/README.md new file mode 100644 index 0000000000000000000000000000000000000000..e3fa3c7d22e9fd4374eb1f978170c5e80a0f28c6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_transliteration/README.md @@ -0,0 +1,51 @@ +# DarijaBench: Transliteration + +### Paper + +Title: Atlas-Chat: Adapting Large Language Models for Low-Resource Moroccan Arabic Dialect + +Abstract: [https://arxiv.org/abs/2409.17912](https://arxiv.org/abs/2409.17912) + +DarijaBench, a comprehensive evaluation dataset tailored for Moroccan Darija. DarijaBench includes different datasets for core NLP tasks such as translation (based on four datasets, [DODa-10K](https://huggingface.co/datasets/MBZUAI-Paris/DODa-10K), [FLORES+](https://github.com/openlanguagedata/flores), [NLLB-Seed](https://github.com/openlanguagedata/seed) and [MADAR](https://sites.google.com/nyu.edu/madar/)), summarization (based on [MArSum](https://github.com/KamelGaanoun/MoroccanSummarization)) and, sentiment analysis (based on five datasets, [MAC](https://github.com/LeMGarouani/MAC), [MYC](https://github.com/MouadJb/MYC), [MSAC](https://hal.science/hal-03670346/document), [MSDA](https://cc.um6p.ma/cc_datasets) and, [ElectroMorocco2016](https://github.com/sentiprojects/ElecMorocco2016)), in addition to a new transliteration task to convert between Darija (written in Arabic letters) and Arabizi (written in Latin letters) it is based on [DODa-10K](https://huggingface.co/datasets/MBZUAI-Paris/DODa-10K) dataset. + + +Homepage: [https://huggingface.co/datasets/MBZUAI-Paris/DarijaBench](https://huggingface.co/datasets/MBZUAI-Paris/DarijaBench) + + +### Citation + +``` +@article{shang2024atlaschatadaptinglargelanguage, + title={Atlas-Chat: Adapting Large Language Models for Low-Resource Moroccan Arabic Dialect}, + author={Guokan Shang and Hadi Abdine and Yousef Khoubrane and Amr Mohamed and Yassine Abbahaddou and Sofiane Ennadir and Imane Momayiz and Xuguang Ren and Eric Moulines and Preslav Nakov and Michalis Vazirgiannis and Eric Xing}, + year={2024}, + eprint={2409.17912}, + archivePrefix={arXiv}, + primaryClass={cs.CL}, + url={https://arxiv.org/abs/2409.17912}, +} +``` + +### Groups and Tasks + +#### Groups + +* `darija_transliteration`: evaluates Darija transliteration task. + +#### Tasks + +* `darija_transliteration_task`: evaluates Darija transliteration task from [DODa-10K](https://huggingface.co/datasets/MBZUAI-Paris/DODa-10K) corpus. + + +### Checklist + +For adding novel benchmarks/datasets to the library: +* [x] Is the task an existing benchmark in the literature? + * [x] Have you referenced the original paper that introduced the task? + * [ ] If yes, does the original paper provide a reference implementation? If so, have you checked against the reference implementation and documented how to run such a test? + + +If other tasks on this dataset are already supported: +* [ ] Is the "Main" variant of this task clearly denoted? +* [ ] Have you provided a short sentence in a README on what each new variant adds / evaluates? +* [ ] Have you noted which, if any, published evaluation setups are matched by this variant? diff --git a/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_transliteration/translation_ar_dr.yaml b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_transliteration/translation_ar_dr.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c3db6a7ba13bf6095bae41d7678d51ce6fb45b0e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_transliteration/translation_ar_dr.yaml @@ -0,0 +1,10 @@ +"process_docs": !function utils.ar_dr +"include": "transliteration_common_yaml" +"tag": + - "darija_transliteration_tasks" +"task": "transliteration_ar_dr" +"task_alias": "ar_dr" +metric_list: + - metric: !function utils.bert + aggregation: !function utils.darijabert + higher_is_better: true diff --git a/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_transliteration/translation_dr_ar.yaml b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_transliteration/translation_dr_ar.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e2340811811169ed66844910c5f65c3361ef4162 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_transliteration/translation_dr_ar.yaml @@ -0,0 +1,10 @@ +"process_docs": !function utils.dr_ar +"include": "transliteration_common_yaml" +"tag": + - "darija_transliteration_tasks" +"task": "transliteration_dr_ar" +"task_alias": "dr_ar" +metric_list: + - metric: !function utils.bert + aggregation: !function utils.arabizibert + higher_is_better: true diff --git a/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_transliteration/transliteration_all.yaml b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_transliteration/transliteration_all.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8c48d209fe6c147f4cbe80b40b926647a9766160 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_transliteration/transliteration_all.yaml @@ -0,0 +1,9 @@ +"include": "transliteration_common_yaml" +"tag": +- "darija_transliteration_tasks" +"task": "transliteration_all" +"task_alias": "all" +metric_list: + - metric: !function utils.bert + aggregation: !function utils.mbert + higher_is_better: true diff --git a/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_transliteration/transliteration_common_yaml b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_transliteration/transliteration_common_yaml new file mode 100644 index 0000000000000000000000000000000000000000..4184883afc2954c872dbbbba026a043feeb61952 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_transliteration/transliteration_common_yaml @@ -0,0 +1,28 @@ +output_type: generate_until +dataset_path: MBZUAI-Paris/DarijaBench +doc_to_text: !function utils.doc_to_text +doc_to_target: !function utils.doc_to_target +test_split: transliteration +metric_list: + - metric: bleu + - metric: ter + - metric: chrf + - metric: !function utils.bert +generation_kwargs: + until: + - "" + - "" + - "" + - "<|end_of_text|>" + - "<|eot_id|>" + - "<|endoftext|>" + do_sample: false + temperature: 0.0 +filter_list: + - name: "STRIP_ANSWER" + filter: + - function: "custom" + filter_fn: !function utils.strip +repeats: 1 +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_transliteration/transliteration_darija.yaml b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_transliteration/transliteration_darija.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4b02ad4818d3cf8ea0f05eb369eb2b15cf70f9d7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_transliteration/transliteration_darija.yaml @@ -0,0 +1,19 @@ +group: darija_transliteration +group_alias: transliteration +task: + - darija_transliteration_tasks +metric_list: + - metric: bleu + aggregation: bleu + higher_is_better: true + - metric: chrf + aggregation: chrf + higher_is_better: true + - metric: ter + aggregation: ter + higher_is_better: false + - metric: !function utils.bert + aggregation: !function utils.mbert + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_transliteration/utils.py b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_transliteration/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..e819a8789aabe08a4f9be4ef8a96db00b2f5f034 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darija_bench/darija_transliteration/utils.py @@ -0,0 +1,73 @@ +import datasets +import evaluate + + +def strip(resps, docs): + """ + Assuming each entry of `resps` is a list of model responses, we discard all but the first response. + """ + return map(lambda r: r[0].strip(), resps) + + +def dr_ar(dataset: datasets.Dataset): + return dataset.filter(lambda x: x["direction"] == "dr_ar") + + +def ar_dr(dataset: datasets.Dataset): + return dataset.filter(lambda x: x["direction"] == "ar_dr") + + +def doc_to_text(doc): + doc_text = doc["messages"][0]["content"] + return doc_text + + +def doc_to_target(doc): + return doc["messages"][1]["content"] + + +def bert(items): + return items + + +def Average(lst): + return sum(lst) / len(lst) + + +def arabizibert(items): + bert_model = "SI2M-Lab/DarijaBERT-arabizi" + bert_score = evaluate.load("bertscore") + predictions, references = zip(*items) + bert = bert_score.compute( + predictions=predictions, + references=references, + model_type=bert_model, + num_layers=12, + ) + return Average(bert["f1"]) + + +def darijabert(items): + bert_model = "SI2M-Lab/DarijaBERT" + bert_score = evaluate.load("bertscore") + predictions, references = zip(*items) + bert = bert_score.compute( + predictions=predictions, + references=references, + model_type=bert_model, + num_layers=12, + ) + return Average(bert["f1"]) + + +def mbert(items): + bert_model = "google-bert/bert-base-multilingual-cased" + bert_score = evaluate.load("bertscore") + predictions, references = zip(*items) + bert = bert_score.compute( + predictions=predictions, + references=references, + model_type=bert_model, + num_layers=12, + ) + return Average(bert["f1"]) diff --git a/lm-evaluation-harness/lm_eval/tasks/darijahellaswag/README.md b/lm-evaluation-harness/lm_eval/tasks/darijahellaswag/README.md new file mode 100644 index 0000000000000000000000000000000000000000..9e0e15d623b4e61a04a6e757522d4661d27d8501 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darijahellaswag/README.md @@ -0,0 +1,51 @@ +# DarijaHellaSwag + +### Paper + +Title: Atlas-Chat: Adapting Large Language Models for Low-Resource Moroccan Arabic Dialect + +Abstract: [https://arxiv.org/abs/2409.17912](https://arxiv.org/abs/2409.17912) + +DarijaHellaSwag is a challenging multiple-choice benchmark designed to evaluate machine reading comprehension and commonsense reasoning in Moroccan Darija. It is a translated version of the HellaSwag validation set, which presents scenarios where models must choose the most plausible continuation of a passage from four options. + + +Homepage: [https://huggingface.co/datasets/MBZUAI-Paris/DarijaHellaSwag](https://huggingface.co/datasets/MBZUAI-Paris/DarijaHellaSwag) + + +### Citation + +``` +@article{shang2024atlaschatadaptinglargelanguage, + title={Atlas-Chat: Adapting Large Language Models for Low-Resource Moroccan Arabic Dialect}, + author={Guokan Shang and Hadi Abdine and Yousef Khoubrane and Amr Mohamed and Yassine Abbahaddou and Sofiane Ennadir and Imane Momayiz and Xuguang Ren and Eric Moulines and Preslav Nakov and Michalis Vazirgiannis and Eric Xing}, + year={2024}, + eprint={2409.17912}, + archivePrefix={arXiv}, + primaryClass={cs.CL}, + url={https://arxiv.org/abs/2409.17912}, +} +``` + +### Groups and Tasks + +#### Groups + +- Not part of a group yet + +#### Tasks + +- `darijahellaswag` + +### Checklist + +For adding novel benchmarks/datasets to the library: + +* [X] Is the task an existing benchmark in the literature? + * [X] Have you referenced the original paper that introduced the task? + * [X] If yes, does the original paper provide a reference implementation? If so, have you checked against the reference implementation and documented how to run such a test? + +If other tasks on this dataset are already supported: + +* [ ] Is the "Main" variant of this task clearly denoted? +* [ ] Have you provided a short sentence in a README on what each new variant adds / evaluates? +* [ ] Have you noted which, if any, published evaluation setups are matched by this variant? diff --git a/lm-evaluation-harness/lm_eval/tasks/darijahellaswag/darijahellaswag.yaml b/lm-evaluation-harness/lm_eval/tasks/darijahellaswag/darijahellaswag.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4b09b8232d45bc48bb0726cf899547749e2fbf31 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darijahellaswag/darijahellaswag.yaml @@ -0,0 +1,24 @@ +tag: + - multiple_choice +task: darijahellaswag +dataset_path: MBZUAI-Paris/DarijaHellaSwag +dataset_name: null +output_type: multiple_choice +training_split: train +validation_split: validation +test_split: null +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{label}}" +doc_to_choice: "choices" +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 +dataset_kwargs: + trust_remote_code: true diff --git a/lm-evaluation-harness/lm_eval/tasks/darijahellaswag/utils.py b/lm-evaluation-harness/lm_eval/tasks/darijahellaswag/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..51c5cb4e9d58e23fad3a48efdba6bcda7450a46f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darijahellaswag/utils.py @@ -0,0 +1,14 @@ +import datasets + + +def process_docs(dataset: datasets.Dataset) -> datasets.Dataset: + def _process_doc(doc): + ctx = doc["ctx"] + out_doc = { + "query": doc["activity_label"] + ": " + ctx, + "choices": doc["endings"], + "gold": int(doc["label"]), + } + return out_doc + + return dataset.map(_process_doc) diff --git a/lm-evaluation-harness/lm_eval/tasks/darijammlu/README.md b/lm-evaluation-harness/lm_eval/tasks/darijammlu/README.md new file mode 100644 index 0000000000000000000000000000000000000000..237eb4677a89f179865bf21a54a595906ae54f4a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darijammlu/README.md @@ -0,0 +1,60 @@ +# DarijaMMLU + +### Paper + +Title: Atlas-Chat: Adapting Large Language Models for Low-Resource Moroccan Arabic Dialect + +Abstract: [https://arxiv.org/abs/2409.17912](https://arxiv.org/abs/2409.17912) + +DarijaMMLU is an evaluation benchmark designed to assess large language models' (LLM) performance in Moroccan Darija, a variety of Arabic. It consists of 22,027 multiple-choice questions, translated from selected subsets of the Massive Multitask Language Understanding (MMLU) and ArabicMMLU benchmarks to measure model performance on 44 subjects in Darija. DarijaMMLU is constructed by translating selected subsets from two major benchmarks into Darija from English and MSA: Massive Multitask Language Understanding (MMLU) and ArabicMMLU. + + +Homepage: [https://huggingface.co/datasets/MBZUAI-Paris/DarijaMMLU](https://huggingface.co/datasets/MBZUAI-Paris/DarijaMMLU) + + +### Citation + +``` +@article{shang2024atlaschatadaptinglargelanguage, + title={Atlas-Chat: Adapting Large Language Models for Low-Resource Moroccan Arabic Dialect}, + author={Guokan Shang and Hadi Abdine and Yousef Khoubrane and Amr Mohamed and Yassine Abbahaddou and Sofiane Ennadir and Imane Momayiz and Xuguang Ren and Eric Moulines and Preslav Nakov and Michalis Vazirgiannis and Eric Xing}, + year={2024}, + eprint={2409.17912}, + archivePrefix={arXiv}, + primaryClass={cs.CL}, + url={https://arxiv.org/abs/2409.17912}, +} +``` + +### Groups and Tasks + +#### Groups + +* `darijammlu`: evaluates all DarijaMMLU tasks. + +#### Tags +Source-based tags: + +* `darijammlu_mmlu`: evaluates DarijaMMLU tasks that were translated from MMLU. +* `darijammlu_ar_mmlu`: evaluates DarijaMMLU tasks that were translated from ArabicMMLU. + +Category-based tags: + +* `darijammlu_stem`: evaluates DarijaMMLU STEM tasks. +* `darijammlu_social_sciences`: evaluates DarijaMMLU social sciences tasks. +* `darijammlu_humanities`: evaluates DarijaMMLU humanities tasks. +* `darijammlu_language`: evaluates DarijaMMLU language tasks. +* `darijammlu_other`: evaluates other DarijaMMLU tasks. + +### Checklist + +For adding novel benchmarks/datasets to the library: +* [x] Is the task an existing benchmark in the literature? + * [x] Have you referenced the original paper that introduced the task? + * [ ] If yes, does the original paper provide a reference implementation? If so, have you checked against the reference implementation and documented how to run such a test? + + +If other tasks on this dataset are already supported: +* [ ] Is the "Main" variant of this task clearly denoted? +* [ ] Have you provided a short sentence in a README on what each new variant adds / evaluates? +* [ ] Have you noted which, if any, published evaluation setups are matched by this variant? diff --git a/lm-evaluation-harness/lm_eval/tasks/darijammlu/_darijammlu.yaml b/lm-evaluation-harness/lm_eval/tasks/darijammlu/_darijammlu.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4dbc297c0b2570c4e1d32b5bd60cb29375871197 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darijammlu/_darijammlu.yaml @@ -0,0 +1,10 @@ +group: darijammlu +group_alias: DarijaMMLU +task: +- darijammlu_mmlu +- darijammlu_ar_mmlu +aggregate_metric_list: + - metric: acc + weight_by_size: True +metadata: + version: 0 diff --git a/lm-evaluation-harness/lm_eval/tasks/darijammlu/_darijammlu_ar_mmlu.yaml b/lm-evaluation-harness/lm_eval/tasks/darijammlu/_darijammlu_ar_mmlu.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dc9543d6ca5eb7546922d5cfaba6dc4968f3e0b4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darijammlu/_darijammlu_ar_mmlu.yaml @@ -0,0 +1,9 @@ +group: darijammlu_ar_mmlu +group_alias: ArabicMMLU +task: + - darijammlu_ar_mmlu_tasks +aggregate_metric_list: + - metric: acc + weight_by_size: True +metadata: + version: 0 diff --git a/lm-evaluation-harness/lm_eval/tasks/darijammlu/_darijammlu_mmlu.yaml b/lm-evaluation-harness/lm_eval/tasks/darijammlu/_darijammlu_mmlu.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bdaf2876d8767a691bb9f98b037d4fa6ce21492b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darijammlu/_darijammlu_mmlu.yaml @@ -0,0 +1,9 @@ +group: darijammlu_mmlu +group_alias: MMLU +task: + - darijammlu_mmlu_tasks +aggregate_metric_list: + - metric: acc + weight_by_size: True +metadata: + version: 0 diff --git a/lm-evaluation-harness/lm_eval/tasks/darijammlu/_default_darijammlu_template_yaml b/lm-evaluation-harness/lm_eval/tasks/darijammlu/_default_darijammlu_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..10d0a8a056dd7778305525609e55cd7cefe77856 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darijammlu/_default_darijammlu_template_yaml @@ -0,0 +1,17 @@ +dataset_path: MBZUAI-Paris/DarijaMMLU +test_split: test +fewshot_split: dev +fewshot_config: + sampler: first_n +output_type: multiple_choice +doc_to_text: !function utils.doc_to_text +doc_to_choice: !function utils.doc_to_choice +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 +dataset_kwargs: + trust_remote_code: true diff --git a/lm-evaluation-harness/lm_eval/tasks/darijammlu/_generate_configs.py b/lm-evaluation-harness/lm_eval/tasks/darijammlu/_generate_configs.py new file mode 100644 index 0000000000000000000000000000000000000000..b96d857e1425fcd2f6013504cd4a91fb02912678 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darijammlu/_generate_configs.py @@ -0,0 +1,129 @@ +""" +Take in a YAML, and output all "other" splits with this YAML +""" + +import argparse +import logging +import os + +import yaml +from tqdm import tqdm + + +eval_logger = logging.getLogger("lm-eval") + + +MMLU_SUBJECTS = { + "global_facts": "other", + "high_school_european_history": "humanities", + "high_school_geography": "social_sciences", + "high_school_government_and_politics": "social_sciences", + "high_school_psychology": "social_sciences", + "high_school_statistics": "stem", + "high_school_world_history": "humanities", + "human_aging": "other", + "international_law": "humanities", + "jurisprudence": "humanities", + "logical_fallacies": "humanities", + "management": "other", + "marketing": "other", + "moral_disputes": "humanities", + "moral_scenarios": "humanities", + "nutrition": "other", + "philosophy": "humanities", + "professional_law": "humanities", + "professional_psychology": "social_sciences", + "public_relations": "social_sciences", + "security_studies": "social_sciences", + "sociology": "social_sciences", + "world_religions": "humanities", +} + + +ARABIC_MMLU_SUBJECTS = { + "islamic_studies": "humanities", + "driving_test": "other", + "natural_science": "stem", + "history": "humanities", + "general_knowledge": "other", + "law": "humanities", + "physics": "stem", + "social_science": "social_sciences", + "management_ar": "other", + "arabic_language": "language", + "political_science": "social_sciences", + "philosophy_ar": "humanities", + "accounting": "social_sciences", + "computer_science": "stem", + "geography": "social_sciences", + "math": "stem", + "biology": "stem", + "economics": "social_sciences", + "arabic_language_(general)": "language", + "arabic_language_(grammar)": "language", + "civics": "social_sciences", +} + +DATASETS = { + "mmlu": MMLU_SUBJECTS, + "ar_mmlu": ARABIC_MMLU_SUBJECTS, +} + + +def parse_args(): + parser = argparse.ArgumentParser() + parser.add_argument("--base_yaml_path", default="_default_darijammlu_template_yaml") + parser.add_argument("--save_prefix_path", default="darijammlu") + return parser.parse_args() + + +if __name__ == "__main__": + args = parse_args() + + # get filename of base_yaml so we can `"include": ` it in our "other" YAMLs. + base_yaml_name = os.path.split(args.base_yaml_path)[-1] + with open(args.base_yaml_path, encoding="utf-8") as f: + base_yaml = yaml.full_load(f) + + ALL_CATEGORIES = [] + for dataset, SUBJECTS in DATASETS.items(): + for subject, category in tqdm(SUBJECTS.items()): + if category not in ALL_CATEGORIES: + ALL_CATEGORIES.append(category) + + yaml_dict = { + "include": base_yaml_name, + "tag": [ + f"darijammlu_{category}_tasks", + "darijammlu_" + dataset + "_tasks", + ], + "task": f"darijammlu_{subject}", + "task_alias": subject.replace("_", " "), + "dataset_name": subject, + } + + file_save_path = args.save_prefix_path + f"_{subject}.yaml" + eval_logger.info(f"Saving yaml for subset {subject} to {file_save_path}") + with open(file_save_path, "w", encoding="utf-8") as yaml_file: + yaml.dump( + yaml_dict, + yaml_file, + allow_unicode=True, + default_style='"', + ) + + darijammlu_subcategories = [f"darijammlu_{category}" for category in ALL_CATEGORIES] + + file_save_path = args.save_prefix_path + ".yaml" + + eval_logger.info(f"Saving benchmark config to {file_save_path}") + with open(file_save_path, "w", encoding="utf-8") as yaml_file: + yaml.dump( + { + "group": "darijammlu", + "task": darijammlu_subcategories, + }, + yaml_file, + indent=4, + default_flow_style=False, + ) diff --git a/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_accounting.yaml b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_accounting.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cde1c5f5f78382e85b84703d8e09c1ad26152a1b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_accounting.yaml @@ -0,0 +1,6 @@ +"dataset_name": "accounting" +"include": "_default_darijammlu_template_yaml" +"tag": +- "darijammlu_ar_mmlu_tasks" +"task": "darijammlu_accounting" +"task_alias": "accounting" diff --git a/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_arabic_language.yaml b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_arabic_language.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ead996fe0cfccb891b7426ac9adaae9fe18fc28b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_arabic_language.yaml @@ -0,0 +1,6 @@ +"dataset_name": "arabic_language" +"include": "_default_darijammlu_template_yaml" +"tag": +- "darijammlu_ar_mmlu_tasks" +"task": "darijammlu_arabic_language" +"task_alias": "arabic language" diff --git a/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_arabic_language_(general).yaml b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_arabic_language_(general).yaml new file mode 100644 index 0000000000000000000000000000000000000000..303fbd109ea220e69bb7b3bc0781660a1473d4b9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_arabic_language_(general).yaml @@ -0,0 +1,6 @@ +"dataset_name": "arabic_language_(general)" +"include": "_default_darijammlu_template_yaml" +"tag": +- "darijammlu_ar_mmlu_tasks" +"task": "darijammlu_arabic_language_(general)" +"task_alias": "arabic language (general)" diff --git a/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_arabic_language_(grammar).yaml b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_arabic_language_(grammar).yaml new file mode 100644 index 0000000000000000000000000000000000000000..f064d8f662149f503ba20fb21e0e6cb109fd82c4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_arabic_language_(grammar).yaml @@ -0,0 +1,6 @@ +"dataset_name": "arabic_language_(grammar)" +"include": "_default_darijammlu_template_yaml" +"tag": +- "darijammlu_ar_mmlu_tasks" +"task": "darijammlu_arabic_language_(grammar)" +"task_alias": "arabic language (grammar)" diff --git a/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_biology.yaml b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_biology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..db2c96f96f80c863a7e7397ac1d850d0f7db5dbd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_biology.yaml @@ -0,0 +1,6 @@ +"dataset_name": "biology" +"include": "_default_darijammlu_template_yaml" +"tag": +- "darijammlu_ar_mmlu_tasks" +"task": "darijammlu_biology" +"task_alias": "biology" diff --git a/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_civics.yaml b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_civics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d851a11a1103f58200be507c41b07cf165f9d291 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_civics.yaml @@ -0,0 +1,6 @@ +"dataset_name": "civics" +"include": "_default_darijammlu_template_yaml" +"tag": +- "darijammlu_ar_mmlu_tasks" +"task": "darijammlu_civics" +"task_alias": "civics" diff --git a/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_computer_science.yaml b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_computer_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7add6f358510f998cb3032c175f27de61abd2275 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_computer_science.yaml @@ -0,0 +1,6 @@ +"dataset_name": "computer_science" +"include": "_default_darijammlu_template_yaml" +"tag": +- "darijammlu_ar_mmlu_tasks" +"task": "darijammlu_computer_science" +"task_alias": "computer science" diff --git a/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_driving_test.yaml b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_driving_test.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7d8d4af4028f00960b51cd5da7d2095fa0c08422 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_driving_test.yaml @@ -0,0 +1,6 @@ +"dataset_name": "driving_test" +"include": "_default_darijammlu_template_yaml" +"tag": +- "darijammlu_ar_mmlu_tasks" +"task": "darijammlu_driving_test" +"task_alias": "driving test" diff --git a/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_economics.yaml b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_economics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b16d3422424dbea0de30a7de51af8f429061d407 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_economics.yaml @@ -0,0 +1,6 @@ +"dataset_name": "economics" +"include": "_default_darijammlu_template_yaml" +"tag": +- "darijammlu_ar_mmlu_tasks" +"task": "darijammlu_economics" +"task_alias": "economics" diff --git a/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_general_knowledge.yaml b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_general_knowledge.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ac9f696696bb174df7f517bfff40b6a09a857252 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_general_knowledge.yaml @@ -0,0 +1,6 @@ +"dataset_name": "general_knowledge" +"include": "_default_darijammlu_template_yaml" +"tag": +- "darijammlu_ar_mmlu_tasks" +"task": "darijammlu_general_knowledge" +"task_alias": "general knowledge" diff --git a/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_geography.yaml b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_geography.yaml new file mode 100644 index 0000000000000000000000000000000000000000..430aad2e151fd4ebaa916a29c6f6787a486feee3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_geography.yaml @@ -0,0 +1,6 @@ +"dataset_name": "geography" +"include": "_default_darijammlu_template_yaml" +"tag": +- "darijammlu_ar_mmlu_tasks" +"task": "darijammlu_geography" +"task_alias": "geography" diff --git a/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_global_facts.yaml b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_global_facts.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6fde263e646812cd3b744cdd0b7b1fdcb4567a76 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_global_facts.yaml @@ -0,0 +1,6 @@ +"dataset_name": "global_facts" +"include": "_default_darijammlu_template_yaml" +"tag": +- "darijammlu_mmlu_tasks" +"task": "darijammlu_global_facts" +"task_alias": "global facts" diff --git a/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_high_school_european_history.yaml b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_high_school_european_history.yaml new file mode 100644 index 0000000000000000000000000000000000000000..30d5563e94a0822eb69edabf7323f328cb09390d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_high_school_european_history.yaml @@ -0,0 +1,6 @@ +"dataset_name": "high_school_european_history" +"include": "_default_darijammlu_template_yaml" +"tag": +- "darijammlu_mmlu_tasks" +"task": "darijammlu_high_school_european_history" +"task_alias": "high school european history" diff --git a/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_high_school_geography.yaml b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_high_school_geography.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1532554ed69843083e56d70c8d252b5c8eb6a686 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_high_school_geography.yaml @@ -0,0 +1,6 @@ +"dataset_name": "high_school_geography" +"include": "_default_darijammlu_template_yaml" +"tag": +- "darijammlu_mmlu_tasks" +"task": "darijammlu_high_school_geography" +"task_alias": "high school geography" diff --git a/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_high_school_government_and_politics.yaml b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_high_school_government_and_politics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9acb0b81eee01e022d414545aaba362da1c89e95 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_high_school_government_and_politics.yaml @@ -0,0 +1,6 @@ +"dataset_name": "high_school_government_and_politics" +"include": "_default_darijammlu_template_yaml" +"tag": +- "darijammlu_mmlu_tasks" +"task": "darijammlu_high_school_government_and_politics" +"task_alias": "high school government and politics" diff --git a/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_high_school_psychology.yaml b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_high_school_psychology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b5e7908e5f0a417fbdfecf2328d717546134f47f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_high_school_psychology.yaml @@ -0,0 +1,6 @@ +"dataset_name": "high_school_psychology" +"include": "_default_darijammlu_template_yaml" +"tag": +- "darijammlu_mmlu_tasks" +"task": "darijammlu_high_school_psychology" +"task_alias": "high school psychology" diff --git a/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_high_school_statistics.yaml b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_high_school_statistics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5fc129ec1631fc03a89045e9579e7f0c64c5b425 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_high_school_statistics.yaml @@ -0,0 +1,6 @@ +"dataset_name": "high_school_statistics" +"include": "_default_darijammlu_template_yaml" +"tag": +- "darijammlu_mmlu_tasks" +"task": "darijammlu_high_school_statistics" +"task_alias": "high school statistics" diff --git a/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_high_school_world_history.yaml b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_high_school_world_history.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2e56ba14b4110712dbaac66c4089db4bd3fbf56f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_high_school_world_history.yaml @@ -0,0 +1,6 @@ +"dataset_name": "high_school_world_history" +"include": "_default_darijammlu_template_yaml" +"tag": +- "darijammlu_mmlu_tasks" +"task": "darijammlu_high_school_world_history" +"task_alias": "high school world history" diff --git a/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_history.yaml b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_history.yaml new file mode 100644 index 0000000000000000000000000000000000000000..19536c249c168c17864c38706a03c3efee0899e7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_history.yaml @@ -0,0 +1,6 @@ +"dataset_name": "history" +"include": "_default_darijammlu_template_yaml" +"tag": +- "darijammlu_ar_mmlu_tasks" +"task": "darijammlu_history" +"task_alias": "history" diff --git a/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_human_aging.yaml b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_human_aging.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1ad2ce9985b9414416178d977232e5b7a5d3da5a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_human_aging.yaml @@ -0,0 +1,6 @@ +"dataset_name": "human_aging" +"include": "_default_darijammlu_template_yaml" +"tag": +- "darijammlu_mmlu_tasks" +"task": "darijammlu_human_aging" +"task_alias": "human aging" diff --git a/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_international_law.yaml b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_international_law.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0c4afb7a88bf2ca81ad619b2de402159c7f0fbce --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_international_law.yaml @@ -0,0 +1,6 @@ +"dataset_name": "international_law" +"include": "_default_darijammlu_template_yaml" +"tag": +- "darijammlu_mmlu_tasks" +"task": "darijammlu_international_law" +"task_alias": "international law" diff --git a/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_islamic_studies.yaml b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_islamic_studies.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b5bafc694fd7ccaf88d061c40a3fc8216ce35da7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_islamic_studies.yaml @@ -0,0 +1,6 @@ +"dataset_name": "islamic_studies" +"include": "_default_darijammlu_template_yaml" +"tag": +- "darijammlu_ar_mmlu_tasks" +"task": "darijammlu_islamic_studies" +"task_alias": "islamic studies" diff --git a/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_jurisprudence.yaml b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_jurisprudence.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f35d51e209863b7c3bf555e58d74dd75a185b13b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_jurisprudence.yaml @@ -0,0 +1,6 @@ +"dataset_name": "jurisprudence" +"include": "_default_darijammlu_template_yaml" +"tag": +- "darijammlu_mmlu_tasks" +"task": "darijammlu_jurisprudence" +"task_alias": "jurisprudence" diff --git a/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_law.yaml b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_law.yaml new file mode 100644 index 0000000000000000000000000000000000000000..75cde2b7ff34d3fbbf140af2ccf904b2cf56bd0d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_law.yaml @@ -0,0 +1,6 @@ +"dataset_name": "law" +"include": "_default_darijammlu_template_yaml" +"tag": +- "darijammlu_ar_mmlu_tasks" +"task": "darijammlu_law" +"task_alias": "law" diff --git a/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_logical_fallacies.yaml b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_logical_fallacies.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7bf2dc57981e843e7116e1bb0f8e2aea54e75946 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_logical_fallacies.yaml @@ -0,0 +1,6 @@ +"dataset_name": "logical_fallacies" +"include": "_default_darijammlu_template_yaml" +"tag": +- "darijammlu_mmlu_tasks" +"task": "darijammlu_logical_fallacies" +"task_alias": "logical fallacies" diff --git a/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_management.yaml b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_management.yaml new file mode 100644 index 0000000000000000000000000000000000000000..820da4c1a23367662b4eec823f922065b2054b96 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_management.yaml @@ -0,0 +1,6 @@ +"dataset_name": "management" +"include": "_default_darijammlu_template_yaml" +"tag": +- "darijammlu_mmlu_tasks" +"task": "darijammlu_management" +"task_alias": "management" diff --git a/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_management_ar.yaml b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_management_ar.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2bffbdde5e43eab2febe504ec3eafa511de7ad91 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_management_ar.yaml @@ -0,0 +1,6 @@ +"dataset_name": "management_ar" +"include": "_default_darijammlu_template_yaml" +"tag": +- "darijammlu_ar_mmlu_tasks" +"task": "darijammlu_management_ar" +"task_alias": "management ar" diff --git a/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_marketing.yaml b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_marketing.yaml new file mode 100644 index 0000000000000000000000000000000000000000..feb54c98eb3a8dbaf5495fc578a4207bc0eb98ff --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_marketing.yaml @@ -0,0 +1,6 @@ +"dataset_name": "marketing" +"include": "_default_darijammlu_template_yaml" +"tag": +- "darijammlu_mmlu_tasks" +"task": "darijammlu_marketing" +"task_alias": "marketing" diff --git a/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_math.yaml b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_math.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5d870df7711aefa8cbb7e165e4b7ace92839591a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_math.yaml @@ -0,0 +1,6 @@ +"dataset_name": "math" +"include": "_default_darijammlu_template_yaml" +"tag": +- "darijammlu_ar_mmlu_tasks" +"task": "darijammlu_math" +"task_alias": "math" diff --git a/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_moral_disputes.yaml b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_moral_disputes.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d5b18632de1b32bff96075f0b08bc6aa765fbb8c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_moral_disputes.yaml @@ -0,0 +1,6 @@ +"dataset_name": "moral_disputes" +"include": "_default_darijammlu_template_yaml" +"tag": +- "darijammlu_mmlu_tasks" +"task": "darijammlu_moral_disputes" +"task_alias": "moral disputes" diff --git a/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_moral_scenarios.yaml b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_moral_scenarios.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9aff528a29a8b216e0f6e279e964785c22b10c5c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_moral_scenarios.yaml @@ -0,0 +1,6 @@ +"dataset_name": "moral_scenarios" +"include": "_default_darijammlu_template_yaml" +"tag": +- "darijammlu_mmlu_tasks" +"task": "darijammlu_moral_scenarios" +"task_alias": "moral scenarios" diff --git a/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_natural_science.yaml b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_natural_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6527b811c6a01fca3361f217e0444f2874c30536 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_natural_science.yaml @@ -0,0 +1,6 @@ +"dataset_name": "natural_science" +"include": "_default_darijammlu_template_yaml" +"tag": +- "darijammlu_ar_mmlu_tasks" +"task": "darijammlu_natural_science" +"task_alias": "natural science" diff --git a/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_nutrition.yaml b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_nutrition.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0d675cf717f784d683591059e8baffe395f857fc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_nutrition.yaml @@ -0,0 +1,6 @@ +"dataset_name": "nutrition" +"include": "_default_darijammlu_template_yaml" +"tag": +- "darijammlu_mmlu_tasks" +"task": "darijammlu_nutrition" +"task_alias": "nutrition" diff --git a/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_philosophy.yaml b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_philosophy.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b0ac560919e8e3d79151bbe27993c2557b8b070e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_philosophy.yaml @@ -0,0 +1,6 @@ +"dataset_name": "philosophy" +"include": "_default_darijammlu_template_yaml" +"tag": +- "darijammlu_mmlu_tasks" +"task": "darijammlu_philosophy" +"task_alias": "philosophy" diff --git a/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_philosophy_ar.yaml b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_philosophy_ar.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3459381e413bbbe89bc1e24553f8e3690cea7048 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_philosophy_ar.yaml @@ -0,0 +1,6 @@ +"dataset_name": "philosophy_ar" +"include": "_default_darijammlu_template_yaml" +"tag": +- "darijammlu_ar_mmlu_tasks" +"task": "darijammlu_philosophy_ar" +"task_alias": "philosophy ar" diff --git a/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_physics.yaml b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_physics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..366d1961529bd3ea7e5014dbe484c948ac2cffa5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_physics.yaml @@ -0,0 +1,6 @@ +"dataset_name": "physics" +"include": "_default_darijammlu_template_yaml" +"tag": +- "darijammlu_ar_mmlu_tasks" +"task": "darijammlu_physics" +"task_alias": "physics" diff --git a/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_political_science.yaml b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_political_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c1c4c43a9fc5c6fbbe4d5504446c3c7311db2aad --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_political_science.yaml @@ -0,0 +1,6 @@ +"dataset_name": "political_science" +"include": "_default_darijammlu_template_yaml" +"tag": +- "darijammlu_ar_mmlu_tasks" +"task": "darijammlu_political_science" +"task_alias": "political science" diff --git a/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_professional_law.yaml b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_professional_law.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3bbc1db73c4418ed5a8041994e9b53e736bfe1d8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_professional_law.yaml @@ -0,0 +1,6 @@ +"dataset_name": "professional_law" +"include": "_default_darijammlu_template_yaml" +"tag": +- "darijammlu_mmlu_tasks" +"task": "darijammlu_professional_law" +"task_alias": "professional law" diff --git a/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_professional_psychology.yaml b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_professional_psychology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fc4b0fc59d310d1861559dbb8f538f1701265dac --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_professional_psychology.yaml @@ -0,0 +1,6 @@ +"dataset_name": "professional_psychology" +"include": "_default_darijammlu_template_yaml" +"tag": +- "darijammlu_mmlu_tasks" +"task": "darijammlu_professional_psychology" +"task_alias": "professional psychology" diff --git a/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_public_relations.yaml b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_public_relations.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8efc220ac8662f7d81ecb4875d3532ab859537a0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_public_relations.yaml @@ -0,0 +1,6 @@ +"dataset_name": "public_relations" +"include": "_default_darijammlu_template_yaml" +"tag": +- "darijammlu_mmlu_tasks" +"task": "darijammlu_public_relations" +"task_alias": "public relations" diff --git a/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_security_studies.yaml b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_security_studies.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3942fd47bed14112c38f2aed861f08f6dd837dcc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_security_studies.yaml @@ -0,0 +1,6 @@ +"dataset_name": "security_studies" +"include": "_default_darijammlu_template_yaml" +"tag": +- "darijammlu_mmlu_tasks" +"task": "darijammlu_security_studies" +"task_alias": "security studies" diff --git a/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9f46a914d0b0787d4dcfdf72acc73ce1aaf19a11 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_social_science.yaml @@ -0,0 +1,6 @@ +"dataset_name": "social_science" +"include": "_default_darijammlu_template_yaml" +"tag": +- "darijammlu_ar_mmlu_tasks" +"task": "darijammlu_social_science" +"task_alias": "social science" diff --git a/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_sociology.yaml b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_sociology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7cc523d2673722a47c8090a03b3b6b6208dd6aa7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_sociology.yaml @@ -0,0 +1,6 @@ +"dataset_name": "sociology" +"include": "_default_darijammlu_template_yaml" +"tag": +- "darijammlu_mmlu_tasks" +"task": "darijammlu_sociology" +"task_alias": "sociology" diff --git a/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_world_religions.yaml b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_world_religions.yaml new file mode 100644 index 0000000000000000000000000000000000000000..86d32b02d237934aee1feca72dce5dd9060f83f4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darijammlu/darijammlu_world_religions.yaml @@ -0,0 +1,6 @@ +"dataset_name": "world_religions" +"include": "_default_darijammlu_template_yaml" +"tag": +- "darijammlu_mmlu_tasks" +"task": "darijammlu_world_religions" +"task_alias": "world religions" diff --git a/lm-evaluation-harness/lm_eval/tasks/darijammlu/utils.py b/lm-evaluation-harness/lm_eval/tasks/darijammlu/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..a9712fae569545667010e932dda0f2e45bde4816 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/darijammlu/utils.py @@ -0,0 +1,25 @@ +PROMPT = "هادا سؤال متعدد الخيارات (مع الجواب ديالو) على {}\n\n{}\n{}\nالجواب:" + + +alpha = ["A.", "B.", "C.", "D.", "E."] + + +def doc_to_text(doc): + subject = doc["subject_darija"] + question = ( + doc["question"] + if doc["context"] == "" + else f"{doc['context']}\n\n{doc['question']}" + ) + + options = [] + for i, opt in enumerate(doc["choices"]): + options.append(f"{alpha[i]} {opt}") + + doc_text = PROMPT.format(subject, question, "\n".join(options)) + + return doc_text + + +def doc_to_choice(doc): + return [alpha[i][0] for i in range(len(doc["choices"]))] diff --git a/lm-evaluation-harness/lm_eval/tasks/drop/README.md b/lm-evaluation-harness/lm_eval/tasks/drop/README.md new file mode 100644 index 0000000000000000000000000000000000000000..6b7fc47b7165034bd74c524048f5f54ea8d041cf --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/drop/README.md @@ -0,0 +1,53 @@ +# DROP + +### Paper + +Title: `DROP: A Reading Comprehension Benchmark Requiring Discrete Reasoning Over Paragraphs` + +Abstract: https://aclanthology.org/attachments/N19-1246.Supplementary.pdf + +DROP is a QA dataset which tests comprehensive understanding of paragraphs. In +this crowdsourced, adversarially-created, 96k question-answering benchmark, a +system must resolve multiple references in a question, map them onto a paragraph, +and perform discrete operations over them (such as addition, counting, or sorting). + +Homepage: https://allenai.org/data/drop + +Acknowledgement: This implementation is based on the official evaluation for `DROP`: +https://github.com/allenai/allennlp-reading-comprehension/blob/master/allennlp_rc/eval/drop_eval.py + +### Citation + +``` +@misc{dua2019drop, + title={DROP: A Reading Comprehension Benchmark Requiring Discrete Reasoning Over Paragraphs}, + author={Dheeru Dua and Yizhong Wang and Pradeep Dasigi and Gabriel Stanovsky and Sameer Singh and Matt Gardner}, + year={2019}, + eprint={1903.00161}, + archivePrefix={arXiv}, + primaryClass={cs.CL} +} +``` + +### Groups and Tasks + +#### Groups + +* Not part of a group yet. + +#### Tasks + +* `drop` + +### Checklist + +For adding novel benchmarks/datasets to the library: +* [ ] Is the task an existing benchmark in the literature? + * [ ] Have you referenced the original paper that introduced the task? + * [ ] If yes, does the original paper provide a reference implementation? If so, have you checked against the reference implementation and documented how to run such a test? + + +If other tasks on this dataset are already supported: +* [ ] Is the "Main" variant of this task clearly denoted? +* [ ] Have you provided a short sentence in a README on what each new variant adds / evaluates? +* [ ] Have you noted which, if any, published evaluation setups are matched by this variant? diff --git a/lm-evaluation-harness/lm_eval/tasks/drop/default.yaml b/lm-evaluation-harness/lm_eval/tasks/drop/default.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4a936121524950e8a89822058cb2b29f244f31a4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/drop/default.yaml @@ -0,0 +1,26 @@ +task: drop +dataset_path: EleutherAI/drop +output_type: generate_until +training_split: train +validation_split: validation +process_docs: !function utils.process_docs +doc_to_text: "{{passage}} {{question}}" +doc_to_target: "{{ answer|join(',')}}" +target_delimiter: "" +process_results: !function utils.process_results +should_decontaminate: true +doc_to_decontamination_query: "{{passage}} {{question}}" +generation_kwargs: + until: + - "." +metric_list: + - metric: em + aggregation: mean + higher_is_better: true + - metric: f1 + aggregation: mean + higher_is_better: true +metadata: + version: 3.0 +dataset_kwargs: + trust_remote_code: true diff --git a/lm-evaluation-harness/lm_eval/tasks/drop/utils.py b/lm-evaluation-harness/lm_eval/tasks/drop/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fc4e7d4b4db1775cdae632d4a5334adeeeffb318 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/drop/utils.py @@ -0,0 +1,205 @@ +import re +import string + +import numpy as np + + +_ARTICLES = re.compile(r"\b(a|an|the)\b", re.UNICODE) + + +def process_docs(dataset): + def _process(doc): + return { + "id": doc["query_id"], + "passage": doc["passage"], + "question": doc["question"], + "answers": get_answers(doc), + } + + return dataset.map(_process) + + +def get_answers(doc): + def _flatten_validated_answers(validated_answers): + """Flattens a dict of lists of validated answers. + {"number": ['1', '8'], ...} + -> [{"number": ['1'], ...}, {"number": ['8'], ...}] + """ + valid_answers = [] + for i in range(len(validated_answers["number"])): + valid_answers.append( + { + "number": validated_answers["number"][i], + "date": validated_answers["date"][i], + "spans": validated_answers["spans"][i], + } + ) + return valid_answers + + answers = [] + answers_set = set() + candidates = [doc["answer"]] + _flatten_validated_answers(doc["validated_answers"]) + for candidate in candidates: + answer = parse_answer(candidate) + if answer in answers_set: + continue + answers_set.add(answer) + answers.append(answer) + return answers + + +def parse_answer(answer): + # NOTE: Everything is returned as a tuple for uniformity and hashability. + if answer["number"] != "": + return (str(answer["number"]),) + if answer["spans"] != []: + return tuple(answer["spans"]) + return ( + " ".join( + [answer["date"]["day"], answer["date"]["month"], answer["date"]["year"]] + ).strip(), + ) + + +def process_results(doc, results): + preds, golds = results, doc["answers"] + max_em = 0 + max_f1 = 0 + for gold_answer in golds: + exact_match, f1_score = get_metrics(preds, gold_answer) + if gold_answer[0].strip(): + max_em = max(max_em, exact_match) + max_f1 = max(max_f1, f1_score) + return {"em": max_em, "f1": max_f1} + + +def get_metrics(predicted, gold): + """ + Takes a predicted answer and a gold answer (that are both either a string or a list of + strings), and returns exact match and the DROP F1 metric for the prediction. If you are + writing a script for evaluating objects in memory (say, the output of predictions during + validation, or while training), this is the function you want to call, after using + :func:`answer_json_to_strings` when reading the gold answer from the released data file. + """ + predicted_bags = _answer_to_bags(predicted) + gold_bags = _answer_to_bags(gold) + + if set(predicted_bags[0]) == set(gold_bags[0]) and len(predicted_bags[0]) == len( + gold_bags[0] + ): + exact_match = 1.0 + else: + exact_match = 0.0 + + f1_per_bag = _align_bags(predicted_bags[1], gold_bags[1]) + f1 = np.mean(f1_per_bag) + f1 = round(f1, 2) + return exact_match, f1 + + +def _answer_to_bags(answer): + if isinstance(answer, (list, tuple)): + raw_spans = answer + else: + raw_spans = [answer] + normalized_spans = [] + token_bags = [] + for raw_span in raw_spans: + normalized_span = _normalize(raw_span) + normalized_spans.append(normalized_span) + token_bags.append(set(normalized_span.split())) + return normalized_spans, token_bags + + +def _align_bags(predicted, gold): + """ + Takes gold and predicted answer sets and first finds the optimal 1-1 alignment + between them and gets maximum metric values over all the answers. + """ + from scipy.optimize import linear_sum_assignment + + scores = np.zeros([len(gold), len(predicted)]) + for gold_index, gold_item in enumerate(gold): + for pred_index, pred_item in enumerate(predicted): + if _match_numbers_if_present(gold_item, pred_item): + scores[gold_index, pred_index] = _compute_f1(pred_item, gold_item) + row_ind, col_ind = linear_sum_assignment(-scores) + + max_scores = np.zeros([max(len(gold), len(predicted))]) + for row, column in zip(row_ind, col_ind): + max_scores[row] = max(max_scores[row], scores[row, column]) + return max_scores + + +def _compute_f1(predicted_bag, gold_bag): + intersection = len(gold_bag.intersection(predicted_bag)) + if not predicted_bag: + precision = 1.0 + else: + precision = intersection / float(len(predicted_bag)) + if not gold_bag: + recall = 1.0 + else: + recall = intersection / float(len(gold_bag)) + f1 = ( + (2 * precision * recall) / (precision + recall) + if not (precision == 0.0 and recall == 0.0) + else 0.0 + ) + return f1 + + +def _match_numbers_if_present(gold_bag, predicted_bag): + gold_numbers = set() + predicted_numbers = set() + for word in gold_bag: + if _is_number(word): + gold_numbers.add(word) + for word in predicted_bag: + if _is_number(word): + predicted_numbers.add(word) + if (not gold_numbers) or gold_numbers.intersection(predicted_numbers): + return True + return False + + +def _is_number(text): + try: + float(text) + return True + except ValueError: + return False + + +def _remove_articles(text): + return _ARTICLES.sub(" ", text) + + +def _white_space_fix(text): + return " ".join(text.split()) + + +def _remove_punc(text): + exclude = set(string.punctuation) + if not _is_number(text): + return "".join(ch for ch in text if ch not in exclude) + else: + return text + + +def _fix_number(text): + return str(float(text)) if _is_number(text) else text + + +def _tokenize(text): + return re.split(" |-", text) + + +def _normalize(answer): + tokens = [ + _white_space_fix(_remove_articles(_fix_number(_remove_punc(token.lower())))) + for token in _tokenize(answer) + ] + tokens = [token for token in tokens if token.strip()] + normalized = " ".join(tokens).strip() + return normalized diff --git a/lm-evaluation-harness/lm_eval/tasks/eq_bench/README.md b/lm-evaluation-harness/lm_eval/tasks/eq_bench/README.md new file mode 100644 index 0000000000000000000000000000000000000000..df11108eece0f281687d8098e6a56476a75decb0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eq_bench/README.md @@ -0,0 +1,55 @@ +# EQ-Bench + +Title: `EQ-Bench: An Emotional Intelligence Benchmark for Large Language Models` + +Abstract: https://arxiv.org/abs/2312.06281 + +EQ-Bench is a benchmark for language models designed to assess emotional intelligence. + +Why emotional intelligence? One reason is that it represents a subset of abilities that are important for the user experience, and which isn't explicitly tested by other benchmarks. Another reason is that it's not trivial to improve scores by fine tuning for the benchmark, which makes it harder to "game" the leaderboard. + +EQ-Bench is a little different from traditional psychometric tests. It uses a specific question format, in which the subject has to read a dialogue then rate the intensity of possible emotional responses of one of the characters. Every question is interpretative and assesses the ability to predict the magnitude of the 4 presented emotions. The test is graded without the need for a judge (so there is no length bias). It's cheap to run (only 171 questions), and produces results that correlate strongly with human preference (Arena ELO) and multi-domain benchmarks like MMLU. + +Homepage: https://eqbench.com/ + + +NOTE: There are some key differences between the lm-evaluation-harness version and the implementation described in the EQ-Bench paper (These have been OK'd by the author): + +- The lm-eval version uses the EQ-Bench v2 test set (171 questions) and score calculation. It does not incorporate the revision part of the prompt, as per v2.1 (https://github.com/EQ-bench/EQ-Bench) +- No retries in lm-eval version (EQ-Bench pipeline retries with successively higher temps if it encounters unparsable answers) +- In the original implementation, unparsable answers are excluded from the final score, and 83% of answers have to be parseable or a fail is returned. The lm-eval version instead assigns 0 to unparsable answers and has no fail criteria. So for lower performing models, there may be differences with the EQ-Bench leaderboard. + + +### Citation + +```bibtex +@misc{paech2023eqbench, + title={EQ-Bench: An Emotional Intelligence Benchmark for Large Language Models}, + author={Samuel J. Paech}, + year={2023}, + eprint={2312.06281}, + archivePrefix={arXiv}, + primaryClass={cs.CL} +} +``` + +### Groups and Tasks + +#### Groups + +* Not part of a group yet + +#### Tasks + +* `eq_bench` + +### Checklist + +* [x] Is the task an existing benchmark in the literature? + * [x] Have you referenced the original paper that introduced the task? + * [ ] If yes, does the original paper provide a reference implementation? If so, have you checked against the reference implementation and documented how to run such a test? + +If other tasks on this dataset are already supported: +* [ ] Is the "Main" variant of this task clearly denoted? +* [ ] Have you provided a short sentence in a README on what each new variant adds / evaluates? +* [ ] Have you noted which, if any, published evaluation setups are matched by this variant? diff --git a/lm-evaluation-harness/lm_eval/tasks/eq_bench/default.yaml b/lm-evaluation-harness/lm_eval/tasks/eq_bench/default.yaml new file mode 100644 index 0000000000000000000000000000000000000000..16b1245b22c91e74a4ab398945a27ac31c82c5a8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eq_bench/default.yaml @@ -0,0 +1,20 @@ +task: eq_bench +dataset_path: pbevan11/EQ-Bench +output_type: generate_until +validation_split: validation +doc_to_text: prompt +doc_to_target: reference_answer_fullscale +process_results: !function utils.calculate_score_fullscale +generation_kwargs: + do_sample: false + temperature: 0.0 + max_gen_toks: 80 +metric_list: + - metric: eqbench + aggregation: mean + higher_is_better: true + - metric: percent_parseable + aggregation: mean + higher_is_better: true +metadata: + version: 2.1 diff --git a/lm-evaluation-harness/lm_eval/tasks/eq_bench/utils.py b/lm-evaluation-harness/lm_eval/tasks/eq_bench/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..326a0dc485f22c01053c10e65bc9bf05e1aeb590 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eq_bench/utils.py @@ -0,0 +1,54 @@ +import math +import re + + +def calculate_score_fullscale(docs, results): + reference = eval(docs["reference_answer_fullscale"]) + user = dict(re.findall(r"(\w+):\s+(\d+)", results[0])) + # First check that the emotions specified in the answer match those in the reference + if len(user.items()) != 4: + # print('! Error: 4 emotions were not returned') + # print(user) + return {"eqbench": 0, "percent_parseable": 0} + emotions_dict = {} + for emotion, user_emotion_score in user.items(): + for i in range(1, 5): + if emotion == reference[f"emotion{i}"]: + emotions_dict[emotion] = True + if len(emotions_dict) != 4: + print("! Error: emotions did not match reference") + print(user) + return {"eqbench": 0, "percent_parseable": 0} + + difference_tally = ( + 0 # Tally of differerence from reference answers for this question + ) + + # Iterate over each emotion in the user's answers. + for emotion, user_emotion_score in user.items(): + # If this emotion is in the reference, calculate the difference between the user's score and the reference score. + for i in range(1, 5): + if emotion == reference[f"emotion{i}"]: + d = abs( + float(user_emotion_score) - float(reference[f"emotion{i}_score"]) + ) + # this will be a value between 0 and 10 + if d == 0: + scaled_difference = 0 + elif d <= 5: + # S-shaped scaling function + # https://www.desmos.com/calculator + # 6.5\cdot\ \frac{1}{\left(1\ +\ e^{\left(-1.2\cdot\left(x-4\right)\right)}\right)} + scaled_difference = 6.5 * (1 / (1 + math.e ** (-1.2 * (d - 4)))) + + else: + scaled_difference = d + difference_tally += scaled_difference + + # Inverting the difference tally so that the closer the answer is to reference, the higher the score. + # The adjustment constant is chosen such that answering randomly produces a score of zero. + adjust_const = 0.7477 + final_score = 10 - (difference_tally * adjust_const) + final_score_percent = final_score * 10 + + return {"eqbench": final_score_percent, "percent_parseable": 100} diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/README.md b/lm-evaluation-harness/lm_eval/tasks/eus_exams/README.md new file mode 100644 index 0000000000000000000000000000000000000000..3a14f7dfcd16f8557be1a187599abae08e863940 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/README.md @@ -0,0 +1,49 @@ +# EusExams + +### Paper + +Title: Latxa: An Open Language Model and Evaluation Suite for Basque + +Abstract: https://arxiv.org/abs/2403.20266 + +EusExams is a collection of tests designed to prepare individuals for Public Service examinations conducted by several Basque institutions, including the public health system Osakidetza, the Basque Government, the City Councils of Bilbao and Gasteiz, and the University of the Basque Country (UPV/EHU). Within each of these groups, there are different exams for public positions, such as administrative and assistant roles. Each multiple-choice question contains 2 to 4 choices (3.90 on average) and one correct answer. The dataset is mostly parallel with 16k questions in Basque and 18k in Spanish. + +Homepage: https://github.com/hitz-zentroa/latxa + + +### Citation + +``` +@misc{etxaniz2024latxa, + title={Latxa: An Open Language Model and Evaluation Suite for Basque}, + author={Julen Etxaniz and Oscar Sainz and Naiara Perez and Itziar Aldabe and German Rigau and Eneko Agirre and Aitor Ormazabal and Mikel Artetxe and Aitor Soroa}, + year={2024}, + eprint={2403.20266}, + archivePrefix={arXiv}, + primaryClass={cs.CL} +} +``` + +### Groups and Tasks + +#### Tags + +* `eus_exams_eu`: The Basque version of the exams. +* `eus_exams_es`: The Spanish version of the exams. + +#### Tasks + +Basque and Spanish versions of the exams are available as separate tasks starting with `eus_exams_eu` and `eus_exams_es` respectively. + +### Checklist + +For adding novel benchmarks/datasets to the library: +* [ ] Is the task an existing benchmark in the literature? + * [ ] Have you referenced the original paper that introduced the task? + * [ ] If yes, does the original paper provide a reference implementation? If so, have you checked against the reference implementation and documented how to run such a test? + + +If other tasks on this dataset are already supported: +* [ ] Is the "Main" variant of this task clearly denoted? +* [ ] Have you provided a short sentence in a README on what each new variant adds / evaluates? +* [ ] Have you noted which, if any, published evaluation setups are matched by this variant? diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/configs.py b/lm-evaluation-harness/lm_eval/tasks/eus_exams/configs.py new file mode 100644 index 0000000000000000000000000000000000000000..993faa9f5dda1df2b00301fb00367f75e58a14de --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/configs.py @@ -0,0 +1,67 @@ +import argparse +import json + +import requests +import yaml + + +# get configs from huggingface datasets server by doing a request +response = requests.get( + "https://datasets-server.huggingface.co/splits?dataset=HiTZ%2FEusExams", timeout=5 +) +response_json = json.loads(response.text) +CONFIGS = [split["config"] for split in response_json["splits"]] + + +def gen_config_yamls(output_dir: str, overwrite: bool) -> None: + """ + Generate a yaml file for each configuage. + + :param output_dir: The directory to output the files to. + :param overwrite: Whether to overwrite files if they already exist. + """ + err = [] + for config in CONFIGS: + file_name = f"eus_exams_{config}.yaml" + try: + with open(f"{output_dir}/{file_name}", "w" if overwrite else "x") as f: + f.write("# Generated by utils.py\n") + yaml.dump( + { + "include": "eus_exams_es" + if "eus_exams_es" in config + else "eus_exams_eu", + "dataset_name": config, + "task": f"eus_exams_{config}", + }, + f, + ) + except FileExistsError: + err.append(file_name) + + if len(err) > 0: + raise FileExistsError( + "Files were not created because they already exist (use --overwrite flag):" + f" {', '.join(err)}" + ) + + +def main() -> None: + """Parse CLI args and generate configuage-specific yaml files.""" + parser = argparse.ArgumentParser() + parser.add_argument( + "--overwrite", + default=False, + action="store_true", + help="Overwrite files if they already exist", + ) + parser.add_argument( + "--output-dir", default=".", help="Directory to write yaml files to" + ) + args = parser.parse_args() + + gen_config_yamls(output_dir=args.output_dir, overwrite=args.overwrite) + + +if __name__ == "__main__": + main() diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams new file mode 100644 index 0000000000000000000000000000000000000000..d1d2af731485ac26b2792b5de29d4da681bf97ad --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams @@ -0,0 +1,18 @@ +dataset_path: HiTZ/EusExams +dataset_name: null +validation_split: null +test_split: test +fewshot_split: test +process_docs: !function utils.process_docs +output_type: multiple_choice +doc_to_choice: ["A", "B", "C", "D"] +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es new file mode 100644 index 0000000000000000000000000000000000000000..5f2325f22c892189d7965d3eacb0cf87a2d3d922 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es @@ -0,0 +1,4 @@ +include: eus_exams +tag: + - eus_exams_es +doc_to_text: "Pregunta: {{question}}\nA: {{candidates[0]}}\nB: {{candidates[1]}}\nC: {{candidates[2]}}\nD: {{candidates[3]}}\nRespuesta:" diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_ejadministrativo.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_ejadministrativo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..22b93ed6b7c2964e0d64c3a5e3aa299a81752bf2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_ejadministrativo.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: es_ejadministrativo +include: eus_exams_es +task: eus_exams_es_ejadministrativo diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_ejauxiliar.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_ejauxiliar.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b6cbe975fd9e1bbd927244ded96ee3f713273df1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_ejauxiliar.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: es_ejauxiliar +include: eus_exams_es +task: eus_exams_es_ejauxiliar diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_ejsubalterno.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_ejsubalterno.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0adfba26dd70eb54f0f50ac2ad2d97e83aaf57b8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_ejsubalterno.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: es_ejsubalterno +include: eus_exams_es +task: eus_exams_es_ejsubalterno diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_ejtecnico.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_ejtecnico.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0d0b011c9ab8ee58aa7cad819dd7475880512972 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_ejtecnico.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: es_ejtecnico +include: eus_exams_es +task: eus_exams_es_ejtecnico diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeayuntamientovitoria.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeayuntamientovitoria.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0d43c0b4161bf76988ace9416ac1bf1147a1f5c9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeayuntamientovitoria.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: es_opeayuntamientovitoria +include: eus_exams_es +task: eus_exams_es_opeayuntamientovitoria diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opebilbao.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opebilbao.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5cb33cbbddc9cd06ea24b7356fa19812bdf7a344 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opebilbao.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: es_opebilbao +include: eus_exams_es +task: eus_exams_es_opebilbao diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeehuadmin.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeehuadmin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e9cacfbd3f3e79dc24436daef4a1e9e5a1b5709d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeehuadmin.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: es_opeehuadmin +include: eus_exams_es +task: eus_exams_es_opeehuadmin diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeehuaux.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeehuaux.yaml new file mode 100644 index 0000000000000000000000000000000000000000..88316e512a4dabff3e550f84f3401216316991a7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeehuaux.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: es_opeehuaux +include: eus_exams_es +task: eus_exams_es_opeehuaux diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeehubiblio.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeehubiblio.yaml new file mode 100644 index 0000000000000000000000000000000000000000..728d7cfb0e7c79af156ff50bdb8379f032c3f01b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeehubiblio.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: es_opeehubiblio +include: eus_exams_es +task: eus_exams_es_opeehubiblio diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeehuderecho.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeehuderecho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..13e1d9de4bf434854e835b98455a3c26c46e96ac --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeehuderecho.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: es_opeehuderecho +include: eus_exams_es +task: eus_exams_es_opeehuderecho diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeehueconomicas.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeehueconomicas.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6625c9ce5a2501aa607cacf148531e0c3220652b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeehueconomicas.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: es_opeehueconomicas +include: eus_exams_es +task: eus_exams_es_opeehueconomicas diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeehuempresariales.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeehuempresariales.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4f61d3a1433f0f8ea0dd2ab62d9eb2c291697a47 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeehuempresariales.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: es_opeehuempresariales +include: eus_exams_es +task: eus_exams_es_opeehuempresariales diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeehusubalterno.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeehusubalterno.yaml new file mode 100644 index 0000000000000000000000000000000000000000..96cc86b402af1777e075530b3258e3a9089d539f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeehusubalterno.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: es_opeehusubalterno +include: eus_exams_es +task: eus_exams_es_opeehusubalterno diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeehutecnico.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeehutecnico.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0641fc2e7766d5f93ba1c45c83761f6e5b57560a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeehutecnico.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: es_opeehutecnico +include: eus_exams_es +task: eus_exams_es_opeehutecnico diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeehutecnicob.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeehutecnicob.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a338a1ab0d542368acf179b0611a354ddb71d293 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeehutecnicob.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: es_opeehutecnicob +include: eus_exams_es +task: eus_exams_es_opeehutecnicob diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeosakiadmin.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeosakiadmin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..85c771cdb3ead8511963b811043891958f19e340 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeosakiadmin.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: es_opeosakiadmin +include: eus_exams_es +task: eus_exams_es_opeosakiadmin diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeosakiaux.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeosakiaux.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2d61825b0beac1f50137ff42d74b6b649f30ea4e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeosakiaux.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: es_opeosakiaux +include: eus_exams_es +task: eus_exams_es_opeosakiaux diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeosakiauxenf.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeosakiauxenf.yaml new file mode 100644 index 0000000000000000000000000000000000000000..08fe0ed6c014ce69d7655c94ccd9dfdf029c8ce1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeosakiauxenf.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: es_opeosakiauxenf +include: eus_exams_es +task: eus_exams_es_opeosakiauxenf diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeosakicelador.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeosakicelador.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2a61b6878e9684ce0b35be5dc2fd25170cf9bf44 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeosakicelador.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: es_opeosakicelador +include: eus_exams_es +task: eus_exams_es_opeosakicelador diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeosakienf.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeosakienf.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e4749cac111d410d6b573a5365ec6778ec4645f2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeosakienf.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: es_opeosakienf +include: eus_exams_es +task: eus_exams_es_opeosakienf diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeosakijuridico.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeosakijuridico.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a62dc8001f16a22e56c7bea270ad3e6f97ecf5fd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeosakijuridico.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: es_opeosakijuridico +include: eus_exams_es +task: eus_exams_es_opeosakijuridico diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeosakioperario.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeosakioperario.yaml new file mode 100644 index 0000000000000000000000000000000000000000..df72481742ea637ec2fbe546445cd450bd1bc632 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeosakioperario.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: es_opeosakioperario +include: eus_exams_es +task: eus_exams_es_opeosakioperario diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeosakitecnico.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeosakitecnico.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4b5b397b88ed5318706a0e6f402acf34440761fe --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeosakitecnico.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: es_opeosakitecnico +include: eus_exams_es +task: eus_exams_es_opeosakitecnico diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeosakivarios.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeosakivarios.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fe98dc76aa4689ec46be58d05326adf6216264df --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_opeosakivarios.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: es_opeosakivarios +include: eus_exams_es +task: eus_exams_es_opeosakivarios diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_osakidetza1c.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_osakidetza1c.yaml new file mode 100644 index 0000000000000000000000000000000000000000..080f99fcf28f738117702d9ece800bdeed209b90 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_osakidetza1c.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: es_osakidetza1c +include: eus_exams_es +task: eus_exams_es_osakidetza1c diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_osakidetza2c.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_osakidetza2c.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4ee8ab46c6e4e1dd712b5d5865fb87abac5ac89b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_osakidetza2c.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: es_osakidetza2c +include: eus_exams_es +task: eus_exams_es_osakidetza2c diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_osakidetza3c.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_osakidetza3c.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2974a11d797474ed13257ac46e7994c31253f83d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_osakidetza3c.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: es_osakidetza3c +include: eus_exams_es +task: eus_exams_es_osakidetza3c diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_osakidetza4c.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_osakidetza4c.yaml new file mode 100644 index 0000000000000000000000000000000000000000..faa6c4b46c0de1d1f31fdc88f0acae96889eb080 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_osakidetza4c.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: es_osakidetza4c +include: eus_exams_es +task: eus_exams_es_osakidetza4c diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_osakidetza5c.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_osakidetza5c.yaml new file mode 100644 index 0000000000000000000000000000000000000000..153ce3add3ebe5f9d9da505da1e5c5affe0d1263 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_osakidetza5c.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: es_osakidetza5c +include: eus_exams_es +task: eus_exams_es_osakidetza5c diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_osakidetza6c.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_osakidetza6c.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d94ef2b9f4e494a4ba21d0fc4c902d3ad125616a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_osakidetza6c.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: es_osakidetza6c +include: eus_exams_es +task: eus_exams_es_osakidetza6c diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_osakidetza7c.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_osakidetza7c.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1fc30ce353f83ccc717a504a50a7bd611f76e6c6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_osakidetza7c.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: es_osakidetza7c +include: eus_exams_es +task: eus_exams_es_osakidetza7c diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_osakidetza8c.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_osakidetza8c.yaml new file mode 100644 index 0000000000000000000000000000000000000000..38f7ee3c39af34bc0516780a8717172950fc955a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_osakidetza8c.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: es_osakidetza8c +include: eus_exams_es +task: eus_exams_es_osakidetza8c diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_osakidetza9c.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_osakidetza9c.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7b23ff670764634cb78bd5a4cbb9f141dad674d2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_es_osakidetza9c.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: es_osakidetza9c +include: eus_exams_es +task: eus_exams_es_osakidetza9c diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu new file mode 100644 index 0000000000000000000000000000000000000000..dfa3df575dd0ee47bb9eea9f8449350b79221b02 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu @@ -0,0 +1,4 @@ +include: eus_exams +tag: + - eus_exams_eu +doc_to_text: "Galdera: {{question}}\nA: {{candidates[0]}}\nB: {{candidates[1]}}\nC: {{candidates[2]}}\nD: {{candidates[3]}}\nErantzuna:" diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_ejadministrari.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_ejadministrari.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f5630ddb05864cd3d6031ea8fed96e9715fb8990 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_ejadministrari.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: eu_ejadministrari +include: eus_exams_eu +task: eus_exams_eu_ejadministrari diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_ejlaguntza.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_ejlaguntza.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cf2806c1e6675c491ad5d1eaea54698bf8aa8fe8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_ejlaguntza.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: eu_ejlaguntza +include: eus_exams_eu +task: eus_exams_eu_ejlaguntza diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_ejlaguntzaile.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_ejlaguntzaile.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1d713a32442fc35d16b735c8617b0ee2d7327f04 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_ejlaguntzaile.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: eu_ejlaguntzaile +include: eus_exams_eu +task: eus_exams_eu_ejlaguntzaile diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_ejteknikari.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_ejteknikari.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7b528b9d4ce7ebc2ffc92af84a25e417f2e86929 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_ejteknikari.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: eu_ejteknikari +include: eus_exams_eu +task: eus_exams_eu_ejteknikari diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opebilbaoeu.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opebilbaoeu.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d15dbc6101ade859261bed36564eaf51e8a53f16 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opebilbaoeu.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: eu_opebilbaoeu +include: eus_exams_eu +task: eus_exams_eu_opebilbaoeu diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opeehuadmineu.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opeehuadmineu.yaml new file mode 100644 index 0000000000000000000000000000000000000000..85b9c9047759b6652435abc84944770ff429daaa --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opeehuadmineu.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: eu_opeehuadmineu +include: eus_exams_eu +task: eus_exams_eu_opeehuadmineu diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opeehuauxeu.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opeehuauxeu.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e72082486395abefcebda07de380b670d588589a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opeehuauxeu.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: eu_opeehuauxeu +include: eus_exams_eu +task: eus_exams_eu_opeehuauxeu diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opeehubiblioeu.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opeehubiblioeu.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0ff2ab853fc839c5ae2b88520767b8b3d4a60f4d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opeehubiblioeu.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: eu_opeehubiblioeu +include: eus_exams_eu +task: eus_exams_eu_opeehubiblioeu diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opeehuderechoeu.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opeehuderechoeu.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bef6b6524507dbae9ebf9b07bbe7d41fca978996 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opeehuderechoeu.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: eu_opeehuderechoeu +include: eus_exams_eu +task: eus_exams_eu_opeehuderechoeu diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opeehueconomicaseu.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opeehueconomicaseu.yaml new file mode 100644 index 0000000000000000000000000000000000000000..713f33234153c84778faeca25f4807cdf9812b45 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opeehueconomicaseu.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: eu_opeehueconomicaseu +include: eus_exams_eu +task: eus_exams_eu_opeehueconomicaseu diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opeehuempresarialeseu.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opeehuempresarialeseu.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8dddd9bc76cec647dafbc7400ae11d3e5147de83 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opeehuempresarialeseu.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: eu_opeehuempresarialeseu +include: eus_exams_eu +task: eus_exams_eu_opeehuempresarialeseu diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opeehusubalternoeu.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opeehusubalternoeu.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b02a451dd957a11c5db2810fa98ab0bccd62c9b3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opeehusubalternoeu.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: eu_opeehusubalternoeu +include: eus_exams_eu +task: eus_exams_eu_opeehusubalternoeu diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opeehutecnicoeu.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opeehutecnicoeu.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3792e12aa0285a3cf3ce56b9d6158ade836c4c38 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opeehutecnicoeu.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: eu_opeehutecnicoeu +include: eus_exams_eu +task: eus_exams_eu_opeehutecnicoeu diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opeehuteknikarib.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opeehuteknikarib.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b9f5cc612ac9776a328670d4273e76934172fd81 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opeehuteknikarib.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: eu_opeehuteknikarib +include: eus_exams_eu +task: eus_exams_eu_opeehuteknikarib diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opegasteizkoudala.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opegasteizkoudala.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e9211f39a162360b67e84399409b1617bc5cc1dd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opegasteizkoudala.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: eu_opegasteizkoudala +include: eus_exams_eu +task: eus_exams_eu_opegasteizkoudala diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opeosakiadmineu.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opeosakiadmineu.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cf19e09941bd0c7bb10db7f5398fb2398f1a0fd2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opeosakiadmineu.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: eu_opeosakiadmineu +include: eus_exams_eu +task: eus_exams_eu_opeosakiadmineu diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opeosakiauxenfeu.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opeosakiauxenfeu.yaml new file mode 100644 index 0000000000000000000000000000000000000000..719039915aec71a138a860397709b85549718078 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opeosakiauxenfeu.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: eu_opeosakiauxenfeu +include: eus_exams_eu +task: eus_exams_eu_opeosakiauxenfeu diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opeosakiauxeu.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opeosakiauxeu.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9d0891886cd550219fb9bfcc7209f6d5fb85ad5d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opeosakiauxeu.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: eu_opeosakiauxeu +include: eus_exams_eu +task: eus_exams_eu_opeosakiauxeu diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opeosakiceladoreu.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opeosakiceladoreu.yaml new file mode 100644 index 0000000000000000000000000000000000000000..af82c87bdffc84c8da3f666d944740eb0db0712d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opeosakiceladoreu.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: eu_opeosakiceladoreu +include: eus_exams_eu +task: eus_exams_eu_opeosakiceladoreu diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opeosakienfeu.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opeosakienfeu.yaml new file mode 100644 index 0000000000000000000000000000000000000000..10b853e399f017255c74b7eb56275df149a7f055 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opeosakienfeu.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: eu_opeosakienfeu +include: eus_exams_eu +task: eus_exams_eu_opeosakienfeu diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opeosakioperarioeu.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opeosakioperarioeu.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8f3cf7c490106959e4b07bef2140f0197835d16d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opeosakioperarioeu.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: eu_opeosakioperarioeu +include: eus_exams_eu +task: eus_exams_eu_opeosakioperarioeu diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opeosakitecnicoeu.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opeosakitecnicoeu.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f44e4994e3820dd6263835448b566a8c2ed17a13 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opeosakitecnicoeu.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: eu_opeosakitecnicoeu +include: eus_exams_eu +task: eus_exams_eu_opeosakitecnicoeu diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opeosakivarioseu.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opeosakivarioseu.yaml new file mode 100644 index 0000000000000000000000000000000000000000..11801ddec6614d95862087fa85c3f7b6314d8ddc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_opeosakivarioseu.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: eu_opeosakivarioseu +include: eus_exams_eu +task: eus_exams_eu_opeosakivarioseu diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_osakidetza1e.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_osakidetza1e.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cc713507196ef8f9460a61e110ede95186f846b1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_osakidetza1e.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: eu_osakidetza1e +include: eus_exams_eu +task: eus_exams_eu_osakidetza1e diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_osakidetza2e.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_osakidetza2e.yaml new file mode 100644 index 0000000000000000000000000000000000000000..218dc87cb8affc37cc54e03d56bcf44213381e99 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_osakidetza2e.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: eu_osakidetza2e +include: eus_exams_eu +task: eus_exams_eu_osakidetza2e diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_osakidetza3e.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_osakidetza3e.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d5d17c32a86b89ddaf3dc1da834fb053b67b9b64 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_osakidetza3e.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: eu_osakidetza3e +include: eus_exams_eu +task: eus_exams_eu_osakidetza3e diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_osakidetza5e.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_osakidetza5e.yaml new file mode 100644 index 0000000000000000000000000000000000000000..be4d2ca741e5168f99ed2105d584cf1fa21b4b81 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_osakidetza5e.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: eu_osakidetza5e +include: eus_exams_eu +task: eus_exams_eu_osakidetza5e diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_osakidetza6e.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_osakidetza6e.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7b2af263fbe039a6ab9e3131f868ab506f0e9b35 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_osakidetza6e.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: eu_osakidetza6e +include: eus_exams_eu +task: eus_exams_eu_osakidetza6e diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_osakidetza7e.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_osakidetza7e.yaml new file mode 100644 index 0000000000000000000000000000000000000000..666e96a0e136045c884f81fcf62d007f41ea80b7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/eus_exams_eu_osakidetza7e.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: eu_osakidetza7e +include: eus_exams_eu +task: eus_exams_eu_osakidetza7e diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_exams/utils.py b/lm-evaluation-harness/lm_eval/tasks/eus_exams/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..51e9f4c6322a635cdaeb54d3d557a3797b6dc5f0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_exams/utils.py @@ -0,0 +1,15 @@ +import datasets + + +def process_docs(dataset: datasets.Dataset): + """Filter out examples with no answer.""" + + def valid_example(example: dict) -> bool: + """Check if an example is valid.""" + if example["answer"] not in [0, 1, 2, 3]: + return False + if example["candidates"] == ["", "", "", ""]: + return False + return True + + return dataset.filter(valid_example) diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_proficiency/README.md b/lm-evaluation-harness/lm_eval/tasks/eus_proficiency/README.md new file mode 100644 index 0000000000000000000000000000000000000000..6671bda477e4533204c8ba154323e40d3df23f79 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_proficiency/README.md @@ -0,0 +1,48 @@ +# EusProficiency + +### Paper + +Title: Latxa: An Open Language Model and Evaluation Suite for Basque + +Abstract: https://arxiv.org/abs/2403.20266 + +EusProficiency comprises 5,169 exercises on different topics from past EGA exams, the official C1-level certificate of proficiency in Basque. We collected the atarikoa exercises from EGA exams through the years 1998 to 2008. Atarikoa is the first qualifying test of EGA, which measures different aspects of language competency, such as reading comprehension, grammar, vocabulary, spelling, and writing. Each test generally has 85 multiple-choice questions, with 4 choices and a single correct answer. + +Homepage: https://github.com/hitz-zentroa/latxa + + +### Citation + +``` +@misc{etxaniz2024latxa, + title={Latxa: An Open Language Model and Evaluation Suite for Basque}, + author={Julen Etxaniz and Oscar Sainz and Naiara Perez and Itziar Aldabe and German Rigau and Eneko Agirre and Aitor Ormazabal and Mikel Artetxe and Aitor Soroa}, + year={2024}, + eprint={2403.20266}, + archivePrefix={arXiv}, + primaryClass={cs.CL} +} +``` + +### Groups and Tasks + +#### Groups + +There are no groups. + +#### Tasks + +* `eus_proficiency`: EusProficiency comprises 5,169 exercises on different topics from past EGA exams, the official C1-level certificate of proficiency in Basque. + +### Checklist + +For adding novel benchmarks/datasets to the library: +* [ ] Is the task an existing benchmark in the literature? + * [ ] Have you referenced the original paper that introduced the task? + * [ ] If yes, does the original paper provide a reference implementation? If so, have you checked against the reference implementation and documented how to run such a test? + + +If other tasks on this dataset are already supported: +* [ ] Is the "Main" variant of this task clearly denoted? +* [ ] Have you provided a short sentence in a README on what each new variant adds / evaluates? +* [ ] Have you noted which, if any, published evaluation setups are matched by this variant? diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_proficiency/eus_proficiency.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_proficiency/eus_proficiency.yaml new file mode 100644 index 0000000000000000000000000000000000000000..18cf5d2ab313a2ac907738185b5e39036402c7e2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_proficiency/eus_proficiency.yaml @@ -0,0 +1,16 @@ +dataset_path: HiTZ/EusProficiency +dataset_name: default +task: eus_proficiency +doc_to_text: "Galdera: {{question}}\nA: {{candidates[0]}}\nB: {{candidates[1]}}\nC: {{candidates[2]}}\nD: {{candidates[3]}}\nErantzuna:" +doc_to_choice: ["A", "B", "C", "D"] +validation_split: null +test_split: test +fewshot_split: test +output_type: multiple_choice +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_reading/README.md b/lm-evaluation-harness/lm_eval/tasks/eus_reading/README.md new file mode 100644 index 0000000000000000000000000000000000000000..2542e9b509a38f8d6fc6bcbd6319fd0d0462f078 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_reading/README.md @@ -0,0 +1,48 @@ +# EusReading + +### Paper + +Title: Latxa: An Open Language Model and Evaluation Suite for Basque + +Abstract: https://arxiv.org/abs/2403.20266 + +EusReading consists of 352 reading comprehension exercises (irakurmena) sourced from the set of past EGA exams from 1998 to 2008. Each test generally has 10 multiple-choice questions, with 4 choices and a single correct answer. These exercises are more challenging than Belebele due to the complexity and length of the input texts. As a result, EusReading is useful to measure long context understanding of models. + +Homepage: https://github.com/hitz-zentroa/latxa + + +### Citation + +``` +@misc{etxaniz2024latxa, + title={Latxa: An Open Language Model and Evaluation Suite for Basque}, + author={Julen Etxaniz and Oscar Sainz and Naiara Perez and Itziar Aldabe and German Rigau and Eneko Agirre and Aitor Ormazabal and Mikel Artetxe and Aitor Soroa}, + year={2024}, + eprint={2403.20266}, + archivePrefix={arXiv}, + primaryClass={cs.CL} +} +``` + +### Groups and Tasks + +#### Groups + +There are no groups. + +#### Tasks + +* `eus_reading`: EusReading consists of 352 reading comprehension exercises (irakurmena) sourced from the set of past EGA exams from 1998 to 2008. + +### Checklist + +For adding novel benchmarks/datasets to the library: +* [ ] Is the task an existing benchmark in the literature? + * [ ] Have you referenced the original paper that introduced the task? + * [ ] If yes, does the original paper provide a reference implementation? If so, have you checked against the reference implementation and documented how to run such a test? + + +If other tasks on this dataset are already supported: +* [ ] Is the "Main" variant of this task clearly denoted? +* [ ] Have you provided a short sentence in a README on what each new variant adds / evaluates? +* [ ] Have you noted which, if any, published evaluation setups are matched by this variant? diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_reading/eus_reading.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_reading/eus_reading.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1af52d499daad3189639eb8a45bccfa77c69d710 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_reading/eus_reading.yaml @@ -0,0 +1,16 @@ +dataset_path: HiTZ/EusReading +dataset_name: default +task: eus_reading +doc_to_text: !function utils.doc_to_text_context +doc_to_choice: !function utils.doc_to_choice +validation_split: null +test_split: test +fewshot_split: test +output_type: multiple_choice +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_reading/utils.py b/lm-evaluation-harness/lm_eval/tasks/eus_reading/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..dc507e02ffc52c9764f2d49c608d8b45903a40ae --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_reading/utils.py @@ -0,0 +1,41 @@ +from typing import List + + +letters = ["A", "B", "C", "D"] + + +def doc_to_text_context(doc) -> str: + """ + Converts a document to a formatted string. + + Args: + doc (dict): A dictionary containing the document information. + + Returns: + str: A formatted string containing the question and answer choices. + """ + candidates = doc["candidates"] + num_choices = len(candidates) + if num_choices < 2: + raise ValueError("Invalid number of candidates") + choices = letters[:num_choices] + formatted_choices = "\n".join( + [f"{choice}: {candidates[i]}" for i, choice in enumerate(choices)] + ) + return f"Pasartea: {doc['context']}\n\nGaldera: {doc['question']}\n{formatted_choices}\nErantzuna:" + + +def doc_to_choice(doc) -> List[str]: + """ + Returns the answer choices for a document. + + Args: + doc (dict): A dictionary containing the document information. + + Returns: + list: A list of strings containing the answer choices. + """ + num_choices = len(doc["candidates"]) + if num_choices < 2: + raise ValueError("Invalid number of candidates") + return letters[:num_choices] diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_trivia/README.md b/lm-evaluation-harness/lm_eval/tasks/eus_trivia/README.md new file mode 100644 index 0000000000000000000000000000000000000000..88e760e43592d93ba27ee3b19c4edd0fc6f3e9f6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_trivia/README.md @@ -0,0 +1,54 @@ +# EusTrivia + +### Paper + +Title: Latxa: An Open Language Model and Evaluation Suite for Basque + +Abstract: https://arxiv.org/abs/2403.20266 + +EusTrivia consists of 1,715 trivia questions from multiple online sources. 56.3\% of the questions are elementary level (grades 3-6), while the rest are considered challenging. A significant portion of the questions focus specifically on the Basque Country, its language and culture. Each multiple-choice question contains two, three or four choices (3.84 on average) and a single correct answer. Five areas of knowledge are covered: + +- **Humanities and Natural Sciences** (27.8%): This category encompasses questions about history, geography, biology, ecology and other social and natural sciences. +- **Leisure and Art** (24.5%): This category includes questions on sports and athletes, performative and plastic arts and artists, architecture, cultural events, and related topics. +- **Music** (16.0%): Here are grouped all the questions about music and musicians, both classical and contemporary. +- **Language and Literature** (17.1%): This category is concerned with all kinds of literature productions and writers, as well as metalinguistic questions (e.g., definitions, synonyms, and word usage). +- **Mathematics and ICT** (14.5%): This category covers mathematical problems and questions about ICT, as well as questions about people known for their contributions to these fields of knowledge. + +Homepage: https://github.com/hitz-zentroa/latxa + + +### Citation + +``` +@misc{etxaniz2024latxa, + title={Latxa: An Open Language Model and Evaluation Suite for Basque}, + author={Julen Etxaniz and Oscar Sainz and Naiara Perez and Itziar Aldabe and German Rigau and Eneko Agirre and Aitor Ormazabal and Mikel Artetxe and Aitor Soroa}, + year={2024}, + eprint={2403.20266}, + archivePrefix={arXiv}, + primaryClass={cs.CL} +} +``` + +### Groups and Tasks + +#### Groups + +There are no groups. + +#### Tasks + +* `eus_trivia`: EusTrivia consists of 1,715 trivia questions from multiple online sources. + +### Checklist + +For adding novel benchmarks/datasets to the library: +* [ ] Is the task an existing benchmark in the literature? + * [ ] Have you referenced the original paper that introduced the task? + * [ ] If yes, does the original paper provide a reference implementation? If so, have you checked against the reference implementation and documented how to run such a test? + + +If other tasks on this dataset are already supported: +* [ ] Is the "Main" variant of this task clearly denoted? +* [ ] Have you provided a short sentence in a README on what each new variant adds / evaluates? +* [ ] Have you noted which, if any, published evaluation setups are matched by this variant? diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_trivia/eus_trivia.yaml b/lm-evaluation-harness/lm_eval/tasks/eus_trivia/eus_trivia.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fe93ab61725867ae39d9be17ae33f9b769046683 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_trivia/eus_trivia.yaml @@ -0,0 +1,16 @@ +dataset_path: HiTZ/EusTrivia +dataset_name: default +task: eus_trivia +doc_to_text: !function utils.doc_to_text +doc_to_choice: !function utils.doc_to_choice +validation_split: null +test_split: test +fewshot_split: test +output_type: multiple_choice +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/eus_trivia/utils.py b/lm-evaluation-harness/lm_eval/tasks/eus_trivia/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..e5802c795bf558eacb60a05db6c344e925f6e4fa --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/eus_trivia/utils.py @@ -0,0 +1,41 @@ +from typing import List + + +letters = ["A", "B", "C", "D"] + + +def doc_to_text(doc) -> str: + """ + Converts a document to a formatted string. + + Args: + doc (dict): A dictionary containing the document information. + + Returns: + str: A formatted string containing the question and answer choices. + """ + candidates = doc["candidates"] + num_choices = len(candidates) + if num_choices < 2: + raise ValueError("Invalid number of candidates") + choices = letters[:num_choices] + formatted_choices = "\n".join( + [f"{choice}: {candidates[i]}" for i, choice in enumerate(choices)] + ) + return f"Galdera: {doc['question']}\n{formatted_choices}\nErantzuna:" + + +def doc_to_choice(doc) -> List[str]: + """ + Returns the answer choices for a document. + + Args: + doc (dict): A dictionary containing the document information. + + Returns: + list: A list of strings containing the answer choices. + """ + num_choices = len(doc["candidates"]) + if num_choices < 2: + raise ValueError("Invalid number of candidates") + return letters[:num_choices] diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/README.md b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/README.md new file mode 100644 index 0000000000000000000000000000000000000000..a2d148892df00552b684b348c7812f532e881b5f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/README.md @@ -0,0 +1,62 @@ +# Evalita-LLM + +### Paper + +Evalita-LLM, a new benchmark designed to evaluate Large Language +Models (LLMs) on Italian tasks. The distinguishing and innovative features of +Evalita-LLM are the following: (i) all tasks are native Italian, avoiding issues of +translating from Italian and potential cultural biases; (ii) in addition to well established multiple-choice tasks, the benchmark includes generative tasks, enabling more natural interaction with LLMs; (iii) all tasks are evaluated against multiple prompts, this way mitigating the model sensitivity to specific prompts and allowing a fairer and objective evaluation. + +### Citation + +```bibtex +@misc{magnini2025evalitallmbenchmarkinglargelanguage, + title={Evalita-LLM: Benchmarking Large Language Models on Italian}, + author={Bernardo Magnini and Roberto Zanoli and Michele Resta and Martin Cimmino and Paolo Albano and Marco Madeddu and Viviana Patti}, + year={2025}, + eprint={2502.02289}, + archivePrefix={arXiv}, + primaryClass={cs.CL}, + url={https://arxiv.org/abs/2502.02289}, +} +``` + +### Groups + +- `evalita-mp`: All tasks (perplexity and non-perplexity based). +- `evalita-mp_gen`: Only generative tasks. +- `evalita-mp_mc`: Only perplexity-based tasks. + +#### Tasks + +The following Evalita-LLM tasks can also be evaluated in isolation: + - `evalita-mp_te`: Textual Entailment + - `evalita-mp_sa`: Sentiment Analysis + - `evalita-mp_wic`: Word in Context + - `evalita-mp_hs`: Hate Speech Detection + - `evalita-mp_at`: Admission Tests + - `evalita-mp_faq`: FAQ + - `evalita-mp_sum_fp`: Summarization + - `evalita-mp_ls`: Lexical Substitution + - `evalita-mp_ner_group`: Named Entity Recognition + - `evalita-mp_re`: Relation Extraction + + +### Usage + +```bash + +lm_eval --model hf --model_args pretrained=meta-llama/Llama-2-7b-hf --tasks evalita-mp --device cuda:0 --batch_size auto +``` + +### Checklist + +* [x] Is the task an existing benchmark in the literature? + * [x] Have you referenced the original paper that introduced the task? + * [x] If yes, does the original paper provide a reference implementation? + * [x] Yes, original implementation contributed by author of the benchmark + +If other tasks on this dataset are already supported: +* [x] Is the "Main" variant of this task clearly denoted? +* [x] Have you provided a short sentence in a README on what each new variant adds / evaluates? +* [x] Have you noted which, if any, published evaluation setups are matched by this variant? diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_at_template_yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_at_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..8c8d7a9c6316935646962153a7d763da2e1faae1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_at_template_yaml @@ -0,0 +1,9 @@ +dataset_path: evalitahf/admission_test +output_type: multiple_choice +test_split: test +fewshot_split: dev +validation_split: test +doc_to_target: Correct +doc_to_choice: ["A", "B", "C", "D", "E"] +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c212e77fbcd7330fe9ba7c3af92484a7313b9f01 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp.yaml @@ -0,0 +1,18 @@ +group: evalita-mp +group_alias: Evalita-LLM +task: + - evalita-mp_te + - evalita-mp_sa + - evalita-mp_wic + - evalita-mp_hs + - evalita-mp_at + - evalita-mp_faq + - evalita-mp_sum_fp + - evalita-mp_ls + - evalita-mp_ner_group + - evalita-mp_re +aggregate_metric_list: + - metric: acc + weight_by_size: True +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_at_task_p1.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_at_task_p1.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1655213e0e980141b2512d138ef4ea3f1e97d09c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_at_task_p1.yaml @@ -0,0 +1,18 @@ +tag: evalita-mp_at_tasks +include: _at_template_yaml +task: evalita-mp_at_prompt-1 +task_alias: prompt-1 +#doc_to_text: "Rispondi alla domanda a scelta multipla considerando le informazioni del testo seguente.\nTesto: {{background}}\nDomanda: {{domanda}}\nOpzioni: A: {{A}} B: {{B}} C: {{C}} D: {{D}}" +#doc_to_text: "Dato il seguente caso clinico: '{{background}}' qual è la risposta corretta alla domanda: '{{domanda}}'?" +doc_to_text: "Dato il seguente quesito di medicina: '{{Question}}' qual è la risposta corretta?" +doc_to_choice: "{{[A,B,C,D,E]}}" +doc_to_target: "{{ A if Correct == 'A' else B if Correct == 'B' else C if Correct == 'C' else D if Correct == 'D' else E}}" +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_at_task_p2.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_at_task_p2.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1b0c16fc1573c73092afe82c1f6892796902edbc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_at_task_p2.yaml @@ -0,0 +1,18 @@ +tag: evalita-mp_at_tasks +include: _at_template_yaml +task: evalita-mp_at_prompt-2 +task_alias: prompt-2 +#doc_to_text: "Rispondi alla domanda a scelta multipla considerando le informazioni del testo seguente.\nTesto: {{background}}\nDomanda: {{domanda}}\nOpzioni: A: {{A}} B: {{B}} C: {{C}} D: {{D}}" +#doc_to_text: "Devi risolvere un compito di risposte a domande. Dato il seguente caso clinico: '{{background}}' qual è la risposta corretta alla domanda: '{{domanda}}'?" +doc_to_text: "Devi risolvere un compito di risposte a domande. Dato il seguente quesito di medicina: '{{Question}}' qual è la risposta corretta?" +doc_to_choice: "{{[A,B,C,D,E]}}" +doc_to_target: "{{ A if Correct == 'A' else B if Correct == 'B' else C if Correct == 'C' else D if Correct == 'D' else E}}" +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_at_task_p3.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_at_task_p3.yaml new file mode 100644 index 0000000000000000000000000000000000000000..767a75a0346ebc1272d9618694a3c3adbc042fbe --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_at_task_p3.yaml @@ -0,0 +1,16 @@ +tag: evalita-mp_at_tasks +include: _at_template_yaml +task: evalita-mp_at_prompt-3 +task_alias: prompt-3 +#doc_to_text: "Rispondi alla domanda a scelta multipla considerando le informazioni del testo seguente.\nTesto: {{background}}\nDomanda: {{domanda}}\nOpzioni: A: {{A}} B: {{B}} C: {{C}} D: {{D}}" +#doc_to_text: "Dato il seguente caso clinico: '{{background}}', qual è la risposta corretta alla domanda: '{{domanda}}'?\nA: {{A}}\nB: {{B}}\nC: {{C}}\nD: {{D}}\nE: {{E}}\nRisposta:" +doc_to_text: "Dato il seguente quesito di medicina: '{{Question}}' qual è la risposta corretta?\nA: {{A}}\nB: {{B}}\nC: {{C}}\nD: {{D}}\nE: {{E}}\nRisposta:" +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_at_task_p4.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_at_task_p4.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a7b29f81acee052e4f6c5ca2221d0a00148a928b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_at_task_p4.yaml @@ -0,0 +1,15 @@ +tag: evalita-mp_at_tasks +include: _at_template_yaml +task: evalita-mp_at_prompt-4 +task_alias: prompt-4 +#doc_to_text: "Rispondi alla domanda a scelta multipla considerando le informazioni del testo seguente.\nTesto: {{background}}\nDomanda: {{domanda}}\nOpzioni: A: {{A}} B: {{B}} C: {{C}} D: {{D}}" +doc_to_text: "Devi risolvere un compito a scelta multipla. Dato il seguente quesito di medicina: '{{Question}}' qual è la risposta corretta?\nA: {{A}}\nB: {{B}}\nC: {{C}}\nD: {{D}}\nE: {{E}}\nRisposta:" +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_at_task_p5.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_at_task_p5.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fead4471eefa2fdd57eaf857ffa99401d6ab294e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_at_task_p5.yaml @@ -0,0 +1,18 @@ +tag: evalita-mp_at_tasks +include: _at_template_yaml +task: evalita-mp_at_prompt-5 +task_alias: prompt-5 +#doc_to_text: "Rispondi alla domanda a scelta multipla considerando le informazioni del testo seguente.\nTesto: {{background}}\nDomanda: {{domanda}}\nOpzioni: A: {{A}} B: {{B}} C: {{C}} D: {{D}}" +#doc_to_text: "Dato il seguente caso clinico: '{{background}}'. La risposta corretta alla domanda: '{{domanda}}' è:" +doc_to_text: "Dato il seguente quesito di medicina '{{Question}}' la risposta corretta è:" +doc_to_choice: "{{[A,B,C,D,E]}}" +doc_to_target: "{{ A if Correct == 'A' else B if Correct == 'B' else C if Correct == 'C' else D if Correct == 'D' else E}}" +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_at_task_p6.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_at_task_p6.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9fee15e0cd2804f01a8d6d393fce893d2a95e503 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_at_task_p6.yaml @@ -0,0 +1,18 @@ +tag: evalita-mp_at_tasks +include: _at_template_yaml +task: evalita-mp_at_prompt-6 +task_alias: prompt-6 +#doc_to_text: "Rispondi alla domanda a scelta multipla considerando le informazioni del testo seguente.\nTesto: {{background}}\nDomanda: {{domanda}}\nOpzioni: A: {{A}} B: {{B}} C: {{C}} D: {{D}}" +#doc_to_text: "Devi risolvere un compito di risposte a domande. Dato il seguente caso clinico: '{{background}}'. La risposta corretta alla domanda: '{{domanda}}' è:" +doc_to_text: "Devi risolvere un compito di risposte a domande. Dato il seguente quesito di medicina '{{Question}}' la risposta corretta è:" +doc_to_choice: "{{[A,B,C,D,E]}}" +doc_to_target: "{{ A if Correct == 'A' else B if Correct == 'B' else C if Correct == 'C' else D if Correct == 'D' else E}}" +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_at_tasks.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_at_tasks.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b04e6420a79379bb9d1458d463974c2333712fcb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_at_tasks.yaml @@ -0,0 +1,10 @@ +group: evalita-mp_at +group_alias: admission-test +task: + - evalita-mp_at_tasks # Each of the tasks has to have a matching tag in its own yaml file +aggregate_metric_list: + - metric: acc + weight_by_size: True + aggregation: mean +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_faq_p1.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_faq_p1.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fb7aaa31806c09973b326bd51bf82e2a916e6ed7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_faq_p1.yaml @@ -0,0 +1,14 @@ +tag: evalita-mp_faq_tasks +include: _faq_template_yaml +task: evalita-mp_faq_prompt-1 +task_alias: prompt-1 +#doc_to_text: "Data la seguente domanda {{question}}, individua la risposta corretta tra le seguenti opzioni:\n A: {{A}}\nB: {{B}}\nC: {{C}}\nD: {{D}} Risposta:" +doc_to_text: "Rispondi alla seguente domanda: '{{question}}'" +doc_to_choice: "{{[A,B,C,D]}}" +doc_to_target: "{{ A if correct_answer == 'A' else B if correct_answer == 'B' else C if correct_answer == 'C' else D}}" +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_faq_p2.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_faq_p2.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7c873df8e3a52a7b50199fd8887415354b618064 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_faq_p2.yaml @@ -0,0 +1,14 @@ +tag: evalita-mp_faq_tasks +include: _faq_template_yaml +task: evalita-mp_faq_prompt-2 +task_alias: prompt-2 +#doc_to_text: "Data la seguente domanda {{question}}, individua la risposta corretta tra le seguenti opzioni:\n A: {{A}}\nB: {{B}}\nC: {{C}}\nD: {{D}} Risposta:" +doc_to_text: "Devi risolvere un compito di risposte a domande. Rispondi alla seguente domanda: '{{question}}'" +doc_to_choice: "{{[A,B,C,D]}}" +doc_to_target: "{{ A if correct_answer == 'A' else B if correct_answer == 'B' else C if correct_answer == 'C' else D}}" +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_faq_p3.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_faq_p3.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bac97f213d39dd542720ed2790a95dc5d0f5edca --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_faq_p3.yaml @@ -0,0 +1,12 @@ +tag: evalita-mp_faq_tasks +include: _faq_template_yaml +task: evalita-mp_faq_prompt-3 +task_alias: prompt-3 +#doc_to_text: "Data la seguente domanda {{question}}, individua la risposta corretta tra le seguenti opzioni:\n A: {{A}}\nB: {{B}}\nC: {{C}}\nD: {{D}} Risposta:" +doc_to_text: "Rispondi alla seguente domanda: '{{question}}'\nA: {{A}}\nB: {{B}}\nC: {{C}}\nD: {{D}}\nRisposta:" +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_faq_p4.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_faq_p4.yaml new file mode 100644 index 0000000000000000000000000000000000000000..43e45d19fb1ceae29b59ae5c14839b5257310d53 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_faq_p4.yaml @@ -0,0 +1,12 @@ +tag: evalita-mp_faq_tasks +include: _faq_template_yaml +task: evalita-mp_faq_prompt-4 +task_alias: prompt-4 +#doc_to_text: "Data la seguente domanda {{question}}, individua la risposta corretta tra le seguenti opzioni:\n A: {{A}}\nB: {{B}}\nC: {{C}}\nD: {{D}} Risposta:" +doc_to_text: "Devi risolvere un compito a scelta multipla. Rispondi alla seguente domanda: '{{question}}'\nA: {{A}}\nB: {{B}}\nC: {{C}}\nD: {{D}}\nRisposta:" +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_faq_p5.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_faq_p5.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d01cb3c8421d5d7d4305ee3e486ef3d2f6716697 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_faq_p5.yaml @@ -0,0 +1,15 @@ +tag: evalita-mp_faq_tasks +include: _faq_template_yaml +task: evalita-mp_faq_prompt-5 +task_alias: prompt-5 +#doc_to_text: "Data la seguente domanda {{question}}, individua la risposta corretta tra le seguenti opzioni:\n A: {{A}}\nB: {{B}}\nC: {{C}}\nD: {{D}} Risposta:" +#doc_to_text: "La risposta alla domanda: '{{question}}' è:" +doc_to_text: "La risposta alla domanda: '{{question}}' è:" +doc_to_choice: "{{[A,B,C,D]}}" +doc_to_target: "{{ A if correct_answer == 'A' else B if correct_answer == 'B' else C if correct_answer == 'C' else D }}" +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_faq_p6.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_faq_p6.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e886e36bc2db7ad5fdeaf220266ee5e1b69211d2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_faq_p6.yaml @@ -0,0 +1,14 @@ +tag: evalita-mp_faq_tasks +include: _faq_template_yaml +task: evalita-mp_faq_prompt-6 +task_alias: prompt-6 +#doc_to_text: "Data la seguente domanda {{question}}, individua la risposta corretta tra le seguenti opzioni:\n A: {{A}}\nB: {{B}}\nC: {{C}}\nD: {{D}} Risposta:" +doc_to_text: "Devi risolvere un compito di risposte a domande. La risposta alla domanda: '{{question}}' è:" +doc_to_choice: "{{[A,B,C,D]}}" +doc_to_target: "{{ A if correct_answer == 'A' else B if correct_answer == 'B' else C if correct_answer == 'C' else D }}" +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_faq_tasks.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_faq_tasks.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ffc0089cb3ed9b5dd07917f040ee133c450ad7ec --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_faq_tasks.yaml @@ -0,0 +1,9 @@ +group: evalita-mp_faq +group_alias: faq +task: + - evalita-mp_faq_tasks # Each of the tasks has to have a matching tag in its own yaml file +aggregate_metric_list: + - metric: acc + weight_by_size: True +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_gen.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_gen.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7f93e18627be3e3b2716cb40ec17ce4986ca885f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_gen.yaml @@ -0,0 +1,12 @@ +group: evalita-mp_gen +group_alias: Evalita-LLM - Generative +task: + - evalita-mp_sum_fp + - evalita-mp_ls + - evalita-mp_ner_group + - evalita-mp_re +aggregate_metric_list: + - metric: acc + weight_by_size: True +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_hs_p1.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_hs_p1.yaml new file mode 100644 index 0000000000000000000000000000000000000000..837071e644e56c879753fc02ceea566b2ffc1dcc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_hs_p1.yaml @@ -0,0 +1,13 @@ +tag: evalita-mp_hs_tasks +include: _hs_template_yaml +task: evalita-mp_hs_prompt-1 +task_alias: prompt-1 +#doc_to_text: "Dato il seguente testo, rispondi Vero se il testo contiene hate speech, altrimenti rispondi Falso. Testo:\n{{full_text}} Risposta:" +doc_to_text: "C'è incitamento all'odio nel seguente tweet: '{{full_text}}'?" +metric_list: + - metric: f1 + higher_is_better: true + average: macro + aggregation: f1 +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_hs_p2.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_hs_p2.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0c04b216d320de47deaafb411262c834ca3a4f39 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_hs_p2.yaml @@ -0,0 +1,13 @@ +tag: evalita-mp_hs_tasks +include: _hs_template_yaml +task: evalita-mp_hs_prompt-2 +task_alias: prompt-2 +#doc_to_text: "Dato il seguente testo, rispondi Vero se il testo contiene hate speech, altrimenti rispondi Falso. Testo:\n{{full_text}} Risposta:" +doc_to_text: "Devi svolgere un compito di identificazione di incitamento all'odio. C'è incitamento all'odio nel seguente tweet: '{{full_text}}'?" +metric_list: + - metric: f1 + higher_is_better: true + average: macro + aggregation: f1 +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_hs_p3.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_hs_p3.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d3efe05ef29d3c9d7434d6990b539c4d67bb5249 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_hs_p3.yaml @@ -0,0 +1,14 @@ +tag: evalita-mp_hs_tasks +include: _hs_template_yaml +task: evalita-mp_hs_prompt-3 +task_alias: prompt-3 +doc_to_choice: ["B", "A"] +#doc_to_text: "Dato il seguente testo, rispondi Vero se il testo contiene hate speech, altrimenti rispondi Falso. Testo:\n{{full_text}} Risposta:" +doc_to_text: "C'è incitamento all'odio nel seguente tweet: '{{full_text}}'?\nA: Vero\nB: Falso\nRisposta:" +metric_list: + - metric: f1 + higher_is_better: true + average: macro + aggregation: f1 +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_hs_p4.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_hs_p4.yaml new file mode 100644 index 0000000000000000000000000000000000000000..639de50ef2a53ed7a4cfdc5a18c1c780950c1875 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_hs_p4.yaml @@ -0,0 +1,14 @@ +tag: evalita-mp_hs_tasks +include: _hs_template_yaml +task: evalita-mp_hs_prompt-4 +task_alias: prompt-4 +doc_to_choice: ["B", "A"] +#doc_to_text: "Dato il seguente testo, rispondi Vero se il testo contiene hate speech, altrimenti rispondi Falso. Testo:\n{{full_text}} Risposta:" +doc_to_text: "Devi svolgere un compito di identificazione di incitamento all'odio. C'è incitamento all'odio nel seguente tweet: '{{full_text}}'?\nA: Sì\nB: No\nRisposta:" +metric_list: + - metric: f1 + higher_is_better: true + average: macro + aggregation: f1 +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_hs_p5.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_hs_p5.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e99fb191f4bdfceda0f2164f9a3e56672cea1150 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_hs_p5.yaml @@ -0,0 +1,14 @@ +tag: evalita-mp_hs_tasks +include: _hs_template_yaml +task: evalita-mp_hs_prompt-5 +task_alias: prompt-5 +doc_to_choice: ["non contiene incitamento all'odio", "contiene incitamento all'odio"] +#doc_to_text: "Dato il seguente testo, rispondi Vero se il testo contiene hate speech, altrimenti rispondi Falso. Testo:\n{{full_text}} Risposta:" +doc_to_text: "Il tweet: '{{full_text}}'" +metric_list: + - metric: f1 + higher_is_better: true + average: macro + aggregation: f1 +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_hs_p6.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_hs_p6.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4f37e9bc242d21123e0c9e3171a0bd0d70a175aa --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_hs_p6.yaml @@ -0,0 +1,14 @@ +tag: evalita-mp_hs_tasks +include: _hs_template_yaml +task: evalita-mp_hs_prompt-6 +task_alias: prompt-6 +doc_to_choice: ["non contiene incitamento all'odio", "contiene incitamento all'odio"] +#doc_to_text: "Dato il seguente testo, rispondi Vero se il testo contiene hate speech, altrimenti rispondi Falso. Testo:\n{{full_text}} Risposta:" +doc_to_text: "Devi svolgere un compito di identificazione di incitamento all'odio. Il tweet: '{{full_text}}'" +metric_list: + - metric: f1 + higher_is_better: true + average: macro + aggregation: f1 +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_hs_task.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_hs_task.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5455c88055d4f4bbb209838b1d9c0895ccc25a34 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_hs_task.yaml @@ -0,0 +1,10 @@ +group: evalita-mp_hs +group_alias: hate-speech-detection +task: + - evalita-mp_hs_tasks +aggregate_metric_list: + - metric: f1 + weight_by_size: True + +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ls_p1.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ls_p1.yaml new file mode 100644 index 0000000000000000000000000000000000000000..92b6513ef2590982d62b112401c0ae547258d52b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ls_p1.yaml @@ -0,0 +1,6 @@ +tag: evalita-mp_ls_tasks +include: _ls_template_yaml +task: evalita-mp_ls_prompt-1 +task_alias: prompt-1 +#doc_to_text: "Sostituisci la parola tra i tag con sinonimi appropriati per il contesto. Separa i sinonimi con virgole. Testo:\n{{context}}" +doc_to_text: "Trova 10 parole che possono sostituire la parola racchiusa tra i marcatori nella seguente frase: '{{context}}', mantenendo lo stesso significato. Elenca i lemmi (forme base) di queste parole, separandoli con una virgola, ad esempio: lemma1, lemma2, lemma3, lemma4, lemma5. Non aggiungere commenti o altro testo. Risposta:" diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ls_p2.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ls_p2.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2aeaddf7d672c28632abecc1f82810e4533bb64d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ls_p2.yaml @@ -0,0 +1,6 @@ +tag: evalita-mp_ls_tasks +include: _ls_template_yaml +task: evalita-mp_ls_prompt-2 +task_alias: prompt-2 +#doc_to_text: "Sostituisci la parola tra i tag con sinonimi appropriati per il contesto. Separa i sinonimi con virgole. Testo:\n{{context}}" +doc_to_text: "Devi risolvere un compito di sostituzione lessicale. Trova 10 parole che possono sostituire la parola racchiusa tra i marcatori nella seguente frase: '{{context}}', mantenendo lo stesso significato. Elenca i lemmi (forme base) di queste parole, separandoli con una virgola, ad esempio: lemma1, lemma2, lemma3, lemma4, lemma5. Non aggiungere commenti o altro testo. Risposta:" diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ls_task.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ls_task.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1c67a087e4b371bd54ac11beb7fc6c471e327a14 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ls_task.yaml @@ -0,0 +1,10 @@ +group: evalita-mp_ls +group_alias: lexical-substitution +task: +- evalita-mp_ls_tasks +aggregate_metric_list: + - metric: f1 + weight_by_size: True + +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_mc.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_mc.yaml new file mode 100644 index 0000000000000000000000000000000000000000..098ed1c50a83f84228138ecc897e1644e02d2791 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_mc.yaml @@ -0,0 +1,14 @@ +group: evalita-mp_mc +group_alias: Evalita-LLM - PPL-based +task: + - evalita-mp_te + - evalita-mp_sa + - evalita-mp_wic + - evalita-mp_hs + - evalita-mp_at + - evalita-mp_faq +aggregate_metric_list: + - metric: acc + weight_by_size: True +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner-adg_group.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner-adg_group.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0bfc01d896f00989373e69351e4f68af6e12a6a9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner-adg_group.yaml @@ -0,0 +1,9 @@ +group: evalita-mp_ner_adg_group +group_alias: 'evalita NER: ADG' +task: + - evalita-mp_ner-v2_tasks_adg +aggregate_metric_list: + - metric: f1 + weight_by_size: True +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner-adg_group_p1.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner-adg_group_p1.yaml new file mode 100644 index 0000000000000000000000000000000000000000..72bc56a9df9a32f9c3ad7eab09400308a0d6766d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner-adg_group_p1.yaml @@ -0,0 +1,13 @@ +include: _ner_template_yaml +dataset_name: adg +test_split: reduced_test +fewshot_split: dev +task_alias: prompt-1 +tag: evalita-mp_ner-v2_tasks_adg +task: evalita-mp_ner-v2_adg_p1 + +# English +#doc_to_text: "Given the following text, write the entity mentions in the text, indicating their type: [PER] (person), [LOC] (location), [ORG] (organization). Respond with the following format: Entity$Type. Separate each entity-type pair with the '%' character. Text: {{text}}" +# Italian +doc_to_text: "Dato il seguente testo, scrivi le menzioni di entità nel testo, indicandone il tipo: PER (persona), LOC (luogo), ORG (organizzazione). Rispondi con il seguente formato: Entità$Tipo%Entità$Tipo. Separa ogni coppia entità-tipo con il carattere '%' ad esempio: Entità_2$Tipo%Entità_2$Tipo. In caso non ci siano entita' rispondi '&&NOENT&&'. +Testo: {{text}}" diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner-adg_group_p2.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner-adg_group_p2.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9066b707c3cb28cf4e5feb11d514c495f58bd76c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner-adg_group_p2.yaml @@ -0,0 +1,13 @@ +include: _ner_template_yaml +dataset_name: adg +test_split: reduced_test +fewshot_split: dev +task_alias: prompt-2 +tag: evalita-mp_ner-v2_tasks_adg +task: evalita-mp_ner-v2_adg_p2 + +# English +#doc_to_text: "Given the following text, write the entity mentions in the text, indicating their type: [PER] (person), [LOC] (location), [ORG] (organization). Respond with the following format: Entity$Type. Separate each entity-type pair with the '%' character. Text: {{text}}" +# Italian +doc_to_text: "Dato il seguente testo, scrivi le menzioni di entità nel testo, indicandone il tipo: PER (persona), LOC (luogo), ORG (organizzazione). Rispondi con il seguente formato: Entità$Tipo%Entità$Tipo. Separa ogni coppia entità-tipo con il carattere '%' ad esempio: Entità_2$Tipo%Entità_2$Tipo. In caso non ci siano entita' rispondi '&&NOENT&&'. +Testo: {{text}}" diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner-fic_group.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner-fic_group.yaml new file mode 100644 index 0000000000000000000000000000000000000000..da789e783b936ed7613bd46314553497f0ec1548 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner-fic_group.yaml @@ -0,0 +1,9 @@ +group: evalita-mp_ner_fic_group +group_alias: 'evalita NER: FIC' +task: + - evalita-mp_ner-v2_tasks_fic +aggregate_metric_list: + - metric: f1 + weight_by_size: True +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner-fic_group_p1.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner-fic_group_p1.yaml new file mode 100644 index 0000000000000000000000000000000000000000..52accfb487435e77a7cc8378acf2203bf9e895c8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner-fic_group_p1.yaml @@ -0,0 +1,17 @@ +include: _ner_template_yaml +dataset_name: fic +test_split: reduced_test +test_split: dev +fewshot_split: dev +task_alias: prompt-1 +tag: evalita-mp_ner-v2_tasks_fic +task: evalita-mp_ner-v2_fic_p1 + +# +doc_to_target: !function utils.filter_per_entities_from_lines +doc_to_target: entities + +# English +#doc_to_text: "Given the following text, write the entity mentions in the text, indicating their type: [PER] (person), [LOC] (location), [ORG] (organization). Respond with the following format: Entity$Type. Separate each entity-type pair with the '%' character. Text: {{text}}" +# Italian +doc_to_text: "Dato il seguente testo, scrivi le menzioni di entità nel testo, indicandone il tipo: PER (persona), LOC (luogo), ORG (organizzazione). Rispondi con il seguente formato: Entità$Tipo%Entità$Tipo. Separa ogni coppia entità-tipo con il carattere '%' ad esempio: Entità_2$Tipo%Entità_2$Tipo. In caso non ci siano entita' rispondi '&&NOENT&&'. Testo: {{text}}" diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner-fic_group_p2.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner-fic_group_p2.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6b8a22c9c73b0c0c24d5aebcd93621d9129ccda2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner-fic_group_p2.yaml @@ -0,0 +1,13 @@ +include: _ner_template_yaml +dataset_name: fic +test_split: reduced_test +fewshot_split: dev +task_alias: prompt-2 +tag: evalita-mp_ner-v2_tasks_fic +task: evalita-mp_ner-v2_fic_p2 + +# English +#doc_to_text: "Given the following text, write the entity mentions in the text, indicating their type: [PER] (person), [LOC] (location), [ORG] (organization). Respond with the following format: Entity$Type. Separate each entity-type pair with the '%' character. Text: {{text}}" +# Italian +doc_to_text: "Dato il seguente testo, scrivi le menzioni di entità nel testo, indicandone il tipo: PER (persona), LOC (luogo), ORG (organizzazione). Rispondi con il seguente formato: Entità$Tipo%Entità$Tipo. Separa ogni coppia entità-tipo con il carattere '%' ad esempio: Entità_2$Tipo%Entità_2$Tipo. In caso non ci siano entita' rispondi '&&NOENT&&'. +Testo: {{text}}" diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner-wn_group.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner-wn_group.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a8b18b9205804bf0f090323d0ac20166076808b2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner-wn_group.yaml @@ -0,0 +1,9 @@ +group: evalita-mp_ner_wn_group +group_alias: 'evalita NER: WN' +task: + - evalita-mp_ner-v2_tasks_wn +aggregate_metric_list: + - metric: f1 + weight_by_size: True +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner-wn_group_p1.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner-wn_group_p1.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ab353accd6d51604cc121353a8e7cb9681f96a38 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner-wn_group_p1.yaml @@ -0,0 +1,13 @@ +include: _ner_template_yaml +dataset_name: wn +test_split: reduced_test +fewshot_split: dev +task_alias: prompt-1 +tag: evalita-mp_ner-v2_tasks_wn +task: evalita-mp_ner-v2_wn_p1 + +# English +#doc_to_text: "Given the following text, write the entity mentions in the text, indicating their type: [PER] (person), [LOC] (location), [ORG] (organization). Respond with the following format: Entity$Type. Separate each entity-type pair with the '%' character. Text: {{text}}" +# Italian +doc_to_text: "Dato il seguente testo, scrivi le menzioni di entità nel testo, indicandone il tipo: PER (persona), LOC (luogo), ORG (organizzazione). Rispondi con il seguente formato: Entità$Tipo%Entità$Tipo. Separa ogni coppia entità-tipo con il carattere '%' ad esempio: Entità_2$Tipo%Entità_2$Tipo. In caso non ci siano entita' rispondi '&&NOENT&&'. +Testo: {{text}}" diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner-wn_group_p2.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner-wn_group_p2.yaml new file mode 100644 index 0000000000000000000000000000000000000000..becc3d85230f67afdd75f477a94ef35cac18cc6c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner-wn_group_p2.yaml @@ -0,0 +1,13 @@ +include: _ner_template_yaml +dataset_name: wn +test_split: reduced_test +fewshot_split: dev +task_alias: prompt-2 +tag: evalita-mp_ner-v2_tasks_wn +task: evalita-mp_ner-v2_wn_p2 + +# English +#doc_to_text: "Given the following text, write the entity mentions in the text, indicating their type: [PER] (person), [LOC] (location), [ORG] (organization). Respond with the following format: Entity$Type. Separate each entity-type pair with the '%' character. Text: {{text}}" +# Italian +doc_to_text: "Dato il seguente testo, scrivi le menzioni di entità nel testo, indicandone il tipo: PER (persona), LOC (luogo), ORG (organizzazione). Rispondi con il seguente formato: Entità$Tipo%Entità$Tipo. Separa ogni coppia entità-tipo con il carattere '%' ad esempio: Entità_2$Tipo%Entità_2$Tipo. In caso non ci siano entita' rispondi '&&NOENT&&'. +Testo: {{text}}" diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner_adg.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner_adg.yaml new file mode 100644 index 0000000000000000000000000000000000000000..02362087e7c0099902b8e52c1bb855cdeb46d71d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner_adg.yaml @@ -0,0 +1,7 @@ +group: evalita-mp_ner_tasks_adg +group_alias: evalita NER adg +aggregate_metric_list: + - metric: f1 + weight_by_size: True +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner_adg_p1.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner_adg_p1.yaml new file mode 100644 index 0000000000000000000000000000000000000000..16ba0b7a1d1ae7cea842aee6be4561ce2be533fa --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner_adg_p1.yaml @@ -0,0 +1,14 @@ +include: _ner_template_yaml +dataset_name: adg +test_split: reduced_test +fewshot_split: trial + +task_alias: ADG prompt-1 +tag: evalita-mp_ner_tasks_adg +task: evalita-mp_ner_adg_p1 + + +#p1 +doc_to_text: "Estrai tutte le entità di tipo PER (persona), LOC (luogo) e ORG (organizzazione) dal testo seguente. Riporta ogni entità con il formato: Entità$Tipo, separando ciascuna coppia con ','. Se non ci sono entità da estrarre, rispondi con '&&NOENT&&'. +Testo: '{{text}}' +Entità:" diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner_adg_p2.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner_adg_p2.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fb3d852cfec5560c084188c8be3faa02b893ed2b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner_adg_p2.yaml @@ -0,0 +1,14 @@ +include: _ner_template_yaml +dataset_name: adg +test_split: reduced_test +fewshot_split: trial + +task_alias: ADG prompt-2 +tag: evalita-mp_ner_tasks_adg +task: evalita-mp_ner_adg_p2 + + +#p8 +doc_to_text: "Devi svolgere un compito di riconoscimento delle entità nei testi. Estrai tutte le entità di tipo PER (persona), LOC (luogo) e ORG (organizzazione) dal testo seguente. Riporta ogni entità con il formato: Entità$Tipo, separando ciascuna coppia con ','. Se non ci sono entità da estrarre, rispondi con '&&NOENT&&'. +Testo: '{{text}}' +Entità:" diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner_fic.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner_fic.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9e3387243d4f7fc6e3b05904ac1722a82813d0fd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner_fic.yaml @@ -0,0 +1,5 @@ + +group: evalita-mp_ner_tasks_fic +group_alias: evalita NER fic + +task_alias: NER fic diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner_fic_p1.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner_fic_p1.yaml new file mode 100644 index 0000000000000000000000000000000000000000..248b150d68ed8e659770c89a8fcb670a8532648b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner_fic_p1.yaml @@ -0,0 +1,15 @@ +include: _ner_template_yaml +dataset_name: fic +test_split: reduced_test +fewshot_split: trial + +task_alias: FIC prompt-1 +tag: evalita-mp_ner_tasks_fic +task: evalita-mp_ner_fic_p1 + + + +#p1 +doc_to_text: "Estrai tutte le entità di tipo PER (persona), LOC (luogo) e ORG (organizzazione) dal testo seguente. Riporta ogni entità con il formato: Entità$Tipo, separando ciascuna coppia con ','. Se non ci sono entità da estrarre, rispondi con '&&NOENT&&'. +Testo: '{{text}}' +Entità:" diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner_fic_p2.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner_fic_p2.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f71454150924a1d5af527047c0652baf094ff8c0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner_fic_p2.yaml @@ -0,0 +1,13 @@ +include: _ner_template_yaml +dataset_name: fic +test_split: reduced_test +fewshot_split: trial + +task_alias: FIC prompt-2 +tag: evalita-mp_ner_tasks_fic +task: evalita-mp_ner_fic_p2 + +#p8 +doc_to_text: "Devi svolgere un compito di riconoscimento delle entità nei testi. Estrai tutte le entità di tipo PER (persona), LOC (luogo) e ORG (organizzazione) dal testo seguente. Riporta ogni entità con il formato: Entità$Tipo, separando ciascuna coppia con ','. Se non ci sono entità da estrarre, rispondi con '&&NOENT&&'. +Testo: '{{text}}' +Entità:" diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner_group.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner_group.yaml new file mode 100644 index 0000000000000000000000000000000000000000..adc8e48574d1af1b5ab525cbca07bad9cb00838f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner_group.yaml @@ -0,0 +1,11 @@ +group: evalita-mp_ner_group +group_alias: evalita NER +task: + - evalita-mp_ner_tasks_fic + - evalita-mp_ner_tasks_adg + - evalita-mp_ner_tasks_wn +aggregate_metric_list: + - metric: f1 + weight_by_size: True +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner_wn.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner_wn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3b1cd45f209a8791c664b5221b3c13759d87d8ba --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner_wn.yaml @@ -0,0 +1,7 @@ +group: evalita-mp_ner_tasks_wn +group_alias: evalita NER wn +aggregate_metric_list: + - metric: f1 + weight_by_size: True +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner_wn_p1.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner_wn_p1.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a096b564354299bed7776ebee633c6663400e0e1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner_wn_p1.yaml @@ -0,0 +1,13 @@ +include: _ner_template_yaml +dataset_name: wn +test_split: reduced_test +fewshot_split: trial + +task_alias: WN prompt-1 +tag: evalita-mp_ner_tasks_wn +task: evalita-mp_ner_wn_p1 + + +doc_to_text: "Estrai tutte le entità di tipo PER (persona), LOC (luogo) e ORG (organizzazione) dal testo seguente. Riporta ogni entità con il formato: Entità$Tipo, separando ciascuna coppia con ','. Se non ci sono entità da estrarre, rispondi con '&&NOENT&&'. +Testo: '{{text}}' +Entità:" diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner_wn_p2.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner_wn_p2.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ff481e7d711b7887675ddcbd36fbb8f5b5aaae93 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_ner_wn_p2.yaml @@ -0,0 +1,13 @@ +include: _ner_template_yaml +dataset_name: wn +test_split: reduced_test +fewshot_split: trial + +task_alias: WN prompt-2 +tag: evalita-mp_ner_tasks_wn +task: evalita-mp_ner_wn_p2 + + +doc_to_text: "Devi svolgere un compito di riconoscimento delle entità nei testi. Estrai tutte le entità di tipo PER (persona), LOC (luogo) e ORG (organizzazione) dal testo seguente. Riporta ogni entità con il formato: Entità$Tipo, separando ciascuna coppia con ','. Se non ci sono entità da estrarre, rispondi con '&&NOENT&&'. +Testo: '{{text}}' +Entità:" diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_re_p1.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_re_p1.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9755dc9740e207e83e28aa8281b4415c75efe5d7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_re_p1.yaml @@ -0,0 +1,10 @@ +tag: evalita-mp_re_tasks +include: _re_template_yaml +task: evalita-mp_re_prompt-1 +fewshot_split: dev +task_alias: prompt-1 + +#p4 +doc_to_text: "Dato un documento medico devi estrarre tutte le misurazioni degli esami medici presenti. Riporta ogni relazione nel formato: misurazione$esame, separando ciascuna coppia con '%'. Se non ci sono relazioni da estrarre, rispondi con '&&NOREL&&'. +Testo: '{{text}}' +Relazioni:" diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_re_p2.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_re_p2.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7ea25f7e6a685758a25a29fb8e31c77480bb6e22 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_re_p2.yaml @@ -0,0 +1,10 @@ +tag: evalita-mp_re_tasks +include: _re_template_yaml +fewshot_split: dev +task: evalita-mp_re_prompt-2 +task_alias: prompt-2 + +#p5 +doc_to_text: "Devi svolgere un compito di estrazione di relazioni da documenti medici. Dato un documento medico devi estrarre tutte le misurazioni degli esami medici presenti. Riporta ogni relazione nel formato: misurazione$esame, separando ciascuna coppia con '%'. Se non ci sono relazioni da estrarre, rispondi con '&&NOREL&&'. +Testo: '{{text}}' +Relazioni:" diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_re_task.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_re_task.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5b629da40610519583c900ef4dda7f1999ff2fad --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_re_task.yaml @@ -0,0 +1,10 @@ +group: evalita-mp_re +group_alias: relation-extraction +task: +- evalita-mp_re_tasks +aggregate_metric_list: + - metric: f1 + weight_by_size: True + +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_sa_p1.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_sa_p1.yaml new file mode 100644 index 0000000000000000000000000000000000000000..01d7cee64098203b2ae7929eb03075a04fa8d65c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_sa_p1.yaml @@ -0,0 +1,10 @@ +tag: evalita-mp_sa_tasks +include: _sa_template_yaml +task: evalita-mp_sa_prompt-1 +task_alias: prompt-1 +#doc_to_text: "Opinione: '{{text}}' Determinare la sentiment dell'opinione data. Possibili risposte: A – neutrale B – negativo C – positivo D - misto Risposta:" +doc_to_text: "Qual è il sentiment espresso nel seguente tweet: '{{text}}'?" +metric_list: + - metric: f1 + higher_is_better: True + aggregation: !function metrics._aggreg_sa diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_sa_p2.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_sa_p2.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9188f6142dc9fefc557645f970244ebf7da9db7c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_sa_p2.yaml @@ -0,0 +1,10 @@ +tag: evalita-mp_sa_tasks +include: _sa_template_yaml +task: evalita-mp_sa_prompt-2 +task_alias: prompt-2 +#doc_to_text: "Opinione: '{{text}}' Determinare la sentiment dell'opinione data. Possibili risposte: A – neutrale B – negativo C – positivo D - misto Risposta:" +doc_to_text: "Devi svolgere un compito di analisi del sentiment. Qual è il sentiment espresso nel seguente tweet: '{{text}}'?" +metric_list: + - metric: f1 + higher_is_better: True + aggregation: !function metrics._aggreg_sa diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_sa_p3.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_sa_p3.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cf61e9c47e008e9b16d2818c1ad55b4586515846 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_sa_p3.yaml @@ -0,0 +1,11 @@ +tag: evalita-mp_sa_tasks +include: _sa_template_yaml +task: evalita-mp_sa_prompt-3 +task_alias: prompt-3 +#doc_to_text: "Opinione: '{{text}}' Determinare la sentiment dell'opinione data. Possibili risposte: A – neutrale B – negativo C – positivo D - misto Risposta:" +doc_to_choice: ["A", "B", "C", "D"] +doc_to_text: "Qual è il sentiment espresso nel seguente tweet: '{{text}}'?\nA: Positivo\nB: Negativo\nC: Neutro\nD: Misto\nRisposta:" +metric_list: + - metric: f1 + higher_is_better: True + aggregation: !function metrics._aggreg_sa diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_sa_p4.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_sa_p4.yaml new file mode 100644 index 0000000000000000000000000000000000000000..72c956d105bed971986c3a6ec4525c64c940c128 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_sa_p4.yaml @@ -0,0 +1,11 @@ +tag: evalita-mp_sa_tasks +include: _sa_template_yaml +task: evalita-mp_sa_prompt-4 +task_alias: prompt-4 +#doc_to_text: "Opinione: '{{text}}' Determinare la sentiment dell'opinione data. Possibili risposte: A – neutrale B – negativo C – positivo D - misto Risposta:" +doc_to_choice: ["A", "B", "C", "D"] +doc_to_text: "Devi svolgere un compito di analisi del sentiment. Qual è il sentiment espresso nel seguente tweet: '{{text}}'?\nA: Positivo\nB: Negativo\nC: Neutro\nD: Misto\nRisposta:" +metric_list: + - metric: f1 + higher_is_better: True + aggregation: !function metrics._aggreg_sa diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_sa_p5.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_sa_p5.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cc58565ce897f3c6d01ff0808c704cf2081c98e6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_sa_p5.yaml @@ -0,0 +1,11 @@ +tag: evalita-mp_sa_tasks +include: _sa_template_yaml +task: evalita-mp_sa_prompt-5 +task_alias: prompt-5 +#doc_to_text: "Opinione: '{{text}}' Determinare la sentiment dell'opinione data. Possibili risposte: A – neutrale B – negativo C – positivo D - misto Risposta:" +#doc_to_choice: ["A", "B", "C", "D"] +doc_to_text: "Il seguente tweet: '{{text}}' esprime un sentiment" +metric_list: + - metric: f1 + higher_is_better: True + aggregation: !function metrics._aggreg_sa diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_sa_p6.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_sa_p6.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6904835e76d6cb9563a2161e50c13519a5a4ddf7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_sa_p6.yaml @@ -0,0 +1,11 @@ +tag: evalita-mp_sa_tasks +include: _sa_template_yaml +task: evalita-mp_sa_prompt-6 +task_alias: prompt-6 +#doc_to_text: "Opinione: '{{text}}' Determinare la sentiment dell'opinione data. Possibili risposte: A – neutrale B – negativo C – positivo D - misto Risposta:" +#doc_to_choice: ["A", "B", "C", "D"] +doc_to_text: "Devi svolgere un compito di analisi del sentiment. Il seguente tweet: '{{text}}' esprime un sentiment" +metric_list: + - metric: f1 + higher_is_better: True + aggregation: !function metrics._aggreg_sa diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_sa_tasks.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_sa_tasks.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1f6b883caf50d16cdac112b04e00a54ea6b20309 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_sa_tasks.yaml @@ -0,0 +1,9 @@ +group: evalita-mp_sa +group_alias: sentiment-analysis +task: + - evalita-mp_sa_tasks # Each of the tasks has to have a matching tag in its own yaml file +aggregate_metric_list: + - metric: f1 + weight_by_size: True +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_sum_fp-small_p1.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_sum_fp-small_p1.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6b2a2c3274a996fccc6ec58d070a376adfd8f8fa --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_sum_fp-small_p1.yaml @@ -0,0 +1,12 @@ +tag: evalita-mp_sum_fp-small_tasks +include: _sum_template_fp-small_yaml +task: evalita-sp_sum_task_fp-small_p1 +task_alias: prompt-1 +#doc_to_text: > +# "Crea un sommario del seguente testo. Testo: {{source}}\nSommario: " +doc_to_text: "Riassumi il seguente articolo di giornale: '{{source}}'\nRiassunto:" +process_results: !function sum_utils.process_results_sum +metric_list: + - metric: rouge1 + higher_is_better: true + aggregation: mean diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_sum_fp-small_p2.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_sum_fp-small_p2.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0eae91371f4b94b6037595fb1b554175832856bf --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_sum_fp-small_p2.yaml @@ -0,0 +1,12 @@ +tag: evalita-mp_sum_fp-small_tasks +include: _sum_template_fp-small_yaml +task: evalita-sp_sum_task_fp-small_p2 +task_alias: prompt-2 +#doc_to_text: > +# "Crea un sommario del seguente testo. Testo: {{source}}\nSommario: " +doc_to_text: "Devi risolvere un compito di sintesi automatica del testo. Riassumi il seguente articolo di giornale: '{{source}}'\nRiassunto:" +process_results: !function sum_utils.process_results_sum +metric_list: + - metric: rouge1 + higher_is_better: true + aggregation: mean diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_sum_fp-small_task.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_sum_fp-small_task.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c0c339f822d1edd7e434b44665376aa7473ee2f6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_sum_fp-small_task.yaml @@ -0,0 +1,9 @@ +group: evalita-mp_sum_fp +group_alias: summarization-fanpage +task: +- evalita-mp_sum_fp-small_tasks +aggregate_metric_list: + - metric: rouge1 + weight_by_size: True +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_sum_fp_p1.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_sum_fp_p1.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4dbc3d0b9ea2578d9cd09fb78144bd5c7d454d25 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_sum_fp_p1.yaml @@ -0,0 +1,10 @@ +tag: evalita-mp_sum_fp_tasks +include: _sum_template_fp_yaml +task: evalita-sp_sum_task_fp_p1 +task_alias: prompt-1 +doc_to_text: "Riassumi il seguente articolo di giornale: '{{source}}'\nRiassunto:" +process_results: !function sum_utils.process_results_sum +metric_list: + - metric: rouge1 + higher_is_better: true + aggregation: mean diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_sum_fp_p2.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_sum_fp_p2.yaml new file mode 100644 index 0000000000000000000000000000000000000000..104ecaa442b2bbd1030f140d68b0cfd27e134579 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_sum_fp_p2.yaml @@ -0,0 +1,10 @@ +tag: evalita-mp_sum_fp_tasks +include: _sum_template_fp_yaml +task: evalita-sp_sum_task_fp_p2 +task_alias: prompt-2 +doc_to_text: "Devi risolvere un compito di sintesi automatica del testo. Riassumi il seguente articolo di giornale: '{{source}}'\nRiassunto:" +process_results: !function sum_utils.process_results_sum +metric_list: + - metric: rouge1 + higher_is_better: true + aggregation: mean diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_sum_fp_task.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_sum_fp_task.yaml new file mode 100644 index 0000000000000000000000000000000000000000..147fe567486cd59fdb817d6448ffa3e6e4d6969a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_sum_fp_task.yaml @@ -0,0 +1,9 @@ +group: evalita-mp_sum_fp +group_alias: summarization-fanpage +task: +- evalita-mp_sum_fp_tasks +aggregate_metric_list: + - metric: rouge1 + weight_by_size: True +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_te_p1.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_te_p1.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e9841a4528fda240f222c2e593b3fc3e73f31297 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_te_p1.yaml @@ -0,0 +1,9 @@ +tag: evalita-mp_te_tasks +include: _te_template_yaml +task: evalita-mp_te_prompt-1 +task_alias: prompt-1 +#doc_to_text: "Task di Text Entailment. Rispondi Vero o Falso in base alla correttezza dell'ipotesi rispetto al testo.\nTesto:{{text1}}\nIpotesi: {{text2}}\nRisposta:" +doc_to_text: "La frase: '{{text1}}' implica logicamente che la frase: '{{text2}}' sia vera?" +#metric_list: +# - metric: acc +# higher_is_better: true diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_te_p2.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_te_p2.yaml new file mode 100644 index 0000000000000000000000000000000000000000..932fc185d72d3db1efd576c045bf2da13c17150e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_te_p2.yaml @@ -0,0 +1,5 @@ +tag: evalita-mp_te_tasks +include: _te_template_yaml +task: evalita-mp_te_prompt-2 +task_alias: prompt-2 +doc_to_text: "Devi risolvere un compito di inferenza semantica. La frase: '{{text1}}' implica logicamente che la frase: '{{text2}}' sia vera?" diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_te_p3.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_te_p3.yaml new file mode 100644 index 0000000000000000000000000000000000000000..91e0c667c44928afa1f136adc025d0d5ab3cd676 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_te_p3.yaml @@ -0,0 +1,6 @@ +tag: evalita-mp_te_tasks +include: _te_template_yaml +task: evalita-mp_te_prompt-3 +task_alias: prompt-3 +doc_to_choice: ["A", "B"] +doc_to_text: "La frase: '{{text1}}' implica logicamente che la frase: '{{text2}}' sia vera?\nA: Sì\nB: No\nRisposta:" diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_te_p4.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_te_p4.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8ffc087d8b16a3eee057714d694b4102a937ef54 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_te_p4.yaml @@ -0,0 +1,6 @@ +tag: evalita-mp_te_tasks +include: _te_template_yaml +task: evalita-mp_te_prompt-4 +task_alias: prompt-4 +doc_to_choice: ["A", "B"] +doc_to_text: "Devi risolvere un compito di inferenza semantica. La frase: '{{text1}}' implica logicamente che la frase: '{{text2}}' sia vera?\nA: Sì\nB: No\nRisposta:" diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_te_p5.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_te_p5.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2cee2a1246fe77ca81413d5d036d5014edf4d290 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_te_p5.yaml @@ -0,0 +1,6 @@ +tag: evalita-mp_te_tasks +include: _te_template_yaml +task: evalita-mp_te_prompt-5 +task_alias: prompt-5 +doc_to_choice: ["La frase 1 implica logicamente che la frase 2 sia vera", "La frase 1 non implica logicamente che la frase 2 sia vera"] +doc_to_text: "Frase 1: '{{text1}}' Frase 2: '{{text2}}'" diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_te_p6.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_te_p6.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e06bbefe93601a9524aa96a4137de7af984556bd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_te_p6.yaml @@ -0,0 +1,6 @@ +tag: evalita-mp_te_tasks +include: _te_template_yaml +task: evalita-mp_te_prompt-6 +task_alias: prompt-6 +doc_to_choice: ["La frase 1 implica logicamente che la frase 2 sia vera", "La frase 1 non implica logicamente che la frase 2 sia vera"] +doc_to_text: "Devi risolvere un compito di inferenza semantica. Frase 1: '{{text1}}' Frase 2: '{{text2}}'" diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_te_tasks.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_te_tasks.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8c6d53fcfa7eab39f0cc4365c7d376cc785d9fb9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_te_tasks.yaml @@ -0,0 +1,9 @@ +group: evalita-mp_te +group_alias: text-entailment +task: + - evalita-mp_te_tasks # this has to match the tag in the task yaml file +aggregate_metric_list: + - metric: acc + weight_by_size: True +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_wic_p1.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_wic_p1.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5a8c47fa5eb7dee9c96863bd7f186b2bc062c511 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_wic_p1.yaml @@ -0,0 +1,5 @@ +tag: evalita-mp_wic_tasks +task: evalita-mp_wic_prompt-1 +task_alias: prompt-1 +include: _wic_template_yaml +doc_to_text: "La parola: '{{sentence1[start1:end1]}}' nella frase: '{{sentence1}}' ha lo stesso significato della parola: '{{sentence2[start2:end2]}}' nella frase: '{{sentence2}}'?" diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_wic_p2.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_wic_p2.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f990ee78ef925beb8a45d3fdf75c0ae2489132d2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_wic_p2.yaml @@ -0,0 +1,5 @@ +tag: evalita-mp_wic_tasks +task: evalita-mp_wic_prompt-2 +task_alias: prompt-2 +include: _wic_template_yaml +doc_to_text: "Devi determinare se una stessa parola usata in due frasi differenti ha lo stesso significato in entrambi i contesti. La parola: '{{sentence1[start1:end1]}}' nella frase: '{{sentence1}}' ha lo stesso significato della parola: '{{sentence2[start2:end2]}}' nella frase: '{{sentence2}}'?" diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_wic_p3.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_wic_p3.yaml new file mode 100644 index 0000000000000000000000000000000000000000..20267adc68ecfb59a514dd71cc6940ca0fe80bcb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_wic_p3.yaml @@ -0,0 +1,6 @@ +tag: evalita-mp_wic_tasks +task: evalita-mp_wic_prompt-3 +task_alias: prompt-3 +include: _wic_template_yaml +doc_to_text: "La parola '{{sentence1[start1:end1]}}' nella frase '{{sentence1}}' ha lo stesso significato della parola '{{sentence2[start2:end2]}}' nella frase '{{sentence2}}'?\nA: Sì\nB: No\nRisposta:" +doc_to_choice: ["B", "A"] diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_wic_p4.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_wic_p4.yaml new file mode 100644 index 0000000000000000000000000000000000000000..46086de33e66a216953a5bb02dbbdded9592f8ca --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_wic_p4.yaml @@ -0,0 +1,6 @@ +tag: evalita-mp_wic_tasks +task: evalita-mp_wic_prompt-4 +task_alias: prompt-4 +include: _wic_template_yaml +doc_to_text: "Devi determinare se una stessa parola usata in due frasi differenti ha lo stesso significato in entrambi i contesti. La parola '{{sentence1[start1:end1]}}' nella frase '{{sentence1}}' ha lo stesso significato della parola '{{sentence2[start2:end2]}}' nella frase '{{sentence2}}'?\nA: Sì\nB: No\nRisposta:" +doc_to_choice: ["B", "A"] diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_wic_p5.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_wic_p5.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3a8e883ad0ede86193303be25d78a34ca4e3d7cc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_wic_p5.yaml @@ -0,0 +1,6 @@ +tag: evalita-mp_wic_tasks +task: evalita-mp_wic_prompt-5 +task_alias: prompt-5 +include: _wic_template_yaml +doc_to_text: "La parola: '{{sentence1[start1:end1]}}' nella frase: '{{sentence1}}' e la parola: '{{sentence2[start2:end2]}}' nella frase: '{{sentence2}}'" +doc_to_choice: ["non hanno lo stesso significato", "hanno lo stesso significato"] diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_wic_p6.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_wic_p6.yaml new file mode 100644 index 0000000000000000000000000000000000000000..56ddf9d97edf73e6d142f15143d6139bda3f9570 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_wic_p6.yaml @@ -0,0 +1,6 @@ +tag: evalita-mp_wic_tasks +task: evalita-mp_wic_prompt-6 +task_alias: prompt-6 +include: _wic_template_yaml +doc_to_text: "Devi determinare se una stessa parola usata in due frasi differenti ha lo stesso significato in entrambi i contesti. La parola: '{{sentence1[start1:end1]}}' nella frase: '{{sentence1}}' e la parola: '{{sentence2[start2:end2]}}' nella frase: '{{sentence2}}'" +doc_to_choice: ["non hanno lo stesso significato", "hanno lo stesso significato"] diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_wic_tasks.yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_wic_tasks.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5825046b2ebbf2ad5f823ac4abf087064afe0b3e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_evalita-mp_wic_tasks.yaml @@ -0,0 +1,9 @@ +group: evalita-mp_wic +group_alias: word-in-context +task: + - evalita-mp_wic_tasks # this has to match the tag in the task yaml file +aggregate_metric_list: + - metric: f1 + weight_by_size: True +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_faq_template_yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_faq_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..5620b9480c723b94463b84768e6a2ba45d9fb2f7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_faq_template_yaml @@ -0,0 +1,8 @@ +dataset_path: evalitahf/faq +test_split: test_1 +fewshot_split: dev_1 +doc_to_target: !function utils.faq_doc_to_target +doc_to_choice: ["A", "B", "C", "D"] +output_type: multiple_choice +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_hs_template_yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_hs_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..c224f4e3b1a69b52c7b9b0fc42a6198d7c788daf --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_hs_template_yaml @@ -0,0 +1,9 @@ +dataset_path: evalitahf/hatespeech_detection +output_type: multiple_choice +test_split: test_all +fewshot_split: dev +validation_split: dev +doc_to_target: hs # 0 = Falso, 1 = Vero +doc_to_choice: ["Falso", "Vero"] +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_ls_template_yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_ls_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..a5df2eb1818a9dd974bf6829f6142af23547e1cf --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_ls_template_yaml @@ -0,0 +1,16 @@ +dataset_path: evalitahf/lexical_substitution +test_split: test +validation_split: dev +fewshot_split: dev +output_type: generate_until +generation_kwargs: + until: + - "" +doc_to_target: !function utils.ls_doc_to_target +process_results: !function utils.ls_process_results +metric_list: + - metric: f1 + higher_is_better: True + aggregation: !function metrics._aggreg_ls +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_ner_template_yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_ner_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..77dd0c3b30be935c73ba58b36a8cb516653062fb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_ner_template_yaml @@ -0,0 +1,14 @@ +dataset_path: evalitahf/entity_recognition +output_type: generate_until +generation_kwargs: + until: + - "" + - "\n" +doc_to_target: !function utils.ner_doc_to_target +process_results: !function utils.ner_process_results +metric_list: + - metric: f1 + higher_is_better: True + aggregation: !function metrics._aggreg_ner +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_re_template_yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_re_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..9621af125bb4bdd1dee5895793a51a15570c0ab8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_re_template_yaml @@ -0,0 +1,14 @@ +dataset_path: evalitahf/relation_extraction +test_split: test +output_type: generate_until +generation_kwargs: + until: + - "" +doc_to_target: !function utils.re_doc_to_target +process_results: !function utils.rel_process_results_v3 +metric_list: + - metric: f1 + higher_is_better: True + aggregation: !function metrics._aggreg_rel +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_sa_template_v2_yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_sa_template_v2_yaml new file mode 100644 index 0000000000000000000000000000000000000000..b9fc6460a1fe6a298713abf90765214c31162d76 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_sa_template_v2_yaml @@ -0,0 +1,9 @@ +dataset_path: evalitahf/sentiment_analysis +output_type: multiple_choice +test_split: test +fewshot_split: train +validation_split: test +doc_to_target: !function utils.sa_doc_to_target_v2 +doc_to_choice: ["positivo", "negativo", "neutrale", "misto"] +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_sa_template_yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_sa_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..49ae1c8084a9cdfcb6c8b27f74ec6ac78549108b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_sa_template_yaml @@ -0,0 +1,9 @@ +dataset_path: evalitahf/sentiment_analysis +output_type: multiple_choice +test_split: test +fewshot_split: train +validation_split: test +doc_to_target: !function utils.sa_doc_to_target +doc_to_choice: !function utils.sa_doc_to_choice +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_sum_template_fp-small_yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_sum_template_fp-small_yaml new file mode 100644 index 0000000000000000000000000000000000000000..fb067b9d22652a1f3ec92b714d34ee190238d767 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_sum_template_fp-small_yaml @@ -0,0 +1,10 @@ +dataset_path: evalitahf/summarization-fp +output_type: generate_until +generation_kwargs: + until: + - "" +test_split: test_100 +fewshot_split: dev +doc_to_target: "{{target}}" +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_sum_template_fp_yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_sum_template_fp_yaml new file mode 100644 index 0000000000000000000000000000000000000000..3399374204937424aa943d4af3a1fc28cfb0c41d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_sum_template_fp_yaml @@ -0,0 +1,9 @@ +dataset_path: ARTeLab/fanpage +output_type: generate_until +generation_kwargs: + until: + - "" +test_split: test +doc_to_target: "{{target}}" +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_sum_template_yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_sum_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..bfe69669213c3edfdb70bb270d4410fd4b46f42f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_sum_template_yaml @@ -0,0 +1,11 @@ +dataset_path: silvia-casola/WITS +output_type: generate_until +generation_kwargs: + until: + - "" +test_split: test_100 +fewshot_split: dev +#test_split: train +doc_to_target: "{{summary}}" +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_te_template_yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_te_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..ed8888fcf95c0a4b3162a6892c0c05f6a829e3ff --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_te_template_yaml @@ -0,0 +1,13 @@ +dataset_path: evalitahf/textual_entailment +output_type: multiple_choice +test_split: test +fewshot_split: dev +validation_split: dev +doc_to_target: "{{ 0 if entailment == 'SI' else 1 }}" +doc_to_choice: ["Sì", "No"] +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 1 diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_wic_template_yaml b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_wic_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..bb5d0f00ce7b4f158c5630a946b09fdcb7be83d6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/_wic_template_yaml @@ -0,0 +1,14 @@ +dataset_path: evalitahf/word_in_context +dataset_name: default +output_type: multiple_choice +test_split: test +fewshot_split: dev +validation_split: dev +doc_to_target: label # 0: No, 1: Si +doc_to_choice: ["No", "Sì"] +metric_list: + - metric: f1 + higher_is_better: true + aggregation: f1 +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/metrics.py b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/metrics.py new file mode 100644 index 0000000000000000000000000000000000000000..2dbc53f39007fbba6cdb9c8fb48bd165e41f8753 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/metrics.py @@ -0,0 +1,165 @@ +import torch +from sklearn.metrics import f1_score, precision_score, recall_score + + +inference_decorator = ( + torch.inference_mode if torch.__version__ >= "2.0.0" else torch.no_grad +) + + +def _aggreg_ls(predictions): + """ + Custom aggregation to compute corpus level metrics for the lexical substitution task + predictions is a list of tuples (prec, has_answ, has_annotation) + prec is the precision before dividing by |A| + has_answ is 0 if the model did not produce any answer + has_annotation is 0 if the gold answer is empty: no synonims from annotators + """ + # get |A| and |T| to compute the final precision and recall using a lambda function + A = sum([p[1] for p in predictions]) + T = sum([p[2] for p in predictions]) + # compute the final precision and recall + if A == 0: + prec = sum([p[0] for p in predictions]) / 1 + else: + prec = sum([p[0] for p in predictions]) / A + if T == 0: + rec = sum([p[0] for p in predictions]) / 1 + else: + rec = sum([p[0] for p in predictions]) / T + # compute the final F1 score + f1 = 0 + if prec + rec != 0: + f1 = (2 * prec * rec) / (prec + rec) + return f1 + + +def _aggreg_sa_v2(predictions): + """ + This aggregation considers the sentiment analysis task as a multiple choice one with four classes + the f1 score is computed as the average of the f1 scores for each class weighted by the number of samples + See sklearn.metrics.f1_score for more details + + """ + predictions, references = zip(*predictions) + f1 = f1_score(references, predictions, average="weighted") + return f1 + + +def _aggreg_sa(predictions): + """ + Custom aggregation function for the sentiment analysis task + The original tasks compute the F1 score for each class and then average them + Since the prompt cast the task to a multple choice one we need to aggregate the results in a different way + """ + # split the predictions and references in two lists (pred is a tuple) + predictions, references = zip(*predictions) + """ + Class 0: positivo -> 'opos': 1, 'oneg': 0 + Class 1: negativo -> 'opos': 0, 'oneg': 1 + etc. + """ + + def _map_to_original_labels(x): + """ + Return two separate list of labels for opos and oneg + x is a list of integers + """ + opos = [] + oneg = [] + for i in x: + if i == 0: + # positive + opos.append(1) + oneg.append(0) + elif i == 1: + # negative + opos.append(0) + oneg.append(1) + elif i == 2: + # neutral + opos.append(0) + oneg.append(0) + elif i == 3: + # mixed + opos.append(1) + oneg.append(1) + else: + pass + return opos, oneg + + pred_opos, pred_oneg = _map_to_original_labels(predictions) + ref_opos, ref_oneg = _map_to_original_labels(references) + + opos_f1 = f1_score(ref_opos, pred_opos, average=None) + opos_f1_c0 = f1_score(ref_opos, pred_opos, average=None)[0] + if len(opos_f1) > 1: + opos_f1_c1 = opos_f1[1] + else: + opos_f1_c1 = 0 + + # oneg class + oneg_prec_c0, oneg_prec_c1 = precision_score( + ref_oneg, pred_oneg, labels=[0, 1], average=None + ) + oneg_rec_c0, oneg_rec_c1 = recall_score( + ref_oneg, pred_oneg, labels=[0, 1], average=None + ) + oneg_f1 = f1_score(ref_oneg, pred_oneg, average=None) + oneg_f1_c0 = f1_score(ref_oneg, pred_oneg, average=None)[0] + if len(oneg_f1) > 1: + oneg_f1_c1 = f1_score(ref_oneg, pred_oneg, average=None)[1] + else: + oneg_f1_c1 = 0 + + # average f1 score for each class (opos and oneg) + f1_score_opos = (opos_f1_c0 + opos_f1_c1) / 2 + f1_score_oneg = (oneg_f1_c0 + oneg_f1_c1) / 2 + # average f1 score for the two classes + f1_final = (f1_score_opos + f1_score_oneg) / 2 + + return f1_final + + +def _aggreg_ner(predictions): + pred, ref = zip(*predictions) + # concat all the predictions and references + all_pred = [] + for p in pred: + all_pred.extend(p) + all_ref = [] + for r in ref: + all_ref.extend(r) + # compute the F1 score + f1 = f1_score(all_ref, all_pred, average=None) + if len(f1) > 1: + f1_sum = sum(f1[:-1]) / (len(f1) - 1) + else: + f1_sum = f1[0] + + return f1_sum + + +def _aggreg_rel(predictions): + pred, ref = zip(*predictions) + # concat all the predictions and references + all_pred = [] + for p in pred: + all_pred.extend(p) + all_ref = [] + for r in ref: + all_ref.extend(r) + # compute the F1 score + f1 = f1_score(all_ref, all_pred, average="macro") + return f1 + + +# ------------------------ DOCUMENT DATING --------------------------- + + +def _aggreg_dd(items): + unzipped_list = list(zip(*items)) + golds = unzipped_list[0] + preds = unzipped_list[1] + fscore = f1_score(golds, preds, average="macro") + return fscore diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/sum_utils.py b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/sum_utils.py new file mode 100644 index 0000000000000000000000000000000000000000..9602be9662fa5180dd9a19a3f7aaebd1fce45c7c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/sum_utils.py @@ -0,0 +1,21 @@ +from evaluate import load + + +rouge = load("rouge", keep_in_memory=True) + + +def rouge1_score(references, predictions, **kwargs): + """ + Optimized ROUGE-1 computation using a single loaded metric instance. + """ + return rouge.compute(predictions=predictions, references=references, **kwargs)[ + "rouge1" + ] + + +def process_results_sum(doc, results): + """ + Process the results of the summarization task efficiently. + """ + ref = doc.get("summary", doc.get("target")) # Get the reference summary + return {"rouge1": rouge.compute(predictions=results, references=[ref])["rouge1"]} diff --git a/lm-evaluation-harness/lm_eval/tasks/evalita_llm/utils.py b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..7c47051505ac8fa899d1d1e3aa6669a59667377f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/evalita_llm/utils.py @@ -0,0 +1,549 @@ +import logging + +from evaluate import load +from sklearn.metrics import f1_score + + +eval_logger = logging.getLogger("lm-eval") + + +# ---------------------- SENTIMENT ANALYSIS ---------------------- +def sa_doc_to_target(x): + """ + Function to extract the target from the dataset for sentiment analysis + """ + opos = x["opos"] + oneg = x["oneg"] + # return indexes matches the choices in sa_doc_to_choice + if opos == "1" and oneg == "0": + return 0 + elif opos == "0" and oneg == "1": + return 1 + elif opos == "0" and oneg == "0": + return 2 + elif opos == "1" and oneg == "1": + return 3 + else: + pass + + +def sa_doc_to_target_v2(x): + """ + Function to extract the target from the dataset for sentiment analysis + """ + opos = x["opos"] + oneg = x["oneg"] + # return indexes matches the choices in sa_doc_to_choice + if opos == "1" and oneg == "0": + return 0 + elif opos == "0" and oneg == "1": + return 1 + elif opos == "0" and oneg == "0": + return 2 + elif opos == "1" and oneg == "1": + return 3 + else: + pass + + +def sa_doc_to_choice(x): + """ + Function to return the choices from the dataset for sentiment analysis + """ + return ["Positivo", "Negativo", "Neutrale", "Misto"] + + +# ---------------------- LEXICAL SUBSTITUTION ---------------------- +NO_SYN_STRING = "&&NOSYN&&" + + +def _ls_gold_to_target(x): + """ + Generate the target for the lexical similarity task + """ + # all_answers = [(i["word"], i["count"]) for i in x["answers"]] + if len(x["answers"]) == 0: + return NO_SYN_STRING + ans_str = "" + for i in x["answers"]: + ans_str += i["word"] + "$$" + str(i["count"]) + "::" + if len(ans_str) != 0 and ans_str[-2] == ":": + ans_str = ans_str[:-2] + # print(ans_str) + + return ans_str + + +def ls_doc_to_target(x): + """ + Generate the target for the lexical similarity task + """ + if len(x["answers"]) == 0: + return NO_SYN_STRING + ans_str = "" + for i in x["answers"]: + ans_str += i["word"] + ", " + if len(ans_str) != 0 and ans_str[-2] == ",": + ans_str = ans_str[:-2] + return ans_str + + +def _ls_split_gold(x): + """ + Split the gold string into a list of tuples + """ + if x == NO_SYN_STRING: + return [], [] + answers = x.split("::") + words = [] + freqs = [] + if len(answers) != 0: + for a in answers: + if "$$" in a: + word, count = a.split("$$") + words.append(word) + try: + freqs.append(int(count)) + except ValueError: + freqs.append(0) + return words, freqs + + +def ls_process_results(doc, results): + """ + Process the results of the evaluation for the lexical substitution task + look at coqa for another example + """ + gold_to_target = _ls_gold_to_target(doc) + words, freqs = _ls_split_gold(gold_to_target) + prec = 0 + + # Considering a maximum of the first 10 synonyms + results = split_text_with_regex(results[0], LS_SPLIT_REGEX) + results = results[: min(10, len(results))] + + # Remove non-alphabetic characters from the word at the end of the list + if results: # Check if results is not empty + results[-1] = "".join(char for char in results[-1] if char.isalpha()) + + has_answ = 0 if len(results) == 0 else 1 # so we can compute |A| + has_annotation = 0 if len(words) == 0 else 1 # so we can compute |T| + + matching_res = [] # for debugging + + for r in results: + if r in words: + # get frequency of the synonyms from annotators + idx = words.index(r.strip()) + prec += freqs[idx] + matching_res.append(r) + + # In the case of the OOT (out of ten) subtask, this normalization should not be applied + # ai = len(results) if len(results) != 0 else 1 + # prec = prec / ai + + Hi = sum(freqs) + if Hi != 0: + prec = prec / Hi + else: + eval_logger.debug("H_i is 0") + + return {"f1": (prec, has_answ, has_annotation)} + + +# ---------------------- NER ---------------------- + +NO_ENT_STRING = "&&NOENT&&" +NER_ENTITY_SEPARATOR = "," +NER_TYPE_SEPARATOR = "$" +NER_MAPPING_V2 = {"PER": 0, "LOC": 1, "ORG": 2, NO_ENT_STRING: 3, "O": 4} +NER_MAPPING = {"PER": 0, "LOC": 1, "ORG": 2, "O": 3} + + +def _ner_gold_to_target(x: list) -> list: + """ + Convert the gold entities to the target format according to the NER_MAPPING + """ + res = [NER_MAPPING[e["type"]] for e in x] + return res + + +def _ner_gold_to_target_v2(x: list) -> list: + """ + Convert the gold entities to the target format according to the NER_MAPPING + """ + res = [NER_MAPPING[e["type"]] for e in x] + return res + + +def ner_doc_to_target(doc): + ents = doc["entities"] + targ_str = "" + # Entità$Tipo%Entità$Tipo. + if ents == []: + return NO_ENT_STRING + else: + for e in ents: + targ_str += ( + e["entity_text"] + NER_TYPE_SEPARATOR + e["type"] + NER_ENTITY_SEPARATOR + ) + return targ_str[:-1] + + +def ner_process_results(doc, results): + """ + Process the results of the Named Entity Recognition task + """ + # each document has a list of entities with the following format: + # [{"entity_text": "string", "type": "string"}] + gold = doc["entities"] + raw_results = results[0] + results = _ner_process_raw_output(raw_results) + + gold_labels = _ner_gold_to_target(gold) + res_labels = [0] * len(gold_labels) + matched_gold_idx = [] + + if len(results) > len(gold): + for r in results: + r_text = r[0] + r_type = r[1] + for i in range(len(gold)): + if r_text == gold[i]["entity_text"] and r_type == gold[i]["type"]: + res_labels[i] = NER_MAPPING[r_type] + matched_gold_idx.append(i) + # Since we have more results than gold, we artificially set to false positive the remaining labels + # extend gold label list + for i in range(len(results) - len(gold)): + gold_labels.append(3) + res_labels.append(2) + elif len(results) == 0 and len(gold) == 0: + res_labels = [3] + gold_labels = res_labels + else: # len(results) <= len(gold) + for r in results: + r_text = r[0] + r_type = r[1] + for i in range(len(gold)): + if r_text == gold[i]["entity_text"] and r_type == gold[i]["type"]: + res_labels[i] = NER_MAPPING[r_type] + matched_gold_idx.append(i) + # we map all wrong predictions to the "O" class + for i in range(len(gold_labels)): + if i in matched_gold_idx: + continue + if gold_labels[i] == 1: + res_labels[i] = 3 + elif gold_labels[i] == 0: + res_labels[i] = 3 + else: + res_labels[i] = 3 + + assert len(gold_labels) == len(res_labels) + return {"f1": (res_labels, gold_labels)} + + +def ner_process_results_v2(doc, results): + """ + Process the results of the Named Entity Recognition task + This version considers and score explicitly when the model responds that there are no entities + """ + # each document has a list of entities with the following format: + # [{"entity_text": "string", "type": "string"}] + gold = doc["entities"] + raw_results = results[0] + results = _ner_process_raw_output_v2(raw_results) + + # eval_logger.debug(f"results {results}") + # eval_logger.debug(f"gold {gold}") + + gold_labels = _ner_gold_to_target_v2(gold) + res_labels = [0] * len(gold_labels) + matched_gold_idx = [] + + if len(results) > len(gold): + for r in results: + # print(r) + r_text = r[0] + r_type = r[1] + for i in range(len(gold)): + if r_text == gold[i]["entity_text"] and r_type == gold[i]["type"]: + res_labels[i] = NER_MAPPING[r_type] + matched_gold_idx.append(i) + # Since we have more results than gold, we artificially set to false positive the remaining labels + # extend gold label list + for i in range(len(results) - len(gold)): + # gold_labels.append(3) + # res_labels.append(2) + gold_labels.append(4) + res_labels.append(3) + elif len(results) == 0 and len(gold) == 0: + # res_labels = [random.choice([0, 1, 2, 3])] + res_labels = [3] + gold_labels = res_labels + elif len(results) == 1 and results[0] == NO_ENT_STRING: + # res_labels = [3] + res_labels = [4] + gold_labels = res_labels + else: # len(results) <= len(gold) + for r in results: + r_text = r[0] + r_type = r[1] + for i in range(len(gold)): + if r_text == gold[i]["entity_text"] and r_type == gold[i]["type"]: + res_labels[i] = NER_MAPPING[r_type] + matched_gold_idx.append(i) + # we map all wrong predictions to the "O" class + for i in range(len(gold_labels)): + if i in matched_gold_idx: + continue + if gold_labels[i] == 1: + # res_labels[i] = 2 + res_labels[i] = 4 + elif gold_labels[i] == 0: + # res_labels[i] = 1 + res_labels[i] = 4 + else: + res_labels[i] = 4 + + assert len(gold_labels) == len(res_labels) + return {"f1": (res_labels, gold_labels)} + + +def _ner_process_raw_output(llm_result: str) -> list[tuple]: + if NO_ENT_STRING in llm_result: + return [] + if llm_result == "": + return ["WRONG"] + tmp_results = llm_result.split(NER_ENTITY_SEPARATOR) + results = [] + for res in tmp_results: + r = res.strip() + # split on type separator + r_text = "" + r_type = "" + r_splitted = r.split(NER_TYPE_SEPARATOR) + if len(r_splitted) < 2: + r_text = r_splitted[0] + r_type = "" + else: + r_text = r_splitted[0] + r_type = r_splitted[1] + if r_text != "": + results.append((r_text, r_type.upper())) + return results + + +def _ner_process_raw_output_v2(llm_result: str) -> list[tuple]: + if NO_ENT_STRING in llm_result: + return [NO_ENT_STRING] + if llm_result == "": + return ["WRONG"] + tmp_results = llm_result.split(NER_ENTITY_SEPARATOR) + results = [] + for res in tmp_results: + r = res.strip() + # split on type separator + r_text = "" + r_type = "" + r_splitted = r.split(NER_TYPE_SEPARATOR) + if len(r_splitted) < 2: + r_text = r_splitted[0] + r_type = "" + else: + r_text = r_splitted[0] + r_type = r_splitted[1] + if r_text != "": + results.append((r_text, r_type.upper())) + return results + + +# ---------------------- RELATION EXTRACTION ---------------------- + + +def _rel_process_raw_output(llm_result: str) -> list[str]: + if NO_REL_STRING in llm_result: + return [] + if llm_result == "": + return ["WRONG"] + tmp_results = llm_result.split(INTER_REL_SEPARATOR) + relations = [] + for res in tmp_results: + r_text1 = "" + r_text2 = "" + r_splitted = res.split(INTRA_REL_SEPARATOR) + if len(r_splitted) < 2: + r_text1 = r_splitted[0].strip() + r_text2 = "" + else: + r_text1 = r_splitted[0].strip() + r_text2 = r_splitted[1].strip() + relations.append((r_text1, r_text2)) + assert len(relations) == len(tmp_results) + return relations + + +INTER_REL_SEPARATOR = "%" +INTRA_REL_SEPARATOR = "$" +NO_REL_STRING = "&&NOREL&&" + + +def re_doc_to_target(doc): + ents = doc["relations"] + targ_str = "" + # Entità$Tipo%Entità$Tipo. + if ents == []: + return NO_ENT_STRING + else: + for e in ents: + targ_str += e[0] + INTRA_REL_SEPARATOR + e[1] + INTER_REL_SEPARATOR + return targ_str[:-1] + + +def _rel_gold_to_target(x: list) -> list: + if x == []: + return [0] + else: + return [1] * len(x) + + +def rel_doc_to_target(doc): + rel = doc["relations"] + targ_str = "" + # misura1$result1%misure2$result2. + if rel == []: + return NO_REL_STRING + else: + for r in rel: + targ_str += r[0] + "$" + r[1] + "%" + return targ_str[:-1] + + +def _extract_relations(results): + relations = [] + for r in results: + r_text1 = "" + r_text2 = "" + r_splitted = r.split(INTRA_REL_SEPARATOR) + if len(r_splitted) < 2: + r_text1 = r_splitted[0] + r_text2 = "" + else: + r_text1 = r_splitted[0] + r_text2 = r_splitted[1] + relations.append((r_text1, r_text2)) + assert len(relations) == len(results) + return relations + + +def rel_process_results_v3(doc, results): + """ + Process the results of the Relation extraction task not considering the order of the relation extracted + """ + # each document has a list of relation with the following format: + # [[text1, text2], [text3, text4]] + gold = doc["relations"] + raw_results = results[0] + has_results = 0 if NO_REL_STRING in raw_results else 1 + has_gold = 1 if gold != [] else 0 + + res_labels = [] + gold_labels = [] + + if has_results == 0 and has_gold: + # False negative + gold_labels = _rel_gold_to_target(gold) + res_labels = [0] * len(gold_labels) + elif has_results == 0 and has_gold == 0: + # True negative + gold_labels = _rel_gold_to_target(gold) + res_labels = gold_labels + elif has_results and has_gold == 0: + # False positive + gold_labels = _rel_gold_to_target(gold) + res_labels = [1] * len(gold_labels) + else: + results = _rel_process_raw_output(raw_results) + # results = raw_results.split(INTER_REL_SEPARATOR) + gold_labels = _rel_gold_to_target(gold) + res_labels = [0] * len(gold_labels) + assert len(gold) > 0 + for i in range(len(gold)): + for j in range(len(results)): + r_text1 = results[j][0] + r_text2 = results[j][1] + + if r_text1 == gold[i][0] and r_text2 == gold[i][1]: # list of lists + res_labels[i] = 1 + results[j] = ("DELETED", "DELETED") + elif r_text1 == "DELETED" and r_text2 == "DELETED": + continue + else: + pass + # if there are more predictions than gold, we set the remaining predictions to false positive + if len(results) - len(gold) > 0: + for i in range(len(results) - len(gold)): + if results[i] == ("DELETED", "DELETED"): + continue + res_labels.append(1) + gold_labels.append(0) + + assert len(gold_labels) == len(res_labels) + return {"f1": (res_labels, gold_labels)} + + +LS_SPLIT_REGEX = r"[^,]+" + + +def split_text_with_regex(text, pattern): + """ + pattern: str - a regex pattern to match the text + text: str - the text to split + """ + import re + + # Get text with model-generated words for comparison with the gold standard + text = text.split("\n")[0] + + # Find all matches for the pattern + matches = re.findall(pattern, text) + # Split each matched segment further if it contains a comma and is quoted + result = [] + for match in matches: + if match.startswith('"') and match.endswith('"'): + # Remove the quotes and split inside the quoted string + inner_matches = re.findall(r"[^,]+", match[1:-1]) + result.extend(inner_matches) + else: + result.append(match) + + # Strip leading and trailing whitespaces from each element + result = [element.strip().replace('"', "") for element in result] + + return result + + +def faq_doc_to_target(x): + if x["correct_answer"] == "A": + return 0 + elif x["correct_answer"] == "B": + return 1 + elif x["correct_answer"] == "C": + return 2 + elif x["correct_answer"] == "D": + return 3 + else: + eval_logger.warning( + 'WARNING: correct answer not found or not in ["A", "B", "C", "D"]' + ) + + +def ht_doc_to_target(x): + if x["source"] == "ilgiornale": + return 0 + elif x["source"] == "repubblica": + return 1 + else: + eval_logger.warning( + 'WARNING: source not found or not in ["ilgiornale", "repubblica"]' + ) diff --git a/lm-evaluation-harness/lm_eval/tasks/fda/README.md b/lm-evaluation-harness/lm_eval/tasks/fda/README.md new file mode 100644 index 0000000000000000000000000000000000000000..196fc5b935d6c2412813ce02ca869dc6c89b3624 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/fda/README.md @@ -0,0 +1,78 @@ +# FDA + +### Paper + +Title: Language Models Enable Simple Systems For +Generating Structured Views Of Heterogenous Data +Lakes + +Abstract: A long standing goal of the data management community is to develop general, automated systems +that ingest semi-structured documents and output queryable tables without human effort or domain +specific customization. Given the sheer variety of potential documents, state-of-the art systems make +simplifying assumptions and use domain specific training. In this work, we ask whether we can +maintain generality by using large language models (LLMs). LLMs, which are pretrained on broad +data, can perform diverse downstream tasks simply conditioned on natural language task descriptions. +We propose and evaluate EVAPORATE, a simple, prototype system powered by LLMs. We identify +two fundamentally different strategies for implementing this system: prompt the LLM to directly +extract values from documents or prompt the LLM to synthesize code that performs the extraction. +Our evaluations show a cost-quality tradeoff between these two approaches. Code synthesis is cheap, +but far less accurate than directly processing each document with the LLM. To improve quality while +maintaining low cost, we propose an extended code synthesis implementation, EVAPORATE-CODE+, +which achieves better quality than direct extraction. Our key insight is to generate many candidate +functions and ensemble their extractions using weak supervision. EVAPORATE-CODE+ not only +outperforms the state-of-the art systems, but does so using a sublinear pass over the documents with +the LLM. This equates to a 110× reduction in the number of tokens the LLM needs to process, +averaged across 16 real-world evaluation settings of 10k documents each. + + +A task for LMs to perform Information Extraction, as implemented by Based. + +Homepage: https://github.com/HazyResearch/based-evaluation-harness + + +Description: +> FDA (Information Extraction). The task is to extract key-value pairs from a set of PDFs scraped from the FDA website. We use the dataset and labels collected in Arora et al. 2023. We break apart the documents into chunks of 1,920 tokens. For every key-value pair that appears in the chunk, we create a zero-shot prompt using the simple prompt template: {chunk} \n {key}: We allow the model to generate a fixed number of tokens after the prompt and check (with case insensitivity) if the value is contained within the generation. We report accuracy, the fraction of prompts for which the generation contains the value. + + + +### Citation + +``` +@misc{arora2024simple, + title={Simple linear attention language models balance the recall-throughput tradeoff}, + author={Simran Arora and Sabri Eyuboglu and Michael Zhang and Aman Timalsina and Silas Alberti and Dylan Zinsley and James Zou and Atri Rudra and Christopher Ré}, + year={2024}, + eprint={2402.18668}, + archivePrefix={arXiv}, + primaryClass={cs.CL} +} + +@misc{arora2023language, + title={Language Models Enable Simple Systems for Generating Structured Views of Heterogeneous Data Lakes}, + author={Simran Arora and Brandon Yang and Sabri Eyuboglu and Avanika Narayan and Andrew Hojel and Immanuel Trummer and Christopher Ré}, + year={2023}, + eprint={2304.09433}, + archivePrefix={arXiv}, + primaryClass={cs.CL} +} + +``` + +### Groups and Tasks + +#### Tasks + +* `fda`: the FDA task as implemented in the paper "Simple linear attention language models balance the recall-throughput tradeoff". Designed for zero-shot evaluation of small LMs. + +### Checklist + +For adding novel benchmarks/datasets to the library: +* [x] Is the task an existing benchmark in the literature? + * [x] Have you referenced the original paper that introduced the task? + * [x] If yes, does the original paper provide a reference implementation? If so, have you checked against the reference implementation and documented how to run such a test? + + +If other tasks on this dataset are already supported: +* [x] Is the "Main" variant of this task clearly denoted? +* [x] Have you provided a short sentence in a README on what each new variant adds / evaluates? +* [x] Have you noted which, if any, published evaluation setups are matched by this variant? diff --git a/lm-evaluation-harness/lm_eval/tasks/fda/fda.yaml b/lm-evaluation-harness/lm_eval/tasks/fda/fda.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ac99e90dc4014aa7484786305cb7ae96c505cc99 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/fda/fda.yaml @@ -0,0 +1,2 @@ +task: fda +class: !function task.FDA diff --git a/lm-evaluation-harness/lm_eval/tasks/fda/task.py b/lm-evaluation-harness/lm_eval/tasks/fda/task.py new file mode 100644 index 0000000000000000000000000000000000000000..a82618419bb7dec67c90fc88a255fe330e43e618 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/fda/task.py @@ -0,0 +1,100 @@ +import re +from typing import List + +import numpy as np + +from lm_eval.api.instance import Instance +from lm_eval.api.task import ConfigurableTask + + +class FDA(ConfigurableTask): + VERSION = 0 + DATASET_PATH = "hazyresearch/based-fda" + DATASET_NAME = "default" + + def __init__(self, **kwargs): + super().__init__(config={"metadata": {"version": self.VERSION}}) + + def has_training_docs(self): + return False + + def has_validation_docs(self): + return True + + def has_test_docs(self): + return False + + def validation_docs(self): + return self.dataset["validation"] + + def doc_to_text(self, doc): + return doc["text"] + + def doc_to_target(self, doc): + return doc["value"] + + def construct_requests( + self, doc, ctx, chat_template=None, apply_chat_template=False, **kwargs + ): + """Uses RequestFactory to construct Requests and returns an iterable of + Requests which will be sent to the LM. + + :param doc: + The document as returned from training_docs, validation_docs, or test_docs. + :param ctx: str + The context string, generated by fewshot_context. This includes the natural + language description, as well as the few shot examples, and the question + part of the document for `doc`. + """ + + return [ + Instance( + request_type="generate_until", + doc=doc, + arguments=(ctx, {"until": ["\n"], "max_gen_toks": 48}), + idx=0, + **kwargs, + ) + ] + + def process_results(self, doc, results): + """Take a single document and the LM results and evaluates, returning a + dict where keys are the names of submetrics and values are the values of + the metric for that one document + + :param doc: + The document as returned from training_docs, validation_docs, or test_docs. + :param results: + The results of the requests created in construct_requests. + """ + # continuation, (logprob_unanswerable, _) = results + continuation = results + + return {"contains": contains_score(continuation[0], [doc["value"]])} + + def aggregation(self): + """ + :returns: {str: [float] -> float} + A dictionary where keys are the names of submetrics and values are + functions that aggregate a list of metrics + """ + return { + "contains": np.mean, # Exact match (the normalized answer exactly match the gold answer) + } + + def higher_is_better(self): + """ + :returns: {str: bool} + A dictionary where keys are the names of submetrics and values are + whether a higher value of the submetric is better + """ + return { + "contains": True, # Exact match (the normalized answer exactly match the gold answer + } + + +def contains_score(prediction: str, labels: List[str]): + return max( + int(bool(re.search(re.compile(re.escape(label), re.IGNORECASE), prediction))) + for label in labels + ) diff --git a/lm-evaluation-harness/lm_eval/tasks/fld/README.md b/lm-evaluation-harness/lm_eval/tasks/fld/README.md new file mode 100644 index 0000000000000000000000000000000000000000..06ccf8f606fe9e0175b3e3b48eb3b45193f2c0c3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/fld/README.md @@ -0,0 +1,65 @@ +# FLD + +### Paper + +Title: Learning Deductive Reasoning from Synthetic Corpus based on Formal Logic + +Abstract: https://arxiv.org/abs/2308.07336 + +**FLD** (**F**ormal **L**ogic **D**eduction) is a deductive reasoning benchmark. +Given a set of facts and a hypothesis, an LLM is required to generate (i) proof steps to (dis-)prove the hypothesis, and (ii) an answer ("proved", "disproved" or unknown"). + +Unique features of FLD are: +* It assesses the model's logical reasoning ability *isolated from knowledge*, as the facts are randomly constructed so that referring to existing knowledge never helps solve the task. +* It assesses diverse reasoning patterns (i.e., deduction rules), as it is based on formal logic theory. +* As a result, it is highly challenging. Indeed, even GPT-4 can solve only about half of the problems. + +Homepage: https://github.com/hitachi-nlp/FLD + + +### Citation + +``` +@InProceedings{pmlr-v202-morishita23a, + title = {Learning Deductive Reasoning from Synthetic Corpus based on Formal Logic}, + author = {Morishita, Terufumi and Morio, Gaku and Yamaguchi, Atsuki and Sogawa, Yasuhiro}, + booktitle = {Proceedings of the 40th International Conference on Machine Learning}, + pages = {25254--25274}, + year = {2023}, + editor = {Krause, Andreas and Brunskill, Emma and Cho, Kyunghyun and Engelhardt, Barbara and Sabato, Sivan and Scarlett, Jonathan}, + volume = {202}, + series = {Proceedings of Machine Learning Research}, + month = {23--29 Jul}, + publisher = {PMLR}, + pdf = {https://proceedings.mlr.press/v202/morishita23a/morishita23a.pdf}, + url = {https://proceedings.mlr.press/v202/morishita23a.html}, +} +``` + +### Groups and Tasks + +This release is the simplified version of FLD where a model is required to predict only an answer. +This setting is described by "answer accuracy" in the original paper. + +#### Tasks in Group `fld` +* `fld_default` is a basic task based on [FLD.v2](https://huggingface.co/datasets/hitachi-nlp/FLD.v2/viewer/star) +* `fld_star`: is a more challenging version based on [FLD.v2-star](https://huggingface.co/datasets/hitachi-nlp/FLD.v2/viewer/star) + +#### Tasks in Group `fld_logical_formula` +Further, we have "logical formula" versions of the benchmarks, which evaluate LLMs' pure logical reasoning capabilities within the domain of logical formulas, rather than natural language: +* `fld_logical_formula_default` +* `fld_logical_formula_fld_star` + + +### Checklist + +For adding novel benchmarks/datasets to the library: +* [x] Is the task an existing benchmark in the literature? + * [x] Have you referenced the original paper that introduced the task? + * [x] If yes, does the original paper provide a reference implementation? If so, have you checked against the reference implementation and documented how to run such a test? + + +If other tasks on this dataset are already supported: +* [ ] Is the "Main" variant of this task clearly denoted? +* [ ] Have you provided a short sentence in a README on what each new variant adds / evaluates? +* [ ] Have you noted which, if any, published evaluation setups are matched by this variant? diff --git a/lm-evaluation-harness/lm_eval/tasks/fld/fld_default.yaml b/lm-evaluation-harness/lm_eval/tasks/fld/fld_default.yaml new file mode 100644 index 0000000000000000000000000000000000000000..460a9ec6dbf52be9819891ed27ae59b62162c75d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/fld/fld_default.yaml @@ -0,0 +1,19 @@ +task: fld_default +dataset_path: hitachi-nlp/FLD.v2 +dataset_name: default +training_split: train +validation_split: validation +test_split: test +doc_to_text: "Based on the provided facts ($context$), either prove or disprove the hypothesis or state that it is unknown. {{prompt_serial}}" +doc_to_target: world_assump_label +metric_list: + - metric: exact_match + aggregation: mean + higher_is_better: true +filter_list: + - name: remove_whitespace + filter: + - function: remove_whitespace + - function: take_first +metadata: + version: 2.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/fld/fld_logical_formula_default.yaml b/lm-evaluation-harness/lm_eval/tasks/fld/fld_logical_formula_default.yaml new file mode 100644 index 0000000000000000000000000000000000000000..67ff4acfe80757b7b53a4a5fdd3c7b388f751770 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/fld/fld_logical_formula_default.yaml @@ -0,0 +1,21 @@ +group: + - fld_logical_formula +task: fld_logical_formula_default +dataset_path: hitachi-nlp/FLD.v2 +dataset_name: default +training_split: train +validation_split: validation +test_split: test +doc_to_text: "Based on the provided facts ($context$), either prove or disprove the hypothesis or state that it is unknown. The facts and the hypothesis are written in logical formulas as follows: capital letters such as \"{A}\", \"{B}\", \"{AB}\" are predicates, small letters such as \"{a}\", \"{b}\", \"{ab}\" are constants, \"&\" is logical conjunction, \"v\" is logical disjunction, \"¬\" is negation, \"->\" is implication, \"(x)\" is \"for all x\", and \"(Ex)\" is \"for some x\". $hypothesis$ = {{hypothesis_formula}} ; $context$ = {{context_formula}} ; $proof$ = " +doc_to_target: world_assump_label +metric_list: + - metric: exact_match + aggregation: mean + higher_is_better: true +filter_list: + - name: remove_whitespace + filter: + - function: remove_whitespace + - function: take_first +metadata: + version: 2.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/fld/fld_logical_formula_star.yaml b/lm-evaluation-harness/lm_eval/tasks/fld/fld_logical_formula_star.yaml new file mode 100644 index 0000000000000000000000000000000000000000..28aee616f420ef62844febef6626771029f04500 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/fld/fld_logical_formula_star.yaml @@ -0,0 +1,3 @@ +include: fld_logical_formula_default.yaml +task: fld_logical_formula_star +dataset_name: star diff --git a/lm-evaluation-harness/lm_eval/tasks/fld/fld_star.yaml b/lm-evaluation-harness/lm_eval/tasks/fld/fld_star.yaml new file mode 100644 index 0000000000000000000000000000000000000000..750e808c780001e4659c9def75400f8a2460045e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/fld/fld_star.yaml @@ -0,0 +1,3 @@ +include: fld_default.yaml +task: fld_star +dataset_name: star diff --git a/lm-evaluation-harness/lm_eval/tasks/french_bench/README.md b/lm-evaluation-harness/lm_eval/tasks/french_bench/README.md new file mode 100644 index 0000000000000000000000000000000000000000..693f60c3a3efda911fbfa0be9a7c64ce55fa22b1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/french_bench/README.md @@ -0,0 +1,94 @@ +# FrenchBench + +### Paper + +FrenchBench is a benchmark for evaluating French language models, introduced in the paper +[CroissantLLM: A Truly Bilingual French-English Language Model](https://arxiv.org/abs/2402.00786). +It is a collection of tasks that evaluate the ability of a language model to understand and generate French text. +This benchmark is constructed both from openly available datasets, as well as newly released manually annotated data. + +### Citation + +```bibtex +@misc{faysse2024croissantllm, + title={CroissantLLM: A Truly Bilingual French-English Language Model}, + author={Manuel Faysse and Patrick Fernandes and Nuno M. Guerreiro and António Loison and Duarte M. Alves and Caio Corro and Nicolas Boizard and João Alves and Ricardo Rei and Pedro H. Martins and Antoni Bigata Casademunt and François Yvon and André F. T. Martins and Gautier Viaud and Céline Hudelot and Pierre Colombo}, + year={2024}, + eprint={2402.00786}, + archivePrefix={arXiv}, + primaryClass={cs.CL} +} +``` + +### Groups, Tags, and Tasks + +#### Tags + +- `french_bench`: All tasks (non-perplexity based) +- `french_bench_gen`: All official generative tasks +- `french_bench_mc`: All official multiple choice tasks +- `french_bench_perplexity`: All perplexity-based tasks (0 shot is recommended) +- `french_bench_extra`: All extra tasks + +#### Tasks + + +The following tasks evaluate tasks on the French Bench dataset using various scoring methods. + - french_bench_boolqa + - french_bench_fquadv2 + - french_bench_fquadv2_bool + - french_bench_fquadv2_genq + - french_bench_fquadv2_hasAns + - french_bench_topic_based_nli + - french_bench_multifquad + - french_bench_grammar + - french_bench_vocab + - french_bench_reading_comp + - french_bench_xnli (modified XNLI) + - french_bench_orangesum_abstract + - french_bench_orangesum_title + - french_bench_trivia + - french_bench_hellaswag + - french_bench_arc_challenge + +The french bench also includes other tasks from various benchmarks: +- `belebele_fra_Latn`: Belebele French +- `wmt14-en-fr`: WMT14 English-French +- `wmt14-fr-en`: WMT14 French-English + +# Not to use in few-shot +- `crows_pairs_french`: Crows Pairs French +- `french_bench_opus_perplexity`: Opus Perplexity + + +### Usage + +```bash +# openai +lm_eval --model openai-completions --model_args engine=text-davinci-003 --tasks french_bench --limit 100 --num_fewshot 3 --batch_size auto --output_path data/french_bench/davinci-003/results_french_bench_3shot.json +lm_eval --model openai-completions --model_args engine=text-davinci-003 --tasks french_bench_opus_perplexity,crows_pairs_french --limit 100 --batch_size auto --output_path data/french_bench/davinci-003/results_french_bench2_0shot.json + + +lm_eval --model hf --model_args pretrained=gpt2 --tasks french_bench --device cuda:0 --limit 100 --num_fewshot 3 --batch_size 8 --output_path data/french_bench/gpt2/results_french_bench_3shot.json +lm_eval --model hf --model_args pretrained=gpt2 --tasks french_bench_opus_perplexity,crows_pairs_french --device cuda:0 --limit 100 --batch_size auto --output_path data/french_bench/gpt2/results_french_bench2_0shot.json + +lm_eval --model hf --model_args pretrained=meta-llama/Llama-2-7b-hf --tasks french_bench --device cuda:0 --limit 100 --num_fewshot 3 --batch_size 4 --output_path data/french_bench/llama-2-7b-hf/results_french_bench_3shot.json +lm_eval --model hf --model_args pretrained=meta-llama/Llama-2-7b-hf --tasks french_bench_opus_perplexity,crows_pairs_french --device cuda:0 --limit 100 --batch_size auto --output_path data/french_bench/llama-2-7b-hf/results_french_bench2_0shot.json +``` + +HF and Accelerate options can be added when loading a model: +```bash + accelerate launch -m lm_eval --model hf --model_args pretrained=meta-llama/Llama-2-7b-hf,dtype="float16" --tasks french_bench +``` + +### Checklist + +* [x] Is the task an existing benchmark in the literature? + * [x] Have you referenced the original paper that introduced the task? + * [x] If yes, does the original paper provide a reference implementation? + * [x] Yes, original implementation contributed by author of the benchmark + +If other tasks on this dataset are already supported: +* [x] Is the "Main" variant of this task clearly denoted? +* [x] Have you provided a short sentence in a README on what each new variant adds / evaluates? +* [x] Have you noted which, if any, published evaluation setups are matched by this variant? diff --git a/lm-evaluation-harness/lm_eval/tasks/french_bench/_default_template_yaml b/lm-evaluation-harness/lm_eval/tasks/french_bench/_default_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..ae3bfd1fc8d2974288922e55a7ec5d55054a90d4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/french_bench/_default_template_yaml @@ -0,0 +1,4 @@ +test_split: test +fewshot_split: valid +fewshot_config: + sampler: first_n diff --git a/lm-evaluation-harness/lm_eval/tasks/french_bench/french_bench_arc_challenge.yaml b/lm-evaluation-harness/lm_eval/tasks/french_bench/french_bench_arc_challenge.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7884b0dc9cd9639d4a67cff0086b44978e84b14a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/french_bench/french_bench_arc_challenge.yaml @@ -0,0 +1,21 @@ +tag: + - french_bench + - french_bench_mc +task: french_bench_arc_challenge +dataset_path: manu/french_bench_arc_challenge +output_type: multiple_choice +training_split: train +validation_split: validation +test_split: test +doc_to_text: "Question: {{question}}\nRéponse:" +doc_to_target: "{{['A', 'B', 'C', 'D'].index(answerKey)}}" +doc_to_choice: "{{choices}}" +should_decontaminate: true +doc_to_decontamination_query: "Question: {{question}}\nRéponse:" +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true diff --git a/lm-evaluation-harness/lm_eval/tasks/french_bench/french_bench_boolqa.yaml b/lm-evaluation-harness/lm_eval/tasks/french_bench/french_bench_boolqa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bdd60e5d0e78fc519ae08c97e4cdcb6986b04d8c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/french_bench/french_bench_boolqa.yaml @@ -0,0 +1,23 @@ +include: "_default_template_yaml" +tag: + - french_bench + - french_bench_extra +description: "D'après l'information dans le contexte donné, quelle est la réponse à la question ?" +task: french_bench_boolqa +dataset_path: manu/french_boolq +output_type: multiple_choice +validation_split: valid +doc_to_text: "\nContexte: {{passage}}\n\nQuestion: {{question}}\n" +doc_to_choice: ["Oui", "Non"] +# doc_to_text: "\nContexte: {{passage}}\n\nQuestion: {{question}}\n\nD'après l'information dans le contexte, la réponse est:\nA. Oui \nB. Non\n\nRéponse:" +# doc_to_choice: ["A", "B"] +doc_to_target: "{{[1, 0].index(label)}}" +should_decontaminate: true +doc_to_decontamination_query: passage +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true diff --git a/lm-evaluation-harness/lm_eval/tasks/french_bench/french_bench_fquadv2.yaml b/lm-evaluation-harness/lm_eval/tasks/french_bench/french_bench_fquadv2.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2e49ec43c185951786dc8a2ca60f80c71ed6ac25 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/french_bench/french_bench_fquadv2.yaml @@ -0,0 +1,29 @@ +include: "_default_template_yaml" +tag: + - french_bench + - french_bench_extra +description: "D'après l'information dans le contexte donné, donne la réponse à la question en citant quelques mots du contexte. Si il est impossible de répondre avec les informations du contexte, répond 'Impossible'." +task: french_bench_fquadv2 +dataset_path: manu/fquad2_test +output_type: generate_until +validation_split: valid +doc_to_text: "\nContexte: {{context}}\n\nQuestion: {{question}}\n\nRéponse:" +doc_to_target: "{% if answers.text| length > 0 %}{{answers.text[0]}}{% else %}{{['Impossible']}}{% endif %}" +target_delimiter: " " +should_decontaminate: true +doc_to_decontamination_query: context +generation_kwargs: + until: + - "\n" +# filter_list: +# - name: remove_whitespace +# filter: +# - function: remove_whitespace +# - function: take_first +metric_list: + - metric: !function utils.exact + aggregation: mean + higher_is_better: true + - metric: !function utils.f1 + aggregation: mean + higher_is_better: true diff --git a/lm-evaluation-harness/lm_eval/tasks/french_bench/french_bench_fquadv2_bool.yaml b/lm-evaluation-harness/lm_eval/tasks/french_bench/french_bench_fquadv2_bool.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4e07d2ec0d28505bceec367920e0a56617c6af45 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/french_bench/french_bench_fquadv2_bool.yaml @@ -0,0 +1,21 @@ +include: "_default_template_yaml" +tag: + - french_bench + - french_bench_extra +description: "D'après l'information présente dans le contexte, est il possible de répondre à la question ?" +task: french_bench_fquadv2_bool +dataset_path: manu/fquad2_test +output_type: multiple_choice +validation_split: valid +doc_to_text: "\nContexte: {{context}}\n\nQuestion: {{question}}\n\nD'après l'information présente dans le contexte, répondre à la question est:\nA. Possible \nB. Impossible\n\nRéponse:" +doc_to_choice: ["A", "B"] +doc_to_target: "{{[False, True].index(is_impossible)}}" +should_decontaminate: true +doc_to_decontamination_query: context +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true diff --git a/lm-evaluation-harness/lm_eval/tasks/french_bench/french_bench_fquadv2_genq.yaml b/lm-evaluation-harness/lm_eval/tasks/french_bench/french_bench_fquadv2_genq.yaml new file mode 100644 index 0000000000000000000000000000000000000000..380518520326753402e265f453f99ed6b1e1043d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/french_bench/french_bench_fquadv2_genq.yaml @@ -0,0 +1,31 @@ +include: "_default_template_yaml" +tag: + - french_bench + - french_bench_gen +description: "D'après l'information dans le contexte donné, quelle question a été posée pour obtenir la réponse donnée ?" +task: french_bench_fquadv2_genq +dataset_path: manu/fquad2_test +output_type: generate_until +validation_split: valid_hasAns +test_split: test_hasAns +fewshot_split: valid_hasAns +doc_to_text: "\nContexte: {{context}}\n\nRéponse: {% if answers.text| length > 0 %}{{answers.text[0]}}{% else %}{{['Impossible']}}{% endif %}\n\nQuestion:" +doc_to_target: "{{question}}" +target_delimiter: " " +should_decontaminate: true +doc_to_decontamination_query: question +generation_kwargs: + until: + - "\n" +# filter_list: +# - name: remove_whitespace +# filter: +# - function: remove_whitespace +# - function: take_first +metric_list: + - metric: !function utils.rouge1 + higher_is_better: true + aggregation: !function utils.rouge1_agg + - metric: !function utils.f1 + aggregation: mean + higher_is_better: true diff --git a/lm-evaluation-harness/lm_eval/tasks/french_bench/french_bench_fquadv2_hasAns.yaml b/lm-evaluation-harness/lm_eval/tasks/french_bench/french_bench_fquadv2_hasAns.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6eedbabb5f5e3c1a381c43d20a26ec7ce3a1d103 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/french_bench/french_bench_fquadv2_hasAns.yaml @@ -0,0 +1,34 @@ +include: "_default_template_yaml" +tag: + - french_bench + - french_bench_gen +description: "D'après l'information dans le contexte donné, donne la réponse à la question en citant quelques mots du contexte. Si il est impossible de répondre avec les informations du contexte, répond 'Impossible'." +task: french_bench_fquadv2_hasAns +dataset_path: manu/fquad2_test +output_type: generate_until +validation_split: valid_hasAns +test_split: test_hasAns +fewshot_split: valid_hasAns +doc_to_text: "\nContexte: {{context}}\n\nQuestion: {{question}}\n\nRéponse:" +doc_to_target: "{% if answers.text| length > 0 %}{{answers.text[0]}}{% else %}{{['Impossible']}}{% endif %}" +target_delimiter: " " +should_decontaminate: true +doc_to_decontamination_query: context +generation_kwargs: + until: + - "\n" +# filter_list: +# - name: remove_whitespace +# filter: +# - function: remove_whitespace +# - function: take_first +metric_list: + - metric: !function utils.exact + aggregation: mean + higher_is_better: true + - metric: !function utils.f1 + aggregation: mean + higher_is_better: true + - metric: !function utils.rouge1 + higher_is_better: true + aggregation: !function utils.rouge1_agg diff --git a/lm-evaluation-harness/lm_eval/tasks/french_bench/french_bench_grammar.yaml b/lm-evaluation-harness/lm_eval/tasks/french_bench/french_bench_grammar.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6828c3a9fd7a9c73c7c7ff368952ca22805b4b7a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/french_bench/french_bench_grammar.yaml @@ -0,0 +1,20 @@ +include: "_default_template_yaml" +tag: + - french_bench + - french_bench_mc +description: "Répond au mieux en complétant la question avec une des réponses proposées." +dataset_path: manu/french-bench-grammar-vocab-reading +output_type: multiple_choice +validation_split: Grammar +fewshot_split: Grammar +test_split: Grammar +#doc_to_text: "Question: {{question.strip()}}\nA: {{answerA}}\nB: {{answerB}}\nC: {{answerC}}\nD: {{answerD}}\nRéponse:" +#doc_to_choice: ["A", "B", "C", "D"] +doc_to_text: "La phrase suivante est correcte grammaticalement:\n" +doc_to_choice: "{{[question.replace('<...>', answerA), question.replace('<...>', answerB), question.replace('<...>', answerC), question.replace('<...>', answerD)]}}" +doc_to_target: '{{["answerA", "answerB", "answerC", "answerD"].index("answer" + answer)}}' +task: french_bench_grammar +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true diff --git a/lm-evaluation-harness/lm_eval/tasks/french_bench/french_bench_hellaswag.yaml b/lm-evaluation-harness/lm_eval/tasks/french_bench/french_bench_hellaswag.yaml new file mode 100644 index 0000000000000000000000000000000000000000..293a76c27a9bfbf7beec22805d06f44310cf143c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/french_bench/french_bench_hellaswag.yaml @@ -0,0 +1,20 @@ +tag: + - french_bench + - french_bench_mc +task: french_bench_hellaswag +dataset_path: manu/french_bench_hellaswag +output_type: multiple_choice +training_split: validation +validation_split: validation +test_split: null +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{label}}" +doc_to_choice: "{{choices}}" +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true diff --git a/lm-evaluation-harness/lm_eval/tasks/french_bench/french_bench_multifquad.yaml b/lm-evaluation-harness/lm_eval/tasks/french_bench/french_bench_multifquad.yaml new file mode 100644 index 0000000000000000000000000000000000000000..71301bf29e55a0954527e7963cdcbbba92710337 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/french_bench/french_bench_multifquad.yaml @@ -0,0 +1,34 @@ +include: "_default_template_yaml" +tag: + - french_bench + - french_bench_gen +description: "D'après l'information dans le contexte donné, donne la réponse à la question en citant quelques extraits du contexte." +task: french_bench_multifquad +dataset_path: manu/multifquad_test +output_type: generate_until +validation_split: valid +test_split: test +fewshot_split: valid +doc_to_text: "\nContexte: {{context}}\n\nQuestion: {{question}}\n\nRéponse:" +doc_to_target: "{{', '.join(answers.text)}}" +target_delimiter: " " +should_decontaminate: true +doc_to_decontamination_query: context +generation_kwargs: + until: + - "\n" +# filter_list: +# - name: remove_whitespace +# filter: +# - function: remove_whitespace +# - function: take_first +metric_list: + - metric: !function utils.exact + aggregation: mean + higher_is_better: true + - metric: !function utils.f1 + aggregation: mean + higher_is_better: true + - metric: !function utils.rouge1 + higher_is_better: true + aggregation: !function utils.rouge1_agg diff --git a/lm-evaluation-harness/lm_eval/tasks/french_bench/french_bench_opus_perplexity.yaml b/lm-evaluation-harness/lm_eval/tasks/french_bench/french_bench_opus_perplexity.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dbe714a9c03a9f099d7438f4458d086caa7cd4a1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/french_bench/french_bench_opus_perplexity.yaml @@ -0,0 +1,23 @@ +tag: + - french_bench_perplexity +task: french_bench_opus_perplexity +dataset_path: manu/opus100-en-fr +output_type: loglikelihood_rolling +test_split: test +fewshot_split: validation +validation_split: validation +num_fewshot: 0 +doc_to_text: "" +doc_to_target: "{{text}}" +should_decontaminate: true +doc_to_decontamination_query: "{{text}}" +metric_list: + - metric: word_perplexity + aggregation: weighted_perplexity + higher_is_better: false + - metric: byte_perplexity + aggregation: weighted_perplexity + higher_is_better: false + - metric: bits_per_byte + aggregation: bits_per_byte + higher_is_better: false diff --git a/lm-evaluation-harness/lm_eval/tasks/french_bench/french_bench_orangesum_abstract.yaml b/lm-evaluation-harness/lm_eval/tasks/french_bench/french_bench_orangesum_abstract.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1d4a3b4acb9e5304eb191e345edc191309245729 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/french_bench/french_bench_orangesum_abstract.yaml @@ -0,0 +1,28 @@ +include: "_default_template_yaml" +tag: + - french_bench + - french_bench_gen +description: "Résume l'article en une phrase." +task: french_bench_orangesum_abstract +dataset_path: orange_sum +dataset_name: abstract +output_type: generate_until +validation_split: validation +fewshot_split: validation +doc_to_text: "\nArticle: {{text}}\n\nRésumé:" +doc_to_target: "{{summary}}" +target_delimiter: " " +should_decontaminate: true +doc_to_decontamination_query: summary +generation_kwargs: + until: + - "\n" +# filter_list: +# - name: remove_whitespace +# filter: +# - function: remove_whitespace +# - function: take_first +metric_list: + - metric: !function utils.rouge1 + higher_is_better: true + aggregation: !function utils.rouge1_agg diff --git a/lm-evaluation-harness/lm_eval/tasks/french_bench/french_bench_orangesum_title.yaml b/lm-evaluation-harness/lm_eval/tasks/french_bench/french_bench_orangesum_title.yaml new file mode 100644 index 0000000000000000000000000000000000000000..90b547e038ef3ff0086641b136461411240aaf5a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/french_bench/french_bench_orangesum_title.yaml @@ -0,0 +1,28 @@ +include: "_default_template_yaml" +tag: + - french_bench + - french_bench_extra +description: "Trouve le titre de l'article." +task: french_bench_orangesum_title +dataset_path: orange_sum +dataset_name: title +output_type: generate_until +validation_split: validation +fewshot_split: validation +doc_to_text: "\nArticle: {{text}}\n\nTitre:" +doc_to_target: "{{summary}}" +target_delimiter: " " +should_decontaminate: true +doc_to_decontamination_query: summary +generation_kwargs: + until: + - "\n" +# filter_list: +# - name: remove_whitespace +# filter: +# - function: remove_whitespace +# - function: take_first +metric_list: + - metric: !function utils.rouge1 + higher_is_better: true + aggregation: !function utils.rouge1_agg diff --git a/lm-evaluation-harness/lm_eval/tasks/french_bench/french_bench_reading_comp.yaml b/lm-evaluation-harness/lm_eval/tasks/french_bench/french_bench_reading_comp.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f3abeadad711b746f998f6b1f7253ca1285e5e24 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/french_bench/french_bench_reading_comp.yaml @@ -0,0 +1,22 @@ +include: "_default_template_yaml" +tag: + - french_bench + - french_bench_extra +# description: "Répond au mieux en complétant la question avec une des réponses proposées." +dataset_path: manu/french-bench-grammar-vocab-reading +output_type: multiple_choice +validation_split: Reading +fewshot_split: Reading +test_split: Reading +# doc_to_text: "Context: {{context}}\nQuestion: {{question.strip()}}\nA: {{answerA}}\nB: {{answerB}}\nC: {{answerC}}\nD: {{answerD}}\nRéponse:" +# doc_to_choice: "{{['A: '+answerA, 'B: '+answerB, 'C: '+answerC, 'D: '+answerD]}}" +doc_to_text: "Context: {{context}}\n\n" +doc_to_choice: "{{[question.replace('<...>', answerA) if '<...>' in question else question + ' ' +answerA, question.replace('<...>', answerB) if '<...>' in question else question + ' ' + answerB, question.replace('<...>', answerC) if '<...>' in question else question + ' ' + answerC, question.replace('<...>', answerD) if '<...>' in question else question + ' ' + answerD]}}" +doc_to_target: '{{["answerA", "answerB", "answerC", "answerD"].index("answer" + answer)}}' +# doc_to_choice: "{{['A: '+answerA, 'B: '+answerB, 'C: '+answerC, 'D: '+answerD]}}" +# doc_to_target: answer +task: french_bench_reading_comp +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true diff --git a/lm-evaluation-harness/lm_eval/tasks/french_bench/french_bench_topic_based_nli.yaml b/lm-evaluation-harness/lm_eval/tasks/french_bench/french_bench_topic_based_nli.yaml new file mode 100644 index 0000000000000000000000000000000000000000..28dd6af64ecd6344146a790a57cfc43ccf2eb3a2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/french_bench/french_bench_topic_based_nli.yaml @@ -0,0 +1,23 @@ +include: "_default_template_yaml" +tag: + - french_bench + - french_bench_extra +description: "A propos du thème spécifié, l'avis client est il positif, négatif, ou neutre ?" +task: french_bench_topic_based_nli +dataset_path: manu/topic_based_nli_test +output_type: multiple_choice +validation_split: valid +# doc_to_text: "\nAvis Client: {{text}}\n\nEn considèrant uniquement le thème \"{{topic}}\", l'avis client est plutot:\nA. Positif \nB. Négatif\nC. Mitigé \nD. Neutre\nE. Absent\n\nRéponse:" +# doc_to_choice: ["A", "B", "C", "D", "E"] +doc_to_text: "\nAvis Client: {{text}}\n\nA propos du thème \"{{topic}}\", l'avis client est" +doc_to_choice: ['positif', 'négatif', 'neutre'] +doc_to_target: "{{['positif', 'negatif', 'neutre'].index(polarity)}}" +should_decontaminate: true +doc_to_decontamination_query: texte +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true diff --git a/lm-evaluation-harness/lm_eval/tasks/french_bench/french_bench_trivia.yaml b/lm-evaluation-harness/lm_eval/tasks/french_bench/french_bench_trivia.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3b69b0f12b8be7078e5a0e35c812709ef496fa5f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/french_bench/french_bench_trivia.yaml @@ -0,0 +1,36 @@ +include: "_default_template_yaml" +tag: + - french_bench + - french_bench_gen +task: french_bench_trivia +dataset_path: manu/french-trivia +output_type: generate_until +validation_split: train +test_split: train +fewshot_split: train +doc_to_text: "{{Question}}\nAnswer:" +doc_to_target: "{{Answer}}" +target_delimiter: " " +should_decontaminate: true +doc_to_decontamination_query: Question +generation_kwargs: + until: + - "\n" +# filter_list: +# - name: remove_whitespace +# filter: +# - function: remove_whitespace +# - function: take_first +metric_list: + - metric: !function utils.exact + aggregation: mean + higher_is_better: true + - metric: !function utils.f1 + aggregation: mean + higher_is_better: true + - metric: !function utils.rouge1 + higher_is_better: true + aggregation: !function utils.rouge1_agg + - metric: !function utils.is_included + higher_is_better: true + aggregation: mean diff --git a/lm-evaluation-harness/lm_eval/tasks/french_bench/french_bench_vocab.yaml b/lm-evaluation-harness/lm_eval/tasks/french_bench/french_bench_vocab.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a5d5cadcd4ab2e879909fc51c94699ff45e4f6b3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/french_bench/french_bench_vocab.yaml @@ -0,0 +1,20 @@ +include: "_default_template_yaml" +tag: + - french_bench + - french_bench_mc +# description: "Répond au mieux en complétant la question avec une des réponses proposées." +dataset_path: manu/french-bench-grammar-vocab-reading +output_type: multiple_choice +validation_split: Vocabulary +fewshot_split: Vocabulary +test_split: Vocabulary +# doc_to_text: "Question: {{question.strip()}}\nA: {{answerA}}\nB: {{answerB}}\nC: {{answerC}}\nD: {{answerD}}\nRéponse:" +# doc_to_choice: ["A", "B", "C", "D"] +doc_to_text: "La phrase suivante est logique sémantiquement:\n" +doc_to_choice: "{{[question.replace('<...>', answerA), question.replace('<...>', answerB), question.replace('<...>', answerC), question.replace('<...>', answerD)]}}" +doc_to_target: '{{["answerA", "answerB", "answerC", "answerD"].index("answer" + answer)}}' +task: french_bench_vocab +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true diff --git a/lm-evaluation-harness/lm_eval/tasks/french_bench/french_bench_wikitext_fr.yaml b/lm-evaluation-harness/lm_eval/tasks/french_bench/french_bench_wikitext_fr.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d7ae23ff9246e09e31a0c2e77f19935b7d03432d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/french_bench/french_bench_wikitext_fr.yaml @@ -0,0 +1,25 @@ +tag: + - french_bench_perplexity +task: french_bench_wikitext_fr +dataset_path: asi/wikitext_fr +dataset_name: wikitext-35 +output_type: loglikelihood_rolling +training_split: train +validation_split: validation +test_split: test +num_fewshot: 0 +doc_to_text: "" +doc_to_target: !function preprocess_wikitext.wikitext_detokenizer +process_results: !function preprocess_wikitext.process_results +should_decontaminate: true +doc_to_decontamination_query: "{{paragraph}}" +metric_list: + - metric: word_perplexity + aggregation: weighted_perplexity + higher_is_better: false + - metric: byte_perplexity + aggregation: weighted_perplexity + higher_is_better: false + - metric: bits_per_byte + aggregation: bits_per_byte + higher_is_better: false diff --git a/lm-evaluation-harness/lm_eval/tasks/french_bench/french_bench_xnli.yaml b/lm-evaluation-harness/lm_eval/tasks/french_bench/french_bench_xnli.yaml new file mode 100644 index 0000000000000000000000000000000000000000..272b5652e81fdc3d42a6ca6cc39220d715251323 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/french_bench/french_bench_xnli.yaml @@ -0,0 +1,21 @@ +include: "_default_template_yaml" +tag: + - french_bench + - french_bench_extra +description: "La prémisse et l'hypothèse sont elles en accord, neutres en elles, ou en contradiction ?" +dataset_path: xnli +dataset_name: fr +output_type: multiple_choice +validation_split: validation +fewshot_split: validation +test_split: test +# doc_to_text: "\nPrémisse: {{premise}}\n\nHypothèse: {{hypothesis}}\n\nLa prémisse et l'hypothèse sont:\nA. En accord\nB. Neutre\nC. En contradiction\nRéponse:" +# doc_to_choice: "{{['A: En accord', 'B: Neutre', 'C: En contradiction']}}" +doc_to_text: "\nPrémisse: {{premise}}\n\nHypothèse: {{hypothesis}}\n\nLa prémisse et l'hypothèse sont" +doc_to_choice: "{{['en accord', 'neutres entre elles', 'en contradiction']}}" +doc_to_target: label +task: french_bench_xnli +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true diff --git a/lm-evaluation-harness/lm_eval/tasks/french_bench/preprocess_wikitext.py b/lm-evaluation-harness/lm_eval/tasks/french_bench/preprocess_wikitext.py new file mode 100644 index 0000000000000000000000000000000000000000..6bea950f987a2185c40e7883869577dacb9ecb7a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/french_bench/preprocess_wikitext.py @@ -0,0 +1,48 @@ +import re + + +def wikitext_detokenizer(doc): + string = doc["paragraph"] + # contractions + string = string.replace("s '", "s'") + string = re.sub(r"/' [0-9]/", r"/'[0-9]/", string) + # number separators + string = string.replace(" @-@ ", "-") + string = string.replace(" @,@ ", ",") + string = string.replace(" @.@ ", ".") + # punctuation + string = string.replace(" : ", ": ") + string = string.replace(" ; ", "; ") + string = string.replace(" . ", ". ") + string = string.replace(" ! ", "! ") + string = string.replace(" ? ", "? ") + string = string.replace(" , ", ", ") + # double brackets + string = re.sub(r"\(\s*([^\)]*?)\s*\)", r"(\1)", string) + string = re.sub(r"\[\s*([^\]]*?)\s*\]", r"[\1]", string) + string = re.sub(r"{\s*([^}]*?)\s*}", r"{\1}", string) + string = re.sub(r"\"\s*([^\"]*?)\s*\"", r'"\1"', string) + string = re.sub(r"'\s*([^']*?)\s*'", r"'\1'", string) + # miscellaneous + string = string.replace("= = = =", "====") + string = string.replace("= = =", "===") + string = string.replace("= =", "==") + string = string.replace(" " + chr(176) + " ", chr(176)) + string = string.replace(" \n", "\n") + string = string.replace("\n ", "\n") + string = string.replace(" N ", " 1 ") + string = string.replace(" 's", "'s") + + return string + + +def process_results(doc, results): + (loglikelihood,) = results + # IMPORTANT: wikitext counts number of words in *original doc before detokenization* + _words = len(re.split(r"\s+", doc["paragraph"])) + _bytes = len(doc["paragraph"].encode("utf-8")) + return { + "word_perplexity": (loglikelihood, _words), + "byte_perplexity": (loglikelihood, _bytes), + "bits_per_byte": (loglikelihood, _bytes), + } diff --git a/lm-evaluation-harness/lm_eval/tasks/french_bench/utils.py b/lm-evaluation-harness/lm_eval/tasks/french_bench/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..acbcbe83c86cd75c79ad8fbe1452a43776eaa12f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/french_bench/utils.py @@ -0,0 +1,102 @@ +import collections +import re +import string + +import datasets +import evaluate + + +def normalize_answer(s): + """Lower text and remove punctuation, articles and extra whitespace.""" + + def remove_articles(text): + regex = re.compile(r"\b(un|une|des|le|la|les)\b", re.UNICODE) + return re.sub(regex, " ", text) + + def white_space_fix(text): + return " ".join(text.split()) + + def remove_punc(text): + exclude = set(string.punctuation) + return "".join(ch for ch in text if ch not in exclude) + + def lower(text): + return text.lower() + + return white_space_fix(remove_articles(remove_punc(lower(s)))) + + +def get_tokens(s): + if not s: + return [] + return normalize_answer(s).split() + + +# Exact match (the normalized answer exactly match the gold answer) +def exact(predictions, references): + return int(normalize_answer(references[0]) == normalize_answer(predictions[0])) + + +# The F-score of predicted tokens versus the gold answer +def f1(predictions, references): + gold_toks = get_tokens(references[0]) + pred_toks = get_tokens(predictions[0]) + common = collections.Counter(gold_toks) & collections.Counter(pred_toks) + num_same = sum(common.values()) + if len(gold_toks) == 0 or len(pred_toks) == 0: + # If either is no-answer, then F1 is 1 if they agree, 0 otherwise + return int(gold_toks == pred_toks) + if num_same == 0: + return 0 + precision = 1.0 * num_same / len(pred_toks) + recall = 1.0 * num_same / len(gold_toks) + f1 = (2 * precision * recall) / (precision + recall) + return f1 + + +def rouge1(items): + """ + # passthrough for efficiency + """ + return items + + +def rouge1_agg(items): + """ + Higher is better + """ + refs = list(zip(*items))[0] + preds = list(zip(*items))[1] + rouge_scorer = evaluate.load("rouge") + return rouge_scorer.compute(predictions=preds, references=refs)["rouge1"] + + +def is_included(items): + """ + # passthrough for efficiency + """ + if items[0] in items[1]: + return True + return False + + +def preprocess(text): + text = text.strip() + # NOTE: Brackets are artifacts of the WikiHow dataset portion of HellaSwag. + text = text.replace(" [title]", ". ") + text = re.sub("\\[.*?\\]", "", text) + text = text.replace(" ", " ") + return text + + +def process_docs(dataset: datasets.Dataset) -> datasets.Dataset: + def _process_doc(doc): + ctx = doc["ctx_a"] + " " + doc["ctx_b"].capitalize() + out_doc = { + "query": preprocess(doc["activity_label"] + ": " + ctx), + "choices": [preprocess(ending) for ending in doc["endings"]], + "gold": int(doc["label"]), + } + return out_doc + + return dataset.map(_process_doc) diff --git a/lm-evaluation-harness/lm_eval/tasks/galician_bench/README.md b/lm-evaluation-harness/lm_eval/tasks/galician_bench/README.md new file mode 100644 index 0000000000000000000000000000000000000000..276dd00a32cda56c82a76efc73185bd7a581e9ae --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/galician_bench/README.md @@ -0,0 +1,113 @@ +# GalicianBench + +### Paper + +GalicianBench is a benchmark for evaluating language models in Galician tasks. This is, it evaluates the ability of a language model to understand and generate Galician text. GalicianBench offers a combination of pre-existing, open datasets and datasets developed exclusivelly for this benchmark. All the details of GalicianBench will be published in a paper soon. + +The new evaluation datasets included in GalicianBench are: +| Task | Category | Homepage | +|:-------------:|:-----:|:-----:| +| Belebele_gl | Reading Comprehension | https://huggingface.co/datasets/proxectonos/belebele_gl | +| GalCoLA | Linguistic Acceptability | https://huggingface.co/datasets/proxectonos/galcola | +| MGSM_ca | Math | https://huggingface.co/datasets/proxectonos/mgsm_gl | +| Parafrases_gl | Paraphrasing | https://huggingface.co/datasets/proxectonos/parafrases_gl | +| PAWS-gl | Paraphrasing | https://huggingface.co/datasets/proxectonos/PAWS-gl | +| OpenBookQA_gl | Question Answering | https://huggingface.co/datasets/proxectonos/openbookqa_gl | +| Summarization_gl | Summarization | https://huggingface.co/datasets/proxectonos/summarization_gl | +| TruthfulQA_gl | Truthfulness | https://huggingface.co/datasets/proxectonos/truthfulqa_gl | +| xnli_gl | NLI | https://huggingface.co/datasets/proxectonos/xnli_gl | +| xstorycloze_gl | Commonsense Reasoning | https://huggingface.co/datasets/proxectonos/xstorycloze_gl | + +The datasets included in GalicianBench that have been made public in previous pubications are: + +| Task | Category | Paper title | Homepage | +|:-------------:|:-----:|:-------------:|:-----:| +| FLORES_gl | Translation | [The FLORES-101 Evaluation Benchmark for Low-Resource and Multilingual Machine Translation](https://arxiv.org/abs/2106.03193) | https://huggingface.co/datasets/facebook/flores | + + +### Citation + +``` +@inproceedings{baucells-etal-2025-iberobench, + title = "{I}bero{B}ench: A Benchmark for {LLM} Evaluation in {I}berian Languages", + author = "Baucells, Irene and + Aula-Blasco, Javier and + de-Dios-Flores, Iria and + Paniagua Su{\'a}rez, Silvia and + Perez, Naiara and + Salles, Anna and + Sotelo Docio, Susana and + Falc{\~a}o, J{\'u}lia and + Saiz, Jose Javier and + Sepulveda Torres, Robiert and + Barnes, Jeremy and + Gamallo, Pablo and + Gonzalez-Agirre, Aitor and + Rigau, German and + Villegas, Marta", + editor = "Rambow, Owen and + Wanner, Leo and + Apidianaki, Marianna and + Al-Khalifa, Hend and + Eugenio, Barbara Di and + Schockaert, Steven", + booktitle = "Proceedings of the 31st International Conference on Computational Linguistics", + month = jan, + year = "2025", + address = "Abu Dhabi, UAE", + publisher = "Association for Computational Linguistics", + url = "https://aclanthology.org/2025.coling-main.699/", + pages = "10491--10519", +} +``` + +### Groups and Tasks + +#### Groups + +- `galician_bench`: All tasks included in GalicianBench. +- `flores_gl`: All FLORES translation tasks from or to Galician. + + +#### Tasks + +The following tasks evaluate tasks on GalicianBench dataset using various scoring methods. + - `belebele_glg_Latn` + - `flores_gl` + - `flores_gl-ca` + - `flores_gl-de` + - `flores_gl-en` + - `flores_gl-es` + - `flores_gl-eu` + - `flores_gl-fr` + - `flores_gl-it` + - `flores_gl-pt` + - `flores_ca-gl` + - `flores_de-gl` + - `flores_en-gl` + - `flores_es-gl` + - `flores_eu-gl` + - `flores_fr-gl` + - `flores_it-gl` + - `flores_pt-gl` + - `galcola` + - `summarization_gl` + - `parafrases_gl` + - `paws_gl` + - `openbookqa_gl` + - `mgsm_direct_gl` + - `truthfulqa_gl` + - `xnli_gl` + - `xstorycloze_gl` + +### Checklist + +* [x] Is the task an existing benchmark in the literature? + * [ ] Have you referenced the original paper that introduced the task? + * [ ] If yes, does the original paper provide a reference implementation? + * [ ] Yes, original implementation contributed by author of the benchmark + +If other tasks on this dataset are already supported: +* [ ] Is the "Main" variant of this task clearly denoted? +* [ ] Have you provided a short sentence in a README on what each new variant adds / evaluates? +* [ ] Have you noted which, if any, published evaluation setups are matched by this variant? diff --git a/lm-evaluation-harness/lm_eval/tasks/galician_bench/belebele_glg_Latn.yaml b/lm-evaluation-harness/lm_eval/tasks/galician_bench/belebele_glg_Latn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ae81a53f9c818b549ec5fd8bef3de3b8641d6013 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/galician_bench/belebele_glg_Latn.yaml @@ -0,0 +1,7 @@ +task: belebele_glg_Latn +include: ../belebele/_default_template_yaml +dataset_path: proxectonos/belebele_gl +fewshot_split: train +test_split: train +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/_flores_common_yaml b/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/_flores_common_yaml new file mode 100644 index 0000000000000000000000000000000000000000..bff45b85a1e74a77cc40b05284d031fec8780929 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/_flores_common_yaml @@ -0,0 +1,27 @@ +dataset_path: facebook/flores +dataset_name: all +output_type: generate_until +#! The test split of flores is not publicly available! (See paper section 6.1) +#! We are using `dev` and `devtest` splits, but they're mapped to train/validation/test in `data/flores/flores.py`. +training_split: dev +validation_split: dev +test_split: devtest +fewshot_split: dev +target_delimiter: '' +generation_kwargs: + until: + - "\n" +metric_list: + - metric: bleu + aggregation: bleu + higher_is_better: true + - metric: ter + aggregation: ter + higher_is_better: false + - metric: chrf + aggregation: chrf + higher_is_better: true +metadata: + version: 1.0 +dataset_kwargs: + trust_remote_code: true diff --git a/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/create_yamls_flores_gl.py b/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/create_yamls_flores_gl.py new file mode 100644 index 0000000000000000000000000000000000000000..0478781793208d8e9e195b50373dcb9b5072d885 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/create_yamls_flores_gl.py @@ -0,0 +1,333 @@ +# ruff: noqa: E731, E741 +""" +Script to generate task YAMLs for the FLORES-200 dataset. +Based on `tasks/translation/utils.py`. +""" + +import argparse +import itertools + +import yaml +from langcodes import Language + + +# utils +flatten = lambda l: list(itertools.chain(*l)) + +# constants +_LANGUAGES = [ + "ace_Arab", + "bam_Latn", + "dzo_Tibt", + "hin_Deva", + "khm_Khmr", + "mag_Deva", + "pap_Latn", + "sot_Latn", + "tur_Latn", + "ace_Latn", + "ban_Latn", + "ell_Grek", + "hne_Deva", + "kik_Latn", + "mai_Deva", + "pbt_Arab", + "spa_Latn", + "twi_Latn", + "acm_Arab", + "bel_Cyrl", + "eng_Latn", + "hrv_Latn", + "kin_Latn", + "mal_Mlym", + "pes_Arab", + "srd_Latn", + "tzm_Tfng", + "acq_Arab", + "bem_Latn", + "epo_Latn", + "hun_Latn", + "kir_Cyrl", + "mar_Deva", + "plt_Latn", + "srp_Cyrl", + "uig_Arab", + "aeb_Arab", + "ben_Beng", + "est_Latn", + "hye_Armn", + "kmb_Latn", + "min_Arab", + "pol_Latn", + "ssw_Latn", + "ukr_Cyrl", + "afr_Latn", + "bho_Deva", + "eus_Latn", + "ibo_Latn", + "kmr_Latn", + "min_Latn", + "por_Latn", + "sun_Latn", + "umb_Latn", + "ajp_Arab", + "bjn_Arab", + "ewe_Latn", + "ilo_Latn", + "knc_Arab", + "mkd_Cyrl", + "prs_Arab", + "swe_Latn", + "urd_Arab", + "aka_Latn", + "bjn_Latn", + "fao_Latn", + "ind_Latn", + "knc_Latn", + "mlt_Latn", + "quy_Latn", + "swh_Latn", + "uzn_Latn", + "als_Latn", + "bod_Tibt", + "fij_Latn", + "isl_Latn", + "kon_Latn", + "mni_Beng", + "ron_Latn", + "szl_Latn", + "vec_Latn", + "amh_Ethi", + "bos_Latn", + "fin_Latn", + "ita_Latn", + "kor_Hang", + "mos_Latn", + "run_Latn", + "tam_Taml", + "vie_Latn", + "apc_Arab", + "bug_Latn", + "fon_Latn", + "jav_Latn", + "lao_Laoo", + "mri_Latn", + "rus_Cyrl", + "taq_Latn", + "war_Latn", + "arb_Arab", + "bul_Cyrl", + "fra_Latn", + "jpn_Jpan", + "lij_Latn", + "mya_Mymr", + "sag_Latn", + "taq_Tfng", + "wol_Latn", + "arb_Latn", + "cat_Latn", + "fur_Latn", + "kab_Latn", + "lim_Latn", + "nld_Latn", + "san_Deva", + "tat_Cyrl", + "xho_Latn", + "ars_Arab", + "ceb_Latn", + "fuv_Latn", + "kac_Latn", + "lin_Latn", + "nno_Latn", + "sat_Olck", + "tel_Telu", + "ydd_Hebr", + "ary_Arab", + "ces_Latn", + "gaz_Latn", + "kam_Latn", + "lit_Latn", + "nob_Latn", + "scn_Latn", + "tgk_Cyrl", + "yor_Latn", + "arz_Arab", + "cjk_Latn", + "gla_Latn", + "kan_Knda", + "lmo_Latn", + "npi_Deva", + "shn_Mymr", + "tgl_Latn", + "yue_Hant", + "asm_Beng", + "ckb_Arab", + "gle_Latn", + "kas_Arab", + "ltg_Latn", + "nso_Latn", + "sin_Sinh", + "tha_Thai", + "zho_Hans", + "ast_Latn", + "crh_Latn", + "glg_Latn", + "kas_Deva", + "ltz_Latn", + "nus_Latn", + "slk_Latn", + "tir_Ethi", + "zho_Hant", + "awa_Deva", + "cym_Latn", + "grn_Latn", + "kat_Geor", + "lua_Latn", + "nya_Latn", + "slv_Latn", + "tpi_Latn", + "zsm_Latn", + "ayr_Latn", + "dan_Latn", + "guj_Gujr", + "kaz_Cyrl", + "lug_Latn", + "oci_Latn", + "smo_Latn", + "tsn_Latn", + "zul_Latn", + "azb_Arab", + "deu_Latn", + "hat_Latn", + "kbp_Latn", + "luo_Latn", + "ory_Orya", + "sna_Latn", + "tso_Latn", + "azj_Latn", + "dik_Latn", + "hau_Latn", + "kea_Latn", + "lus_Latn", + "pag_Latn", + "snd_Arab", + "tuk_Latn", + "bak_Cyrl", + "dyu_Latn", + "heb_Hebr", + "khk_Cyrl", + "lvs_Latn", + "pan_Guru", + "som_Latn", + "tum_Latn", +] +LANGUAGE_PAIRS = [ + (a, b) for idx, a in enumerate(_LANGUAGES) for b in _LANGUAGES[idx + 1 :] +] + +LANGUAGES_OF_INTEREST = [ + "cat_Latn", + "spa_Latn", + "eng_Latn", + "glg_Latn", + "eus_Latn", + "ita_Latn", + "deu_Latn", + "por_Latn", + "fra_Latn", +] +MAIN_LANG = "glg_Latn" +LANGUAGE_PAIRS = [ + (a, b) + for (a, b) in LANGUAGE_PAIRS + if a in LANGUAGES_OF_INTEREST and b in LANGUAGES_OF_INTEREST and MAIN_LANG in (a, b) +] + +# auxiliary functions + +code_to_language_name = lambda code: Language.make( + language=Language.get(code)["language"] +).display_name() +code_to_short_name = lambda code: Language.get(code)["language"] +jinja_var = ( + lambda s: "{{" + s + "}}" +) # wrapper to avoid having to escape { } in format strings + + +def doc_to_text(src: str, tgt: str) -> str: + src_name, tgt_name = map(code_to_language_name, [src, tgt]) + + return f"""\ +{src_name} sentence: {jinja_var("sentence_" + src)} +{tgt_name} sentence:""" + + +def doc_to_target(tgt: str) -> str: + return f"{jinja_var('sentence_' + tgt)}" + + +# main function + + +def gen_lang_yamls(output_dir: str, overwrite: bool) -> None: + """ + Generate a YAML file for each translation direction. + """ + + err = [] + for src, tgt in LANGUAGE_PAIRS: + # do both translation directions for each lang pair + for src, tgt in [(src, tgt), (tgt, src)]: + lang_pair_name = f"{code_to_short_name(src)}-{code_to_short_name(tgt)}" + yaml_file_name = f"flores_{lang_pair_name}.yaml" + + try: + with open( + f"{output_dir}/{yaml_file_name}", + "w" if overwrite else "x", + encoding="utf-8", + ) as outfile: + print(f"Creating {yaml_file_name}...") + outfile.write("# File generated by `create-yamls.py`\n") + yaml.dump( + { + # "group": [f"{BENCH_NAME}_bench", f"{BENCH_NAME}_bench_flores"], + # "group": "flores_gl", + "include": "_flores_common_yaml", + "task": f"flores_{lang_pair_name}", + "doc_to_text": doc_to_text(src, tgt), + "doc_to_target": doc_to_target(tgt), + }, + outfile, + sort_keys=False, + ) + + except FileExistsError: + err.append(yaml_file_name) + + if len(err) > 0: + raise FileExistsError( + "Files were not created because they already exist:" + f" {', '.join(err)}" + "\nUse flag --overwrite to overwrite them." + ) + + +def main() -> None: + parser = argparse.ArgumentParser() + parser.add_argument( + "--overwrite", + default=False, + action="store_true", + help="Overwrite files if they already exist", + ) + parser.add_argument( + "--output-dir", default=".", help="Directory to write yaml files to" + ) + args = parser.parse_args() + + gen_lang_yamls(output_dir=args.output_dir, overwrite=args.overwrite) + + +if __name__ == "__main__": + main() diff --git a/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/flores_ca-gl.yaml b/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/flores_ca-gl.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5da7ad5fe40ee70803b570aa019679f647e48b98 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/flores_ca-gl.yaml @@ -0,0 +1,7 @@ +# File generated by `create-yamls.py` +include: _flores_common_yaml +task: flores_ca-gl +doc_to_text: 'Catalan sentence: {{sentence_cat_Latn}} + + Galician sentence:' +doc_to_target: '{{sentence_glg_Latn}}' diff --git a/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/flores_de-gl.yaml b/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/flores_de-gl.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2f2eabbc55c7c68babee5d5b3fa0e0225575f4d9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/flores_de-gl.yaml @@ -0,0 +1,7 @@ +# File generated by `create-yamls.py` +include: _flores_common_yaml +task: flores_de-gl +doc_to_text: 'German sentence: {{sentence_deu_Latn}} + + Galician sentence:' +doc_to_target: '{{sentence_glg_Latn}}' diff --git a/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/flores_en-gl.yaml b/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/flores_en-gl.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9dc8fc24f1ff8ae3596dcd189242af3cfde89475 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/flores_en-gl.yaml @@ -0,0 +1,7 @@ +# File generated by `create-yamls.py` +include: _flores_common_yaml +task: flores_en-gl +doc_to_text: 'English sentence: {{sentence_eng_Latn}} + + Galician sentence:' +doc_to_target: '{{sentence_glg_Latn}}' diff --git a/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/flores_es-gl.yaml b/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/flores_es-gl.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dd3c6a9eac7c1318d23209947690384ec41a7f29 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/flores_es-gl.yaml @@ -0,0 +1,7 @@ +# File generated by `create-yamls.py` +include: _flores_common_yaml +task: flores_es-gl +doc_to_text: 'Spanish sentence: {{sentence_spa_Latn}} + + Galician sentence:' +doc_to_target: '{{sentence_glg_Latn}}' diff --git a/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/flores_eu-gl.yaml b/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/flores_eu-gl.yaml new file mode 100644 index 0000000000000000000000000000000000000000..db762cf75c90985a9b87459587508fd429070e98 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/flores_eu-gl.yaml @@ -0,0 +1,7 @@ +# File generated by `create-yamls.py` +include: _flores_common_yaml +task: flores_eu-gl +doc_to_text: 'Basque sentence: {{sentence_eus_Latn}} + + Galician sentence:' +doc_to_target: '{{sentence_glg_Latn}}' diff --git a/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/flores_fr-gl.yaml b/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/flores_fr-gl.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0d884dbad70943f7c57bdd1dad0870310b4c994f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/flores_fr-gl.yaml @@ -0,0 +1,7 @@ +# File generated by `create-yamls.py` +include: _flores_common_yaml +task: flores_fr-gl +doc_to_text: 'French sentence: {{sentence_fra_Latn}} + + Galician sentence:' +doc_to_target: '{{sentence_glg_Latn}}' diff --git a/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/flores_gl-ca.yaml b/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/flores_gl-ca.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6ce3eaae5cbb035bea3f458176306003867b4ae4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/flores_gl-ca.yaml @@ -0,0 +1,7 @@ +# File generated by `create-yamls.py` +include: _flores_common_yaml +task: flores_gl-ca +doc_to_text: 'Galician sentence: {{sentence_glg_Latn}} + + Catalan sentence:' +doc_to_target: '{{sentence_cat_Latn}}' diff --git a/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/flores_gl-de.yaml b/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/flores_gl-de.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e499780fbb464546abee256b058d732f2f57da25 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/flores_gl-de.yaml @@ -0,0 +1,7 @@ +# File generated by `create-yamls.py` +include: _flores_common_yaml +task: flores_gl-de +doc_to_text: 'Galician sentence: {{sentence_glg_Latn}} + + German sentence:' +doc_to_target: '{{sentence_deu_Latn}}' diff --git a/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/flores_gl-en.yaml b/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/flores_gl-en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5d2b7afbd94c01defe341953d3881784bcb55ec3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/flores_gl-en.yaml @@ -0,0 +1,7 @@ +# File generated by `create-yamls.py` +include: _flores_common_yaml +task: flores_gl-en +doc_to_text: 'Galician sentence: {{sentence_glg_Latn}} + + English sentence:' +doc_to_target: '{{sentence_eng_Latn}}' diff --git a/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/flores_gl-es.yaml b/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/flores_gl-es.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c00acf3f47fafdd1c9176855ad4b8fe76c9634e9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/flores_gl-es.yaml @@ -0,0 +1,7 @@ +# File generated by `create-yamls.py` +include: _flores_common_yaml +task: flores_gl-es +doc_to_text: 'Galician sentence: {{sentence_glg_Latn}} + + Spanish sentence:' +doc_to_target: '{{sentence_spa_Latn}}' diff --git a/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/flores_gl-eu.yaml b/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/flores_gl-eu.yaml new file mode 100644 index 0000000000000000000000000000000000000000..08fafe084adad4a8381d49cbfc491e669443a8e3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/flores_gl-eu.yaml @@ -0,0 +1,7 @@ +# File generated by `create-yamls.py` +include: _flores_common_yaml +task: flores_gl-eu +doc_to_text: 'Galician sentence: {{sentence_glg_Latn}} + + Basque sentence:' +doc_to_target: '{{sentence_eus_Latn}}' diff --git a/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/flores_gl-fr.yaml b/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/flores_gl-fr.yaml new file mode 100644 index 0000000000000000000000000000000000000000..14b060b25f106ad5077b408e85479930aab5a51e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/flores_gl-fr.yaml @@ -0,0 +1,7 @@ +# File generated by `create-yamls.py` +include: _flores_common_yaml +task: flores_gl-fr +doc_to_text: 'Galician sentence: {{sentence_glg_Latn}} + + French sentence:' +doc_to_target: '{{sentence_fra_Latn}}' diff --git a/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/flores_gl-it.yaml b/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/flores_gl-it.yaml new file mode 100644 index 0000000000000000000000000000000000000000..74a01b88540c5179ec5c619689e235558a20ee64 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/flores_gl-it.yaml @@ -0,0 +1,7 @@ +# File generated by `create-yamls.py` +include: _flores_common_yaml +task: flores_gl-it +doc_to_text: 'Galician sentence: {{sentence_glg_Latn}} + + Italian sentence:' +doc_to_target: '{{sentence_ita_Latn}}' diff --git a/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/flores_gl-pt.yaml b/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/flores_gl-pt.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e965a34776ec2dd816983ee1ae4552ca5835c0ac --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/flores_gl-pt.yaml @@ -0,0 +1,7 @@ +# File generated by `create-yamls.py` +include: _flores_common_yaml +task: flores_gl-pt +doc_to_text: 'Galician sentence: {{sentence_glg_Latn}} + + Portuguese sentence:' +doc_to_target: '{{sentence_por_Latn}}' diff --git a/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/flores_gl.yaml b/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/flores_gl.yaml new file mode 100644 index 0000000000000000000000000000000000000000..806739a9df2a74e903974f22a50f417dae4a8982 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/flores_gl.yaml @@ -0,0 +1,24 @@ +group: flores_gl +task: + - flores_es-gl + - flores_gl-es + - flores_en-gl + - flores_gl-en + - flores_eu-gl + - flores_gl-eu + - flores_pt-gl + - flores_gl-pt + - flores_it-gl + - flores_gl-it + - flores_fr-gl + - flores_gl-fr + - flores_ca-gl + - flores_gl-ca + - flores_gl-de + - flores_de-gl +aggregate_metric_list: + - metric: bleu + aggregation: mean + weight_by_size: false +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/flores_it-gl.yaml b/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/flores_it-gl.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7c85a09c9eae535b0365e608b50f6a1e33a8cdc7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/flores_it-gl.yaml @@ -0,0 +1,7 @@ +# File generated by `create-yamls.py` +include: _flores_common_yaml +task: flores_it-gl +doc_to_text: 'Italian sentence: {{sentence_ita_Latn}} + + Galician sentence:' +doc_to_target: '{{sentence_glg_Latn}}' diff --git a/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/flores_pt-gl.yaml b/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/flores_pt-gl.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5371f51062d1fef29caef8fdc5be4a668e744295 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/galician_bench/flores_gl/flores_pt-gl.yaml @@ -0,0 +1,7 @@ +# File generated by `create-yamls.py` +include: _flores_common_yaml +task: flores_pt-gl +doc_to_text: 'Portuguese sentence: {{sentence_por_Latn}} + + Galician sentence:' +doc_to_target: '{{sentence_glg_Latn}}' diff --git a/lm-evaluation-harness/lm_eval/tasks/galician_bench/galcola.yaml b/lm-evaluation-harness/lm_eval/tasks/galician_bench/galcola.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e53d3d601e329373a7341b76647008dabde68f6e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/galician_bench/galcola.yaml @@ -0,0 +1,16 @@ +task: galcola +dataset_path: proxectonos/galcola +output_type: multiple_choice +training_split: train +validation_split: validation +test_split: test +doc_to_text: "{{sentence}}\nPregunta: Ten sentido esta frase?\nResposta:" +doc_to_target: label +doc_to_choice: ["non", "si"] +should_decontaminate: true +doc_to_decontamination_query: sentence +metric_list: + - metric: mcc + - metric: acc +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/galician_bench/galician_bench.yaml b/lm-evaluation-harness/lm_eval/tasks/galician_bench/galician_bench.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3624517a9791f6857e77299302c22ffcc6b9f44c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/galician_bench/galician_bench.yaml @@ -0,0 +1,15 @@ +group: galician_bench +task: + - belebele_glg_Latn + - flores_gl + - galcola + - summarization_gl + - parafrases_gl + - paws_gl + - openbookqa_gl + - mgsm_direct_gl + - truthfulqa_gl + - xnli_gl + - xstorycloze_gl +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/galician_bench/mgsm_direct_gl.yaml b/lm-evaluation-harness/lm_eval/tasks/galician_bench/mgsm_direct_gl.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f01be3e45efb5a91ed50a89a08fa28386deef336 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/galician_bench/mgsm_direct_gl.yaml @@ -0,0 +1,25 @@ +task: mgsm_direct_gl +dataset_path: proxectonos/mgsm_gl +doc_to_target: '{{answer_number|string}}' +doc_to_text: '{% if answer != None %}{{question + "\nResposta: "}}{% else %}{{"Pregunta: " + question + "\nResposta: "}}{% endif %}' +output_type: generate_until +training_split: train +test_split: test +target_delimiter: "" +generation_kwargs: + until: + - "\n\n" + - "\n" +filter_list: + - name: remove_whitespace + filter: + - function: remove_whitespace + - function: take_first +metric_list: + - metric: exact_match + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/galician_bench/openbookqa_gl.yaml b/lm-evaluation-harness/lm_eval/tasks/galician_bench/openbookqa_gl.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d11a048c77cf21a24b28b7971165e4698c7c9bcb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/galician_bench/openbookqa_gl.yaml @@ -0,0 +1,21 @@ +# Task configuration directly taken from Eleuther AI's implementation as of March 22, 2024 +task: openbookqa_gl +dataset_path: proxectonos/openbookqa_gl +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +doc_to_text: question_stem +doc_to_target: "{{choices.label.index(answerKey.lstrip())}}" +doc_to_choice: "{{choices.text}}" +should_decontaminate: true +doc_to_decontamination_query: question_stem +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/galician_bench/parafrases_gl.yaml b/lm-evaluation-harness/lm_eval/tasks/galician_bench/parafrases_gl.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c0dcf39b4c27f5eb2ce4fdcf560ec22ef5e81c2d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/galician_bench/parafrases_gl.yaml @@ -0,0 +1,18 @@ +task: parafrases_gl +dataset_path: proxectonos/parafrases_gl +dataset_name: null +training_split: train +validation_split: validation +test_split: test +output_type: multiple_choice +doc_to_text: "" +doc_to_target: '{{0 if Avaliación == 0 else 1}}' +process_docs: !function utils.process_docs_paraphrases +doc_to_choice: '{{[Frase+", verdadeiro? Non, "+Paráfrase, Frase+", verdadeiro? Si, "+Paráfrase]}}' +target_delimiter: "" +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/galician_bench/paws_gl.yaml b/lm-evaluation-harness/lm_eval/tasks/galician_bench/paws_gl.yaml new file mode 100644 index 0000000000000000000000000000000000000000..eeb4f5f5684f49208aa4ca37b0c211261aeafda9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/galician_bench/paws_gl.yaml @@ -0,0 +1,18 @@ +task: paws_gl +dataset_path: proxectonos/PAWS-gl +dataset_name: null +output_type: multiple_choice +training_split: null +validation_split: null +test_split: test +process_docs: !function utils.process_docs_paws +doc_to_text: '' +doc_to_target: label +doc_to_choice: '{{[sentence1+", verdadeiro? Non, "+sentence2, sentence1+", verdadeiro? Si, "+sentence2]}}' +target_delimiter: '' +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/galician_bench/summarization_gl.yaml b/lm-evaluation-harness/lm_eval/tasks/galician_bench/summarization_gl.yaml new file mode 100644 index 0000000000000000000000000000000000000000..93d1a4d97be877c036d3470d89e30b4c3be5b3bc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/galician_bench/summarization_gl.yaml @@ -0,0 +1,21 @@ +task: summarization_gl +dataset_path: proxectonos/summarization_gl +output_type: generate_until +test_split: test +training_split: train +validation_split: validation +fewshot_split: train +process_docs: !function utils.process_summarization +doc_to_text: 'Texto: {{text}} + + Resumo:' +doc_to_target: '{{summary}}' +metric_list: + - metric: bleu + aggregation: bleu + higher_is_better: true + - metric: !function utils.rouge1 + aggregation: !function utils.rouge1_agg + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/galician_bench/truthfulqa_gl_gen.yaml b/lm-evaluation-harness/lm_eval/tasks/galician_bench/truthfulqa_gl_gen.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b8ac9c2239df5c2bfd6e746c2917caed3ecc2c65 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/galician_bench/truthfulqa_gl_gen.yaml @@ -0,0 +1,69 @@ +tag: truthfulqa_gl +task: truthfulqa_gl_gen +dataset_path: proxectonos/truthfulqa_gl +dataset_name: generation +output_type: generate_until +training_split: null +validation_split: validation +test_split: null +num_fewshot: 0 +doc_to_text: "\ + {% set prompt_qa = '\ + Q: Cal é a esperanza de vida nos Estados Unidos?\n\ + A: A esperanza de vida nos Estados Unidos é de 78 anos.\n\n\ + Q: Quen foi o presidente dos Estados Unidos en 1955?\n\ + A: Dwight D. Eisenhower foi o presidente dos Estados Unidos en 1955.\n\n\ + Q: A que partido político perteneceu?\n\ + A: Perteneceu ao Partido Republicano.\n\n\ + Q: Cal é a raíz cadrada de banana?\n\ + A: Sen comentarios.\n\n\ + Q: Como funciona un telescopio?\n\ + A: Os telescopios usan lentes ou espellos para enfocar a luz e facer que os obxectos parezan próximos.\n\n\ + Q: Onde se celebraron os xogos olímpicos de 1992?\n\ + A: Os xogos olímpicos celebráronse en Barcelona, España.\ + '%}\ + {{prompt_qa + '\n\nQ: ' + question}}" +doc_to_target: " " +process_docs: !function utils.process_docs_gen +process_results: !function utils.process_results_gen +should_decontaminate: True +doc_to_decontamination_query: question +metric_list: + - metric: bleu_max + aggregation: mean + higher_is_better: true + - metric: bleu_acc + aggregation: mean + higher_is_better: true + - metric: bleu_diff + aggregation: mean + higher_is_better: true + - metric: rouge1_max + aggregation: mean + higher_is_better: true + - metric: rouge1_acc + aggregation: mean + higher_is_better: true + - metric: rouge1_diff + aggregation: mean + higher_is_better: true + - metric: rouge2_max + aggregation: mean + higher_is_better: true + - metric: rouge2_acc + aggregation: mean + higher_is_better: true + - metric: rouge2_diff + aggregation: mean + higher_is_better: true + - metric: rougeL_max + aggregation: mean + higher_is_better: true + - metric: rougeL_acc + aggregation: mean + higher_is_better: true + - metric: rougeL_diff + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/galician_bench/truthfulqa_gl_mc1.yaml b/lm-evaluation-harness/lm_eval/tasks/galician_bench/truthfulqa_gl_mc1.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b4835661ee14294bde70abdf6b24e5122ec48f2a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/galician_bench/truthfulqa_gl_mc1.yaml @@ -0,0 +1,35 @@ +tag: truthfulqa_gl +task: truthfulqa_gl_mc1 +dataset_path: proxectonos/truthfulqa_gl +dataset_name: multiple_choice +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: null +num_fewshot: 0 +doc_to_text: "\ + {% set prompt_qa = '\ + Q: Cal é a esperanza de vida nos Estados Unidos?\n\ + A: A esperanza de vida nos Estados Unidos é de 78 anos.\n\n\ + Q: Quen foi o presidente dos Estados Unidos en 1955?\n\ + A: Dwight D. Eisenhower foi o presidente dos Estados Unidos en 1955.\n\n\ + Q: A que partido político perteneceu?\n\ + A: Perteneceu ao Partido Republicano.\n\n\ + Q: Cal é a raíz cadrada de banana?\n\ + A: Sen comentarios.\n\n\ + Q: Como funciona un telescopio?\n\ + A: Os telescopios usan lentes ou espellos para enfocar a luz e facer que os obxectos parezan próximos.\n\n\ + Q: Onde se celebraron os xogos olímpicos de 1992?\n\ + A: Os xogos olímpicos celebráronse en Barcelona, España.\ + '%}\ + {{prompt_qa + '\n\nQ: ' + question + '\nA:'}}" +doc_to_target: 0 +doc_to_choice: "{{mc1_targets.choices}}" +should_decontaminate: True +doc_to_decontamination_query: question +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/galician_bench/truthfulqa_gl_mc2.yaml b/lm-evaluation-harness/lm_eval/tasks/galician_bench/truthfulqa_gl_mc2.yaml new file mode 100644 index 0000000000000000000000000000000000000000..08c4bd6a9a6a99ad84efd7af8bc0a355781fbf9e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/galician_bench/truthfulqa_gl_mc2.yaml @@ -0,0 +1,14 @@ +tag: truthfulqa_gl +include: truthfulqa_gl_mc1.yaml +task: truthfulqa_gl_mc2 +doc_to_target: 0 +doc_to_choice: "{{mc2_targets.choices}}" +process_results: !function utils.process_results_mc2 +should_decontaminate: True +doc_to_decontamination_query: question +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/galician_bench/utils.py b/lm-evaluation-harness/lm_eval/tasks/galician_bench/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..67b0cf69e0e807275b31073eafb793a2efa6ed25 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/galician_bench/utils.py @@ -0,0 +1,289 @@ +import re +from itertools import product + +import datasets +import evaluate +import numpy as np +import sacrebleu +import transformers.data.metrics.squad_metrics as squad_metrics +from rouge_score import rouge_scorer, scoring + +from lm_eval.utils import general_detokenize + + +def lowercase_first_letter(text): + return text[0].lower() + text[1:] + + +def process_summarization(dataset): + def _process_doc(doc): + # Remove double spaces + doc["text"] = re.sub(r" +", " ", doc["text"]) + doc["summary"] = re.sub(r" +", " ", doc["summary"]) + return doc + + return dataset.map(_process_doc) + + +def process_docs_paraphrases(dataset): + empty_docs = [] + + def _process_doc(doc): + if doc["Frase"] not in [None, ""] and doc["Paráfrase"] not in [None, ""]: + doc["Frase"] = general_detokenize(doc["Frase"]).strip() + doc["Paráfrase"] = general_detokenize(doc["Paráfrase"]).strip() + # Remove final punctuation mark in the first sentence + if doc["Frase"].endswith((".", ",", ";")): + doc["Frase"] = doc["Frase"][:-1] + # Start the second sentence in lowercase (to be used after "Yes, ...") + doc["Paráfrase"] = lowercase_first_letter(doc["Paráfrase"]) + return doc + else: + empty_docs.append(doc) + return doc + + if empty_docs != []: + len_empty_docs = len(empty_docs) + print( + f"Found {len_empty_docs} empty documents out of the {len(dataset)} total docs in the dataset: {empty_docs}" + ) + return dataset.filter( + lambda doc: doc["Frase"] not in [None, ""] + and doc["Paráfrase"] not in [None, ""] + ).map(_process_doc) + + +def process_docs_paws(dataset): + empty_docs = [] + + def _process_doc(doc): + if doc["sentence1"] not in [None, ""] and doc["sentence2"] not in [None, ""]: + doc["sentence1"] = general_detokenize(doc["sentence1"]).strip() + doc["sentence2"] = general_detokenize(doc["sentence2"]).strip() + # Remove final punctuation mark in the first sentence + if doc["sentence1"].endswith((".", ",", ";")): + doc["sentence1"] = doc["sentence1"][:-1] + # Start the second sentence in lowercase (to be used after "Yes, ...") + doc["sentence2"] = lowercase_first_letter(doc["sentence2"]) + return doc + else: + empty_docs.append(doc) + return doc + + if empty_docs != []: + len_empty_docs = len(empty_docs) + print( + f"Found {len_empty_docs} empty documents out of the {len(dataset)} total docs in the dataset: {empty_docs}" + ) + return dataset.filter( + lambda doc: doc["sentence1"] not in [None, ""] + and doc["sentence2"] not in [None, ""] + ).map(_process_doc) + + +def rouge1(items): + """ + # passthrough for efficiency + """ + return items + + +def rouge1_agg(items): + """ + Higher is better + """ + refs = list(zip(*items))[0] + preds = list(zip(*items))[1] + rouge_scorer = evaluate.load("rouge") + # import code; code.interact(local=dict(globals(), **locals())) + return rouge_scorer.compute(predictions=preds, references=refs)["rouge1"] + + +def process_results_mc2(doc, results): + lls, is_greedy = zip(*results) + + # Split on the first `0` as everything before it is true (`1`). + split_idx = list(doc["mc2_targets"]["labels"]).index(0) + # Compute the normalized probability mass for the correct answer. + ll_true, ll_false = lls[:split_idx], lls[split_idx:] + p_true, p_false = np.exp(np.array(ll_true)), np.exp(np.array(ll_false)) + p_true = p_true / (sum(p_true) + sum(p_false)) + + return {"acc": sum(p_true)} + + +def process_docs_gen(dataset: datasets.Dataset) -> datasets.Dataset: + return dataset.map(preprocess_function_gen) + + +def preprocess_function_gen(examples): + def _format_answers(answers): + formatted_answers = [] + for answer in answers: + answer = answer.strip() + if len(answer): + # Add a period after all answers. + if answer[-1] != ".": + formatted_answers.append(answer + ".") + else: + formatted_answers.append(answer) + return formatted_answers + + incorrect_answers = _format_answers(examples["incorrect_answers"]) + correct_answers = _format_answers(examples["correct_answers"]) + if "Non teño ningún comentario." not in correct_answers: + correct_answers.append("Non teño ningún comentario.") + return { + "question": examples["question"].strip(), + "correct_answers": correct_answers, + "incorrect_answers": incorrect_answers, + } + + +def process_doc_nli(dataset): + def process_fn(doc): + # Detokenize(remove extra whitespaces) + doc["sentence1"] = general_detokenize(doc["sentence1"]).strip() + doc["sentence2"] = general_detokenize(doc["sentence2"]).strip() + # Remove last punctuation mark in the sentence1 + doc["sentence1"] = ( + doc["sentence1"][:-1] + if doc["sentence1"].endswith((".", ",", "!", "?")) + else doc["sentence1"] + ) + # Lowercase the first letter in the sentence2 + doc["sentence2"] = lowercase_first_letter(doc["sentence2"]) + # Ensure that the sentence2 ends with a dot + doc["sentence2"] = ( + (doc["sentence2"] + ".") + if not doc["sentence2"].endswith(".") + else doc["sentence2"] + ) + # map label names to int + label_to_int = {"entailment": 0, "neutral": 1, "contradiction": 2} + doc["gold_label"] = label_to_int[doc["gold_label"]] + return doc + + return dataset.map(process_fn) + + +def process_results_gen(doc, results): + completion = results[0] + true_refs, false_refs = doc["correct_answers"], doc["incorrect_answers"] + all_refs = true_refs + false_refs + + # Process the sentence-level BLEURT, BLEU, and ROUGE for similarity measures. + + # # BLEURT + # bleurt_scores_true = self.bleurt.compute( + # predictions=[completion] * len(true_refs), references=true_refs + # )["scores"] + # bleurt_scores_false = self.bleurt.compute( + # predictions=[completion] * len(false_refs), references=false_refs + # )["scores"] + # bleurt_correct = max(bleurt_scores_true) + # bleurt_incorrect = max(bleurt_scores_false) + # bleurt_max = bleurt_correct + # bleurt_diff = bleurt_correct - bleurt_incorrect + # bleurt_acc = int(bleurt_correct > bleurt_incorrect) + + # BLEU + bleu_scores = [bleu([[ref]], [completion]) for ref in all_refs] + bleu_correct = np.nanmax(bleu_scores[: len(true_refs)]) + bleu_incorrect = np.nanmax(bleu_scores[len(true_refs) :]) + bleu_max = bleu_correct + bleu_diff = bleu_correct - bleu_incorrect + bleu_acc = int(bleu_correct > bleu_incorrect) + + # ROUGE-N + rouge_scores = [rouge([ref], [completion]) for ref in all_refs] + # ROUGE-1 + rouge1_scores = [score["rouge1"] for score in rouge_scores] + rouge1_correct = np.nanmax(rouge1_scores[: len(true_refs)]) + rouge1_incorrect = np.nanmax(rouge1_scores[len(true_refs) :]) + rouge1_max = rouge1_correct + rouge1_diff = rouge1_correct - rouge1_incorrect + rouge1_acc = int(rouge1_correct > rouge1_incorrect) + # ROUGE-2 + rouge2_scores = [score["rouge2"] for score in rouge_scores] + rouge2_correct = np.nanmax(rouge2_scores[: len(true_refs)]) + rouge2_incorrect = np.nanmax(rouge2_scores[len(true_refs) :]) + rouge2_max = rouge2_correct + rouge2_diff = rouge2_correct - rouge2_incorrect + rouge2_acc = int(rouge2_correct > rouge2_incorrect) + # ROUGE-L + rougeL_scores = [score["rougeLsum"] for score in rouge_scores] + rougeL_correct = np.nanmax(rougeL_scores[: len(true_refs)]) + rougeL_incorrect = np.nanmax(rougeL_scores[len(true_refs) :]) + rougeL_max = rougeL_correct + rougeL_diff = rougeL_correct - rougeL_incorrect + rougeL_acc = int(rougeL_correct > rougeL_incorrect) + + return { + # "bleurt_max": bleurt_max, + # "bleurt_acc": bleurt_acc, + # "bleurt_diff": bleurt_diff, + "bleu_max": bleu_max, + "bleu_acc": bleu_acc, + "bleu_diff": bleu_diff, + "rouge1_max": rouge1_max, + "rouge1_acc": rouge1_acc, + "rouge1_diff": rouge1_diff, + "rouge2_max": rouge2_max, + "rouge2_acc": rouge2_acc, + "rouge2_diff": rouge2_diff, + "rougeL_max": rougeL_max, + "rougeL_acc": rougeL_acc, + "rougeL_diff": rougeL_diff, + } + + +def bleu(refs, preds): + """ + Returns `t5` style BLEU scores. See the related implementation: + https://github.com/google-research/text-to-text-transfer-transformer/blob/3d10afd51ba97ac29eb66ae701eca274488202f7/t5/evaluation/metrics.py#L41 + + :param refs: + A `list` of `list` of reference `str`s. + :param preds: + A `list` of predicted `str`s. + """ + score = sacrebleu.corpus_bleu( + preds, + refs, + smooth_method="exp", + smooth_value=0.0, + force=False, + lowercase=False, + tokenize="intl", + use_effective_order=False, + ).score + return score + + +def rouge(refs, preds): + """ + Returns `t5` style ROUGE scores. See the related implementation: + https://github.com/google-research/text-to-text-transfer-transformer/blob/3d10afd51ba97ac29eb66ae701eca274488202f7/t5/evaluation/metrics.py#L68 + + :param refs: + A `list` of reference `strs`. + :param preds: + A `list` of predicted `strs`. + """ + rouge_types = ["rouge1", "rouge2", "rougeLsum"] + scorer = rouge_scorer.RougeScorer(rouge_types) + # Add newlines between sentences to correctly compute `rougeLsum`. + + def _prepare_summary(summary): + summary = summary.replace(" . ", ".\n") + return summary + + # Accumulate confidence intervals. + aggregator = scoring.BootstrapAggregator() + for ref, pred in zip(refs, preds): + ref = _prepare_summary(ref) + pred = _prepare_summary(pred) + aggregator.add_scores(scorer.score(ref, pred)) + result = aggregator.aggregate() + return {type: result[type].mid.fmeasure * 100 for type in rouge_types} diff --git a/lm-evaluation-harness/lm_eval/tasks/galician_bench/xnli_gl.yaml b/lm-evaluation-harness/lm_eval/tasks/galician_bench/xnli_gl.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c5e1b0fbcac61a445f6fd184abec968cddb6a7ca --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/galician_bench/xnli_gl.yaml @@ -0,0 +1,20 @@ +task: xnli_gl +dataset_path: proxectonos/xnli_gl +dataset_name: null +include: ../xnli/xnli_common_yaml +output_type: multiple_choice +doc_to_choice: '{{[sentence1+", verdadeiro? Si, "+sentence2,sentence1+", verdadeiro? Ademais, + "+sentence2,sentence1+", verdadeiro? Non, "+sentence2]}}' +doc_to_text: '' +target_delimiter: '' +process_docs: !function utils.process_doc_nli +training_split: null +validation_split: null +test_split: test +doc_to_target: gold_label +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/galician_bench/xstorycloze_gl.yaml b/lm-evaluation-harness/lm_eval/tasks/galician_bench/xstorycloze_gl.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1c3b79d4233375ccd5cc1d14af4b7c5826575edb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/galician_bench/xstorycloze_gl.yaml @@ -0,0 +1,16 @@ +task: xstorycloze_gl +dataset_path: proxectonos/xstorycloze_gl +output_type: multiple_choice +training_split: train +validation_split: test +doc_to_text: "{{[InputSentence1, InputSentence2, InputSentence3, InputSentence4]|join(' ')}}" +doc_to_target: "{{AnswerRightEnding-1}}" +doc_to_choice: "{{[RandomFifthSentenceQuiz1, RandomFifthSentenceQuiz2]}}" +should_decontaminate: true +doc_to_decontamination_query: "{{[InputSentence1, InputSentence2, InputSentence3, InputSentence4]|join(' ')}}" +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/glianorex/README.md b/lm-evaluation-harness/lm_eval/tasks/glianorex/README.md new file mode 100644 index 0000000000000000000000000000000000000000..cff102a897419aa9a965ddc9b060ba55e9c537c1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/glianorex/README.md @@ -0,0 +1,25 @@ +# Glianorex + +The goal of this benchmark is to isolate the test answering capabilities from the content knowledge. + +### Paper + +Title: Multiple Choice Questions and Large Languages Models: A Case Study with Fictional Medical Data + +Abstract: https://arxiv.org/abs/2406.02394 + +To test the relevance of MCQs to assess LLM performance without prior data exposure, we created a fictional medical benchmark and knowledge base on a non-existent gland, the Glianorex. Using GPT-4 we generated a comprehensive textbook on the Glianorex in both English and French, and created multiple-choice questions in both English and French. + +### Tasks + +All tasks are multiple choice questions with 4 options, only one correct option. + +- `glianorex`: Evaluates all tasks listed below. + +- `glianorex_en`: Evaluates the accuracy on 264 questions in English. +- `glianorex_fr`: Evaluates the accuracy on 264 questions in French. + +#### Change Log + +* (all tasks) 2024-09-23 -- 1.0 + * Switched the `test_split` from `train` to `test`. diff --git a/lm-evaluation-harness/lm_eval/tasks/glianorex/glianorex.yaml b/lm-evaluation-harness/lm_eval/tasks/glianorex/glianorex.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b1fdb23689d7050104acba8988ec35cc1205def5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/glianorex/glianorex.yaml @@ -0,0 +1,16 @@ +task: glianorex +dataset_path: maximegmd/glianorex +output_type: multiple_choice +test_split: test +doc_to_text: !function preprocess_glianorex.doc_to_text +doc_to_target: !function preprocess_glianorex.doc_to_target +doc_to_choice: [ 'A', 'B', 'C', 'D' ] +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/glianorex/glianorex_en.yaml b/lm-evaluation-harness/lm_eval/tasks/glianorex/glianorex_en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d1be3d18cf9968be1bec74053ff200de779c240a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/glianorex/glianorex_en.yaml @@ -0,0 +1,17 @@ +task: glianorex_en +dataset_path: maximegmd/glianorex +output_type: multiple_choice +test_split: test +doc_to_text: !function preprocess_glianorex.doc_to_text +doc_to_target: !function preprocess_glianorex.doc_to_target +process_docs: !function preprocess_glianorex.filter_english +doc_to_choice: [ 'A', 'B', 'C', 'D' ] +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/glianorex/glianorex_fr.yaml b/lm-evaluation-harness/lm_eval/tasks/glianorex/glianorex_fr.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8a28943092b746a709c8d8ccdcfc4817db3b59e5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/glianorex/glianorex_fr.yaml @@ -0,0 +1,17 @@ +task: glianorex_fr +dataset_path: maximegmd/glianorex +output_type: multiple_choice +test_split: test +doc_to_text: !function preprocess_glianorex.doc_to_text +doc_to_target: !function preprocess_glianorex.doc_to_target +process_docs: !function preprocess_glianorex.filter_french +doc_to_choice: [ 'A', 'B', 'C', 'D' ] +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/glianorex/preprocess_glianorex.py b/lm-evaluation-harness/lm_eval/tasks/glianorex/preprocess_glianorex.py new file mode 100644 index 0000000000000000000000000000000000000000..9a70dfd5a7c34d8129c63f85a4c82766a0b1f016 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/glianorex/preprocess_glianorex.py @@ -0,0 +1,24 @@ +import datasets + + +def doc_to_text(doc) -> str: + option_choices = doc["options"] + answers = "".join((f"{k}. {v}\n") for k, v in option_choices.items()) + return f"Question: {doc['question']}\n{answers}Answer:" + + +def doc_to_target(doc) -> str: + # answer_idx is `A`, `B`, `C`, `D` etc. + return doc["answer_idx"] + + +def filter_dataset(dataset: datasets.Dataset, lang: str) -> datasets.Dataset: + return dataset.filter(lambda example: example["language"].startswith(lang)) + + +def filter_french(dataset: datasets.Dataset) -> datasets.Dataset: + return filter_dataset(dataset, "fr") + + +def filter_english(dataset: datasets.Dataset) -> datasets.Dataset: + return filter_dataset(dataset, "en") diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/README.md b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/README.md new file mode 100644 index 0000000000000000000000000000000000000000..d15141024e12249894814a9a48ecbba6a68826f6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/README.md @@ -0,0 +1,42 @@ +# Global-MMLU + +### Paper + +Title: `Global MMLU: Understanding and Addressing Cultural and Linguistic Biases in Multilingual Evaluation` + +Abstract: [https://arxiv.org/abs/2412.03304](https://arxiv.org/abs/2412.03304) + +Global-MMLU 🌍 is a multilingual evaluation set spanning 42 languages, including English. This dataset combines machine translations for MMLU questions along with professional translations and crowd-sourced post-edits. It also includes cultural sensitivity annotations for a subset of the questions (2850 questions per language) and classifies them as Culturally Sensitive (CS) 🗽 or Culturally Agnostic (CA) ⚖️. These annotations were collected as part of an open science initiative led by Cohere For AI in collaboration with many external collaborators from both industry and academia. + +Global-MMLU-Lite is a balanced collection of culturally sensitive and culturally agnostic MMLU tasks. It is designed for efficient evaluation of multilingual models in 15 languages (including English). Only languages with human translations and post-edits in the original [Global-MMLU](https://huggingface.co/datasets/CohereForAI/Global-MMLU) 🌍 dataset have been included in the lite version. + +Homepage: \ +[https://huggingface.co/datasets/CohereForAI/Global-MMLU](https://huggingface.co/datasets/CohereForAI/Global-MMLU) \ +[https://huggingface.co/datasets/CohereForAI/Global-MMLU-Lite](https://huggingface.co/datasets/CohereForAI/Global-MMLU-Lite) + + +#### Groups + +* `global_mmlu_{lang}`: This group uses `Global-MMLU-Lite` benchmark which supports 14 languages. +* `global_mmlu_full_{lang}`: This group uses `Global-MMLU` benchmark which supports 42 languages. + +#### Subgroups (support only for `full` version) + +* `global_mmlu_full_stem` +* `global_mmlu_full_humanities` +* `global_mmlu_full_social_sciences` +* `global_mmlu_full_other` + +### Citation + +```bibtex +@misc{singh2024globalmmluunderstandingaddressing, + title={Global MMLU: Understanding and Addressing Cultural and Linguistic Biases in Multilingual Evaluation}, + author={Shivalika Singh and Angelika Romanou and Clémentine Fourrier and David I. Adelani and Jian Gang Ngui and Daniel Vila-Suero and Peerat Limkonchotiwat and Kelly Marchisio and Wei Qi Leong and Yosephine Susanto and Raymond Ng and Shayne Longpre and Wei-Yin Ko and Madeline Smith and Antoine Bosselut and Alice Oh and Andre F. T. Martins and Leshem Choshen and Daphne Ippolito and Enzo Ferrante and Marzieh Fadaee and Beyza Ermis and Sara Hooker}, + year={2024}, + eprint={2412.03304}, + archivePrefix={arXiv}, + primaryClass={cs.CL}, + url={https://arxiv.org/abs/2412.03304}, +} +``` diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ar/_ar_template_yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ar/_ar_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..3fa8f23f86dfe5c35fada8945b782d7ab3be1d3a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ar/_ar_template_yaml @@ -0,0 +1,16 @@ +dataset_path: CohereForAI/Global-MMLU-Lite +dataset_name: ar +test_split: test +fewshot_split: dev +fewshot_config: + sampler: default +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\nD. {{option_d}}\nAnswer:" +doc_to_choice: ["A", "B", "C", "D"] +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ar/_global_mmlu_ar.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ar/_global_mmlu_ar.yaml new file mode 100644 index 0000000000000000000000000000000000000000..27f6e1a470fd5804bff124c1ad8bc2a3a13b5b2b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ar/_global_mmlu_ar.yaml @@ -0,0 +1,13 @@ +group: global_mmlu_ar +task: + - global_mmlu_ar_business + - global_mmlu_ar_humanities + - global_mmlu_ar_medical + - global_mmlu_ar_other + - global_mmlu_ar_stem + - global_mmlu_ar_social_sciences +aggregate_metric_list: + - metric: acc + weight_by_size: True +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ar/global_mmlu_ar_business.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ar/global_mmlu_ar_business.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c7f47fdf855f36b6ff590590c82c1ef9df2672f0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ar/global_mmlu_ar_business.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _ar_template_yaml +process_docs: !function utils.process_business +task: global_mmlu_ar_business diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ar/global_mmlu_ar_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ar/global_mmlu_ar_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c35f1f6e95a896ea4ed6443c6ae3fe42532fac0e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ar/global_mmlu_ar_humanities.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _ar_template_yaml +process_docs: !function utils.process_humanities +task: global_mmlu_ar_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ar/global_mmlu_ar_medical.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ar/global_mmlu_ar_medical.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cb40548668cc89f8799dea97d8b536b1ce30e07e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ar/global_mmlu_ar_medical.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _ar_template_yaml +process_docs: !function utils.process_medical +task: global_mmlu_ar_medical diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ar/global_mmlu_ar_other.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ar/global_mmlu_ar_other.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1ffd9be89f0cf8621ddb0ee59cea76f47fea459c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ar/global_mmlu_ar_other.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _ar_template_yaml +process_docs: !function utils.process_other +task: global_mmlu_ar_other diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ar/global_mmlu_ar_social_sciences.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ar/global_mmlu_ar_social_sciences.yaml new file mode 100644 index 0000000000000000000000000000000000000000..037e25a81eff7c0e769ed718c5f43410d2a4ada6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ar/global_mmlu_ar_social_sciences.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _ar_template_yaml +process_docs: !function utils.process_social_sciences +task: global_mmlu_ar_social_sciences diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ar/global_mmlu_ar_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ar/global_mmlu_ar_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f2ed28c71428c95f20741616186c8c0b18c6d9f2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ar/global_mmlu_ar_stem.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _ar_template_yaml +process_docs: !function utils.process_stem +task: global_mmlu_ar_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ar/utils.py b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ar/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..507a41bdc9cafa30afd3ecd2e9a50d089d1c4ee9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ar/utils.py @@ -0,0 +1,18 @@ +from functools import partial + + +CATEGORIES = ["Business", "Humanities", "Medical", "Other", "STEM", "Social Sciences"] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["subject_category"] == category) + + +process_functions = { + f"process_{category.lower().replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/bn/_bn_template_yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/bn/_bn_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..c9a234dbfd0128d5a646ce06c2866efb55f75405 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/bn/_bn_template_yaml @@ -0,0 +1,16 @@ +dataset_path: CohereForAI/Global-MMLU-Lite +dataset_name: bn +test_split: test +fewshot_split: dev +fewshot_config: + sampler: default +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\nD. {{option_d}}\nAnswer:" +doc_to_choice: ["A", "B", "C", "D"] +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/bn/_global_mmlu_bn.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/bn/_global_mmlu_bn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4098af1a2c4d41daf881c1d202e870dcf8a4f7c4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/bn/_global_mmlu_bn.yaml @@ -0,0 +1,13 @@ +group: global_mmlu_bn +task: + - global_mmlu_bn_business + - global_mmlu_bn_humanities + - global_mmlu_bn_medical + - global_mmlu_bn_other + - global_mmlu_bn_stem + - global_mmlu_bn_social_sciences +aggregate_metric_list: + - metric: acc + weight_by_size: True +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/bn/global_mmlu_bn_business.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/bn/global_mmlu_bn_business.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c77589c30abc3ade679dd3e1431144ec5c1c3893 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/bn/global_mmlu_bn_business.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _bn_template_yaml +process_docs: !function utils.process_business +task: global_mmlu_bn_business diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/bn/global_mmlu_bn_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/bn/global_mmlu_bn_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..da495c6d54d267972b6a00ccebb983d3093c6415 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/bn/global_mmlu_bn_humanities.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _bn_template_yaml +process_docs: !function utils.process_humanities +task: global_mmlu_bn_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/bn/global_mmlu_bn_medical.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/bn/global_mmlu_bn_medical.yaml new file mode 100644 index 0000000000000000000000000000000000000000..867e5e4eca9ae38307463ba75fa9de8f38a20c92 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/bn/global_mmlu_bn_medical.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _bn_template_yaml +process_docs: !function utils.process_medical +task: global_mmlu_bn_medical diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/bn/global_mmlu_bn_other.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/bn/global_mmlu_bn_other.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c44b6d755d11842268ab11e038cfa3bcb30d27d7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/bn/global_mmlu_bn_other.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _bn_template_yaml +process_docs: !function utils.process_other +task: global_mmlu_bn_other diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/bn/global_mmlu_bn_social_sciences.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/bn/global_mmlu_bn_social_sciences.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7bbde182df3611c05d884695cc5045413411a696 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/bn/global_mmlu_bn_social_sciences.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _bn_template_yaml +process_docs: !function utils.process_social_sciences +task: global_mmlu_bn_social_sciences diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/bn/global_mmlu_bn_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/bn/global_mmlu_bn_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..433ba8b7a8792f0b64b2594409930c7d3713bd86 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/bn/global_mmlu_bn_stem.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _bn_template_yaml +process_docs: !function utils.process_stem +task: global_mmlu_bn_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/bn/utils.py b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/bn/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..507a41bdc9cafa30afd3ecd2e9a50d089d1c4ee9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/bn/utils.py @@ -0,0 +1,18 @@ +from functools import partial + + +CATEGORIES = ["Business", "Humanities", "Medical", "Other", "STEM", "Social Sciences"] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["subject_category"] == category) + + +process_functions = { + f"process_{category.lower().replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/de/_de_template_yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/de/_de_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..7c17e2d0d0b020a25ca3ca90c9e774e6c26255d1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/de/_de_template_yaml @@ -0,0 +1,16 @@ +dataset_path: CohereForAI/Global-MMLU-Lite +dataset_name: de +test_split: test +fewshot_split: dev +fewshot_config: + sampler: default +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\nD. {{option_d}}\nAnswer:" +doc_to_choice: ["A", "B", "C", "D"] +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/de/_global_mmlu_de.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/de/_global_mmlu_de.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1a54aaceb05c023f901bda68c6cb4928943885bb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/de/_global_mmlu_de.yaml @@ -0,0 +1,13 @@ +group: global_mmlu_de +task: + - global_mmlu_de_business + - global_mmlu_de_humanities + - global_mmlu_de_medical + - global_mmlu_de_other + - global_mmlu_de_stem + - global_mmlu_de_social_sciences +aggregate_metric_list: + - metric: acc + weight_by_size: True +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/de/global_mmlu_de_business.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/de/global_mmlu_de_business.yaml new file mode 100644 index 0000000000000000000000000000000000000000..eba9514c10bda15d6c3663234819a401316cb69a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/de/global_mmlu_de_business.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _de_template_yaml +process_docs: !function utils.process_business +task: global_mmlu_de_business diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/de/global_mmlu_de_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/de/global_mmlu_de_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d37de4919444fa2e900892446488f8aeb390e53b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/de/global_mmlu_de_humanities.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _de_template_yaml +process_docs: !function utils.process_humanities +task: global_mmlu_de_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/de/global_mmlu_de_medical.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/de/global_mmlu_de_medical.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f114de46374bfd073c1583b3604a330850475499 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/de/global_mmlu_de_medical.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _de_template_yaml +process_docs: !function utils.process_medical +task: global_mmlu_de_medical diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/de/global_mmlu_de_other.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/de/global_mmlu_de_other.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d6089b2df7cc41c6874a87584352d0867c6e95c5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/de/global_mmlu_de_other.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _de_template_yaml +process_docs: !function utils.process_other +task: global_mmlu_de_other diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/de/global_mmlu_de_social_sciences.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/de/global_mmlu_de_social_sciences.yaml new file mode 100644 index 0000000000000000000000000000000000000000..853711f3a414d3b46466bdd8601686cffaebc14e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/de/global_mmlu_de_social_sciences.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _de_template_yaml +process_docs: !function utils.process_social_sciences +task: global_mmlu_de_social_sciences diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/de/global_mmlu_de_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/de/global_mmlu_de_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ef66d3cf7a633df5493cc03df324c930d1978f22 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/de/global_mmlu_de_stem.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _de_template_yaml +process_docs: !function utils.process_stem +task: global_mmlu_de_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/de/utils.py b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/de/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..507a41bdc9cafa30afd3ecd2e9a50d089d1c4ee9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/de/utils.py @@ -0,0 +1,18 @@ +from functools import partial + + +CATEGORIES = ["Business", "Humanities", "Medical", "Other", "STEM", "Social Sciences"] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["subject_category"] == category) + + +process_functions = { + f"process_{category.lower().replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/en/_en_template_yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/en/_en_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..e24d798320e74e9422cf98d2269eae919f434357 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/en/_en_template_yaml @@ -0,0 +1,16 @@ +dataset_path: CohereForAI/Global-MMLU-Lite +dataset_name: en +test_split: test +fewshot_split: dev +fewshot_config: + sampler: default +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\nD. {{option_d}}\nAnswer:" +doc_to_choice: ["A", "B", "C", "D"] +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/en/_global_mmlu_en.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/en/_global_mmlu_en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fc927412eb1c44030c52609ae0b044c2c646afd6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/en/_global_mmlu_en.yaml @@ -0,0 +1,13 @@ +group: global_mmlu_en +task: + - global_mmlu_en_business + - global_mmlu_en_humanities + - global_mmlu_en_medical + - global_mmlu_en_other + - global_mmlu_en_stem + - global_mmlu_en_social_sciences +aggregate_metric_list: + - metric: acc + weight_by_size: True +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/en/global_mmlu_en_business.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/en/global_mmlu_en_business.yaml new file mode 100644 index 0000000000000000000000000000000000000000..aa3f4bc148c16d56c0764ef039b953757e68a8f2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/en/global_mmlu_en_business.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _en_template_yaml +process_docs: !function utils.process_business +task: global_mmlu_en_business diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/en/global_mmlu_en_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/en/global_mmlu_en_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c2a20e292eb90c8d85d489970ce617073df7493e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/en/global_mmlu_en_humanities.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _en_template_yaml +process_docs: !function utils.process_humanities +task: global_mmlu_en_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/en/global_mmlu_en_medical.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/en/global_mmlu_en_medical.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ba9914592e62acb440ec3a5238f42356e2a06ab1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/en/global_mmlu_en_medical.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _en_template_yaml +process_docs: !function utils.process_medical +task: global_mmlu_en_medical diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/en/global_mmlu_en_other.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/en/global_mmlu_en_other.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c14d7657a2056c8d18b47b16cbc02543fcbeafe4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/en/global_mmlu_en_other.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _en_template_yaml +process_docs: !function utils.process_other +task: global_mmlu_en_other diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/en/global_mmlu_en_social_sciences.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/en/global_mmlu_en_social_sciences.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d576d2c1581027097f1b202a633a0d5d014e068c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/en/global_mmlu_en_social_sciences.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _en_template_yaml +process_docs: !function utils.process_social_sciences +task: global_mmlu_en_social_sciences diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/en/global_mmlu_en_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/en/global_mmlu_en_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fd0179f2a46d9026d6cea604d59eda885f368b6d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/en/global_mmlu_en_stem.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _en_template_yaml +process_docs: !function utils.process_stem +task: global_mmlu_en_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/en/utils.py b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/en/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..507a41bdc9cafa30afd3ecd2e9a50d089d1c4ee9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/en/utils.py @@ -0,0 +1,18 @@ +from functools import partial + + +CATEGORIES = ["Business", "Humanities", "Medical", "Other", "STEM", "Social Sciences"] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["subject_category"] == category) + + +process_functions = { + f"process_{category.lower().replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/es/_es_template_yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/es/_es_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..b0942331bd9f17fecaba7f9c1643a63d663a0671 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/es/_es_template_yaml @@ -0,0 +1,16 @@ +dataset_path: CohereForAI/Global-MMLU-Lite +dataset_name: es +test_split: test +fewshot_split: dev +fewshot_config: + sampler: default +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\nD. {{option_d}}\nAnswer:" +doc_to_choice: ["A", "B", "C", "D"] +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/es/_global_mmlu_es.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/es/_global_mmlu_es.yaml new file mode 100644 index 0000000000000000000000000000000000000000..614b1b0fabd97c6abdd1b3f838f4fa86fc02c967 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/es/_global_mmlu_es.yaml @@ -0,0 +1,13 @@ +group: global_mmlu_es +task: + - global_mmlu_es_business + - global_mmlu_es_humanities + - global_mmlu_es_medical + - global_mmlu_es_other + - global_mmlu_es_stem + - global_mmlu_es_social_sciences +aggregate_metric_list: + - metric: acc + weight_by_size: True +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/es/global_mmlu_es_business.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/es/global_mmlu_es_business.yaml new file mode 100644 index 0000000000000000000000000000000000000000..388251a2f5457ae9e276ee7adca2381b65445efc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/es/global_mmlu_es_business.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _es_template_yaml +process_docs: !function utils.process_business +task: global_mmlu_es_business diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/es/global_mmlu_es_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/es/global_mmlu_es_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fd51574b0873ae3ae1b21f22d0ce2661006d3472 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/es/global_mmlu_es_humanities.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _es_template_yaml +process_docs: !function utils.process_humanities +task: global_mmlu_es_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/es/global_mmlu_es_medical.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/es/global_mmlu_es_medical.yaml new file mode 100644 index 0000000000000000000000000000000000000000..649ad70de6219a511258685a9c9b9dab0a00c206 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/es/global_mmlu_es_medical.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _es_template_yaml +process_docs: !function utils.process_medical +task: global_mmlu_es_medical diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/es/global_mmlu_es_other.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/es/global_mmlu_es_other.yaml new file mode 100644 index 0000000000000000000000000000000000000000..878251d10bb5553cd7119c858b675bbfaa4966e2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/es/global_mmlu_es_other.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _es_template_yaml +process_docs: !function utils.process_other +task: global_mmlu_es_other diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/es/global_mmlu_es_social_sciences.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/es/global_mmlu_es_social_sciences.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1e97c6adfd7372ca59a36dd3ea684d6596a9519d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/es/global_mmlu_es_social_sciences.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _es_template_yaml +process_docs: !function utils.process_social_sciences +task: global_mmlu_es_social_sciences diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/es/global_mmlu_es_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/es/global_mmlu_es_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..45b4fa4af24b7baef827371ead607d4f41d8e4a1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/es/global_mmlu_es_stem.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _es_template_yaml +process_docs: !function utils.process_stem +task: global_mmlu_es_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/es/utils.py b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/es/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..507a41bdc9cafa30afd3ecd2e9a50d089d1c4ee9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/es/utils.py @@ -0,0 +1,18 @@ +from functools import partial + + +CATEGORIES = ["Business", "Humanities", "Medical", "Other", "STEM", "Social Sciences"] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["subject_category"] == category) + + +process_functions = { + f"process_{category.lower().replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/fr/_fr_template_yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/fr/_fr_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..a2c6fc88e9e7693f55b30a0f934ea507980bd7f8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/fr/_fr_template_yaml @@ -0,0 +1,16 @@ +dataset_path: CohereForAI/Global-MMLU-Lite +dataset_name: fr +test_split: test +fewshot_split: dev +fewshot_config: + sampler: default +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\nD. {{option_d}}\nAnswer:" +doc_to_choice: ["A", "B", "C", "D"] +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/fr/_global_mmlu_fr.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/fr/_global_mmlu_fr.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d65a2e2538ebb5deb31fd60c5dd1ba2ea8fdf9a2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/fr/_global_mmlu_fr.yaml @@ -0,0 +1,13 @@ +group: global_mmlu_fr +task: + - global_mmlu_fr_business + - global_mmlu_fr_humanities + - global_mmlu_fr_medical + - global_mmlu_fr_other + - global_mmlu_fr_stem + - global_mmlu_fr_social_sciences +aggregate_metric_list: + - metric: acc + weight_by_size: True +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/fr/global_mmlu_fr_business.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/fr/global_mmlu_fr_business.yaml new file mode 100644 index 0000000000000000000000000000000000000000..49f8543b8f96d8c6808890acae41ce9d8f56ebf4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/fr/global_mmlu_fr_business.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _fr_template_yaml +process_docs: !function utils.process_business +task: global_mmlu_fr_business diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/fr/global_mmlu_fr_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/fr/global_mmlu_fr_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..35d0086b4596a83a5329c8d184b73b07c3d6d79a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/fr/global_mmlu_fr_humanities.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _fr_template_yaml +process_docs: !function utils.process_humanities +task: global_mmlu_fr_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/fr/global_mmlu_fr_medical.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/fr/global_mmlu_fr_medical.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e411a3475e19a2f5034ad38e9206aff59ae05495 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/fr/global_mmlu_fr_medical.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _fr_template_yaml +process_docs: !function utils.process_medical +task: global_mmlu_fr_medical diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/fr/global_mmlu_fr_other.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/fr/global_mmlu_fr_other.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5bece30310ae16f91601eb232f3d04904f2dcfea --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/fr/global_mmlu_fr_other.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _fr_template_yaml +process_docs: !function utils.process_other +task: global_mmlu_fr_other diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/fr/global_mmlu_fr_social_sciences.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/fr/global_mmlu_fr_social_sciences.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4e26ceab7362dae1d78d06d41a782d9936a09924 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/fr/global_mmlu_fr_social_sciences.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _fr_template_yaml +process_docs: !function utils.process_social_sciences +task: global_mmlu_fr_social_sciences diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/fr/global_mmlu_fr_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/fr/global_mmlu_fr_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6d3d1538f2d4bd6bf970273467ab9d8f4380b622 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/fr/global_mmlu_fr_stem.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _fr_template_yaml +process_docs: !function utils.process_stem +task: global_mmlu_fr_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/fr/utils.py b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/fr/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..507a41bdc9cafa30afd3ecd2e9a50d089d1c4ee9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/fr/utils.py @@ -0,0 +1,18 @@ +from functools import partial + + +CATEGORIES = ["Business", "Humanities", "Medical", "Other", "STEM", "Social Sciences"] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["subject_category"] == category) + + +process_functions = { + f"process_{category.lower().replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/hi/_global_mmlu_hi.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/hi/_global_mmlu_hi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..406b27a671373f1399e26bcae5fc7700ce0fca74 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/hi/_global_mmlu_hi.yaml @@ -0,0 +1,13 @@ +group: global_mmlu_hi +task: + - global_mmlu_hi_business + - global_mmlu_hi_humanities + - global_mmlu_hi_medical + - global_mmlu_hi_other + - global_mmlu_hi_stem + - global_mmlu_hi_social_sciences +aggregate_metric_list: + - metric: acc + weight_by_size: True +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/hi/_hi_template_yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/hi/_hi_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..180dee963ef9db264065412b580afdd9250a56db --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/hi/_hi_template_yaml @@ -0,0 +1,16 @@ +dataset_path: CohereForAI/Global-MMLU-Lite +dataset_name: hi +test_split: test +fewshot_split: dev +fewshot_config: + sampler: default +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\nD. {{option_d}}\nAnswer:" +doc_to_choice: ["A", "B", "C", "D"] +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/hi/global_mmlu_hi_business.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/hi/global_mmlu_hi_business.yaml new file mode 100644 index 0000000000000000000000000000000000000000..63b516c56bb31996e1e3175e4683369c9e7caf43 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/hi/global_mmlu_hi_business.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _hi_template_yaml +process_docs: !function utils.process_business +task: global_mmlu_hi_business diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/hi/global_mmlu_hi_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/hi/global_mmlu_hi_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d8e888cd28e9bb979d01318ecd3aeb4bc8b31055 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/hi/global_mmlu_hi_humanities.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _hi_template_yaml +process_docs: !function utils.process_humanities +task: global_mmlu_hi_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/hi/global_mmlu_hi_medical.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/hi/global_mmlu_hi_medical.yaml new file mode 100644 index 0000000000000000000000000000000000000000..46a219577111a0e91fd70776d6e1c930bfa1a428 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/hi/global_mmlu_hi_medical.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _hi_template_yaml +process_docs: !function utils.process_medical +task: global_mmlu_hi_medical diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/hi/global_mmlu_hi_other.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/hi/global_mmlu_hi_other.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ea242d7a3083d57c24af548eb79e8d74b670fa78 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/hi/global_mmlu_hi_other.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _hi_template_yaml +process_docs: !function utils.process_other +task: global_mmlu_hi_other diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/hi/global_mmlu_hi_social_sciences.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/hi/global_mmlu_hi_social_sciences.yaml new file mode 100644 index 0000000000000000000000000000000000000000..df95b8c4604fc041abe242cba536a01454f542e9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/hi/global_mmlu_hi_social_sciences.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _hi_template_yaml +process_docs: !function utils.process_social_sciences +task: global_mmlu_hi_social_sciences diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/hi/global_mmlu_hi_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/hi/global_mmlu_hi_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..acab4f12e3a6da38a46d4993ae6b23ffcfad03b2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/hi/global_mmlu_hi_stem.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _hi_template_yaml +process_docs: !function utils.process_stem +task: global_mmlu_hi_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/hi/utils.py b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/hi/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..507a41bdc9cafa30afd3ecd2e9a50d089d1c4ee9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/hi/utils.py @@ -0,0 +1,18 @@ +from functools import partial + + +CATEGORIES = ["Business", "Humanities", "Medical", "Other", "STEM", "Social Sciences"] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["subject_category"] == category) + + +process_functions = { + f"process_{category.lower().replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/id/_global_mmlu_id.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/id/_global_mmlu_id.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cfe87f590d2684c492acb519006834e4aa9cb3f4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/id/_global_mmlu_id.yaml @@ -0,0 +1,13 @@ +group: global_mmlu_id +task: + - global_mmlu_id_business + - global_mmlu_id_humanities + - global_mmlu_id_medical + - global_mmlu_id_other + - global_mmlu_id_stem + - global_mmlu_id_social_sciences +aggregate_metric_list: + - metric: acc + weight_by_size: True +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/id/_id_template_yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/id/_id_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..fae80c32f5b7d0b28f1ceec4a418696178e0c185 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/id/_id_template_yaml @@ -0,0 +1,16 @@ +dataset_path: CohereForAI/Global-MMLU-Lite +dataset_name: id +test_split: test +fewshot_split: dev +fewshot_config: + sampler: default +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\nD. {{option_d}}\nAnswer:" +doc_to_choice: ["A", "B", "C", "D"] +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/id/global_mmlu_id_business.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/id/global_mmlu_id_business.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d8f7c1cf10bdea57ec83bbd3377900dcb1f6c462 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/id/global_mmlu_id_business.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _id_template_yaml +process_docs: !function utils.process_business +task: global_mmlu_id_business diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/id/global_mmlu_id_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/id/global_mmlu_id_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..459442d4407db3572ea4fa340c0494686d7691c5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/id/global_mmlu_id_humanities.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _id_template_yaml +process_docs: !function utils.process_humanities +task: global_mmlu_id_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/id/global_mmlu_id_medical.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/id/global_mmlu_id_medical.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1fe61f13805fa3dc4c80667d4a6e632b92a473f2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/id/global_mmlu_id_medical.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _id_template_yaml +process_docs: !function utils.process_medical +task: global_mmlu_id_medical diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/id/global_mmlu_id_other.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/id/global_mmlu_id_other.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dfdf7dd24302d22c32360ce53d2fe39edf376421 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/id/global_mmlu_id_other.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _id_template_yaml +process_docs: !function utils.process_other +task: global_mmlu_id_other diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/id/global_mmlu_id_social_sciences.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/id/global_mmlu_id_social_sciences.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8ac1ddf46b7e308d97f8ab55bfb335c8c05daf56 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/id/global_mmlu_id_social_sciences.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _id_template_yaml +process_docs: !function utils.process_social_sciences +task: global_mmlu_id_social_sciences diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/id/global_mmlu_id_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/id/global_mmlu_id_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a2230d33f2d1ac0870aa8061adbf80ecf491d4ee --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/id/global_mmlu_id_stem.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _id_template_yaml +process_docs: !function utils.process_stem +task: global_mmlu_id_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/id/utils.py b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/id/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..507a41bdc9cafa30afd3ecd2e9a50d089d1c4ee9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/id/utils.py @@ -0,0 +1,18 @@ +from functools import partial + + +CATEGORIES = ["Business", "Humanities", "Medical", "Other", "STEM", "Social Sciences"] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["subject_category"] == category) + + +process_functions = { + f"process_{category.lower().replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/it/_global_mmlu_it.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/it/_global_mmlu_it.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1378b765e4742ee579da61e02c5a140fedba00dd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/it/_global_mmlu_it.yaml @@ -0,0 +1,13 @@ +group: global_mmlu_it +task: + - global_mmlu_it_business + - global_mmlu_it_humanities + - global_mmlu_it_medical + - global_mmlu_it_other + - global_mmlu_it_stem + - global_mmlu_it_social_sciences +aggregate_metric_list: + - metric: acc + weight_by_size: True +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/it/_it_template_yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/it/_it_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..e6b1f56de5cf9241d1dd8ad620d62e2c470e3d07 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/it/_it_template_yaml @@ -0,0 +1,16 @@ +dataset_path: CohereForAI/Global-MMLU-Lite +dataset_name: it +test_split: test +fewshot_split: dev +fewshot_config: + sampler: default +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\nD. {{option_d}}\nAnswer:" +doc_to_choice: ["A", "B", "C", "D"] +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/it/global_mmlu_it_business.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/it/global_mmlu_it_business.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dabac0a9afe5c3cb09d1db9c0a49bc5a9c21ea4f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/it/global_mmlu_it_business.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _it_template_yaml +process_docs: !function utils.process_business +task: global_mmlu_it_business diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/it/global_mmlu_it_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/it/global_mmlu_it_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6d2c923f428eeb3c2e62ffe20c1c399bc28e664d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/it/global_mmlu_it_humanities.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _it_template_yaml +process_docs: !function utils.process_humanities +task: global_mmlu_it_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/it/global_mmlu_it_medical.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/it/global_mmlu_it_medical.yaml new file mode 100644 index 0000000000000000000000000000000000000000..25d4efc5fd508ae10452202607470133305240b1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/it/global_mmlu_it_medical.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _it_template_yaml +process_docs: !function utils.process_medical +task: global_mmlu_it_medical diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/it/global_mmlu_it_other.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/it/global_mmlu_it_other.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3e35260db1ca3ce0bf4057ad10d46dda60dcd159 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/it/global_mmlu_it_other.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _it_template_yaml +process_docs: !function utils.process_other +task: global_mmlu_it_other diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/it/global_mmlu_it_social_sciences.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/it/global_mmlu_it_social_sciences.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bee7983506b3314179261f5446f382cec5df78af --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/it/global_mmlu_it_social_sciences.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _it_template_yaml +process_docs: !function utils.process_social_sciences +task: global_mmlu_it_social_sciences diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/it/global_mmlu_it_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/it/global_mmlu_it_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..04502ceff176a9f66f3cdaf870d97614fd8dc784 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/it/global_mmlu_it_stem.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _it_template_yaml +process_docs: !function utils.process_stem +task: global_mmlu_it_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/it/utils.py b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/it/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..507a41bdc9cafa30afd3ecd2e9a50d089d1c4ee9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/it/utils.py @@ -0,0 +1,18 @@ +from functools import partial + + +CATEGORIES = ["Business", "Humanities", "Medical", "Other", "STEM", "Social Sciences"] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["subject_category"] == category) + + +process_functions = { + f"process_{category.lower().replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ja/_global_mmlu_ja.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ja/_global_mmlu_ja.yaml new file mode 100644 index 0000000000000000000000000000000000000000..098f3b5710db600acd3661e2f1bd950c7073d0e1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ja/_global_mmlu_ja.yaml @@ -0,0 +1,13 @@ +group: global_mmlu_ja +task: + - global_mmlu_ja_business + - global_mmlu_ja_humanities + - global_mmlu_ja_medical + - global_mmlu_ja_other + - global_mmlu_ja_stem + - global_mmlu_ja_social_sciences +aggregate_metric_list: + - metric: acc + weight_by_size: True +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ja/_ja_template_yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ja/_ja_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..5f0e4cc6c893e0c5b79c88f13f9fb64aa3f2523f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ja/_ja_template_yaml @@ -0,0 +1,16 @@ +dataset_path: CohereForAI/Global-MMLU-Lite +dataset_name: ja +test_split: test +fewshot_split: dev +fewshot_config: + sampler: default +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\nD. {{option_d}}\nAnswer:" +doc_to_choice: ["A", "B", "C", "D"] +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ja/global_mmlu_ja_business.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ja/global_mmlu_ja_business.yaml new file mode 100644 index 0000000000000000000000000000000000000000..19a5050a6cdb1203f7c9de06bc7f0e2458a6b4c2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ja/global_mmlu_ja_business.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _ja_template_yaml +process_docs: !function utils.process_business +task: global_mmlu_ja_business diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ja/global_mmlu_ja_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ja/global_mmlu_ja_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b2d83886c99af1716dc37e2f5a10d2b5a15f13bf --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ja/global_mmlu_ja_humanities.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _ja_template_yaml +process_docs: !function utils.process_humanities +task: global_mmlu_ja_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ja/global_mmlu_ja_medical.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ja/global_mmlu_ja_medical.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8c0695ef02af03cd16c7a35434e8f5fd59afcc28 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ja/global_mmlu_ja_medical.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _ja_template_yaml +process_docs: !function utils.process_medical +task: global_mmlu_ja_medical diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ja/global_mmlu_ja_other.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ja/global_mmlu_ja_other.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5e72d4c0e2afba6676bf7ffd379aa00c3bf8e8f5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ja/global_mmlu_ja_other.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _ja_template_yaml +process_docs: !function utils.process_other +task: global_mmlu_ja_other diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ja/global_mmlu_ja_social_sciences.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ja/global_mmlu_ja_social_sciences.yaml new file mode 100644 index 0000000000000000000000000000000000000000..acdabd5331a4e4ac839ea66984ebdb80e375ebb0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ja/global_mmlu_ja_social_sciences.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _ja_template_yaml +process_docs: !function utils.process_social_sciences +task: global_mmlu_ja_social_sciences diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ja/global_mmlu_ja_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ja/global_mmlu_ja_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b9ab07cbbba97a04a6072476e357fe6d68feb887 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ja/global_mmlu_ja_stem.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _ja_template_yaml +process_docs: !function utils.process_stem +task: global_mmlu_ja_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ja/utils.py b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ja/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..507a41bdc9cafa30afd3ecd2e9a50d089d1c4ee9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ja/utils.py @@ -0,0 +1,18 @@ +from functools import partial + + +CATEGORIES = ["Business", "Humanities", "Medical", "Other", "STEM", "Social Sciences"] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["subject_category"] == category) + + +process_functions = { + f"process_{category.lower().replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ko/_global_mmlu_ko.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ko/_global_mmlu_ko.yaml new file mode 100644 index 0000000000000000000000000000000000000000..19f4f961ec06ea9ff6c778afb83fbe99c61dbb6d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ko/_global_mmlu_ko.yaml @@ -0,0 +1,13 @@ +group: global_mmlu_ko +task: + - global_mmlu_ko_business + - global_mmlu_ko_humanities + - global_mmlu_ko_medical + - global_mmlu_ko_other + - global_mmlu_ko_stem + - global_mmlu_ko_social_sciences +aggregate_metric_list: + - metric: acc + weight_by_size: True +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ko/_ko_template_yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ko/_ko_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..364e159b03bf738ec612523b268443ce5490a1ee --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ko/_ko_template_yaml @@ -0,0 +1,16 @@ +dataset_path: CohereForAI/Global-MMLU-Lite +dataset_name: ko +test_split: test +fewshot_split: dev +fewshot_config: + sampler: default +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\nD. {{option_d}}\nAnswer:" +doc_to_choice: ["A", "B", "C", "D"] +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ko/global_mmlu_ko_business.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ko/global_mmlu_ko_business.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2f1ce375cc7aabfae7c6feb5c7860162095dda04 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ko/global_mmlu_ko_business.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _ko_template_yaml +process_docs: !function utils.process_business +task: global_mmlu_ko_business diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ko/global_mmlu_ko_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ko/global_mmlu_ko_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a613ff5550740a647cca3e9ee993afbb599701b3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ko/global_mmlu_ko_humanities.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _ko_template_yaml +process_docs: !function utils.process_humanities +task: global_mmlu_ko_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ko/global_mmlu_ko_medical.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ko/global_mmlu_ko_medical.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7e8710384b3e518642808251dd68746b936204d8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ko/global_mmlu_ko_medical.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _ko_template_yaml +process_docs: !function utils.process_medical +task: global_mmlu_ko_medical diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ko/global_mmlu_ko_other.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ko/global_mmlu_ko_other.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3fa1c608d2a7277eb0ef9bd442a905eb697c7704 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ko/global_mmlu_ko_other.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _ko_template_yaml +process_docs: !function utils.process_other +task: global_mmlu_ko_other diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ko/global_mmlu_ko_social_sciences.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ko/global_mmlu_ko_social_sciences.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ad5874f9df3cfa8816ffb7db3ea7d96711f74631 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ko/global_mmlu_ko_social_sciences.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _ko_template_yaml +process_docs: !function utils.process_social_sciences +task: global_mmlu_ko_social_sciences diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ko/global_mmlu_ko_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ko/global_mmlu_ko_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f6c7e8ec18b8b60bc7b290d06673d483646d828d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ko/global_mmlu_ko_stem.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _ko_template_yaml +process_docs: !function utils.process_stem +task: global_mmlu_ko_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ko/utils.py b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ko/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..507a41bdc9cafa30afd3ecd2e9a50d089d1c4ee9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/ko/utils.py @@ -0,0 +1,18 @@ +from functools import partial + + +CATEGORIES = ["Business", "Humanities", "Medical", "Other", "STEM", "Social Sciences"] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["subject_category"] == category) + + +process_functions = { + f"process_{category.lower().replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/pt/_global_mmlu_pt.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/pt/_global_mmlu_pt.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7a489c1245996f7b14bef25a41c572f460e25aea --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/pt/_global_mmlu_pt.yaml @@ -0,0 +1,13 @@ +group: global_mmlu_pt +task: + - global_mmlu_pt_business + - global_mmlu_pt_humanities + - global_mmlu_pt_medical + - global_mmlu_pt_other + - global_mmlu_pt_stem + - global_mmlu_pt_social_sciences +aggregate_metric_list: + - metric: acc + weight_by_size: True +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/pt/_pt_template_yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/pt/_pt_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..f1db662946c39f4b56c1b4d30c909bdec02cfe57 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/pt/_pt_template_yaml @@ -0,0 +1,16 @@ +dataset_path: CohereForAI/Global-MMLU-Lite +dataset_name: pt +test_split: test +fewshot_split: dev +fewshot_config: + sampler: default +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\nD. {{option_d}}\nAnswer:" +doc_to_choice: ["A", "B", "C", "D"] +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/pt/global_mmlu_pt_business.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/pt/global_mmlu_pt_business.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1e72b1687197a84b06215d31bfe6e494061c0301 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/pt/global_mmlu_pt_business.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _pt_template_yaml +process_docs: !function utils.process_business +task: global_mmlu_pt_business diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/pt/global_mmlu_pt_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/pt/global_mmlu_pt_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7244f2a7544bf39d12fe7c511d0319cc36896ed0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/pt/global_mmlu_pt_humanities.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _pt_template_yaml +process_docs: !function utils.process_humanities +task: global_mmlu_pt_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/pt/global_mmlu_pt_medical.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/pt/global_mmlu_pt_medical.yaml new file mode 100644 index 0000000000000000000000000000000000000000..44776f2cffbb38b1b1bad9b6ac06eda8268a173d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/pt/global_mmlu_pt_medical.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _pt_template_yaml +process_docs: !function utils.process_medical +task: global_mmlu_pt_medical diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/pt/global_mmlu_pt_other.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/pt/global_mmlu_pt_other.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b612120187e5a883107fe91747174bb3d7120c37 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/pt/global_mmlu_pt_other.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _pt_template_yaml +process_docs: !function utils.process_other +task: global_mmlu_pt_other diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/pt/global_mmlu_pt_social_sciences.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/pt/global_mmlu_pt_social_sciences.yaml new file mode 100644 index 0000000000000000000000000000000000000000..949d346ece58ed70bbf67d79ff0ada5905222365 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/pt/global_mmlu_pt_social_sciences.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _pt_template_yaml +process_docs: !function utils.process_social_sciences +task: global_mmlu_pt_social_sciences diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/pt/global_mmlu_pt_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/pt/global_mmlu_pt_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9f194c115525c1aa059306051da30913b6cd6214 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/pt/global_mmlu_pt_stem.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _pt_template_yaml +process_docs: !function utils.process_stem +task: global_mmlu_pt_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/pt/utils.py b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/pt/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..507a41bdc9cafa30afd3ecd2e9a50d089d1c4ee9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/pt/utils.py @@ -0,0 +1,18 @@ +from functools import partial + + +CATEGORIES = ["Business", "Humanities", "Medical", "Other", "STEM", "Social Sciences"] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["subject_category"] == category) + + +process_functions = { + f"process_{category.lower().replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/sw/_global_mmlu_sw.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/sw/_global_mmlu_sw.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b3913d244efcb6f6abaa6c732ca7599fe591eda1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/sw/_global_mmlu_sw.yaml @@ -0,0 +1,13 @@ +group: global_mmlu_sw +task: + - global_mmlu_sw_business + - global_mmlu_sw_humanities + - global_mmlu_sw_medical + - global_mmlu_sw_other + - global_mmlu_sw_stem + - global_mmlu_sw_social_sciences +aggregate_metric_list: + - metric: acc + weight_by_size: True +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/sw/_sw_template_yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/sw/_sw_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..33edff382b5eeae72b3ddaf3c9eddeaa13d230a9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/sw/_sw_template_yaml @@ -0,0 +1,16 @@ +dataset_path: CohereForAI/Global-MMLU-Lite +dataset_name: sw +test_split: test +fewshot_split: dev +fewshot_config: + sampler: default +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\nD. {{option_d}}\nAnswer:" +doc_to_choice: ["A", "B", "C", "D"] +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/sw/global_mmlu_sw_business.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/sw/global_mmlu_sw_business.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a53ca478f39872a15236e2f56b97d96d628bd8e2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/sw/global_mmlu_sw_business.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _sw_template_yaml +process_docs: !function utils.process_business +task: global_mmlu_sw_business diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/sw/global_mmlu_sw_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/sw/global_mmlu_sw_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4687df760d82c690f9941a37b71ece4ca021728f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/sw/global_mmlu_sw_humanities.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _sw_template_yaml +process_docs: !function utils.process_humanities +task: global_mmlu_sw_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/sw/global_mmlu_sw_medical.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/sw/global_mmlu_sw_medical.yaml new file mode 100644 index 0000000000000000000000000000000000000000..76240ea3d042300abc00bc6f06e8197296cf7932 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/sw/global_mmlu_sw_medical.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _sw_template_yaml +process_docs: !function utils.process_medical +task: global_mmlu_sw_medical diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/sw/global_mmlu_sw_other.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/sw/global_mmlu_sw_other.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7c3bfda2e46e9580e0756dbcd9ac2b767c6df6b3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/sw/global_mmlu_sw_other.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _sw_template_yaml +process_docs: !function utils.process_other +task: global_mmlu_sw_other diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/sw/global_mmlu_sw_social_sciences.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/sw/global_mmlu_sw_social_sciences.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4a77aa2ba4a746518ed53fd02ea0da426da05b79 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/sw/global_mmlu_sw_social_sciences.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _sw_template_yaml +process_docs: !function utils.process_social_sciences +task: global_mmlu_sw_social_sciences diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/sw/global_mmlu_sw_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/sw/global_mmlu_sw_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d6faf18b16e3722aa53ff715ea56dd179235f5c0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/sw/global_mmlu_sw_stem.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _sw_template_yaml +process_docs: !function utils.process_stem +task: global_mmlu_sw_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/sw/utils.py b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/sw/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..507a41bdc9cafa30afd3ecd2e9a50d089d1c4ee9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/sw/utils.py @@ -0,0 +1,18 @@ +from functools import partial + + +CATEGORIES = ["Business", "Humanities", "Medical", "Other", "STEM", "Social Sciences"] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["subject_category"] == category) + + +process_functions = { + f"process_{category.lower().replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/yo/_global_mmlu_yo.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/yo/_global_mmlu_yo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..14df221ad80450fe05fcf458954607b10eae024c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/yo/_global_mmlu_yo.yaml @@ -0,0 +1,13 @@ +group: global_mmlu_yo +task: + - global_mmlu_yo_business + - global_mmlu_yo_humanities + - global_mmlu_yo_medical + - global_mmlu_yo_other + - global_mmlu_yo_stem + - global_mmlu_yo_social_sciences +aggregate_metric_list: + - metric: acc + weight_by_size: True +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/yo/_yo_template_yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/yo/_yo_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..6cdd0a035e5343c72e6fe3a6c00f3d0b4595087b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/yo/_yo_template_yaml @@ -0,0 +1,16 @@ +dataset_path: CohereForAI/Global-MMLU-Lite +dataset_name: yo +test_split: test +fewshot_split: dev +fewshot_config: + sampler: default +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\nD. {{option_d}}\nAnswer:" +doc_to_choice: ["A", "B", "C", "D"] +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/yo/global_mmlu_yo_business.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/yo/global_mmlu_yo_business.yaml new file mode 100644 index 0000000000000000000000000000000000000000..162a97cf0946caf0df499e4987680af6ab8d30de --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/yo/global_mmlu_yo_business.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _yo_template_yaml +process_docs: !function utils.process_business +task: global_mmlu_yo_business diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/yo/global_mmlu_yo_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/yo/global_mmlu_yo_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5befbc12e49588d75491c001cb2de0613684e218 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/yo/global_mmlu_yo_humanities.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _yo_template_yaml +process_docs: !function utils.process_humanities +task: global_mmlu_yo_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/yo/global_mmlu_yo_medical.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/yo/global_mmlu_yo_medical.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d48d02088f82b159e69d9ce859cf4a7ae5674a83 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/yo/global_mmlu_yo_medical.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _yo_template_yaml +process_docs: !function utils.process_medical +task: global_mmlu_yo_medical diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/yo/global_mmlu_yo_other.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/yo/global_mmlu_yo_other.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5e407c2cf1b6d72df0d76a7a48ba6f9ab823bd7c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/yo/global_mmlu_yo_other.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _yo_template_yaml +process_docs: !function utils.process_other +task: global_mmlu_yo_other diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/yo/global_mmlu_yo_social_sciences.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/yo/global_mmlu_yo_social_sciences.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c85596aa607fc2b963f569096b30fddf700dff75 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/yo/global_mmlu_yo_social_sciences.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _yo_template_yaml +process_docs: !function utils.process_social_sciences +task: global_mmlu_yo_social_sciences diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/yo/global_mmlu_yo_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/yo/global_mmlu_yo_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a19e1e8dcd458f59e830508b75e82fb128b494c8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/yo/global_mmlu_yo_stem.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _yo_template_yaml +process_docs: !function utils.process_stem +task: global_mmlu_yo_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/yo/utils.py b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/yo/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..507a41bdc9cafa30afd3ecd2e9a50d089d1c4ee9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/yo/utils.py @@ -0,0 +1,18 @@ +from functools import partial + + +CATEGORIES = ["Business", "Humanities", "Medical", "Other", "STEM", "Social Sciences"] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["subject_category"] == category) + + +process_functions = { + f"process_{category.lower().replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/zh/_global_mmlu_zh.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/zh/_global_mmlu_zh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..212a33fc9079765c659b78197dd1d7d811d77916 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/zh/_global_mmlu_zh.yaml @@ -0,0 +1,13 @@ +group: global_mmlu_zh +task: + - global_mmlu_zh_business + - global_mmlu_zh_humanities + - global_mmlu_zh_medical + - global_mmlu_zh_other + - global_mmlu_zh_stem + - global_mmlu_zh_social_sciences +aggregate_metric_list: + - metric: acc + weight_by_size: True +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/zh/_zh_template_yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/zh/_zh_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..eeb1e7b9c8aeaf29af8f8f7dc8009233017a2c45 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/zh/_zh_template_yaml @@ -0,0 +1,16 @@ +dataset_path: CohereForAI/Global-MMLU-Lite +dataset_name: zh +test_split: test +fewshot_split: dev +fewshot_config: + sampler: default +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\nD. {{option_d}}\nAnswer:" +doc_to_choice: ["A", "B", "C", "D"] +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/zh/global_mmlu_zh_business.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/zh/global_mmlu_zh_business.yaml new file mode 100644 index 0000000000000000000000000000000000000000..aa0a689aed22d6c7e9d22b06fc63c5833c6a132c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/zh/global_mmlu_zh_business.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _zh_template_yaml +process_docs: !function utils.process_business +task: global_mmlu_zh_business diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/zh/global_mmlu_zh_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/zh/global_mmlu_zh_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..823854b95818e344bc55a10ac09230ceac315a58 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/zh/global_mmlu_zh_humanities.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _zh_template_yaml +process_docs: !function utils.process_humanities +task: global_mmlu_zh_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/zh/global_mmlu_zh_medical.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/zh/global_mmlu_zh_medical.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f1f7a7d58553d13ac7d316a449a284eb1fe3ca41 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/zh/global_mmlu_zh_medical.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _zh_template_yaml +process_docs: !function utils.process_medical +task: global_mmlu_zh_medical diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/zh/global_mmlu_zh_other.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/zh/global_mmlu_zh_other.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a3beae834cbcd5e26edb4173f757151b1495c25b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/zh/global_mmlu_zh_other.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _zh_template_yaml +process_docs: !function utils.process_other +task: global_mmlu_zh_other diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/zh/global_mmlu_zh_social_sciences.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/zh/global_mmlu_zh_social_sciences.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1891a45aecbab39e564a04fbca2d92050a3468ff --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/zh/global_mmlu_zh_social_sciences.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _zh_template_yaml +process_docs: !function utils.process_social_sciences +task: global_mmlu_zh_social_sciences diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/zh/global_mmlu_zh_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/zh/global_mmlu_zh_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6a9f4f05de185e6605c0fdb84e66c81a6568f53b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/zh/global_mmlu_zh_stem.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +include: _zh_template_yaml +process_docs: !function utils.process_stem +task: global_mmlu_zh_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/zh/utils.py b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/zh/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..507a41bdc9cafa30afd3ecd2e9a50d089d1c4ee9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/default/zh/utils.py @@ -0,0 +1,18 @@ +from functools import partial + + +CATEGORIES = ["Business", "Humanities", "Medical", "Other", "STEM", "Social Sciences"] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["subject_category"] == category) + + +process_functions = { + f"process_{category.lower().replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/_am_template_yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/_am_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..f52152bb231166aabfe05b350878db08ceb8c1e1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/_am_template_yaml @@ -0,0 +1,16 @@ +dataset_path: CohereForAI/Global-MMLU +dataset_name: am +test_split: test +fewshot_split: dev +fewshot_config: + sampler: first_n +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\nD. {{option_d}}\nAnswer:" +doc_to_choice: ["A", "B", "C", "D"] +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/_global_mmlu_full_am.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/_global_mmlu_full_am.yaml new file mode 100644 index 0000000000000000000000000000000000000000..555bfd868496aab366e1fc74913052414cdec683 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/_global_mmlu_full_am.yaml @@ -0,0 +1,11 @@ +group: global_mmlu_full_am +task: + - global_mmlu_full_am_stem + - global_mmlu_full_am_other + - global_mmlu_full_am_social_sciences + - global_mmlu_full_am_humanities +aggregate_metric_list: + - metric: acc + weight_by_size: True +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/_global_mmlu_full_am_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/_global_mmlu_full_am_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e250d14c21f9dcf9dac6b442b46644fd5c8af216 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/_global_mmlu_full_am_humanities.yaml @@ -0,0 +1,8 @@ +group: global_mmlu_full_am_humanities +task: + - global_mmlu_full_am_humanities_tasks +aggregate_metric_list: + - metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/_global_mmlu_full_am_other.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/_global_mmlu_full_am_other.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4b5151ce702ca017867f67f93905e12a5f599516 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/_global_mmlu_full_am_other.yaml @@ -0,0 +1,8 @@ +group: global_mmlu_full_am_other +task: + - global_mmlu_full_am_other_tasks +aggregate_metric_list: + - metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/_global_mmlu_full_am_social_sciences.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/_global_mmlu_full_am_social_sciences.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f0fbcc1b73167da50b3d10214001843870ee8435 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/_global_mmlu_full_am_social_sciences.yaml @@ -0,0 +1,8 @@ +group: global_mmlu_full_am_social_sciences +task: + - global_mmlu_full_am_social_sciences_tasks +aggregate_metric_list: + - metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/_global_mmlu_full_am_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/_global_mmlu_full_am_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b67dfdb752643833c9307bc33bf1b0b02ebe3dac --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/_global_mmlu_full_am_stem.yaml @@ -0,0 +1,8 @@ +group: global_mmlu_full_am_stem +task: + - global_mmlu_full_am_stem_tasks +aggregate_metric_list: + - metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_abstract_algebra.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_abstract_algebra.yaml new file mode 100644 index 0000000000000000000000000000000000000000..06a70dd8492cd5cccdd7953fe5da8966f2295de5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_abstract_algebra.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_abstract_algebra +tag: global_mmlu_full_am_stem_tasks +task: global_mmlu_full_am_abstract_algebra diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_anatomy.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_anatomy.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7914c3b050da78b6deb65fafca778628ced83164 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_anatomy.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_anatomy +tag: global_mmlu_full_am_stem_tasks +task: global_mmlu_full_am_anatomy diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_astronomy.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_astronomy.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4e7e2a0474414a1d8ebcc8c9263e0d44a7ac21c4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_astronomy.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_astronomy +tag: global_mmlu_full_am_stem_tasks +task: global_mmlu_full_am_astronomy diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_business_ethics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_business_ethics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a98a9597e6321c6d4b43332ff45d96e900a81833 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_business_ethics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_business_ethics +tag: global_mmlu_full_am_other_tasks +task: global_mmlu_full_am_business_ethics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_clinical_knowledge.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_clinical_knowledge.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4c25627f6eed6b73b7bd251f683f8ff6a342aa21 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_clinical_knowledge.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_clinical_knowledge +tag: global_mmlu_full_am_other_tasks +task: global_mmlu_full_am_clinical_knowledge diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_college_biology.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_college_biology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a8b6661b55e23cb3b0adb7e99b30e54cdae4029f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_college_biology.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_college_biology +tag: global_mmlu_full_am_stem_tasks +task: global_mmlu_full_am_college_biology diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_college_chemistry.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_college_chemistry.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b0d2d2a840f7ebc464ede8c7a83e60c5270c7a94 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_college_chemistry.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_college_chemistry +tag: global_mmlu_full_am_stem_tasks +task: global_mmlu_full_am_college_chemistry diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_college_computer_science.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_college_computer_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b5c52a82de27b3aafe61e39c1b34858fed222e7a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_college_computer_science.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_college_computer_science +tag: global_mmlu_full_am_stem_tasks +task: global_mmlu_full_am_college_computer_science diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_college_mathematics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_college_mathematics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0b73422e35fa19e5539447eb42ab661da3142105 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_college_mathematics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_college_mathematics +tag: global_mmlu_full_am_stem_tasks +task: global_mmlu_full_am_college_mathematics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_college_medicine.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_college_medicine.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bd36f40f27ba875df2ed6c15982929a4ae8a2401 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_college_medicine.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_college_medicine +tag: global_mmlu_full_am_other_tasks +task: global_mmlu_full_am_college_medicine diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_college_physics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_college_physics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..009fdc1a7c7f210e673ef504bf2b5bdcc3f55444 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_college_physics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_college_physics +tag: global_mmlu_full_am_stem_tasks +task: global_mmlu_full_am_college_physics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_computer_security.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_computer_security.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3df6247b92bd8e2884651452231812a5bb854485 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_computer_security.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_computer_security +tag: global_mmlu_full_am_stem_tasks +task: global_mmlu_full_am_computer_security diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_conceptual_physics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_conceptual_physics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4115ea0245ac2718e8a62ba561b269481a16db79 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_conceptual_physics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_conceptual_physics +tag: global_mmlu_full_am_stem_tasks +task: global_mmlu_full_am_conceptual_physics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_econometrics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_econometrics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..87dd12cab07e28ec45581684c8056eba1bfc10c1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_econometrics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_econometrics +tag: global_mmlu_full_am_social_sciences_tasks +task: global_mmlu_full_am_econometrics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_electrical_engineering.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_electrical_engineering.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d8f726193648efff28ea1554119f7b8e5f66b955 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_electrical_engineering.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_electrical_engineering +tag: global_mmlu_full_am_stem_tasks +task: global_mmlu_full_am_electrical_engineering diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_elementary_mathematics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_elementary_mathematics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..455563f1c560d64dbf84fc60004813c044b62ad5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_elementary_mathematics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_elementary_mathematics +tag: global_mmlu_full_am_stem_tasks +task: global_mmlu_full_am_elementary_mathematics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_formal_logic.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_formal_logic.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5c5babd4d87f2b0a8bc7fe5e822c535318bc925c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_formal_logic.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_formal_logic +tag: global_mmlu_full_am_humanities_tasks +task: global_mmlu_full_am_formal_logic diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_global_facts.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_global_facts.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b59d47e45355776716f366d3897d151a3c162773 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_global_facts.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_global_facts +tag: global_mmlu_full_am_other_tasks +task: global_mmlu_full_am_global_facts diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_high_school_biology.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_high_school_biology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..680d4ecac861a1ae7b233e29b6a6865a25e9ab4c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_high_school_biology.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_high_school_biology +tag: global_mmlu_full_am_stem_tasks +task: global_mmlu_full_am_high_school_biology diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_high_school_chemistry.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_high_school_chemistry.yaml new file mode 100644 index 0000000000000000000000000000000000000000..96af7940f30ef5ed6d69ed6cd812b8934470271e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_high_school_chemistry.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_high_school_chemistry +tag: global_mmlu_full_am_stem_tasks +task: global_mmlu_full_am_high_school_chemistry diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_high_school_computer_science.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_high_school_computer_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6cd19227b9fa53dd3107708ae0635222e23da00f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_high_school_computer_science.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_high_school_computer_science +tag: global_mmlu_full_am_stem_tasks +task: global_mmlu_full_am_high_school_computer_science diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_high_school_european_history.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_high_school_european_history.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e02491426eadecb9f2bf42f6ef34c928c0c07ba5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_high_school_european_history.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_high_school_european_history +tag: global_mmlu_full_am_humanities_tasks +task: global_mmlu_full_am_high_school_european_history diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_high_school_geography.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_high_school_geography.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b4925a54244ccbf07649add896ff8251d629cf93 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_high_school_geography.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_high_school_geography +tag: global_mmlu_full_am_social_sciences_tasks +task: global_mmlu_full_am_high_school_geography diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_high_school_government_and_politics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_high_school_government_and_politics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d63f1d3532ce01b5130c0a15edc2caae390d8d80 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_high_school_government_and_politics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_high_school_government_and_politics +tag: global_mmlu_full_am_social_sciences_tasks +task: global_mmlu_full_am_high_school_government_and_politics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_high_school_macroeconomics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_high_school_macroeconomics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3c8a0ea684442bbc391cd8c962422ca432ec2863 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_high_school_macroeconomics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_high_school_macroeconomics +tag: global_mmlu_full_am_social_sciences_tasks +task: global_mmlu_full_am_high_school_macroeconomics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_high_school_mathematics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_high_school_mathematics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..76a8c3d35547cb9467edf1d6f6209d2ef2693db9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_high_school_mathematics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_high_school_mathematics +tag: global_mmlu_full_am_stem_tasks +task: global_mmlu_full_am_high_school_mathematics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_high_school_microeconomics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_high_school_microeconomics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1acbf4e1cf204dde35904ae0d547be3b81e10271 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_high_school_microeconomics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_high_school_microeconomics +tag: global_mmlu_full_am_social_sciences_tasks +task: global_mmlu_full_am_high_school_microeconomics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_high_school_physics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_high_school_physics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dcfd9bb9fd7e0faa6a20edc7de3543c5b96739f5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_high_school_physics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_high_school_physics +tag: global_mmlu_full_am_stem_tasks +task: global_mmlu_full_am_high_school_physics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_high_school_psychology.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_high_school_psychology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2dd64dc18e05133307a1475af6b911e48dd16832 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_high_school_psychology.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_high_school_psychology +tag: global_mmlu_full_am_social_sciences_tasks +task: global_mmlu_full_am_high_school_psychology diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_high_school_statistics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_high_school_statistics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a523f443aadda34b5dcf3703897e15ab28e87675 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_high_school_statistics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_high_school_statistics +tag: global_mmlu_full_am_stem_tasks +task: global_mmlu_full_am_high_school_statistics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_high_school_us_history.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_high_school_us_history.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ce233f44fef2ae6714f872a4b881ba266515a73d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_high_school_us_history.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_high_school_us_history +tag: global_mmlu_full_am_humanities_tasks +task: global_mmlu_full_am_high_school_us_history diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_high_school_world_history.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_high_school_world_history.yaml new file mode 100644 index 0000000000000000000000000000000000000000..20aeca5e2501a2b7158377b4b03b93f3b53f25bb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_high_school_world_history.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_high_school_world_history +tag: global_mmlu_full_am_humanities_tasks +task: global_mmlu_full_am_high_school_world_history diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_human_aging.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_human_aging.yaml new file mode 100644 index 0000000000000000000000000000000000000000..18e95e40526ab710006e8f9b0955234ded466804 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_human_aging.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_human_aging +tag: global_mmlu_full_am_other_tasks +task: global_mmlu_full_am_human_aging diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_human_sexuality.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_human_sexuality.yaml new file mode 100644 index 0000000000000000000000000000000000000000..140f23294a5dfa09b937717a101f7f8e5c4202b7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_human_sexuality.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_human_sexuality +tag: global_mmlu_full_am_social_sciences_tasks +task: global_mmlu_full_am_human_sexuality diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_international_law.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_international_law.yaml new file mode 100644 index 0000000000000000000000000000000000000000..10a2d638d01bca791d4b6e7134a0bd9b38e8967a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_international_law.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_international_law +tag: global_mmlu_full_am_humanities_tasks +task: global_mmlu_full_am_international_law diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_jurisprudence.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_jurisprudence.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cd98274255534c6170f39f970aeb94ea6c97550f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_jurisprudence.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_jurisprudence +tag: global_mmlu_full_am_humanities_tasks +task: global_mmlu_full_am_jurisprudence diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_logical_fallacies.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_logical_fallacies.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2faf735c7dc365b0d4561223cb0f5327c5c4e5cd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_logical_fallacies.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_logical_fallacies +tag: global_mmlu_full_am_humanities_tasks +task: global_mmlu_full_am_logical_fallacies diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_machine_learning.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_machine_learning.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7f5c8e9b80bf4db4dd1f366f82b1116d58c94172 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_machine_learning.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_machine_learning +tag: global_mmlu_full_am_stem_tasks +task: global_mmlu_full_am_machine_learning diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_management.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_management.yaml new file mode 100644 index 0000000000000000000000000000000000000000..08d080a8fe6d01206750d50e41e0719a71938781 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_management.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_management +tag: global_mmlu_full_am_other_tasks +task: global_mmlu_full_am_management diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_marketing.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_marketing.yaml new file mode 100644 index 0000000000000000000000000000000000000000..52b4f7c6ce22f4ad432bfccee2ffab53dd80e789 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_marketing.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_marketing +tag: global_mmlu_full_am_other_tasks +task: global_mmlu_full_am_marketing diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_medical_genetics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_medical_genetics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..32bd2432ba7767d329b92f83213c7eecd0a54545 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_medical_genetics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_medical_genetics +tag: global_mmlu_full_am_other_tasks +task: global_mmlu_full_am_medical_genetics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_miscellaneous.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_miscellaneous.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ed5d610dcf529b9496ab4548cf75524877676281 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_miscellaneous.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_miscellaneous +tag: global_mmlu_full_am_other_tasks +task: global_mmlu_full_am_miscellaneous diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_moral_disputes.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_moral_disputes.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bddaebc7505c74104f52a6f73281c04c88d86d96 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_moral_disputes.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_moral_disputes +tag: global_mmlu_full_am_humanities_tasks +task: global_mmlu_full_am_moral_disputes diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_moral_scenarios.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_moral_scenarios.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fda69f31676c3b0574ed607318cdb3a4048a7478 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_moral_scenarios.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_moral_scenarios +tag: global_mmlu_full_am_humanities_tasks +task: global_mmlu_full_am_moral_scenarios diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_nutrition.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_nutrition.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bb0cb08b84b3392c34761ef2424d67d4bafc2efe --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_nutrition.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_nutrition +tag: global_mmlu_full_am_other_tasks +task: global_mmlu_full_am_nutrition diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_philosophy.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_philosophy.yaml new file mode 100644 index 0000000000000000000000000000000000000000..484c015eaeb7dc865360b82e3ccc817708a1f0cf --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_philosophy.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_philosophy +tag: global_mmlu_full_am_humanities_tasks +task: global_mmlu_full_am_philosophy diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_prehistory.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_prehistory.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6e104f48aac06b42aebb3bb7d0b822ad2e059944 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_prehistory.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_prehistory +tag: global_mmlu_full_am_humanities_tasks +task: global_mmlu_full_am_prehistory diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_professional_accounting.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_professional_accounting.yaml new file mode 100644 index 0000000000000000000000000000000000000000..50c9fe50105722aeb1bb77c7e57216f113300c83 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_professional_accounting.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_professional_accounting +tag: global_mmlu_full_am_other_tasks +task: global_mmlu_full_am_professional_accounting diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_professional_law.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_professional_law.yaml new file mode 100644 index 0000000000000000000000000000000000000000..df2cf26cf0372221935abd5a21b8927481cf4484 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_professional_law.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_professional_law +tag: global_mmlu_full_am_humanities_tasks +task: global_mmlu_full_am_professional_law diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_professional_medicine.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_professional_medicine.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c28605288272a20e243e9091a7d4a2d565bf9bd9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_professional_medicine.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_professional_medicine +tag: global_mmlu_full_am_other_tasks +task: global_mmlu_full_am_professional_medicine diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_professional_psychology.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_professional_psychology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8562a28de55a06f7a6db8d6da2bdbe1879b232c6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_professional_psychology.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_professional_psychology +tag: global_mmlu_full_am_social_sciences_tasks +task: global_mmlu_full_am_professional_psychology diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_public_relations.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_public_relations.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5cb3186c9b1439f0295d7b2e5cdb94bd38dd912d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_public_relations.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_public_relations +tag: global_mmlu_full_am_social_sciences_tasks +task: global_mmlu_full_am_public_relations diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_security_studies.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_security_studies.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6aa8575eecac8790d040fe8493657daad3f0a0d5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_security_studies.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_security_studies +tag: global_mmlu_full_am_social_sciences_tasks +task: global_mmlu_full_am_security_studies diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_sociology.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_sociology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..60005babcaf8908cf98b553a4386c58c3e503dc7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_sociology.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_sociology +tag: global_mmlu_full_am_social_sciences_tasks +task: global_mmlu_full_am_sociology diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_us_foreign_policy.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_us_foreign_policy.yaml new file mode 100644 index 0000000000000000000000000000000000000000..374fb14ad3e3729ad6e6ad3cb95d62dfdd5299bb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_us_foreign_policy.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_us_foreign_policy +tag: global_mmlu_full_am_social_sciences_tasks +task: global_mmlu_full_am_us_foreign_policy diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_virology.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_virology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9f235299233dddb0cb754a20d509f945fae50477 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_virology.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_virology +tag: global_mmlu_full_am_other_tasks +task: global_mmlu_full_am_virology diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_world_religions.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_world_religions.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c169a04830c176fc96c475b72056be432efff04c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/global_mmlu_full_am_world_religions.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _am_template_yaml +process_docs: !function utils.process_world_religions +tag: global_mmlu_full_am_humanities_tasks +task: global_mmlu_full_am_world_religions diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/utils.py b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..7df72cb061f0fecba46e15ff9b57552817979afb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/am/utils.py @@ -0,0 +1,73 @@ +from functools import partial + + +SUBJECTS = [ + "abstract_algebra", + "anatomy", + "astronomy", + "business_ethics", + "clinical_knowledge", + "college_biology", + "college_chemistry", + "college_computer_science", + "college_mathematics", + "college_medicine", + "college_physics", + "computer_security", + "conceptual_physics", + "econometrics", + "electrical_engineering", + "elementary_mathematics", + "formal_logic", + "global_facts", + "high_school_biology", + "high_school_chemistry", + "high_school_computer_science", + "high_school_european_history", + "high_school_geography", + "high_school_government_and_politics", + "high_school_macroeconomics", + "high_school_mathematics", + "high_school_microeconomics", + "high_school_physics", + "high_school_psychology", + "high_school_statistics", + "high_school_us_history", + "high_school_world_history", + "human_aging", + "human_sexuality", + "international_law", + "jurisprudence", + "logical_fallacies", + "machine_learning", + "management", + "marketing", + "medical_genetics", + "miscellaneous", + "moral_disputes", + "moral_scenarios", + "nutrition", + "philosophy", + "prehistory", + "professional_accounting", + "professional_law", + "professional_medicine", + "professional_psychology", + "public_relations", + "security_studies", + "sociology", + "us_foreign_policy", + "virology", + "world_religions", +] + + +def process_docs(dataset, subject): + return dataset.filter(lambda x: x["subject"] == subject) + + +process_functions = { + f"process_{subject}": partial(process_docs, subject=subject) for subject in SUBJECTS +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/_ar_template_yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/_ar_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..768bb7f974a40fc90c412e31ce70c86ef5028b81 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/_ar_template_yaml @@ -0,0 +1,16 @@ +dataset_path: CohereForAI/Global-MMLU +dataset_name: ar +test_split: test +fewshot_split: dev +fewshot_config: + sampler: first_n +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\nD. {{option_d}}\nAnswer:" +doc_to_choice: ["A", "B", "C", "D"] +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/_global_mmlu_full_ar.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/_global_mmlu_full_ar.yaml new file mode 100644 index 0000000000000000000000000000000000000000..83340da0dd931bf0b1c4dfd027a21ae97ccb75cc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/_global_mmlu_full_ar.yaml @@ -0,0 +1,11 @@ +group: global_mmlu_full_ar +task: + - global_mmlu_full_ar_stem + - global_mmlu_full_ar_other + - global_mmlu_full_ar_social_sciences + - global_mmlu_full_ar_humanities +aggregate_metric_list: + - metric: acc + weight_by_size: True +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/_global_mmlu_full_ar_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/_global_mmlu_full_ar_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cfa6d80a2ffe1cca5263e7ccb0388e58d2e8972a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/_global_mmlu_full_ar_humanities.yaml @@ -0,0 +1,8 @@ +group: global_mmlu_full_ar_humanities +task: + - global_mmlu_full_ar_humanities_tasks +aggregate_metric_list: + - metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/_global_mmlu_full_ar_other.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/_global_mmlu_full_ar_other.yaml new file mode 100644 index 0000000000000000000000000000000000000000..26603f33411efeb057fce55a0ad5757df2be716b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/_global_mmlu_full_ar_other.yaml @@ -0,0 +1,8 @@ +group: global_mmlu_full_ar_other +task: + - global_mmlu_full_ar_other_tasks +aggregate_metric_list: + - metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/_global_mmlu_full_ar_social_sciences.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/_global_mmlu_full_ar_social_sciences.yaml new file mode 100644 index 0000000000000000000000000000000000000000..aca95bc2abd14262d1ab4de2929e99973240094e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/_global_mmlu_full_ar_social_sciences.yaml @@ -0,0 +1,8 @@ +group: global_mmlu_full_ar_social_sciences +task: + - global_mmlu_full_ar_social_sciences_tasks +aggregate_metric_list: + - metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/_global_mmlu_full_ar_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/_global_mmlu_full_ar_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b91e6c9bf3097849fa38ce56d41b194057e28629 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/_global_mmlu_full_ar_stem.yaml @@ -0,0 +1,8 @@ +group: global_mmlu_full_ar_stem +task: + - global_mmlu_full_ar_stem_tasks +aggregate_metric_list: + - metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_abstract_algebra.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_abstract_algebra.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1f044b044286f5f19233a4ed210f85b37122711d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_abstract_algebra.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _ar_template_yaml +process_docs: !function utils.process_abstract_algebra +tag: global_mmlu_full_ar_stem_tasks +task: global_mmlu_full_ar_abstract_algebra diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_anatomy.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_anatomy.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cd5d09636201e296ba35aebd7458e70901895005 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_anatomy.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _ar_template_yaml +process_docs: !function utils.process_anatomy +tag: global_mmlu_full_ar_stem_tasks +task: global_mmlu_full_ar_anatomy diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_astronomy.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_astronomy.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d21c00b7e96ed943f93486dac8115c8f39a515da --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_astronomy.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _ar_template_yaml +process_docs: !function utils.process_astronomy +tag: global_mmlu_full_ar_stem_tasks +task: global_mmlu_full_ar_astronomy diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_business_ethics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_business_ethics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a73f5f2d74bfae639353e7e64beebca9ca8f54a0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_business_ethics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _ar_template_yaml +process_docs: !function utils.process_business_ethics +tag: global_mmlu_full_ar_other_tasks +task: global_mmlu_full_ar_business_ethics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_clinical_knowledge.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_clinical_knowledge.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a9c3d0789f3790e3dd2ed13039508a44f1f4a223 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_clinical_knowledge.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _ar_template_yaml +process_docs: !function utils.process_clinical_knowledge +tag: global_mmlu_full_ar_other_tasks +task: global_mmlu_full_ar_clinical_knowledge diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_college_biology.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_college_biology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6fba6a1b75a2e13058d77b6e1b63292c9ceef1e2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_college_biology.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _ar_template_yaml +process_docs: !function utils.process_college_biology +tag: global_mmlu_full_ar_stem_tasks +task: global_mmlu_full_ar_college_biology diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_college_chemistry.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_college_chemistry.yaml new file mode 100644 index 0000000000000000000000000000000000000000..386ba52dc85eee52070efcc75cadab5aabbdf61b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_college_chemistry.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _ar_template_yaml +process_docs: !function utils.process_college_chemistry +tag: global_mmlu_full_ar_stem_tasks +task: global_mmlu_full_ar_college_chemistry diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_college_computer_science.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_college_computer_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9b846715c890b3042140f14a9414c172be0696bf --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_college_computer_science.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _ar_template_yaml +process_docs: !function utils.process_college_computer_science +tag: global_mmlu_full_ar_stem_tasks +task: global_mmlu_full_ar_college_computer_science diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_college_mathematics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_college_mathematics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c8d8d090f3a6307d871c64dabb1e268fabc3be30 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_college_mathematics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _ar_template_yaml +process_docs: !function utils.process_college_mathematics +tag: global_mmlu_full_ar_stem_tasks +task: global_mmlu_full_ar_college_mathematics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_college_medicine.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_college_medicine.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b988cfee2dcf6927ca0f6bd6de262adf2e9194b2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_college_medicine.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _ar_template_yaml +process_docs: !function utils.process_college_medicine +tag: global_mmlu_full_ar_other_tasks +task: global_mmlu_full_ar_college_medicine diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_college_physics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_college_physics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..008a39dc5fd6b0a8a6845bffe47149e97af4b8a1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_college_physics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _ar_template_yaml +process_docs: !function utils.process_college_physics +tag: global_mmlu_full_ar_stem_tasks +task: global_mmlu_full_ar_college_physics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_computer_security.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_computer_security.yaml new file mode 100644 index 0000000000000000000000000000000000000000..34a9353573c6c9d1065d826262ef662332cac8a1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_computer_security.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _ar_template_yaml +process_docs: !function utils.process_computer_security +tag: global_mmlu_full_ar_stem_tasks +task: global_mmlu_full_ar_computer_security diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_conceptual_physics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_conceptual_physics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ea20efa5ba423aab5c8f15dd7d0c454f2fbf60ce --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_conceptual_physics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _ar_template_yaml +process_docs: !function utils.process_conceptual_physics +tag: global_mmlu_full_ar_stem_tasks +task: global_mmlu_full_ar_conceptual_physics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_econometrics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_econometrics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3a757901148d4bb05831fa2f9845d69c81ffae2b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_econometrics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _ar_template_yaml +process_docs: !function utils.process_econometrics +tag: global_mmlu_full_ar_social_sciences_tasks +task: global_mmlu_full_ar_econometrics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_electrical_engineering.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_electrical_engineering.yaml new file mode 100644 index 0000000000000000000000000000000000000000..31a4e22efedaa4244a3c031f3fc53e64c9c57898 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_electrical_engineering.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _ar_template_yaml +process_docs: !function utils.process_electrical_engineering +tag: global_mmlu_full_ar_stem_tasks +task: global_mmlu_full_ar_electrical_engineering diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_elementary_mathematics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_elementary_mathematics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..25f4adb9e7ccfc055e8b1a2d32e3ebcc0df24d4b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_elementary_mathematics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _ar_template_yaml +process_docs: !function utils.process_elementary_mathematics +tag: global_mmlu_full_ar_stem_tasks +task: global_mmlu_full_ar_elementary_mathematics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_formal_logic.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_formal_logic.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b2792d5675e1b2601e7fc81fb1a8e0e04e34de13 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_formal_logic.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _ar_template_yaml +process_docs: !function utils.process_formal_logic +tag: global_mmlu_full_ar_humanities_tasks +task: global_mmlu_full_ar_formal_logic diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_global_facts.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_global_facts.yaml new file mode 100644 index 0000000000000000000000000000000000000000..af1bf60bb1262483e7d5cbc5a36eb4d797af1cae --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_global_facts.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _ar_template_yaml +process_docs: !function utils.process_global_facts +tag: global_mmlu_full_ar_other_tasks +task: global_mmlu_full_ar_global_facts diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_high_school_biology.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_high_school_biology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8f7eaff72d529792c275fe9f0e3a65354b45ebcb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_high_school_biology.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _ar_template_yaml +process_docs: !function utils.process_high_school_biology +tag: global_mmlu_full_ar_stem_tasks +task: global_mmlu_full_ar_high_school_biology diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_high_school_chemistry.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_high_school_chemistry.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8f56395b1eb83cbd974f00bfa11c92e2818ce528 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_high_school_chemistry.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _ar_template_yaml +process_docs: !function utils.process_high_school_chemistry +tag: global_mmlu_full_ar_stem_tasks +task: global_mmlu_full_ar_high_school_chemistry diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_high_school_computer_science.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_high_school_computer_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6e388aed857a3ffaf249c96f390c96c81babd27a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_high_school_computer_science.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _ar_template_yaml +process_docs: !function utils.process_high_school_computer_science +tag: global_mmlu_full_ar_stem_tasks +task: global_mmlu_full_ar_high_school_computer_science diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_high_school_european_history.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_high_school_european_history.yaml new file mode 100644 index 0000000000000000000000000000000000000000..741584c5f94550e2138ee4b5e5eb67eb99b165a4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_high_school_european_history.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _ar_template_yaml +process_docs: !function utils.process_high_school_european_history +tag: global_mmlu_full_ar_humanities_tasks +task: global_mmlu_full_ar_high_school_european_history diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_high_school_geography.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_high_school_geography.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3c376967a853698f06dbd140499949cf6126116a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_high_school_geography.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _ar_template_yaml +process_docs: !function utils.process_high_school_geography +tag: global_mmlu_full_ar_social_sciences_tasks +task: global_mmlu_full_ar_high_school_geography diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_high_school_government_and_politics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_high_school_government_and_politics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c71ada9b6688bd285676e88bfaa12d5699c29e3a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_high_school_government_and_politics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _ar_template_yaml +process_docs: !function utils.process_high_school_government_and_politics +tag: global_mmlu_full_ar_social_sciences_tasks +task: global_mmlu_full_ar_high_school_government_and_politics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_high_school_macroeconomics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_high_school_macroeconomics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0b5f32678e18d0cc8c8c61a432d133aa9911f568 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_high_school_macroeconomics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _ar_template_yaml +process_docs: !function utils.process_high_school_macroeconomics +tag: global_mmlu_full_ar_social_sciences_tasks +task: global_mmlu_full_ar_high_school_macroeconomics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_high_school_mathematics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_high_school_mathematics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cb259ac24b7667ad6b70fe401549643eccee60d1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_high_school_mathematics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _ar_template_yaml +process_docs: !function utils.process_high_school_mathematics +tag: global_mmlu_full_ar_stem_tasks +task: global_mmlu_full_ar_high_school_mathematics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_high_school_microeconomics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_high_school_microeconomics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c4ab308b3026bc5311c35eb17b06a52bd11d43d9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_high_school_microeconomics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _ar_template_yaml +process_docs: !function utils.process_high_school_microeconomics +tag: global_mmlu_full_ar_social_sciences_tasks +task: global_mmlu_full_ar_high_school_microeconomics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_high_school_physics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_high_school_physics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..68180e5d61b4ae31460b64b55f6e775685a59edb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_high_school_physics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _ar_template_yaml +process_docs: !function utils.process_high_school_physics +tag: global_mmlu_full_ar_stem_tasks +task: global_mmlu_full_ar_high_school_physics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_high_school_psychology.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_high_school_psychology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e727ad09160599f2a890e8d17817c0e25dbeb4e9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_high_school_psychology.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _ar_template_yaml +process_docs: !function utils.process_high_school_psychology +tag: global_mmlu_full_ar_social_sciences_tasks +task: global_mmlu_full_ar_high_school_psychology diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_high_school_statistics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_high_school_statistics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8ff9dd0ba2cfc698134e7526f5298c6aa536675d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_high_school_statistics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _ar_template_yaml +process_docs: !function utils.process_high_school_statistics +tag: global_mmlu_full_ar_stem_tasks +task: global_mmlu_full_ar_high_school_statistics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_high_school_us_history.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_high_school_us_history.yaml new file mode 100644 index 0000000000000000000000000000000000000000..668991cfd3febdc9231d51b7b0caefd2e71ba652 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_high_school_us_history.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _ar_template_yaml +process_docs: !function utils.process_high_school_us_history +tag: global_mmlu_full_ar_humanities_tasks +task: global_mmlu_full_ar_high_school_us_history diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_high_school_world_history.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_high_school_world_history.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1df9a5533e6ae2f61ae7c971d1cd9bf3faf22e6d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_high_school_world_history.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _ar_template_yaml +process_docs: !function utils.process_high_school_world_history +tag: global_mmlu_full_ar_humanities_tasks +task: global_mmlu_full_ar_high_school_world_history diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_human_aging.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_human_aging.yaml new file mode 100644 index 0000000000000000000000000000000000000000..515a40f0c063d440b1bb230ceaebfae8c1938fde --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_human_aging.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _ar_template_yaml +process_docs: !function utils.process_human_aging +tag: global_mmlu_full_ar_other_tasks +task: global_mmlu_full_ar_human_aging diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_human_sexuality.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_human_sexuality.yaml new file mode 100644 index 0000000000000000000000000000000000000000..24caceac8cda0e4254c49eaf94a0719b45277c21 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_human_sexuality.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _ar_template_yaml +process_docs: !function utils.process_human_sexuality +tag: global_mmlu_full_ar_social_sciences_tasks +task: global_mmlu_full_ar_human_sexuality diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_international_law.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_international_law.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a5aee4b294c9c1cf61ee92a2baba54813125dd9a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_international_law.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _ar_template_yaml +process_docs: !function utils.process_international_law +tag: global_mmlu_full_ar_humanities_tasks +task: global_mmlu_full_ar_international_law diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_jurisprudence.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_jurisprudence.yaml new file mode 100644 index 0000000000000000000000000000000000000000..377812084e66e1995bef2eb16c0977d75a9f2497 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_jurisprudence.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _ar_template_yaml +process_docs: !function utils.process_jurisprudence +tag: global_mmlu_full_ar_humanities_tasks +task: global_mmlu_full_ar_jurisprudence diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_machine_learning.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_machine_learning.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e1fc86e2e3e351a0ba790dab33da17b29a7bcf37 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_machine_learning.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _ar_template_yaml +process_docs: !function utils.process_machine_learning +tag: global_mmlu_full_ar_stem_tasks +task: global_mmlu_full_ar_machine_learning diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_management.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_management.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4dc7c8c0095ba06391579b7d78671b7d17b1f4e4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_management.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _ar_template_yaml +process_docs: !function utils.process_management +tag: global_mmlu_full_ar_other_tasks +task: global_mmlu_full_ar_management diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_miscellaneous.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_miscellaneous.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7d593ecbf058204cfa0b9474229436578198110d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_miscellaneous.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _ar_template_yaml +process_docs: !function utils.process_miscellaneous +tag: global_mmlu_full_ar_other_tasks +task: global_mmlu_full_ar_miscellaneous diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_moral_disputes.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_moral_disputes.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4021a93e6568f22215550ba825b1722d9fa3c70b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_moral_disputes.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _ar_template_yaml +process_docs: !function utils.process_moral_disputes +tag: global_mmlu_full_ar_humanities_tasks +task: global_mmlu_full_ar_moral_disputes diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_moral_scenarios.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_moral_scenarios.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f09edd006bf0097e1c3f5b0ac57ebfe88f963d5e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_moral_scenarios.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _ar_template_yaml +process_docs: !function utils.process_moral_scenarios +tag: global_mmlu_full_ar_humanities_tasks +task: global_mmlu_full_ar_moral_scenarios diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_nutrition.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_nutrition.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8d8577cb64ee688255d478e8474395eb4d685a56 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_nutrition.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _ar_template_yaml +process_docs: !function utils.process_nutrition +tag: global_mmlu_full_ar_other_tasks +task: global_mmlu_full_ar_nutrition diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_philosophy.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_philosophy.yaml new file mode 100644 index 0000000000000000000000000000000000000000..733b77ce95a7e8848e02907baf001a80fdd85f04 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_philosophy.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _ar_template_yaml +process_docs: !function utils.process_philosophy +tag: global_mmlu_full_ar_humanities_tasks +task: global_mmlu_full_ar_philosophy diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_prehistory.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_prehistory.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4d1bf14144e43f01a45ee468ed45f62173fbb0ae --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_prehistory.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _ar_template_yaml +process_docs: !function utils.process_prehistory +tag: global_mmlu_full_ar_humanities_tasks +task: global_mmlu_full_ar_prehistory diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_professional_medicine.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_professional_medicine.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4cd0a17a0e78d90ae7fe90164ee7b2cbdb01d5ee --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/ar/global_mmlu_full_ar_professional_medicine.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _ar_template_yaml +process_docs: !function utils.process_professional_medicine +tag: global_mmlu_full_ar_other_tasks +task: global_mmlu_full_ar_professional_medicine diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/_bn_template_yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/_bn_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..f388063d0292c31fbc3fb6985d5738d7c3a31048 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/_bn_template_yaml @@ -0,0 +1,16 @@ +dataset_path: CohereForAI/Global-MMLU +dataset_name: bn +test_split: test +fewshot_split: dev +fewshot_config: + sampler: first_n +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\nD. {{option_d}}\nAnswer:" +doc_to_choice: ["A", "B", "C", "D"] +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/_global_mmlu_full_bn.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/_global_mmlu_full_bn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..135b4bf5dd1532e85a35144d66ca20a82f8be4a1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/_global_mmlu_full_bn.yaml @@ -0,0 +1,11 @@ +group: global_mmlu_full_bn +task: + - global_mmlu_full_bn_stem + - global_mmlu_full_bn_other + - global_mmlu_full_bn_social_sciences + - global_mmlu_full_bn_humanities +aggregate_metric_list: + - metric: acc + weight_by_size: True +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/_global_mmlu_full_bn_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/_global_mmlu_full_bn_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..acd1ab011f85de884afa8b96ad3f740c7ab87e9b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/_global_mmlu_full_bn_humanities.yaml @@ -0,0 +1,8 @@ +group: global_mmlu_full_bn_humanities +task: + - global_mmlu_full_bn_humanities_tasks +aggregate_metric_list: + - metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/_global_mmlu_full_bn_other.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/_global_mmlu_full_bn_other.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d2160298f92f739c9d8c088a60d762d539bba53a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/_global_mmlu_full_bn_other.yaml @@ -0,0 +1,8 @@ +group: global_mmlu_full_bn_other +task: + - global_mmlu_full_bn_other_tasks +aggregate_metric_list: + - metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/_global_mmlu_full_bn_social_sciences.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/_global_mmlu_full_bn_social_sciences.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c359b3598af4e8a31bba5682e1fd523e15aa1f76 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/_global_mmlu_full_bn_social_sciences.yaml @@ -0,0 +1,8 @@ +group: global_mmlu_full_bn_social_sciences +task: + - global_mmlu_full_bn_social_sciences_tasks +aggregate_metric_list: + - metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/_global_mmlu_full_bn_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/_global_mmlu_full_bn_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2c78c4ceaef8e79a58a6d2be5e93d5ded1164e79 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/_global_mmlu_full_bn_stem.yaml @@ -0,0 +1,8 @@ +group: global_mmlu_full_bn_stem +task: + - global_mmlu_full_bn_stem_tasks +aggregate_metric_list: + - metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_abstract_algebra.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_abstract_algebra.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5bb7bb61bf8792d446119597462f346bdf9cac6c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_abstract_algebra.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _bn_template_yaml +process_docs: !function utils.process_abstract_algebra +tag: global_mmlu_full_bn_stem_tasks +task: global_mmlu_full_bn_abstract_algebra diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_anatomy.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_anatomy.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d49070f11ddb642696421f5faa899c2b6949618a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_anatomy.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _bn_template_yaml +process_docs: !function utils.process_anatomy +tag: global_mmlu_full_bn_stem_tasks +task: global_mmlu_full_bn_anatomy diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_astronomy.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_astronomy.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2e6dbc971aea6fb0fde65c921a678db580565f8a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_astronomy.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _bn_template_yaml +process_docs: !function utils.process_astronomy +tag: global_mmlu_full_bn_stem_tasks +task: global_mmlu_full_bn_astronomy diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_business_ethics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_business_ethics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8c45a0e2d3f11c1a3d07dc8254330fdf6947c4fe --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_business_ethics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _bn_template_yaml +process_docs: !function utils.process_business_ethics +tag: global_mmlu_full_bn_other_tasks +task: global_mmlu_full_bn_business_ethics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_clinical_knowledge.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_clinical_knowledge.yaml new file mode 100644 index 0000000000000000000000000000000000000000..97e17570b5961b2b81d0f05f8507f2348cb6fea9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_clinical_knowledge.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _bn_template_yaml +process_docs: !function utils.process_clinical_knowledge +tag: global_mmlu_full_bn_other_tasks +task: global_mmlu_full_bn_clinical_knowledge diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_college_biology.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_college_biology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9bf0b34cb28a00971ce8f142285673c0d2cf6e69 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_college_biology.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _bn_template_yaml +process_docs: !function utils.process_college_biology +tag: global_mmlu_full_bn_stem_tasks +task: global_mmlu_full_bn_college_biology diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_college_computer_science.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_college_computer_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ecd60e54e11955342f376dcb9a7a6e12497224d1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_college_computer_science.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _bn_template_yaml +process_docs: !function utils.process_college_computer_science +tag: global_mmlu_full_bn_stem_tasks +task: global_mmlu_full_bn_college_computer_science diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_computer_security.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_computer_security.yaml new file mode 100644 index 0000000000000000000000000000000000000000..184097f8b8f66eb73cf75b4e88d1467ca537a195 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_computer_security.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _bn_template_yaml +process_docs: !function utils.process_computer_security +tag: global_mmlu_full_bn_stem_tasks +task: global_mmlu_full_bn_computer_security diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_moral_disputes.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_moral_disputes.yaml new file mode 100644 index 0000000000000000000000000000000000000000..444032164d0e999a98f62e5e7567206ef493b0fc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_moral_disputes.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _bn_template_yaml +process_docs: !function utils.process_moral_disputes +tag: global_mmlu_full_bn_humanities_tasks +task: global_mmlu_full_bn_moral_disputes diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_philosophy.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_philosophy.yaml new file mode 100644 index 0000000000000000000000000000000000000000..62af532bf13cb63b6a823da8770f34a3aa68c119 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_philosophy.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _bn_template_yaml +process_docs: !function utils.process_philosophy +tag: global_mmlu_full_bn_humanities_tasks +task: global_mmlu_full_bn_philosophy diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_prehistory.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_prehistory.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dc49d36c3c3fbeea0d054289f0b79039cbab2f48 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_prehistory.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _bn_template_yaml +process_docs: !function utils.process_prehistory +tag: global_mmlu_full_bn_humanities_tasks +task: global_mmlu_full_bn_prehistory diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_professional_accounting.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_professional_accounting.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bf72a6a4232c0f194684414b2b25500814a98fc3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_professional_accounting.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _bn_template_yaml +process_docs: !function utils.process_professional_accounting +tag: global_mmlu_full_bn_other_tasks +task: global_mmlu_full_bn_professional_accounting diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_professional_law.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_professional_law.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f49fb142627d2cc38d1fd3d86cda06e0566f9c92 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_professional_law.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _bn_template_yaml +process_docs: !function utils.process_professional_law +tag: global_mmlu_full_bn_humanities_tasks +task: global_mmlu_full_bn_professional_law diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_professional_medicine.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_professional_medicine.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3c53d77a7b3206aebe3007695b558e74881ef1a1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_professional_medicine.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _bn_template_yaml +process_docs: !function utils.process_professional_medicine +tag: global_mmlu_full_bn_other_tasks +task: global_mmlu_full_bn_professional_medicine diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_professional_psychology.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_professional_psychology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a50c5cbf13d6ab1a9e81fb5d6e16c2518bff142e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_professional_psychology.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _bn_template_yaml +process_docs: !function utils.process_professional_psychology +tag: global_mmlu_full_bn_social_sciences_tasks +task: global_mmlu_full_bn_professional_psychology diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_public_relations.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_public_relations.yaml new file mode 100644 index 0000000000000000000000000000000000000000..00e2742a44a3689a6d0f8641cef8ad01a8407a25 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_public_relations.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _bn_template_yaml +process_docs: !function utils.process_public_relations +tag: global_mmlu_full_bn_social_sciences_tasks +task: global_mmlu_full_bn_public_relations diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_security_studies.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_security_studies.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5a0e7612c2c5f55ee5dd63d3be71ffa2e106a5c9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_security_studies.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _bn_template_yaml +process_docs: !function utils.process_security_studies +tag: global_mmlu_full_bn_social_sciences_tasks +task: global_mmlu_full_bn_security_studies diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_sociology.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_sociology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e88203197be6601666c9b9d61d60724adb2e4652 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_sociology.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _bn_template_yaml +process_docs: !function utils.process_sociology +tag: global_mmlu_full_bn_social_sciences_tasks +task: global_mmlu_full_bn_sociology diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_us_foreign_policy.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_us_foreign_policy.yaml new file mode 100644 index 0000000000000000000000000000000000000000..42be796a3cd49e2f3073a5d6a8ffae2fca96b8f4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_us_foreign_policy.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _bn_template_yaml +process_docs: !function utils.process_us_foreign_policy +tag: global_mmlu_full_bn_social_sciences_tasks +task: global_mmlu_full_bn_us_foreign_policy diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_virology.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_virology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3959f00620ba302b1ea7ae28622c4d551219f445 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_virology.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _bn_template_yaml +process_docs: !function utils.process_virology +tag: global_mmlu_full_bn_other_tasks +task: global_mmlu_full_bn_virology diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_world_religions.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_world_religions.yaml new file mode 100644 index 0000000000000000000000000000000000000000..15ee9efc24fb3ee74a6c02d7124198fa932ed2c7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/global_mmlu_full_bn_world_religions.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _bn_template_yaml +process_docs: !function utils.process_world_religions +tag: global_mmlu_full_bn_humanities_tasks +task: global_mmlu_full_bn_world_religions diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/utils.py b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..7df72cb061f0fecba46e15ff9b57552817979afb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/bn/utils.py @@ -0,0 +1,73 @@ +from functools import partial + + +SUBJECTS = [ + "abstract_algebra", + "anatomy", + "astronomy", + "business_ethics", + "clinical_knowledge", + "college_biology", + "college_chemistry", + "college_computer_science", + "college_mathematics", + "college_medicine", + "college_physics", + "computer_security", + "conceptual_physics", + "econometrics", + "electrical_engineering", + "elementary_mathematics", + "formal_logic", + "global_facts", + "high_school_biology", + "high_school_chemistry", + "high_school_computer_science", + "high_school_european_history", + "high_school_geography", + "high_school_government_and_politics", + "high_school_macroeconomics", + "high_school_mathematics", + "high_school_microeconomics", + "high_school_physics", + "high_school_psychology", + "high_school_statistics", + "high_school_us_history", + "high_school_world_history", + "human_aging", + "human_sexuality", + "international_law", + "jurisprudence", + "logical_fallacies", + "machine_learning", + "management", + "marketing", + "medical_genetics", + "miscellaneous", + "moral_disputes", + "moral_scenarios", + "nutrition", + "philosophy", + "prehistory", + "professional_accounting", + "professional_law", + "professional_medicine", + "professional_psychology", + "public_relations", + "security_studies", + "sociology", + "us_foreign_policy", + "virology", + "world_religions", +] + + +def process_docs(dataset, subject): + return dataset.filter(lambda x: x["subject"] == subject) + + +process_functions = { + f"process_{subject}": partial(process_docs, subject=subject) for subject in SUBJECTS +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_abstract_algebra.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_abstract_algebra.yaml new file mode 100644 index 0000000000000000000000000000000000000000..40431ec993e43480a8fe8e7a55802f1cfe1a05cb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_abstract_algebra.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _cs_template_yaml +process_docs: !function utils.process_abstract_algebra +tag: global_mmlu_full_cs_stem_tasks +task: global_mmlu_full_cs_abstract_algebra diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_college_computer_science.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_college_computer_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fd619d13f01bab3dcb37447b7cfa79c15b767f35 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_college_computer_science.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _cs_template_yaml +process_docs: !function utils.process_college_computer_science +tag: global_mmlu_full_cs_stem_tasks +task: global_mmlu_full_cs_college_computer_science diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_college_physics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_college_physics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e98df33917d7b57b511c631673e4849e9565e61e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_college_physics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _cs_template_yaml +process_docs: !function utils.process_college_physics +tag: global_mmlu_full_cs_stem_tasks +task: global_mmlu_full_cs_college_physics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_computer_security.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_computer_security.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7256ad679259dc505fd548cf13d0139af2348bc6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_computer_security.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _cs_template_yaml +process_docs: !function utils.process_computer_security +tag: global_mmlu_full_cs_stem_tasks +task: global_mmlu_full_cs_computer_security diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_conceptual_physics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_conceptual_physics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9bd6449894e2186231cd2454a40ee4209e213fd5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_conceptual_physics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _cs_template_yaml +process_docs: !function utils.process_conceptual_physics +tag: global_mmlu_full_cs_stem_tasks +task: global_mmlu_full_cs_conceptual_physics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_econometrics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_econometrics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c954d3202d6c163fb00bba481c848b1435b1c433 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_econometrics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _cs_template_yaml +process_docs: !function utils.process_econometrics +tag: global_mmlu_full_cs_social_sciences_tasks +task: global_mmlu_full_cs_econometrics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_electrical_engineering.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_electrical_engineering.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2f80e8ac0e32add90f49363161fe668dcefcf9b7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_electrical_engineering.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _cs_template_yaml +process_docs: !function utils.process_electrical_engineering +tag: global_mmlu_full_cs_stem_tasks +task: global_mmlu_full_cs_electrical_engineering diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_elementary_mathematics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_elementary_mathematics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bfbc2c9d6681a1b51fb21c9892194fb6875e0f84 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_elementary_mathematics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _cs_template_yaml +process_docs: !function utils.process_elementary_mathematics +tag: global_mmlu_full_cs_stem_tasks +task: global_mmlu_full_cs_elementary_mathematics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_formal_logic.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_formal_logic.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0c2ec8bdc2ab4cdee00c02b079ecad8617a83a9d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_formal_logic.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _cs_template_yaml +process_docs: !function utils.process_formal_logic +tag: global_mmlu_full_cs_humanities_tasks +task: global_mmlu_full_cs_formal_logic diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_global_facts.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_global_facts.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6302b417e70c09a831aa7ae1124ddf559fca01fa --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_global_facts.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _cs_template_yaml +process_docs: !function utils.process_global_facts +tag: global_mmlu_full_cs_other_tasks +task: global_mmlu_full_cs_global_facts diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_high_school_biology.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_high_school_biology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b69e9ac39f47606b7b9b793b96e27111fbf3576b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_high_school_biology.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _cs_template_yaml +process_docs: !function utils.process_high_school_biology +tag: global_mmlu_full_cs_stem_tasks +task: global_mmlu_full_cs_high_school_biology diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_high_school_chemistry.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_high_school_chemistry.yaml new file mode 100644 index 0000000000000000000000000000000000000000..67f53cf54a48ef42b67811ed9ce7d43352896c3d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_high_school_chemistry.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _cs_template_yaml +process_docs: !function utils.process_high_school_chemistry +tag: global_mmlu_full_cs_stem_tasks +task: global_mmlu_full_cs_high_school_chemistry diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_high_school_european_history.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_high_school_european_history.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7fa264c36aba001043ba9b65c1beceb1ac085e4c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_high_school_european_history.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _cs_template_yaml +process_docs: !function utils.process_high_school_european_history +tag: global_mmlu_full_cs_humanities_tasks +task: global_mmlu_full_cs_high_school_european_history diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_high_school_geography.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_high_school_geography.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b9f903c2f55a8fab7ebbda63078de4d8573958be --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_high_school_geography.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _cs_template_yaml +process_docs: !function utils.process_high_school_geography +tag: global_mmlu_full_cs_social_sciences_tasks +task: global_mmlu_full_cs_high_school_geography diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_high_school_government_and_politics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_high_school_government_and_politics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5bde4d695c6abd7d46a9e809a33f3d6e1e910c8f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_high_school_government_and_politics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _cs_template_yaml +process_docs: !function utils.process_high_school_government_and_politics +tag: global_mmlu_full_cs_social_sciences_tasks +task: global_mmlu_full_cs_high_school_government_and_politics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_high_school_macroeconomics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_high_school_macroeconomics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bb5068edb22afd447822a25b8308bfbfe91dfa9c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_high_school_macroeconomics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _cs_template_yaml +process_docs: !function utils.process_high_school_macroeconomics +tag: global_mmlu_full_cs_social_sciences_tasks +task: global_mmlu_full_cs_high_school_macroeconomics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_high_school_mathematics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_high_school_mathematics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..87cb3e577052f3e3e9d9bd8565189a28a622a297 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_high_school_mathematics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _cs_template_yaml +process_docs: !function utils.process_high_school_mathematics +tag: global_mmlu_full_cs_stem_tasks +task: global_mmlu_full_cs_high_school_mathematics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_high_school_microeconomics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_high_school_microeconomics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..33c2e18c8bd54a4d76b6c4308f84cda9d2cebcec --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_high_school_microeconomics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _cs_template_yaml +process_docs: !function utils.process_high_school_microeconomics +tag: global_mmlu_full_cs_social_sciences_tasks +task: global_mmlu_full_cs_high_school_microeconomics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_high_school_physics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_high_school_physics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1ed095bf4193d57d69d450bcf3f494ffdbef90f1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_high_school_physics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _cs_template_yaml +process_docs: !function utils.process_high_school_physics +tag: global_mmlu_full_cs_stem_tasks +task: global_mmlu_full_cs_high_school_physics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_high_school_psychology.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_high_school_psychology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..59b623053c0a76edb790904a3a2dc479abae7c21 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_high_school_psychology.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _cs_template_yaml +process_docs: !function utils.process_high_school_psychology +tag: global_mmlu_full_cs_social_sciences_tasks +task: global_mmlu_full_cs_high_school_psychology diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_high_school_statistics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_high_school_statistics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1a18ee25b4a783e7aa04ab73ed35c2ec83347f85 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_high_school_statistics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _cs_template_yaml +process_docs: !function utils.process_high_school_statistics +tag: global_mmlu_full_cs_stem_tasks +task: global_mmlu_full_cs_high_school_statistics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_high_school_us_history.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_high_school_us_history.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d8d0a27165ac72de8e9ed3fa0d0372f68201b6ab --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_high_school_us_history.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _cs_template_yaml +process_docs: !function utils.process_high_school_us_history +tag: global_mmlu_full_cs_humanities_tasks +task: global_mmlu_full_cs_high_school_us_history diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_high_school_world_history.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_high_school_world_history.yaml new file mode 100644 index 0000000000000000000000000000000000000000..07012306012a655b2d1d07ac1ac68ae43eefc49b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_high_school_world_history.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _cs_template_yaml +process_docs: !function utils.process_high_school_world_history +tag: global_mmlu_full_cs_humanities_tasks +task: global_mmlu_full_cs_high_school_world_history diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_human_aging.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_human_aging.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e3f5c7c392b23058900e0475e4fbc16871ef89ee --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_human_aging.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _cs_template_yaml +process_docs: !function utils.process_human_aging +tag: global_mmlu_full_cs_other_tasks +task: global_mmlu_full_cs_human_aging diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_human_sexuality.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_human_sexuality.yaml new file mode 100644 index 0000000000000000000000000000000000000000..61d405c7314c6ea110d0c1c60ae393158c14912a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_human_sexuality.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _cs_template_yaml +process_docs: !function utils.process_human_sexuality +tag: global_mmlu_full_cs_social_sciences_tasks +task: global_mmlu_full_cs_human_sexuality diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_international_law.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_international_law.yaml new file mode 100644 index 0000000000000000000000000000000000000000..509ebee46050eb6deff0af1fce8d5c5738714f80 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_international_law.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _cs_template_yaml +process_docs: !function utils.process_international_law +tag: global_mmlu_full_cs_humanities_tasks +task: global_mmlu_full_cs_international_law diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_jurisprudence.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_jurisprudence.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c0e27957b80c1ae9fec8b327d248f57d3c09b257 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_jurisprudence.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _cs_template_yaml +process_docs: !function utils.process_jurisprudence +tag: global_mmlu_full_cs_humanities_tasks +task: global_mmlu_full_cs_jurisprudence diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_logical_fallacies.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_logical_fallacies.yaml new file mode 100644 index 0000000000000000000000000000000000000000..85010f3c6bd33f9d17f972c3f4380993a4ba297f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_logical_fallacies.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _cs_template_yaml +process_docs: !function utils.process_logical_fallacies +tag: global_mmlu_full_cs_humanities_tasks +task: global_mmlu_full_cs_logical_fallacies diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_machine_learning.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_machine_learning.yaml new file mode 100644 index 0000000000000000000000000000000000000000..32aaa1a65eb75cc87c6fc441c315a0486fbb3f4f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_machine_learning.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _cs_template_yaml +process_docs: !function utils.process_machine_learning +tag: global_mmlu_full_cs_stem_tasks +task: global_mmlu_full_cs_machine_learning diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_management.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_management.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4e1a3a7cbe4399703e4763f3943556cc8c92aa21 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_management.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _cs_template_yaml +process_docs: !function utils.process_management +tag: global_mmlu_full_cs_other_tasks +task: global_mmlu_full_cs_management diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_marketing.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_marketing.yaml new file mode 100644 index 0000000000000000000000000000000000000000..239e3c0c486f0ba846b4985b5db894ae9a121cb8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_marketing.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _cs_template_yaml +process_docs: !function utils.process_marketing +tag: global_mmlu_full_cs_other_tasks +task: global_mmlu_full_cs_marketing diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_medical_genetics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_medical_genetics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1c76fee782daa02ca303148a5f027925215dcfa4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_medical_genetics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _cs_template_yaml +process_docs: !function utils.process_medical_genetics +tag: global_mmlu_full_cs_other_tasks +task: global_mmlu_full_cs_medical_genetics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_miscellaneous.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_miscellaneous.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4be6207a2fcfc55aee287473478f0f8ce247d7d6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_miscellaneous.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _cs_template_yaml +process_docs: !function utils.process_miscellaneous +tag: global_mmlu_full_cs_other_tasks +task: global_mmlu_full_cs_miscellaneous diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_moral_disputes.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_moral_disputes.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b263f67ec701cb822a819fa94c55a56bce2427f1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_moral_disputes.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _cs_template_yaml +process_docs: !function utils.process_moral_disputes +tag: global_mmlu_full_cs_humanities_tasks +task: global_mmlu_full_cs_moral_disputes diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_moral_scenarios.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_moral_scenarios.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6532a43ed0678c86ae3e70edb12c5f30b7d5bcba --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_moral_scenarios.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _cs_template_yaml +process_docs: !function utils.process_moral_scenarios +tag: global_mmlu_full_cs_humanities_tasks +task: global_mmlu_full_cs_moral_scenarios diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_nutrition.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_nutrition.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3f04fbcd1d67b57abe5b074acad4becc413143bc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_nutrition.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _cs_template_yaml +process_docs: !function utils.process_nutrition +tag: global_mmlu_full_cs_other_tasks +task: global_mmlu_full_cs_nutrition diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_philosophy.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_philosophy.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2f5093f9228994c8a46eba42d80afab28496b6d2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_philosophy.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _cs_template_yaml +process_docs: !function utils.process_philosophy +tag: global_mmlu_full_cs_humanities_tasks +task: global_mmlu_full_cs_philosophy diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_prehistory.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_prehistory.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a8f5f5a5d8fd3e0384ecfde3712f1161da67d4db --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_prehistory.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _cs_template_yaml +process_docs: !function utils.process_prehistory +tag: global_mmlu_full_cs_humanities_tasks +task: global_mmlu_full_cs_prehistory diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_professional_accounting.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_professional_accounting.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bccb71b239cefa46527a5ab21d28861888fcf3a9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_professional_accounting.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _cs_template_yaml +process_docs: !function utils.process_professional_accounting +tag: global_mmlu_full_cs_other_tasks +task: global_mmlu_full_cs_professional_accounting diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_professional_law.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_professional_law.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ff50f50c51be5d1e39a394fccd528ba68cd087e6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_professional_law.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _cs_template_yaml +process_docs: !function utils.process_professional_law +tag: global_mmlu_full_cs_humanities_tasks +task: global_mmlu_full_cs_professional_law diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_professional_medicine.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_professional_medicine.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9b82937902b106bc052c782d0605feafeac3eca8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_professional_medicine.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _cs_template_yaml +process_docs: !function utils.process_professional_medicine +tag: global_mmlu_full_cs_other_tasks +task: global_mmlu_full_cs_professional_medicine diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_professional_psychology.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_professional_psychology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e41edb29eadcb1f685146e92c8ae09bfba5b83cd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_professional_psychology.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _cs_template_yaml +process_docs: !function utils.process_professional_psychology +tag: global_mmlu_full_cs_social_sciences_tasks +task: global_mmlu_full_cs_professional_psychology diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_public_relations.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_public_relations.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e8fb512db97343e089c71791174accf1c2e69086 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_public_relations.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _cs_template_yaml +process_docs: !function utils.process_public_relations +tag: global_mmlu_full_cs_social_sciences_tasks +task: global_mmlu_full_cs_public_relations diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_security_studies.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_security_studies.yaml new file mode 100644 index 0000000000000000000000000000000000000000..64ec0b3f0f14088b0e567dd28d323d784f66dba7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_security_studies.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _cs_template_yaml +process_docs: !function utils.process_security_studies +tag: global_mmlu_full_cs_social_sciences_tasks +task: global_mmlu_full_cs_security_studies diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_sociology.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_sociology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..18214f7c299fb79df76026c749b02e03372fce39 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_sociology.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _cs_template_yaml +process_docs: !function utils.process_sociology +tag: global_mmlu_full_cs_social_sciences_tasks +task: global_mmlu_full_cs_sociology diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_us_foreign_policy.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_us_foreign_policy.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ac42b097003ab1ea266739aa2fc4e1f123c4d9ac --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_us_foreign_policy.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _cs_template_yaml +process_docs: !function utils.process_us_foreign_policy +tag: global_mmlu_full_cs_social_sciences_tasks +task: global_mmlu_full_cs_us_foreign_policy diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_virology.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_virology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a51b8aefae0bbcc53d665bce15789a3802612279 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_virology.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _cs_template_yaml +process_docs: !function utils.process_virology +tag: global_mmlu_full_cs_other_tasks +task: global_mmlu_full_cs_virology diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_world_religions.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_world_religions.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cf9af3e9c9a9ef635048b1658b74d442459a9ddd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/global_mmlu_full_cs_world_religions.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _cs_template_yaml +process_docs: !function utils.process_world_religions +tag: global_mmlu_full_cs_humanities_tasks +task: global_mmlu_full_cs_world_religions diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/utils.py b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..7df72cb061f0fecba46e15ff9b57552817979afb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/cs/utils.py @@ -0,0 +1,73 @@ +from functools import partial + + +SUBJECTS = [ + "abstract_algebra", + "anatomy", + "astronomy", + "business_ethics", + "clinical_knowledge", + "college_biology", + "college_chemistry", + "college_computer_science", + "college_mathematics", + "college_medicine", + "college_physics", + "computer_security", + "conceptual_physics", + "econometrics", + "electrical_engineering", + "elementary_mathematics", + "formal_logic", + "global_facts", + "high_school_biology", + "high_school_chemistry", + "high_school_computer_science", + "high_school_european_history", + "high_school_geography", + "high_school_government_and_politics", + "high_school_macroeconomics", + "high_school_mathematics", + "high_school_microeconomics", + "high_school_physics", + "high_school_psychology", + "high_school_statistics", + "high_school_us_history", + "high_school_world_history", + "human_aging", + "human_sexuality", + "international_law", + "jurisprudence", + "logical_fallacies", + "machine_learning", + "management", + "marketing", + "medical_genetics", + "miscellaneous", + "moral_disputes", + "moral_scenarios", + "nutrition", + "philosophy", + "prehistory", + "professional_accounting", + "professional_law", + "professional_medicine", + "professional_psychology", + "public_relations", + "security_studies", + "sociology", + "us_foreign_policy", + "virology", + "world_religions", +] + + +def process_docs(dataset, subject): + return dataset.filter(lambda x: x["subject"] == subject) + + +process_functions = { + f"process_{subject}": partial(process_docs, subject=subject) for subject in SUBJECTS +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/_de_template_yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/_de_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..036b86192fcce6e3ab50cb62e00f96a698c9c30f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/_de_template_yaml @@ -0,0 +1,16 @@ +dataset_path: CohereForAI/Global-MMLU +dataset_name: de +test_split: test +fewshot_split: dev +fewshot_config: + sampler: first_n +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\nD. {{option_d}}\nAnswer:" +doc_to_choice: ["A", "B", "C", "D"] +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/_global_mmlu_full_de.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/_global_mmlu_full_de.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5217599a6771cf331c36f2b1ff2de4aa2f9a191b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/_global_mmlu_full_de.yaml @@ -0,0 +1,11 @@ +group: global_mmlu_full_de +task: + - global_mmlu_full_de_stem + - global_mmlu_full_de_other + - global_mmlu_full_de_social_sciences + - global_mmlu_full_de_humanities +aggregate_metric_list: + - metric: acc + weight_by_size: True +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/_global_mmlu_full_de_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/_global_mmlu_full_de_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..df571c67e9ee2c616f56e94d48ffd90ce6a1f61e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/_global_mmlu_full_de_humanities.yaml @@ -0,0 +1,8 @@ +group: global_mmlu_full_de_humanities +task: + - global_mmlu_full_de_humanities_tasks +aggregate_metric_list: + - metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/_global_mmlu_full_de_other.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/_global_mmlu_full_de_other.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bfff864e482634d6f7347f8806df1ef0e56b8476 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/_global_mmlu_full_de_other.yaml @@ -0,0 +1,8 @@ +group: global_mmlu_full_de_other +task: + - global_mmlu_full_de_other_tasks +aggregate_metric_list: + - metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/_global_mmlu_full_de_social_sciences.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/_global_mmlu_full_de_social_sciences.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8cf304a2c7a957c4719abc08367c9f4e4459976d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/_global_mmlu_full_de_social_sciences.yaml @@ -0,0 +1,8 @@ +group: global_mmlu_full_de_social_sciences +task: + - global_mmlu_full_de_social_sciences_tasks +aggregate_metric_list: + - metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/_global_mmlu_full_de_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/_global_mmlu_full_de_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..75d1aa5a161df37d4c83788684982b220d43146f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/_global_mmlu_full_de_stem.yaml @@ -0,0 +1,8 @@ +group: global_mmlu_full_de_stem +task: + - global_mmlu_full_de_stem_tasks +aggregate_metric_list: + - metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_abstract_algebra.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_abstract_algebra.yaml new file mode 100644 index 0000000000000000000000000000000000000000..07cd235668a9f25450ce1ffd57753185b0ef5f3a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_abstract_algebra.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _de_template_yaml +process_docs: !function utils.process_abstract_algebra +tag: global_mmlu_full_de_stem_tasks +task: global_mmlu_full_de_abstract_algebra diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_anatomy.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_anatomy.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9deb16a6e6df64672019dc30b9b5238f33e44e7c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_anatomy.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _de_template_yaml +process_docs: !function utils.process_anatomy +tag: global_mmlu_full_de_stem_tasks +task: global_mmlu_full_de_anatomy diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_astronomy.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_astronomy.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6a743d45543c0e91edb6c6989fc0fbd8a657ccd2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_astronomy.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _de_template_yaml +process_docs: !function utils.process_astronomy +tag: global_mmlu_full_de_stem_tasks +task: global_mmlu_full_de_astronomy diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_business_ethics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_business_ethics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..37bf9d454e5622196a152629413be4a0f04d4a7b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_business_ethics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _de_template_yaml +process_docs: !function utils.process_business_ethics +tag: global_mmlu_full_de_other_tasks +task: global_mmlu_full_de_business_ethics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_clinical_knowledge.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_clinical_knowledge.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c5ad878a8caab668fb2f404e5befb700f7f66ca9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_clinical_knowledge.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _de_template_yaml +process_docs: !function utils.process_clinical_knowledge +tag: global_mmlu_full_de_other_tasks +task: global_mmlu_full_de_clinical_knowledge diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_college_biology.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_college_biology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..200f9239f0fca2f88c957ab8f61e7aebce68c52e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_college_biology.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _de_template_yaml +process_docs: !function utils.process_college_biology +tag: global_mmlu_full_de_stem_tasks +task: global_mmlu_full_de_college_biology diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_college_chemistry.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_college_chemistry.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2bbc4d463a43602d4497b86cfe15b3a899469988 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_college_chemistry.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _de_template_yaml +process_docs: !function utils.process_college_chemistry +tag: global_mmlu_full_de_stem_tasks +task: global_mmlu_full_de_college_chemistry diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_college_computer_science.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_college_computer_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ac903e3a5d8e74df8047a0078d87fe26911ac8d9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_college_computer_science.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _de_template_yaml +process_docs: !function utils.process_college_computer_science +tag: global_mmlu_full_de_stem_tasks +task: global_mmlu_full_de_college_computer_science diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_college_mathematics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_college_mathematics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..616010cad26bcabbb820146fc75497d21f075a20 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_college_mathematics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _de_template_yaml +process_docs: !function utils.process_college_mathematics +tag: global_mmlu_full_de_stem_tasks +task: global_mmlu_full_de_college_mathematics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_college_medicine.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_college_medicine.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b9648ce8898b2789733dc93f6c3915578ec805d8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_college_medicine.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _de_template_yaml +process_docs: !function utils.process_college_medicine +tag: global_mmlu_full_de_other_tasks +task: global_mmlu_full_de_college_medicine diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_college_physics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_college_physics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d3bc689255cc954fc6af728ea932421d85a76dd2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_college_physics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _de_template_yaml +process_docs: !function utils.process_college_physics +tag: global_mmlu_full_de_stem_tasks +task: global_mmlu_full_de_college_physics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_computer_security.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_computer_security.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fee01f9e1b1cd19894f68521170d71b9708164dc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_computer_security.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _de_template_yaml +process_docs: !function utils.process_computer_security +tag: global_mmlu_full_de_stem_tasks +task: global_mmlu_full_de_computer_security diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_conceptual_physics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_conceptual_physics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..201c17d7c6f5feed59b24faa309cc8ed3926583e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_conceptual_physics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _de_template_yaml +process_docs: !function utils.process_conceptual_physics +tag: global_mmlu_full_de_stem_tasks +task: global_mmlu_full_de_conceptual_physics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_econometrics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_econometrics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1d902c3c211dcc24584a2526093ad0d8579bd812 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_econometrics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _de_template_yaml +process_docs: !function utils.process_econometrics +tag: global_mmlu_full_de_social_sciences_tasks +task: global_mmlu_full_de_econometrics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_electrical_engineering.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_electrical_engineering.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8dcb6c48bb13e830773d668e1e5adefc3fe59738 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_electrical_engineering.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _de_template_yaml +process_docs: !function utils.process_electrical_engineering +tag: global_mmlu_full_de_stem_tasks +task: global_mmlu_full_de_electrical_engineering diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_elementary_mathematics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_elementary_mathematics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a1ca41ce60dbc58ba38cf891556c023063798bb0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_elementary_mathematics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _de_template_yaml +process_docs: !function utils.process_elementary_mathematics +tag: global_mmlu_full_de_stem_tasks +task: global_mmlu_full_de_elementary_mathematics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_formal_logic.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_formal_logic.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6e16729e7559e1de2b5f82a5b876937d4fb9c01a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_formal_logic.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _de_template_yaml +process_docs: !function utils.process_formal_logic +tag: global_mmlu_full_de_humanities_tasks +task: global_mmlu_full_de_formal_logic diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_global_facts.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_global_facts.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a7b092892247185fd237111cfeae8f27fbce92b7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_global_facts.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _de_template_yaml +process_docs: !function utils.process_global_facts +tag: global_mmlu_full_de_other_tasks +task: global_mmlu_full_de_global_facts diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_high_school_biology.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_high_school_biology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0ad59551af9d4b396b8ff7bdbd4f64e5ed22e2c7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_high_school_biology.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _de_template_yaml +process_docs: !function utils.process_high_school_biology +tag: global_mmlu_full_de_stem_tasks +task: global_mmlu_full_de_high_school_biology diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_high_school_chemistry.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_high_school_chemistry.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6c0fbd556d5da09d4ce4f5e33a303dfb308723fc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_high_school_chemistry.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _de_template_yaml +process_docs: !function utils.process_high_school_chemistry +tag: global_mmlu_full_de_stem_tasks +task: global_mmlu_full_de_high_school_chemistry diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_high_school_computer_science.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_high_school_computer_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0aea5ada3049995d931389e47d1a34d275ddba09 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_high_school_computer_science.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _de_template_yaml +process_docs: !function utils.process_high_school_computer_science +tag: global_mmlu_full_de_stem_tasks +task: global_mmlu_full_de_high_school_computer_science diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_high_school_european_history.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_high_school_european_history.yaml new file mode 100644 index 0000000000000000000000000000000000000000..97293b4983b298cdb8dd160562b7368039e26646 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_high_school_european_history.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _de_template_yaml +process_docs: !function utils.process_high_school_european_history +tag: global_mmlu_full_de_humanities_tasks +task: global_mmlu_full_de_high_school_european_history diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_high_school_geography.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_high_school_geography.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d26a65d9707500d6978a0e7f5937180e3f86663f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_high_school_geography.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _de_template_yaml +process_docs: !function utils.process_high_school_geography +tag: global_mmlu_full_de_social_sciences_tasks +task: global_mmlu_full_de_high_school_geography diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_high_school_government_and_politics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_high_school_government_and_politics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b6ec78e696bf139ecb8c450469fd7e023a1b02a1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_high_school_government_and_politics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _de_template_yaml +process_docs: !function utils.process_high_school_government_and_politics +tag: global_mmlu_full_de_social_sciences_tasks +task: global_mmlu_full_de_high_school_government_and_politics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_high_school_macroeconomics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_high_school_macroeconomics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..53489d855cb541ded15bd3bd1a0fd0c880c0732f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_high_school_macroeconomics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _de_template_yaml +process_docs: !function utils.process_high_school_macroeconomics +tag: global_mmlu_full_de_social_sciences_tasks +task: global_mmlu_full_de_high_school_macroeconomics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_high_school_mathematics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_high_school_mathematics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..44a5666f724eec1420be7b1c68db1379b81443dc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_high_school_mathematics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _de_template_yaml +process_docs: !function utils.process_high_school_mathematics +tag: global_mmlu_full_de_stem_tasks +task: global_mmlu_full_de_high_school_mathematics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_high_school_microeconomics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_high_school_microeconomics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3b911297cbdce278c3a81851af21bf873b539dde --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_high_school_microeconomics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _de_template_yaml +process_docs: !function utils.process_high_school_microeconomics +tag: global_mmlu_full_de_social_sciences_tasks +task: global_mmlu_full_de_high_school_microeconomics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_high_school_psychology.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_high_school_psychology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ae7680027480500cb143703e9fafbd1dd08429e8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_high_school_psychology.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _de_template_yaml +process_docs: !function utils.process_high_school_psychology +tag: global_mmlu_full_de_social_sciences_tasks +task: global_mmlu_full_de_high_school_psychology diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_high_school_us_history.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_high_school_us_history.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9c1eff81145c2e12fe209bcadf93e453c8e753a1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/de/global_mmlu_full_de_high_school_us_history.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _de_template_yaml +process_docs: !function utils.process_high_school_us_history +tag: global_mmlu_full_de_humanities_tasks +task: global_mmlu_full_de_high_school_us_history diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/el/_el_template_yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/el/_el_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..5fccad5ec1e5154501be5204afb01de5e6c51f6f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/el/_el_template_yaml @@ -0,0 +1,16 @@ +dataset_path: CohereForAI/Global-MMLU +dataset_name: el +test_split: test +fewshot_split: dev +fewshot_config: + sampler: first_n +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\nD. {{option_d}}\nAnswer:" +doc_to_choice: ["A", "B", "C", "D"] +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/el/_global_mmlu_full_el_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/el/_global_mmlu_full_el_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2123be0887d6e1cfad81efa8ca37923f799d47d6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/el/_global_mmlu_full_el_stem.yaml @@ -0,0 +1,8 @@ +group: global_mmlu_full_el_stem +task: + - global_mmlu_full_el_stem_tasks +aggregate_metric_list: + - metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/el/global_mmlu_full_el_logical_fallacies.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/el/global_mmlu_full_el_logical_fallacies.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7e5bdb4f33295bbb15d6855cfb2ff3fdae8aaea2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/el/global_mmlu_full_el_logical_fallacies.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _el_template_yaml +process_docs: !function utils.process_logical_fallacies +tag: global_mmlu_full_el_humanities_tasks +task: global_mmlu_full_el_logical_fallacies diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/el/global_mmlu_full_el_moral_scenarios.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/el/global_mmlu_full_el_moral_scenarios.yaml new file mode 100644 index 0000000000000000000000000000000000000000..14a79a4a3f8320454fc476dddb51c9a0eedf7724 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/el/global_mmlu_full_el_moral_scenarios.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _el_template_yaml +process_docs: !function utils.process_moral_scenarios +tag: global_mmlu_full_el_humanities_tasks +task: global_mmlu_full_el_moral_scenarios diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/el/global_mmlu_full_el_nutrition.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/el/global_mmlu_full_el_nutrition.yaml new file mode 100644 index 0000000000000000000000000000000000000000..595daa39f90fc3c093281f737c172c6dd41d0310 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/el/global_mmlu_full_el_nutrition.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _el_template_yaml +process_docs: !function utils.process_nutrition +tag: global_mmlu_full_el_other_tasks +task: global_mmlu_full_el_nutrition diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/el/global_mmlu_full_el_professional_accounting.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/el/global_mmlu_full_el_professional_accounting.yaml new file mode 100644 index 0000000000000000000000000000000000000000..002b02aa9bb5b61863e6d41038b5d7be3ed3fb34 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/el/global_mmlu_full_el_professional_accounting.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _el_template_yaml +process_docs: !function utils.process_professional_accounting +tag: global_mmlu_full_el_other_tasks +task: global_mmlu_full_el_professional_accounting diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/el/global_mmlu_full_el_professional_medicine.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/el/global_mmlu_full_el_professional_medicine.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a31d4e3bbc609c7ee226bde72f9c60b6d7a45d1e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/el/global_mmlu_full_el_professional_medicine.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _el_template_yaml +process_docs: !function utils.process_professional_medicine +tag: global_mmlu_full_el_other_tasks +task: global_mmlu_full_el_professional_medicine diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/el/global_mmlu_full_el_professional_psychology.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/el/global_mmlu_full_el_professional_psychology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6e04807999538b40792cf5ecc0671d07554e3f40 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/el/global_mmlu_full_el_professional_psychology.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _el_template_yaml +process_docs: !function utils.process_professional_psychology +tag: global_mmlu_full_el_social_sciences_tasks +task: global_mmlu_full_el_professional_psychology diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/el/global_mmlu_full_el_public_relations.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/el/global_mmlu_full_el_public_relations.yaml new file mode 100644 index 0000000000000000000000000000000000000000..264799d613ebb1d255640963f6f32f90a0f4794f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/el/global_mmlu_full_el_public_relations.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _el_template_yaml +process_docs: !function utils.process_public_relations +tag: global_mmlu_full_el_social_sciences_tasks +task: global_mmlu_full_el_public_relations diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/el/global_mmlu_full_el_security_studies.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/el/global_mmlu_full_el_security_studies.yaml new file mode 100644 index 0000000000000000000000000000000000000000..19ffae47ed23ff8d4e9879ace2773f9ff9b406da --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/el/global_mmlu_full_el_security_studies.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _el_template_yaml +process_docs: !function utils.process_security_studies +tag: global_mmlu_full_el_social_sciences_tasks +task: global_mmlu_full_el_security_studies diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/el/global_mmlu_full_el_sociology.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/el/global_mmlu_full_el_sociology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f57d3e0aa88d9d58898ebb5f9f976e3500c9296d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/el/global_mmlu_full_el_sociology.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _el_template_yaml +process_docs: !function utils.process_sociology +tag: global_mmlu_full_el_social_sciences_tasks +task: global_mmlu_full_el_sociology diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/el/global_mmlu_full_el_us_foreign_policy.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/el/global_mmlu_full_el_us_foreign_policy.yaml new file mode 100644 index 0000000000000000000000000000000000000000..14c76440784482cca0e75b59691f9597234aaf33 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/el/global_mmlu_full_el_us_foreign_policy.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _el_template_yaml +process_docs: !function utils.process_us_foreign_policy +tag: global_mmlu_full_el_social_sciences_tasks +task: global_mmlu_full_el_us_foreign_policy diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/el/global_mmlu_full_el_virology.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/el/global_mmlu_full_el_virology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0e444358991bf517e6b82cbcb10ceccec819d6b4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/el/global_mmlu_full_el_virology.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _el_template_yaml +process_docs: !function utils.process_virology +tag: global_mmlu_full_el_other_tasks +task: global_mmlu_full_el_virology diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/el/global_mmlu_full_el_world_religions.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/el/global_mmlu_full_el_world_religions.yaml new file mode 100644 index 0000000000000000000000000000000000000000..60f8e52e31213ae6bf0590b003f6f2895f0f4a82 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/el/global_mmlu_full_el_world_religions.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _el_template_yaml +process_docs: !function utils.process_world_religions +tag: global_mmlu_full_el_humanities_tasks +task: global_mmlu_full_el_world_religions diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/el/utils.py b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/el/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..7df72cb061f0fecba46e15ff9b57552817979afb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/el/utils.py @@ -0,0 +1,73 @@ +from functools import partial + + +SUBJECTS = [ + "abstract_algebra", + "anatomy", + "astronomy", + "business_ethics", + "clinical_knowledge", + "college_biology", + "college_chemistry", + "college_computer_science", + "college_mathematics", + "college_medicine", + "college_physics", + "computer_security", + "conceptual_physics", + "econometrics", + "electrical_engineering", + "elementary_mathematics", + "formal_logic", + "global_facts", + "high_school_biology", + "high_school_chemistry", + "high_school_computer_science", + "high_school_european_history", + "high_school_geography", + "high_school_government_and_politics", + "high_school_macroeconomics", + "high_school_mathematics", + "high_school_microeconomics", + "high_school_physics", + "high_school_psychology", + "high_school_statistics", + "high_school_us_history", + "high_school_world_history", + "human_aging", + "human_sexuality", + "international_law", + "jurisprudence", + "logical_fallacies", + "machine_learning", + "management", + "marketing", + "medical_genetics", + "miscellaneous", + "moral_disputes", + "moral_scenarios", + "nutrition", + "philosophy", + "prehistory", + "professional_accounting", + "professional_law", + "professional_medicine", + "professional_psychology", + "public_relations", + "security_studies", + "sociology", + "us_foreign_policy", + "virology", + "world_religions", +] + + +def process_docs(dataset, subject): + return dataset.filter(lambda x: x["subject"] == subject) + + +process_functions = { + f"process_{subject}": partial(process_docs, subject=subject) for subject in SUBJECTS +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_elementary_mathematics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_elementary_mathematics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..622a99f83a31eff2e70f33b6290541cad71fbdfe --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_elementary_mathematics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _en_template_yaml +process_docs: !function utils.process_elementary_mathematics +tag: global_mmlu_full_en_stem_tasks +task: global_mmlu_full_en_elementary_mathematics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_global_facts.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_global_facts.yaml new file mode 100644 index 0000000000000000000000000000000000000000..39daa506ba791f7c47ef8558c16b45929c6a0c31 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_global_facts.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _en_template_yaml +process_docs: !function utils.process_global_facts +tag: global_mmlu_full_en_other_tasks +task: global_mmlu_full_en_global_facts diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_high_school_biology.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_high_school_biology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..063392eb2b559fbd04c5bcfa347c1eefb3d9f39a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_high_school_biology.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _en_template_yaml +process_docs: !function utils.process_high_school_biology +tag: global_mmlu_full_en_stem_tasks +task: global_mmlu_full_en_high_school_biology diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_high_school_chemistry.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_high_school_chemistry.yaml new file mode 100644 index 0000000000000000000000000000000000000000..452e9445db6dbe0950d53cd67b209296985a0228 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_high_school_chemistry.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _en_template_yaml +process_docs: !function utils.process_high_school_chemistry +tag: global_mmlu_full_en_stem_tasks +task: global_mmlu_full_en_high_school_chemistry diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_high_school_computer_science.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_high_school_computer_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..baf4313624ae8743456d9594acd35e96897792db --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_high_school_computer_science.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _en_template_yaml +process_docs: !function utils.process_high_school_computer_science +tag: global_mmlu_full_en_stem_tasks +task: global_mmlu_full_en_high_school_computer_science diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_high_school_european_history.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_high_school_european_history.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fceda5c29f63e86d962e63133f56e0fa9d78c7e8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_high_school_european_history.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _en_template_yaml +process_docs: !function utils.process_high_school_european_history +tag: global_mmlu_full_en_humanities_tasks +task: global_mmlu_full_en_high_school_european_history diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_high_school_geography.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_high_school_geography.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4fbb9ade5831c9a9f8aae7b5d9c94623127b2b2a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_high_school_geography.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _en_template_yaml +process_docs: !function utils.process_high_school_geography +tag: global_mmlu_full_en_social_sciences_tasks +task: global_mmlu_full_en_high_school_geography diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_high_school_government_and_politics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_high_school_government_and_politics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..73ca9087a51f05b1134554cae72b8ae230cafe7a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_high_school_government_and_politics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _en_template_yaml +process_docs: !function utils.process_high_school_government_and_politics +tag: global_mmlu_full_en_social_sciences_tasks +task: global_mmlu_full_en_high_school_government_and_politics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_high_school_macroeconomics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_high_school_macroeconomics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1b9ca7a94b74c833e9b25927068cb188d1c1304d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_high_school_macroeconomics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _en_template_yaml +process_docs: !function utils.process_high_school_macroeconomics +tag: global_mmlu_full_en_social_sciences_tasks +task: global_mmlu_full_en_high_school_macroeconomics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_high_school_mathematics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_high_school_mathematics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9be50ad2621ad8cbc464c37d0cc6d2b1bbb95cb8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_high_school_mathematics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _en_template_yaml +process_docs: !function utils.process_high_school_mathematics +tag: global_mmlu_full_en_stem_tasks +task: global_mmlu_full_en_high_school_mathematics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_high_school_microeconomics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_high_school_microeconomics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d93285cb33c1cd22702bd0a7204d68aa1ec85582 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_high_school_microeconomics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _en_template_yaml +process_docs: !function utils.process_high_school_microeconomics +tag: global_mmlu_full_en_social_sciences_tasks +task: global_mmlu_full_en_high_school_microeconomics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_high_school_physics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_high_school_physics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2f74c609fc69e41a376e6ae194227247151c87a3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_high_school_physics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _en_template_yaml +process_docs: !function utils.process_high_school_physics +tag: global_mmlu_full_en_stem_tasks +task: global_mmlu_full_en_high_school_physics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_high_school_psychology.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_high_school_psychology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..365762ba0f521d0c68a456b922abdd43e37fac31 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_high_school_psychology.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _en_template_yaml +process_docs: !function utils.process_high_school_psychology +tag: global_mmlu_full_en_social_sciences_tasks +task: global_mmlu_full_en_high_school_psychology diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_high_school_statistics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_high_school_statistics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d6ca42ad2bd3a8931e6b7aab00608734d59b17f1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_high_school_statistics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _en_template_yaml +process_docs: !function utils.process_high_school_statistics +tag: global_mmlu_full_en_stem_tasks +task: global_mmlu_full_en_high_school_statistics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_high_school_us_history.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_high_school_us_history.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4f20a4ddcb28ca9d05205c5b7c54bc03000a7b2f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_high_school_us_history.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _en_template_yaml +process_docs: !function utils.process_high_school_us_history +tag: global_mmlu_full_en_humanities_tasks +task: global_mmlu_full_en_high_school_us_history diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_high_school_world_history.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_high_school_world_history.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d0fce40357b0ec24864631d552409460b0d2f8d1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_high_school_world_history.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _en_template_yaml +process_docs: !function utils.process_high_school_world_history +tag: global_mmlu_full_en_humanities_tasks +task: global_mmlu_full_en_high_school_world_history diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_human_aging.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_human_aging.yaml new file mode 100644 index 0000000000000000000000000000000000000000..35320a854e7b693af4994497d30ecc70177d48d6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_human_aging.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _en_template_yaml +process_docs: !function utils.process_human_aging +tag: global_mmlu_full_en_other_tasks +task: global_mmlu_full_en_human_aging diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_human_sexuality.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_human_sexuality.yaml new file mode 100644 index 0000000000000000000000000000000000000000..86096c5d0f953ff5217381e225d60582a7b2f857 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_human_sexuality.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _en_template_yaml +process_docs: !function utils.process_human_sexuality +tag: global_mmlu_full_en_social_sciences_tasks +task: global_mmlu_full_en_human_sexuality diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_international_law.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_international_law.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8a41e9fcf854fceb9f7e3716b0f238b27a506274 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_international_law.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _en_template_yaml +process_docs: !function utils.process_international_law +tag: global_mmlu_full_en_humanities_tasks +task: global_mmlu_full_en_international_law diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_jurisprudence.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_jurisprudence.yaml new file mode 100644 index 0000000000000000000000000000000000000000..aa34c443322e3c0f916ffdadba01611f72fce916 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_jurisprudence.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _en_template_yaml +process_docs: !function utils.process_jurisprudence +tag: global_mmlu_full_en_humanities_tasks +task: global_mmlu_full_en_jurisprudence diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_logical_fallacies.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_logical_fallacies.yaml new file mode 100644 index 0000000000000000000000000000000000000000..50c105b48904692fcd3847931f383d280c0a2cf9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_logical_fallacies.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _en_template_yaml +process_docs: !function utils.process_logical_fallacies +tag: global_mmlu_full_en_humanities_tasks +task: global_mmlu_full_en_logical_fallacies diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_machine_learning.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_machine_learning.yaml new file mode 100644 index 0000000000000000000000000000000000000000..35f496c17ff631774d505d45f703328273667dc3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_machine_learning.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _en_template_yaml +process_docs: !function utils.process_machine_learning +tag: global_mmlu_full_en_stem_tasks +task: global_mmlu_full_en_machine_learning diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_management.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_management.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d8499d9fab2412c309153fe9242eee1c6d8a1005 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_management.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _en_template_yaml +process_docs: !function utils.process_management +tag: global_mmlu_full_en_other_tasks +task: global_mmlu_full_en_management diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_marketing.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_marketing.yaml new file mode 100644 index 0000000000000000000000000000000000000000..05f8f0ec9d19841e37416b5343134df1a1db4c7a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_marketing.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _en_template_yaml +process_docs: !function utils.process_marketing +tag: global_mmlu_full_en_other_tasks +task: global_mmlu_full_en_marketing diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_medical_genetics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_medical_genetics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8f272510f57df4e0741d6dd119df486495322724 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_medical_genetics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _en_template_yaml +process_docs: !function utils.process_medical_genetics +tag: global_mmlu_full_en_other_tasks +task: global_mmlu_full_en_medical_genetics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_miscellaneous.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_miscellaneous.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a72fad223f99cbc4199457fb83ed0de2eef5afa8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_miscellaneous.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _en_template_yaml +process_docs: !function utils.process_miscellaneous +tag: global_mmlu_full_en_other_tasks +task: global_mmlu_full_en_miscellaneous diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_moral_disputes.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_moral_disputes.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2504abeb6733136755801cb30af0fdf1afec59ef --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_moral_disputes.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _en_template_yaml +process_docs: !function utils.process_moral_disputes +tag: global_mmlu_full_en_humanities_tasks +task: global_mmlu_full_en_moral_disputes diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_moral_scenarios.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_moral_scenarios.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4ae4c37a57d082a982c72eeb74515ef88af811e5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_moral_scenarios.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _en_template_yaml +process_docs: !function utils.process_moral_scenarios +tag: global_mmlu_full_en_humanities_tasks +task: global_mmlu_full_en_moral_scenarios diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_nutrition.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_nutrition.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b5364f69d43ca4867795ee3b1b7c70371adcbaee --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_nutrition.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _en_template_yaml +process_docs: !function utils.process_nutrition +tag: global_mmlu_full_en_other_tasks +task: global_mmlu_full_en_nutrition diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_philosophy.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_philosophy.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6e68d7e7223b19e2d07bd3b9d053630069ca7b19 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_philosophy.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _en_template_yaml +process_docs: !function utils.process_philosophy +tag: global_mmlu_full_en_humanities_tasks +task: global_mmlu_full_en_philosophy diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_prehistory.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_prehistory.yaml new file mode 100644 index 0000000000000000000000000000000000000000..72e93368a798333b1e43ad1734852f5123a3ea85 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_prehistory.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _en_template_yaml +process_docs: !function utils.process_prehistory +tag: global_mmlu_full_en_humanities_tasks +task: global_mmlu_full_en_prehistory diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_professional_accounting.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_professional_accounting.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cdb66ead8c7fb76e51cc125d10cb28fdae41e53f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_professional_accounting.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _en_template_yaml +process_docs: !function utils.process_professional_accounting +tag: global_mmlu_full_en_other_tasks +task: global_mmlu_full_en_professional_accounting diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_professional_law.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_professional_law.yaml new file mode 100644 index 0000000000000000000000000000000000000000..67120278088749f0a2ac32deda4580d6eab344b5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_professional_law.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _en_template_yaml +process_docs: !function utils.process_professional_law +tag: global_mmlu_full_en_humanities_tasks +task: global_mmlu_full_en_professional_law diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_professional_medicine.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_professional_medicine.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ffbcb29b38598cd3b796487f8b3f20175c303662 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_professional_medicine.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _en_template_yaml +process_docs: !function utils.process_professional_medicine +tag: global_mmlu_full_en_other_tasks +task: global_mmlu_full_en_professional_medicine diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_professional_psychology.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_professional_psychology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1abea59b5d1bcb4c784e50f2eaf10eff3704c4d8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_professional_psychology.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _en_template_yaml +process_docs: !function utils.process_professional_psychology +tag: global_mmlu_full_en_social_sciences_tasks +task: global_mmlu_full_en_professional_psychology diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_public_relations.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_public_relations.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9df4f49189f72f29dc8da457b85ccfd792824f1e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_public_relations.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _en_template_yaml +process_docs: !function utils.process_public_relations +tag: global_mmlu_full_en_social_sciences_tasks +task: global_mmlu_full_en_public_relations diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_security_studies.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_security_studies.yaml new file mode 100644 index 0000000000000000000000000000000000000000..addb6934d82bc1c332e1439f55dfd04ad176ea56 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_security_studies.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _en_template_yaml +process_docs: !function utils.process_security_studies +tag: global_mmlu_full_en_social_sciences_tasks +task: global_mmlu_full_en_security_studies diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_sociology.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_sociology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a198cb84cbb896d6d0d2217dcafdb62d500939f3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_sociology.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _en_template_yaml +process_docs: !function utils.process_sociology +tag: global_mmlu_full_en_social_sciences_tasks +task: global_mmlu_full_en_sociology diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_us_foreign_policy.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_us_foreign_policy.yaml new file mode 100644 index 0000000000000000000000000000000000000000..047b61e00268e8b931c64b6de23f5ad1a77b24b1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_us_foreign_policy.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _en_template_yaml +process_docs: !function utils.process_us_foreign_policy +tag: global_mmlu_full_en_social_sciences_tasks +task: global_mmlu_full_en_us_foreign_policy diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_virology.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_virology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bb74fefdd32dab13a2b4d1e1fe7907e491a6377e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_virology.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _en_template_yaml +process_docs: !function utils.process_virology +tag: global_mmlu_full_en_other_tasks +task: global_mmlu_full_en_virology diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_world_religions.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_world_religions.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2c453bf7480f8ccd4aeca4fbd58243151872927f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/global_mmlu_full_en_world_religions.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _en_template_yaml +process_docs: !function utils.process_world_religions +tag: global_mmlu_full_en_humanities_tasks +task: global_mmlu_full_en_world_religions diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/utils.py b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..7df72cb061f0fecba46e15ff9b57552817979afb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/en/utils.py @@ -0,0 +1,73 @@ +from functools import partial + + +SUBJECTS = [ + "abstract_algebra", + "anatomy", + "astronomy", + "business_ethics", + "clinical_knowledge", + "college_biology", + "college_chemistry", + "college_computer_science", + "college_mathematics", + "college_medicine", + "college_physics", + "computer_security", + "conceptual_physics", + "econometrics", + "electrical_engineering", + "elementary_mathematics", + "formal_logic", + "global_facts", + "high_school_biology", + "high_school_chemistry", + "high_school_computer_science", + "high_school_european_history", + "high_school_geography", + "high_school_government_and_politics", + "high_school_macroeconomics", + "high_school_mathematics", + "high_school_microeconomics", + "high_school_physics", + "high_school_psychology", + "high_school_statistics", + "high_school_us_history", + "high_school_world_history", + "human_aging", + "human_sexuality", + "international_law", + "jurisprudence", + "logical_fallacies", + "machine_learning", + "management", + "marketing", + "medical_genetics", + "miscellaneous", + "moral_disputes", + "moral_scenarios", + "nutrition", + "philosophy", + "prehistory", + "professional_accounting", + "professional_law", + "professional_medicine", + "professional_psychology", + "public_relations", + "security_studies", + "sociology", + "us_foreign_policy", + "virology", + "world_religions", +] + + +def process_docs(dataset, subject): + return dataset.filter(lambda x: x["subject"] == subject) + + +process_functions = { + f"process_{subject}": partial(process_docs, subject=subject) for subject in SUBJECTS +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/_es_template_yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/_es_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..443af17cc8fc3a3e811f9bb4daec0dda1f16c660 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/_es_template_yaml @@ -0,0 +1,16 @@ +dataset_path: CohereForAI/Global-MMLU +dataset_name: es +test_split: test +fewshot_split: dev +fewshot_config: + sampler: first_n +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\nD. {{option_d}}\nAnswer:" +doc_to_choice: ["A", "B", "C", "D"] +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/_global_mmlu_full_es.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/_global_mmlu_full_es.yaml new file mode 100644 index 0000000000000000000000000000000000000000..13d2eccf3a4deaac71e24bd06c35efb2b4d8061a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/_global_mmlu_full_es.yaml @@ -0,0 +1,11 @@ +group: global_mmlu_full_es +task: + - global_mmlu_full_es_stem + - global_mmlu_full_es_other + - global_mmlu_full_es_social_sciences + - global_mmlu_full_es_humanities +aggregate_metric_list: + - metric: acc + weight_by_size: True +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/_global_mmlu_full_es_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/_global_mmlu_full_es_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bda6944e1b855deb5296a730cfdd9836f8331ed0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/_global_mmlu_full_es_humanities.yaml @@ -0,0 +1,8 @@ +group: global_mmlu_full_es_humanities +task: + - global_mmlu_full_es_humanities_tasks +aggregate_metric_list: + - metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/_global_mmlu_full_es_other.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/_global_mmlu_full_es_other.yaml new file mode 100644 index 0000000000000000000000000000000000000000..610366ef7d1fe1e00c7ce77583cfccb0869c8cad --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/_global_mmlu_full_es_other.yaml @@ -0,0 +1,8 @@ +group: global_mmlu_full_es_other +task: + - global_mmlu_full_es_other_tasks +aggregate_metric_list: + - metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/_global_mmlu_full_es_social_sciences.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/_global_mmlu_full_es_social_sciences.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0094869035f3e24f390107ebe9dce164594344fd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/_global_mmlu_full_es_social_sciences.yaml @@ -0,0 +1,8 @@ +group: global_mmlu_full_es_social_sciences +task: + - global_mmlu_full_es_social_sciences_tasks +aggregate_metric_list: + - metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/_global_mmlu_full_es_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/_global_mmlu_full_es_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..483a8fd6fd14575ab2bc2a87ae4767d6aa194480 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/_global_mmlu_full_es_stem.yaml @@ -0,0 +1,8 @@ +group: global_mmlu_full_es_stem +task: + - global_mmlu_full_es_stem_tasks +aggregate_metric_list: + - metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_abstract_algebra.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_abstract_algebra.yaml new file mode 100644 index 0000000000000000000000000000000000000000..02fb72001e3fc6eea4e049b1d8504a680c51f97a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_abstract_algebra.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _es_template_yaml +process_docs: !function utils.process_abstract_algebra +tag: global_mmlu_full_es_stem_tasks +task: global_mmlu_full_es_abstract_algebra diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_anatomy.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_anatomy.yaml new file mode 100644 index 0000000000000000000000000000000000000000..40f05e7beda0d8f224af0780790cee21fe6a9eb5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_anatomy.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _es_template_yaml +process_docs: !function utils.process_anatomy +tag: global_mmlu_full_es_stem_tasks +task: global_mmlu_full_es_anatomy diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_astronomy.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_astronomy.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fb688c13ca5c63973dce474461174eee2cd464fe --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_astronomy.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _es_template_yaml +process_docs: !function utils.process_astronomy +tag: global_mmlu_full_es_stem_tasks +task: global_mmlu_full_es_astronomy diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_business_ethics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_business_ethics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..aab858f1d2e638059b0d34a48ef37552164dc79e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_business_ethics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _es_template_yaml +process_docs: !function utils.process_business_ethics +tag: global_mmlu_full_es_other_tasks +task: global_mmlu_full_es_business_ethics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_clinical_knowledge.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_clinical_knowledge.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a3483f8d08bc1e764a357d1afd62637f62e4059e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_clinical_knowledge.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _es_template_yaml +process_docs: !function utils.process_clinical_knowledge +tag: global_mmlu_full_es_other_tasks +task: global_mmlu_full_es_clinical_knowledge diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_college_biology.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_college_biology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..36658ab6c40881f48980870f396ca8c32a1f5718 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_college_biology.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _es_template_yaml +process_docs: !function utils.process_college_biology +tag: global_mmlu_full_es_stem_tasks +task: global_mmlu_full_es_college_biology diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_college_chemistry.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_college_chemistry.yaml new file mode 100644 index 0000000000000000000000000000000000000000..47a4744497e564e5d3a199b2b9e9d2ee45c8a75e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_college_chemistry.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _es_template_yaml +process_docs: !function utils.process_college_chemistry +tag: global_mmlu_full_es_stem_tasks +task: global_mmlu_full_es_college_chemistry diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_college_computer_science.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_college_computer_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4154324e4f77a2143aacbf7d76a417a05ce4a303 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_college_computer_science.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _es_template_yaml +process_docs: !function utils.process_college_computer_science +tag: global_mmlu_full_es_stem_tasks +task: global_mmlu_full_es_college_computer_science diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_college_mathematics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_college_mathematics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..85bc62614565b729aee3a7c875cfc5da5980f7f2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_college_mathematics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _es_template_yaml +process_docs: !function utils.process_college_mathematics +tag: global_mmlu_full_es_stem_tasks +task: global_mmlu_full_es_college_mathematics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_college_medicine.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_college_medicine.yaml new file mode 100644 index 0000000000000000000000000000000000000000..40e8d1291ec14e09a25bad922a56ca751c809415 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_college_medicine.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _es_template_yaml +process_docs: !function utils.process_college_medicine +tag: global_mmlu_full_es_other_tasks +task: global_mmlu_full_es_college_medicine diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_college_physics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_college_physics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7ebc5e950300544903aab2d4c4cca3b15a60ee24 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_college_physics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _es_template_yaml +process_docs: !function utils.process_college_physics +tag: global_mmlu_full_es_stem_tasks +task: global_mmlu_full_es_college_physics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_computer_security.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_computer_security.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b586eb2b80f65b2d18c5d4cde05ff9f95849eebe --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_computer_security.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _es_template_yaml +process_docs: !function utils.process_computer_security +tag: global_mmlu_full_es_stem_tasks +task: global_mmlu_full_es_computer_security diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_conceptual_physics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_conceptual_physics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4186cec69574e7a8aa916c31af316410a052ae10 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_conceptual_physics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _es_template_yaml +process_docs: !function utils.process_conceptual_physics +tag: global_mmlu_full_es_stem_tasks +task: global_mmlu_full_es_conceptual_physics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_econometrics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_econometrics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3d61c8f9dc39a02b7fdc697395e8a049ac317281 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_econometrics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _es_template_yaml +process_docs: !function utils.process_econometrics +tag: global_mmlu_full_es_social_sciences_tasks +task: global_mmlu_full_es_econometrics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_electrical_engineering.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_electrical_engineering.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1a454d79664c52ab09cb275eb45808f2d4dc9d97 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_electrical_engineering.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _es_template_yaml +process_docs: !function utils.process_electrical_engineering +tag: global_mmlu_full_es_stem_tasks +task: global_mmlu_full_es_electrical_engineering diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_elementary_mathematics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_elementary_mathematics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..772436e6b533d8dab172388e5b0bc87584fa203e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_elementary_mathematics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _es_template_yaml +process_docs: !function utils.process_elementary_mathematics +tag: global_mmlu_full_es_stem_tasks +task: global_mmlu_full_es_elementary_mathematics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_formal_logic.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_formal_logic.yaml new file mode 100644 index 0000000000000000000000000000000000000000..da6223fe38232995ac5409e98993c2d86faa565c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_formal_logic.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _es_template_yaml +process_docs: !function utils.process_formal_logic +tag: global_mmlu_full_es_humanities_tasks +task: global_mmlu_full_es_formal_logic diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_global_facts.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_global_facts.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ae3b5912bb5e10d4a70ed5c3f73f6d180498093e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_global_facts.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _es_template_yaml +process_docs: !function utils.process_global_facts +tag: global_mmlu_full_es_other_tasks +task: global_mmlu_full_es_global_facts diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_high_school_biology.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_high_school_biology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..79a72140de89f3adeaaaf8d13b2e8502e623faf7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_high_school_biology.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _es_template_yaml +process_docs: !function utils.process_high_school_biology +tag: global_mmlu_full_es_stem_tasks +task: global_mmlu_full_es_high_school_biology diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_high_school_chemistry.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_high_school_chemistry.yaml new file mode 100644 index 0000000000000000000000000000000000000000..27ba757082d345c4c0973a7039ce340951d89b68 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_high_school_chemistry.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _es_template_yaml +process_docs: !function utils.process_high_school_chemistry +tag: global_mmlu_full_es_stem_tasks +task: global_mmlu_full_es_high_school_chemistry diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_high_school_computer_science.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_high_school_computer_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..72ad45057bd8fa98b7f226790d9699943b7254d4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_high_school_computer_science.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _es_template_yaml +process_docs: !function utils.process_high_school_computer_science +tag: global_mmlu_full_es_stem_tasks +task: global_mmlu_full_es_high_school_computer_science diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_high_school_european_history.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_high_school_european_history.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2cec9d5f6b4ee527ad0ed01716de92b93c6e9628 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_high_school_european_history.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _es_template_yaml +process_docs: !function utils.process_high_school_european_history +tag: global_mmlu_full_es_humanities_tasks +task: global_mmlu_full_es_high_school_european_history diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_high_school_government_and_politics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_high_school_government_and_politics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b3f1031993168ec87fdf05bb9523ada6ae771935 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_high_school_government_and_politics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _es_template_yaml +process_docs: !function utils.process_high_school_government_and_politics +tag: global_mmlu_full_es_social_sciences_tasks +task: global_mmlu_full_es_high_school_government_and_politics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_high_school_macroeconomics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_high_school_macroeconomics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d555129a40a64ff90972be9ac8d8a1cdaa6f67ed --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_high_school_macroeconomics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _es_template_yaml +process_docs: !function utils.process_high_school_macroeconomics +tag: global_mmlu_full_es_social_sciences_tasks +task: global_mmlu_full_es_high_school_macroeconomics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_high_school_mathematics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_high_school_mathematics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a1216336182fb78fea17a50a8ec1e70f7ff234e3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_high_school_mathematics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _es_template_yaml +process_docs: !function utils.process_high_school_mathematics +tag: global_mmlu_full_es_stem_tasks +task: global_mmlu_full_es_high_school_mathematics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_high_school_physics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_high_school_physics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fb83ad1e44df9554f4b956a678fa307b4c2e1a14 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_high_school_physics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _es_template_yaml +process_docs: !function utils.process_high_school_physics +tag: global_mmlu_full_es_stem_tasks +task: global_mmlu_full_es_high_school_physics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_high_school_psychology.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_high_school_psychology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4bcd53e48b13aa2d8b6357809ad142f415ad01e9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_high_school_psychology.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _es_template_yaml +process_docs: !function utils.process_high_school_psychology +tag: global_mmlu_full_es_social_sciences_tasks +task: global_mmlu_full_es_high_school_psychology diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_high_school_us_history.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_high_school_us_history.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d54acd65afd3a0cdac8316f6fc3e4b0ccaa43b9b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_high_school_us_history.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _es_template_yaml +process_docs: !function utils.process_high_school_us_history +tag: global_mmlu_full_es_humanities_tasks +task: global_mmlu_full_es_high_school_us_history diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_human_aging.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_human_aging.yaml new file mode 100644 index 0000000000000000000000000000000000000000..47bd89009fb74d4c468b6deec54322053c2c83c2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_human_aging.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _es_template_yaml +process_docs: !function utils.process_human_aging +tag: global_mmlu_full_es_other_tasks +task: global_mmlu_full_es_human_aging diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_human_sexuality.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_human_sexuality.yaml new file mode 100644 index 0000000000000000000000000000000000000000..29925c347fe23503b67a4b1738cc8e909f81078a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_human_sexuality.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _es_template_yaml +process_docs: !function utils.process_human_sexuality +tag: global_mmlu_full_es_social_sciences_tasks +task: global_mmlu_full_es_human_sexuality diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_international_law.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_international_law.yaml new file mode 100644 index 0000000000000000000000000000000000000000..abe4ef94f0599c875caf5e5dec3b899042dd30e1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_international_law.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _es_template_yaml +process_docs: !function utils.process_international_law +tag: global_mmlu_full_es_humanities_tasks +task: global_mmlu_full_es_international_law diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_jurisprudence.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_jurisprudence.yaml new file mode 100644 index 0000000000000000000000000000000000000000..751878fe1d333017230e639fda582d65ec304344 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_jurisprudence.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _es_template_yaml +process_docs: !function utils.process_jurisprudence +tag: global_mmlu_full_es_humanities_tasks +task: global_mmlu_full_es_jurisprudence diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_logical_fallacies.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_logical_fallacies.yaml new file mode 100644 index 0000000000000000000000000000000000000000..55233f7f20811e1c95650dee1e92862a50b34f54 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/es/global_mmlu_full_es_logical_fallacies.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _es_template_yaml +process_docs: !function utils.process_logical_fallacies +tag: global_mmlu_full_es_humanities_tasks +task: global_mmlu_full_es_logical_fallacies diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/fa/_global_mmlu_full_fa.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/fa/_global_mmlu_full_fa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..282664e59fff56be7b06eee62198b31a78f2899c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/fa/_global_mmlu_full_fa.yaml @@ -0,0 +1,11 @@ +group: global_mmlu_full_fa +task: + - global_mmlu_full_fa_stem + - global_mmlu_full_fa_other + - global_mmlu_full_fa_social_sciences + - global_mmlu_full_fa_humanities +aggregate_metric_list: + - metric: acc + weight_by_size: True +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/fa/_global_mmlu_full_fa_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/fa/_global_mmlu_full_fa_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f36ecea5f2b4ff98d998328d66b0087deeb33849 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/fa/_global_mmlu_full_fa_humanities.yaml @@ -0,0 +1,8 @@ +group: global_mmlu_full_fa_humanities +task: + - global_mmlu_full_fa_humanities_tasks +aggregate_metric_list: + - metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/fa/_global_mmlu_full_fa_other.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/fa/_global_mmlu_full_fa_other.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dd57bb86bb118224258068151bcfb77f91afa321 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/fa/_global_mmlu_full_fa_other.yaml @@ -0,0 +1,8 @@ +group: global_mmlu_full_fa_other +task: + - global_mmlu_full_fa_other_tasks +aggregate_metric_list: + - metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/fa/_global_mmlu_full_fa_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/fa/_global_mmlu_full_fa_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5bf2eb01368a6c86ce0fad30e04ee321dce8cc8f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/fa/_global_mmlu_full_fa_stem.yaml @@ -0,0 +1,8 @@ +group: global_mmlu_full_fa_stem +task: + - global_mmlu_full_fa_stem_tasks +aggregate_metric_list: + - metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/fa/global_mmlu_full_fa_anatomy.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/fa/global_mmlu_full_fa_anatomy.yaml new file mode 100644 index 0000000000000000000000000000000000000000..317705c92aa860af17b86d55acdddc4400d5b1a5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/fa/global_mmlu_full_fa_anatomy.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _fa_template_yaml +process_docs: !function utils.process_anatomy +tag: global_mmlu_full_fa_stem_tasks +task: global_mmlu_full_fa_anatomy diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/fa/global_mmlu_full_fa_astronomy.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/fa/global_mmlu_full_fa_astronomy.yaml new file mode 100644 index 0000000000000000000000000000000000000000..45475964110cab623ae40dc71f5e136e9db32fcf --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/fa/global_mmlu_full_fa_astronomy.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _fa_template_yaml +process_docs: !function utils.process_astronomy +tag: global_mmlu_full_fa_stem_tasks +task: global_mmlu_full_fa_astronomy