Add files using upload-large-folder tool
Browse filesThis view is limited to 50 files because it contains too many changes. See raw diff
- lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_InfluenceFromChina_light.yaml +23 -0
- lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_InfluenceFromRome_light.yaml +23 -0
- lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Jordan_light.yaml +23 -0
- lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Libya_light.yaml +23 -0
- lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Mauritania_light.yaml +23 -0
- lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Mesopotamia_civilization_light.yaml +23 -0
- lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Morocco_light.yaml +23 -0
- lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Oman_light.yaml +23 -0
- lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Palestine_light.yaml +23 -0
- lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Saudi_Arabia_light.yaml +23 -0
- lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Sudan_light.yaml +23 -0
- lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Syria_light.yaml +23 -0
- lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Tunisia_light.yaml +23 -0
- lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_United_Arab_Emirates_light.yaml +23 -0
- lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Yemen_light.yaml +23 -0
- lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_communication_light.yaml +23 -0
- lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_computer_and_phone_light.yaml +23 -0
- lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_daily_life_light.yaml +23 -0
- lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_entertainment_light.yaml +23 -0
- lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_light.yaml +70 -0
- lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/utils.py +16 -0
- lm-evaluation-harness/lm_eval/tasks/arabicmmlu/_arabicmmlu_social_science.yaml +9 -0
- lm-evaluation-harness/lm_eval/tasks/arabicmmlu/_arabicmmlu_stem.yaml +9 -0
- lm-evaluation-harness/lm_eval/tasks/arabicmmlu/_default_arabicmmlu_template_yaml +15 -0
- lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_arabic_language_grammar.yaml +5 -0
- lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_arabic_language_primary_school.yaml +5 -0
- lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_biology_high_school.yaml +5 -0
- lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_civics_high_school.yaml +5 -0
- lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_civics_middle_school.yaml +5 -0
- lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_computer_science_middle_school.yaml +5 -0
- lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_computer_science_primary_school.yaml +5 -0
- lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_computer_science_university.yaml +5 -0
- lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_driving_test.yaml +5 -0
- lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_economics_high_school.yaml +5 -0
- lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_economics_middle_school.yaml +5 -0
- lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_economics_university.yaml +5 -0
- lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_general_knowledge.yaml +5 -0
- lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_general_knowledge_middle_school.yaml +5 -0
- lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_general_knowledge_primary_school.yaml +5 -0
- lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_geography_high_school.yaml +5 -0
- lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_geography_middle_school.yaml +5 -0
- lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_geography_primary_school.yaml +5 -0
- lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_history_high_school.yaml +5 -0
- lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_history_middle_school.yaml +5 -0
- lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_history_primary_school.yaml +5 -0
- lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_islamic_studies_high_school.yaml +5 -0
- lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_islamic_studies_primary_school.yaml +5 -0
- lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_law_professional.yaml +5 -0
- lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_management_university.yaml +5 -0
- lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_math_primary_school.yaml +5 -0
lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_InfluenceFromChina_light.yaml
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
task: arabic_leaderboard_acva_InfluenceFromChina_light
|
| 2 |
+
dataset_path: arcee-globe/ACVA-10percent
|
| 3 |
+
dataset_name: InfluenceFromChina
|
| 4 |
+
output_type: multiple_choice
|
| 5 |
+
training_split: null
|
| 6 |
+
validation_split: validation
|
| 7 |
+
test_split: test
|
| 8 |
+
process_docs: !function utils.process_docs
|
| 9 |
+
doc_to_text: "{{query}}"
|
| 10 |
+
doc_to_target: "{{gold}}"
|
| 11 |
+
doc_to_choice: "choices"
|
| 12 |
+
fewshot_split: validation
|
| 13 |
+
fewshot_config:
|
| 14 |
+
sampler: first_n
|
| 15 |
+
metric_list:
|
| 16 |
+
- metric: acc
|
| 17 |
+
aggregation: mean
|
| 18 |
+
higher_is_better: true
|
| 19 |
+
- metric: acc_norm
|
| 20 |
+
aggregation: mean
|
| 21 |
+
higher_is_better: true
|
| 22 |
+
metadata:
|
| 23 |
+
version: 1.0
|
lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_InfluenceFromRome_light.yaml
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
task: arabic_leaderboard_acva_InfluenceFromRome_light
|
| 2 |
+
dataset_path: arcee-globe/ACVA-10percent
|
| 3 |
+
dataset_name: InfluenceFromRome
|
| 4 |
+
output_type: multiple_choice
|
| 5 |
+
training_split: null
|
| 6 |
+
validation_split: validation
|
| 7 |
+
test_split: test
|
| 8 |
+
process_docs: !function utils.process_docs
|
| 9 |
+
doc_to_text: "{{query}}"
|
| 10 |
+
doc_to_target: "{{gold}}"
|
| 11 |
+
doc_to_choice: "choices"
|
| 12 |
+
fewshot_split: validation
|
| 13 |
+
fewshot_config:
|
| 14 |
+
sampler: first_n
|
| 15 |
+
metric_list:
|
| 16 |
+
- metric: acc
|
| 17 |
+
aggregation: mean
|
| 18 |
+
higher_is_better: true
|
| 19 |
+
- metric: acc_norm
|
| 20 |
+
aggregation: mean
|
| 21 |
+
higher_is_better: true
|
| 22 |
+
metadata:
|
| 23 |
+
version: 1.0
|
lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Jordan_light.yaml
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
task: arabic_leaderboard_acva_Jordan_light
|
| 2 |
+
dataset_path: arcee-globe/ACVA-10percent
|
| 3 |
+
dataset_name: Jordan
|
| 4 |
+
output_type: multiple_choice
|
| 5 |
+
training_split: null
|
| 6 |
+
validation_split: validation
|
| 7 |
+
test_split: test
|
| 8 |
+
process_docs: !function utils.process_docs
|
| 9 |
+
doc_to_text: "{{query}}"
|
| 10 |
+
doc_to_target: "{{gold}}"
|
| 11 |
+
doc_to_choice: "choices"
|
| 12 |
+
fewshot_split: validation
|
| 13 |
+
fewshot_config:
|
| 14 |
+
sampler: first_n
|
| 15 |
+
metric_list:
|
| 16 |
+
- metric: acc
|
| 17 |
+
aggregation: mean
|
| 18 |
+
higher_is_better: true
|
| 19 |
+
- metric: acc_norm
|
| 20 |
+
aggregation: mean
|
| 21 |
+
higher_is_better: true
|
| 22 |
+
metadata:
|
| 23 |
+
version: 1.0
|
lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Libya_light.yaml
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
task: arabic_leaderboard_acva_Libya_light
|
| 2 |
+
dataset_path: arcee-globe/ACVA-10percent
|
| 3 |
+
dataset_name: Libya
|
| 4 |
+
output_type: multiple_choice
|
| 5 |
+
training_split: null
|
| 6 |
+
validation_split: validation
|
| 7 |
+
test_split: test
|
| 8 |
+
process_docs: !function utils.process_docs
|
| 9 |
+
doc_to_text: "{{query}}"
|
| 10 |
+
doc_to_target: "{{gold}}"
|
| 11 |
+
doc_to_choice: "choices"
|
| 12 |
+
fewshot_split: validation
|
| 13 |
+
fewshot_config:
|
| 14 |
+
sampler: first_n
|
| 15 |
+
metric_list:
|
| 16 |
+
- metric: acc
|
| 17 |
+
aggregation: mean
|
| 18 |
+
higher_is_better: true
|
| 19 |
+
- metric: acc_norm
|
| 20 |
+
aggregation: mean
|
| 21 |
+
higher_is_better: true
|
| 22 |
+
metadata:
|
| 23 |
+
version: 1.0
|
lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Mauritania_light.yaml
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
task: arabic_leaderboard_acva_Mauritania_light
|
| 2 |
+
dataset_path: arcee-globe/ACVA-10percent
|
| 3 |
+
dataset_name: Mauritania
|
| 4 |
+
output_type: multiple_choice
|
| 5 |
+
training_split: null
|
| 6 |
+
validation_split: validation
|
| 7 |
+
test_split: test
|
| 8 |
+
process_docs: !function utils.process_docs
|
| 9 |
+
doc_to_text: "{{query}}"
|
| 10 |
+
doc_to_target: "{{gold}}"
|
| 11 |
+
doc_to_choice: "choices"
|
| 12 |
+
fewshot_split: validation
|
| 13 |
+
fewshot_config:
|
| 14 |
+
sampler: first_n
|
| 15 |
+
metric_list:
|
| 16 |
+
- metric: acc
|
| 17 |
+
aggregation: mean
|
| 18 |
+
higher_is_better: true
|
| 19 |
+
- metric: acc_norm
|
| 20 |
+
aggregation: mean
|
| 21 |
+
higher_is_better: true
|
| 22 |
+
metadata:
|
| 23 |
+
version: 1.0
|
lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Mesopotamia_civilization_light.yaml
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
task: arabic_leaderboard_acva_Mesopotamia_civilization_light
|
| 2 |
+
dataset_path: arcee-globe/ACVA-10percent
|
| 3 |
+
dataset_name: Mesopotamia_civilization
|
| 4 |
+
output_type: multiple_choice
|
| 5 |
+
training_split: null
|
| 6 |
+
validation_split: validation
|
| 7 |
+
test_split: test
|
| 8 |
+
process_docs: !function utils.process_docs
|
| 9 |
+
doc_to_text: "{{query}}"
|
| 10 |
+
doc_to_target: "{{gold}}"
|
| 11 |
+
doc_to_choice: "choices"
|
| 12 |
+
fewshot_split: validation
|
| 13 |
+
fewshot_config:
|
| 14 |
+
sampler: first_n
|
| 15 |
+
metric_list:
|
| 16 |
+
- metric: acc
|
| 17 |
+
aggregation: mean
|
| 18 |
+
higher_is_better: true
|
| 19 |
+
- metric: acc_norm
|
| 20 |
+
aggregation: mean
|
| 21 |
+
higher_is_better: true
|
| 22 |
+
metadata:
|
| 23 |
+
version: 1.0
|
lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Morocco_light.yaml
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
task: arabic_leaderboard_acva_Morocco_light
|
| 2 |
+
dataset_path: arcee-globe/ACVA-10percent
|
| 3 |
+
dataset_name: Morocco
|
| 4 |
+
output_type: multiple_choice
|
| 5 |
+
training_split: null
|
| 6 |
+
validation_split: validation
|
| 7 |
+
test_split: test
|
| 8 |
+
process_docs: !function utils.process_docs
|
| 9 |
+
doc_to_text: "{{query}}"
|
| 10 |
+
doc_to_target: "{{gold}}"
|
| 11 |
+
doc_to_choice: "choices"
|
| 12 |
+
fewshot_split: validation
|
| 13 |
+
fewshot_config:
|
| 14 |
+
sampler: first_n
|
| 15 |
+
metric_list:
|
| 16 |
+
- metric: acc
|
| 17 |
+
aggregation: mean
|
| 18 |
+
higher_is_better: true
|
| 19 |
+
- metric: acc_norm
|
| 20 |
+
aggregation: mean
|
| 21 |
+
higher_is_better: true
|
| 22 |
+
metadata:
|
| 23 |
+
version: 1.0
|
lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Oman_light.yaml
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
task: arabic_leaderboard_acva_Oman_light
|
| 2 |
+
dataset_path: arcee-globe/ACVA-10percent
|
| 3 |
+
dataset_name: Oman
|
| 4 |
+
output_type: multiple_choice
|
| 5 |
+
training_split: null
|
| 6 |
+
validation_split: validation
|
| 7 |
+
test_split: test
|
| 8 |
+
process_docs: !function utils.process_docs
|
| 9 |
+
doc_to_text: "{{query}}"
|
| 10 |
+
doc_to_target: "{{gold}}"
|
| 11 |
+
doc_to_choice: "choices"
|
| 12 |
+
fewshot_split: validation
|
| 13 |
+
fewshot_config:
|
| 14 |
+
sampler: first_n
|
| 15 |
+
metric_list:
|
| 16 |
+
- metric: acc
|
| 17 |
+
aggregation: mean
|
| 18 |
+
higher_is_better: true
|
| 19 |
+
- metric: acc_norm
|
| 20 |
+
aggregation: mean
|
| 21 |
+
higher_is_better: true
|
| 22 |
+
metadata:
|
| 23 |
+
version: 1.0
|
lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Palestine_light.yaml
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
task: arabic_leaderboard_acva_Palestine_light
|
| 2 |
+
dataset_path: arcee-globe/ACVA-10percent
|
| 3 |
+
dataset_name: Palestine
|
| 4 |
+
output_type: multiple_choice
|
| 5 |
+
training_split: null
|
| 6 |
+
validation_split: validation
|
| 7 |
+
test_split: test
|
| 8 |
+
process_docs: !function utils.process_docs
|
| 9 |
+
doc_to_text: "{{query}}"
|
| 10 |
+
doc_to_target: "{{gold}}"
|
| 11 |
+
doc_to_choice: "choices"
|
| 12 |
+
fewshot_split: validation
|
| 13 |
+
fewshot_config:
|
| 14 |
+
sampler: first_n
|
| 15 |
+
metric_list:
|
| 16 |
+
- metric: acc
|
| 17 |
+
aggregation: mean
|
| 18 |
+
higher_is_better: true
|
| 19 |
+
- metric: acc_norm
|
| 20 |
+
aggregation: mean
|
| 21 |
+
higher_is_better: true
|
| 22 |
+
metadata:
|
| 23 |
+
version: 1.0
|
lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Saudi_Arabia_light.yaml
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
task: arabic_leaderboard_acva_Saudi_Arabia_light
|
| 2 |
+
dataset_path: arcee-globe/ACVA-10percent
|
| 3 |
+
dataset_name: Saudi_Arabia
|
| 4 |
+
output_type: multiple_choice
|
| 5 |
+
training_split: null
|
| 6 |
+
validation_split: validation
|
| 7 |
+
test_split: test
|
| 8 |
+
process_docs: !function utils.process_docs
|
| 9 |
+
doc_to_text: "{{query}}"
|
| 10 |
+
doc_to_target: "{{gold}}"
|
| 11 |
+
doc_to_choice: "choices"
|
| 12 |
+
fewshot_split: validation
|
| 13 |
+
fewshot_config:
|
| 14 |
+
sampler: first_n
|
| 15 |
+
metric_list:
|
| 16 |
+
- metric: acc
|
| 17 |
+
aggregation: mean
|
| 18 |
+
higher_is_better: true
|
| 19 |
+
- metric: acc_norm
|
| 20 |
+
aggregation: mean
|
| 21 |
+
higher_is_better: true
|
| 22 |
+
metadata:
|
| 23 |
+
version: 1.0
|
lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Sudan_light.yaml
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
task: arabic_leaderboard_acva_Sudan_light
|
| 2 |
+
dataset_path: arcee-globe/ACVA-10percent
|
| 3 |
+
dataset_name: Sudan
|
| 4 |
+
output_type: multiple_choice
|
| 5 |
+
training_split: null
|
| 6 |
+
validation_split: validation
|
| 7 |
+
test_split: test
|
| 8 |
+
process_docs: !function utils.process_docs
|
| 9 |
+
doc_to_text: "{{query}}"
|
| 10 |
+
doc_to_target: "{{gold}}"
|
| 11 |
+
doc_to_choice: "choices"
|
| 12 |
+
fewshot_split: validation
|
| 13 |
+
fewshot_config:
|
| 14 |
+
sampler: first_n
|
| 15 |
+
metric_list:
|
| 16 |
+
- metric: acc
|
| 17 |
+
aggregation: mean
|
| 18 |
+
higher_is_better: true
|
| 19 |
+
- metric: acc_norm
|
| 20 |
+
aggregation: mean
|
| 21 |
+
higher_is_better: true
|
| 22 |
+
metadata:
|
| 23 |
+
version: 1.0
|
lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Syria_light.yaml
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
task: arabic_leaderboard_acva_Syria_light
|
| 2 |
+
dataset_path: arcee-globe/ACVA-10percent
|
| 3 |
+
dataset_name: Syria
|
| 4 |
+
output_type: multiple_choice
|
| 5 |
+
training_split: null
|
| 6 |
+
validation_split: validation
|
| 7 |
+
test_split: test
|
| 8 |
+
process_docs: !function utils.process_docs
|
| 9 |
+
doc_to_text: "{{query}}"
|
| 10 |
+
doc_to_target: "{{gold}}"
|
| 11 |
+
doc_to_choice: "choices"
|
| 12 |
+
fewshot_split: validation
|
| 13 |
+
fewshot_config:
|
| 14 |
+
sampler: first_n
|
| 15 |
+
metric_list:
|
| 16 |
+
- metric: acc
|
| 17 |
+
aggregation: mean
|
| 18 |
+
higher_is_better: true
|
| 19 |
+
- metric: acc_norm
|
| 20 |
+
aggregation: mean
|
| 21 |
+
higher_is_better: true
|
| 22 |
+
metadata:
|
| 23 |
+
version: 1.0
|
lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Tunisia_light.yaml
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
task: arabic_leaderboard_acva_Tunisia_light
|
| 2 |
+
dataset_path: arcee-globe/ACVA-10percent
|
| 3 |
+
dataset_name: Tunisia
|
| 4 |
+
output_type: multiple_choice
|
| 5 |
+
training_split: null
|
| 6 |
+
validation_split: validation
|
| 7 |
+
test_split: test
|
| 8 |
+
process_docs: !function utils.process_docs
|
| 9 |
+
doc_to_text: "{{query}}"
|
| 10 |
+
doc_to_target: "{{gold}}"
|
| 11 |
+
doc_to_choice: "choices"
|
| 12 |
+
fewshot_split: validation
|
| 13 |
+
fewshot_config:
|
| 14 |
+
sampler: first_n
|
| 15 |
+
metric_list:
|
| 16 |
+
- metric: acc
|
| 17 |
+
aggregation: mean
|
| 18 |
+
higher_is_better: true
|
| 19 |
+
- metric: acc_norm
|
| 20 |
+
aggregation: mean
|
| 21 |
+
higher_is_better: true
|
| 22 |
+
metadata:
|
| 23 |
+
version: 1.0
|
lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_United_Arab_Emirates_light.yaml
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
task: arabic_leaderboard_acva_United_Arab_Emirates_light
|
| 2 |
+
dataset_path: arcee-globe/ACVA-10percent
|
| 3 |
+
dataset_name: United_Arab_Emirates
|
| 4 |
+
output_type: multiple_choice
|
| 5 |
+
training_split: null
|
| 6 |
+
validation_split: validation
|
| 7 |
+
test_split: test
|
| 8 |
+
process_docs: !function utils.process_docs
|
| 9 |
+
doc_to_text: "{{query}}"
|
| 10 |
+
doc_to_target: "{{gold}}"
|
| 11 |
+
doc_to_choice: "choices"
|
| 12 |
+
fewshot_split: validation
|
| 13 |
+
fewshot_config:
|
| 14 |
+
sampler: first_n
|
| 15 |
+
metric_list:
|
| 16 |
+
- metric: acc
|
| 17 |
+
aggregation: mean
|
| 18 |
+
higher_is_better: true
|
| 19 |
+
- metric: acc_norm
|
| 20 |
+
aggregation: mean
|
| 21 |
+
higher_is_better: true
|
| 22 |
+
metadata:
|
| 23 |
+
version: 1.0
|
lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_Yemen_light.yaml
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
task: arabic_leaderboard_acva_Yemen_light
|
| 2 |
+
dataset_path: OALL/ACVA
|
| 3 |
+
dataset_name: Yemen
|
| 4 |
+
output_type: multiple_choice
|
| 5 |
+
training_split: null
|
| 6 |
+
validation_split: validation
|
| 7 |
+
test_split: test
|
| 8 |
+
process_docs: !function utils.process_docs
|
| 9 |
+
doc_to_text: "{{query}}"
|
| 10 |
+
doc_to_target: "{{gold}}"
|
| 11 |
+
doc_to_choice: "choices"
|
| 12 |
+
fewshot_split: validation
|
| 13 |
+
fewshot_config:
|
| 14 |
+
sampler: first_n
|
| 15 |
+
metric_list:
|
| 16 |
+
- metric: acc
|
| 17 |
+
aggregation: mean
|
| 18 |
+
higher_is_better: true
|
| 19 |
+
- metric: acc_norm
|
| 20 |
+
aggregation: mean
|
| 21 |
+
higher_is_better: true
|
| 22 |
+
metadata:
|
| 23 |
+
version: 1.0
|
lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_communication_light.yaml
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
task: arabic_leaderboard_acva_communication_light
|
| 2 |
+
dataset_path: arcee-globe/ACVA-10percent
|
| 3 |
+
dataset_name: communication
|
| 4 |
+
output_type: multiple_choice
|
| 5 |
+
training_split: null
|
| 6 |
+
validation_split: validation
|
| 7 |
+
test_split: test
|
| 8 |
+
process_docs: !function utils.process_docs
|
| 9 |
+
doc_to_text: "{{query}}"
|
| 10 |
+
doc_to_target: "{{gold}}"
|
| 11 |
+
doc_to_choice: "choices"
|
| 12 |
+
fewshot_split: validation
|
| 13 |
+
fewshot_config:
|
| 14 |
+
sampler: first_n
|
| 15 |
+
metric_list:
|
| 16 |
+
- metric: acc
|
| 17 |
+
aggregation: mean
|
| 18 |
+
higher_is_better: true
|
| 19 |
+
- metric: acc_norm
|
| 20 |
+
aggregation: mean
|
| 21 |
+
higher_is_better: true
|
| 22 |
+
metadata:
|
| 23 |
+
version: 1.0
|
lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_computer_and_phone_light.yaml
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
task: arabic_leaderboard_acva_computer_and_phone_light
|
| 2 |
+
dataset_path: arcee-globe/ACVA-10percent
|
| 3 |
+
dataset_name: computer_and_phone
|
| 4 |
+
output_type: multiple_choice
|
| 5 |
+
training_split: null
|
| 6 |
+
validation_split: validation
|
| 7 |
+
test_split: test
|
| 8 |
+
process_docs: !function utils.process_docs
|
| 9 |
+
doc_to_text: "{{query}}"
|
| 10 |
+
doc_to_target: "{{gold}}"
|
| 11 |
+
doc_to_choice: "choices"
|
| 12 |
+
fewshot_split: validation
|
| 13 |
+
fewshot_config:
|
| 14 |
+
sampler: first_n
|
| 15 |
+
metric_list:
|
| 16 |
+
- metric: acc
|
| 17 |
+
aggregation: mean
|
| 18 |
+
higher_is_better: true
|
| 19 |
+
- metric: acc_norm
|
| 20 |
+
aggregation: mean
|
| 21 |
+
higher_is_better: true
|
| 22 |
+
metadata:
|
| 23 |
+
version: 1.0
|
lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_daily_life_light.yaml
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
task: arabic_leaderboard_acva_daily_life_light
|
| 2 |
+
dataset_path: arcee-globe/ACVA-10percent
|
| 3 |
+
dataset_name: daily_life
|
| 4 |
+
output_type: multiple_choice
|
| 5 |
+
training_split: null
|
| 6 |
+
validation_split: validation
|
| 7 |
+
test_split: test
|
| 8 |
+
process_docs: !function utils.process_docs
|
| 9 |
+
doc_to_text: "{{query}}"
|
| 10 |
+
doc_to_target: "{{gold}}"
|
| 11 |
+
doc_to_choice: "choices"
|
| 12 |
+
fewshot_split: validation
|
| 13 |
+
fewshot_config:
|
| 14 |
+
sampler: first_n
|
| 15 |
+
metric_list:
|
| 16 |
+
- metric: acc
|
| 17 |
+
aggregation: mean
|
| 18 |
+
higher_is_better: true
|
| 19 |
+
- metric: acc_norm
|
| 20 |
+
aggregation: mean
|
| 21 |
+
higher_is_better: true
|
| 22 |
+
metadata:
|
| 23 |
+
version: 1.0
|
lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_entertainment_light.yaml
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
task: arabic_leaderboard_acva_entertainment_light
|
| 2 |
+
dataset_path: arcee-globe/ACVA-10percent
|
| 3 |
+
dataset_name: entertainment
|
| 4 |
+
output_type: multiple_choice
|
| 5 |
+
training_split: null
|
| 6 |
+
validation_split: validation
|
| 7 |
+
test_split: test
|
| 8 |
+
process_docs: !function utils.process_docs
|
| 9 |
+
doc_to_text: "{{query}}"
|
| 10 |
+
doc_to_target: "{{gold}}"
|
| 11 |
+
doc_to_choice: "choices"
|
| 12 |
+
fewshot_split: validation
|
| 13 |
+
fewshot_config:
|
| 14 |
+
sampler: first_n
|
| 15 |
+
metric_list:
|
| 16 |
+
- metric: acc
|
| 17 |
+
aggregation: mean
|
| 18 |
+
higher_is_better: true
|
| 19 |
+
- metric: acc_norm
|
| 20 |
+
aggregation: mean
|
| 21 |
+
higher_is_better: true
|
| 22 |
+
metadata:
|
| 23 |
+
version: 1.0
|
lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/arabic_leaderboard_acva_light.yaml
ADDED
|
@@ -0,0 +1,70 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
group: arabic_leaderboard_acva_light
|
| 2 |
+
task:
|
| 3 |
+
- arabic_leaderboard_acva_Algeria_light
|
| 4 |
+
- arabic_leaderboard_acva_Ancient_Egypt_light
|
| 5 |
+
- arabic_leaderboard_acva_Arab_Empire_light
|
| 6 |
+
- arabic_leaderboard_acva_Arabic_Architecture_light
|
| 7 |
+
- arabic_leaderboard_acva_Arabic_Art_light
|
| 8 |
+
- arabic_leaderboard_acva_Arabic_Astronomy_light
|
| 9 |
+
- arabic_leaderboard_acva_Arabic_Calligraphy_light
|
| 10 |
+
- arabic_leaderboard_acva_Arabic_Ceremony_light
|
| 11 |
+
- arabic_leaderboard_acva_Arabic_Clothing_light
|
| 12 |
+
- arabic_leaderboard_acva_Arabic_Culture_light
|
| 13 |
+
- arabic_leaderboard_acva_Arabic_Food_light
|
| 14 |
+
- arabic_leaderboard_acva_Arabic_Funeral_light
|
| 15 |
+
- arabic_leaderboard_acva_Arabic_Geography_light
|
| 16 |
+
- arabic_leaderboard_acva_Arabic_History_light
|
| 17 |
+
- arabic_leaderboard_acva_Arabic_Language_Origin_light
|
| 18 |
+
- arabic_leaderboard_acva_Arabic_Literature_light
|
| 19 |
+
- arabic_leaderboard_acva_Arabic_Math_light
|
| 20 |
+
- arabic_leaderboard_acva_Arabic_Medicine_light
|
| 21 |
+
- arabic_leaderboard_acva_Arabic_Music_light
|
| 22 |
+
- arabic_leaderboard_acva_Arabic_Ornament_light
|
| 23 |
+
- arabic_leaderboard_acva_Arabic_Philosophy_light
|
| 24 |
+
- arabic_leaderboard_acva_Arabic_Physics_and_Chemistry_light
|
| 25 |
+
- arabic_leaderboard_acva_Arabic_Wedding_light
|
| 26 |
+
- arabic_leaderboard_acva_Bahrain_light
|
| 27 |
+
- arabic_leaderboard_acva_Comoros_light
|
| 28 |
+
- arabic_leaderboard_acva_Egypt_modern_light
|
| 29 |
+
- arabic_leaderboard_acva_InfluenceFromAncientEgypt_light
|
| 30 |
+
- arabic_leaderboard_acva_InfluenceFromByzantium_light
|
| 31 |
+
- arabic_leaderboard_acva_InfluenceFromChina_light
|
| 32 |
+
- arabic_leaderboard_acva_InfluenceFromGreece_light
|
| 33 |
+
- arabic_leaderboard_acva_InfluenceFromIslam_light
|
| 34 |
+
- arabic_leaderboard_acva_InfluenceFromPersia_light
|
| 35 |
+
- arabic_leaderboard_acva_InfluenceFromRome_light
|
| 36 |
+
- arabic_leaderboard_acva_Iraq_light
|
| 37 |
+
- arabic_leaderboard_acva_Islam_Education_light
|
| 38 |
+
- arabic_leaderboard_acva_Islam_branches_and_schools_light
|
| 39 |
+
- arabic_leaderboard_acva_Islamic_law_system_light
|
| 40 |
+
- arabic_leaderboard_acva_Jordan_light
|
| 41 |
+
- arabic_leaderboard_acva_Kuwait_light
|
| 42 |
+
- arabic_leaderboard_acva_Lebanon_light
|
| 43 |
+
- arabic_leaderboard_acva_Libya_light
|
| 44 |
+
- arabic_leaderboard_acva_Mauritania_light
|
| 45 |
+
- arabic_leaderboard_acva_Mesopotamia_civilization_light
|
| 46 |
+
- arabic_leaderboard_acva_Morocco_light
|
| 47 |
+
- arabic_leaderboard_acva_Oman_light
|
| 48 |
+
- arabic_leaderboard_acva_Palestine_light
|
| 49 |
+
- arabic_leaderboard_acva_Qatar_light
|
| 50 |
+
- arabic_leaderboard_acva_Saudi_Arabia_light
|
| 51 |
+
- arabic_leaderboard_acva_Somalia_light
|
| 52 |
+
- arabic_leaderboard_acva_Sudan_light
|
| 53 |
+
- arabic_leaderboard_acva_Syria_light
|
| 54 |
+
- arabic_leaderboard_acva_Tunisia_light
|
| 55 |
+
- arabic_leaderboard_acva_United_Arab_Emirates_light
|
| 56 |
+
- arabic_leaderboard_acva_Yemen_light
|
| 57 |
+
- arabic_leaderboard_acva_communication_light
|
| 58 |
+
- arabic_leaderboard_acva_computer_and_phone_light
|
| 59 |
+
- arabic_leaderboard_acva_daily_life_light
|
| 60 |
+
- arabic_leaderboard_acva_entertainment_light
|
| 61 |
+
|
| 62 |
+
aggregate_metric_list:
|
| 63 |
+
- metric: acc
|
| 64 |
+
aggregation: mean
|
| 65 |
+
weight_by_size: true
|
| 66 |
+
- metric: acc_norm
|
| 67 |
+
aggregation: mean
|
| 68 |
+
weight_by_size: true
|
| 69 |
+
metadata:
|
| 70 |
+
version: 1.0
|
lm-evaluation-harness/lm_eval/tasks/arabic_leaderboard_light/arabic_leaderboard_avca_light/utils.py
ADDED
|
@@ -0,0 +1,16 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import datasets
|
| 2 |
+
import numpy as np
|
| 3 |
+
|
| 4 |
+
|
| 5 |
+
def process_docs(dataset: datasets.Dataset):
|
| 6 |
+
def _process_doc(doc):
|
| 7 |
+
question = doc["question"]
|
| 8 |
+
answer = doc["answer"]
|
| 9 |
+
|
| 10 |
+
return {
|
| 11 |
+
"query": f"السؤال: {question}\nالإجابة:",
|
| 12 |
+
"choices": ["صح", "خطأ"],
|
| 13 |
+
"gold": ["صح", "خطأ"].index(answer),
|
| 14 |
+
}
|
| 15 |
+
|
| 16 |
+
return dataset.map(_process_doc)
|
lm-evaluation-harness/lm_eval/tasks/arabicmmlu/_arabicmmlu_social_science.yaml
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
group: arabicmmlu_social_science
|
| 2 |
+
group_alias: Social Science
|
| 3 |
+
task:
|
| 4 |
+
- arabicmmlu_social_science_tasks
|
| 5 |
+
aggregate_metric_list:
|
| 6 |
+
- metric: acc
|
| 7 |
+
weight_by_size: True
|
| 8 |
+
metadata:
|
| 9 |
+
version: 1
|
lm-evaluation-harness/lm_eval/tasks/arabicmmlu/_arabicmmlu_stem.yaml
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
group: arabicmmlu_stem
|
| 2 |
+
group_alias: STEM
|
| 3 |
+
task:
|
| 4 |
+
- arabicmmlu_stem_tasks
|
| 5 |
+
aggregate_metric_list:
|
| 6 |
+
- metric: acc
|
| 7 |
+
weight_by_size: True
|
| 8 |
+
metadata:
|
| 9 |
+
version: 1
|
lm-evaluation-harness/lm_eval/tasks/arabicmmlu/_default_arabicmmlu_template_yaml
ADDED
|
@@ -0,0 +1,15 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
dataset_path: MBZUAI/ArabicMMLU
|
| 2 |
+
test_split: test
|
| 3 |
+
fewshot_split: dev
|
| 4 |
+
fewshot_config:
|
| 5 |
+
sampler: first_n
|
| 6 |
+
output_type: multiple_choice
|
| 7 |
+
doc_to_text: !function utils.doc_to_text
|
| 8 |
+
doc_to_choice: !function utils.doc_to_choice
|
| 9 |
+
doc_to_target: "Answer Key"
|
| 10 |
+
metric_list:
|
| 11 |
+
- metric: acc
|
| 12 |
+
aggregation: mean
|
| 13 |
+
higher_is_better: true
|
| 14 |
+
metadata:
|
| 15 |
+
version: 1.0
|
lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_arabic_language_grammar.yaml
ADDED
|
@@ -0,0 +1,5 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"dataset_name": "Arabic Language (Grammar)"
|
| 2 |
+
"include": "_default_arabicmmlu_template_yaml"
|
| 3 |
+
"tag": "arabicmmlu_language_tasks"
|
| 4 |
+
"task": "arabicmmlu_arabic_language_grammar"
|
| 5 |
+
"task_alias": "Arabic Language (Grammar)"
|
lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_arabic_language_primary_school.yaml
ADDED
|
@@ -0,0 +1,5 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"dataset_name": "Arabic Language (Primary School)"
|
| 2 |
+
"include": "_default_arabicmmlu_template_yaml"
|
| 3 |
+
"tag": "arabicmmlu_language_tasks"
|
| 4 |
+
"task": "arabicmmlu_arabic_language_primary_school"
|
| 5 |
+
"task_alias": "Arabic Language (Primary School)"
|
lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_biology_high_school.yaml
ADDED
|
@@ -0,0 +1,5 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"dataset_name": "Biology (High School)"
|
| 2 |
+
"include": "_default_arabicmmlu_template_yaml"
|
| 3 |
+
"tag": "arabicmmlu_stem_tasks"
|
| 4 |
+
"task": "arabicmmlu_biology_high_school"
|
| 5 |
+
"task_alias": "Biology (High School)"
|
lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_civics_high_school.yaml
ADDED
|
@@ -0,0 +1,5 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"dataset_name": "Civics (High School)"
|
| 2 |
+
"include": "_default_arabicmmlu_template_yaml"
|
| 3 |
+
"tag": "arabicmmlu_social_science_tasks"
|
| 4 |
+
"task": "arabicmmlu_civics_high_school"
|
| 5 |
+
"task_alias": "Civics (High School)"
|
lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_civics_middle_school.yaml
ADDED
|
@@ -0,0 +1,5 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"dataset_name": "Civics (Middle School)"
|
| 2 |
+
"include": "_default_arabicmmlu_template_yaml"
|
| 3 |
+
"tag": "arabicmmlu_social_science_tasks"
|
| 4 |
+
"task": "arabicmmlu_civics_middle_school"
|
| 5 |
+
"task_alias": "Civics (Middle School)"
|
lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_computer_science_middle_school.yaml
ADDED
|
@@ -0,0 +1,5 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"dataset_name": "Computer Science (Middle School)"
|
| 2 |
+
"include": "_default_arabicmmlu_template_yaml"
|
| 3 |
+
"tag": "arabicmmlu_stem_tasks"
|
| 4 |
+
"task": "arabicmmlu_computer_science_middle_school"
|
| 5 |
+
"task_alias": "Computer Science (Middle School)"
|
lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_computer_science_primary_school.yaml
ADDED
|
@@ -0,0 +1,5 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"dataset_name": "Computer Science (Primary School)"
|
| 2 |
+
"include": "_default_arabicmmlu_template_yaml"
|
| 3 |
+
"tag": "arabicmmlu_stem_tasks"
|
| 4 |
+
"task": "arabicmmlu_computer_science_primary_school"
|
| 5 |
+
"task_alias": "Computer Science (Primary School)"
|
lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_computer_science_university.yaml
ADDED
|
@@ -0,0 +1,5 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"dataset_name": "Computer Science (University)"
|
| 2 |
+
"include": "_default_arabicmmlu_template_yaml"
|
| 3 |
+
"tag": "arabicmmlu_stem_tasks"
|
| 4 |
+
"task": "arabicmmlu_computer_science_university"
|
| 5 |
+
"task_alias": "Computer Science (University)"
|
lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_driving_test.yaml
ADDED
|
@@ -0,0 +1,5 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"dataset_name": "Driving Test"
|
| 2 |
+
"include": "_default_arabicmmlu_template_yaml"
|
| 3 |
+
"tag": "arabicmmlu_other_tasks"
|
| 4 |
+
"task": "arabicmmlu_driving_test"
|
| 5 |
+
"task_alias": "Driving Test"
|
lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_economics_high_school.yaml
ADDED
|
@@ -0,0 +1,5 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"dataset_name": "Economics (High School)"
|
| 2 |
+
"include": "_default_arabicmmlu_template_yaml"
|
| 3 |
+
"tag": "arabicmmlu_social_science_tasks"
|
| 4 |
+
"task": "arabicmmlu_economics_high_school"
|
| 5 |
+
"task_alias": "Economics (High School)"
|
lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_economics_middle_school.yaml
ADDED
|
@@ -0,0 +1,5 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"dataset_name": "Economics (Middle School)"
|
| 2 |
+
"include": "_default_arabicmmlu_template_yaml"
|
| 3 |
+
"tag": "arabicmmlu_social_science_tasks"
|
| 4 |
+
"task": "arabicmmlu_economics_middle_school"
|
| 5 |
+
"task_alias": "Economics (Middle School)"
|
lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_economics_university.yaml
ADDED
|
@@ -0,0 +1,5 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"dataset_name": "Economics (University)"
|
| 2 |
+
"include": "_default_arabicmmlu_template_yaml"
|
| 3 |
+
"tag": "arabicmmlu_social_science_tasks"
|
| 4 |
+
"task": "arabicmmlu_economics_university"
|
| 5 |
+
"task_alias": "Economics (University)"
|
lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_general_knowledge.yaml
ADDED
|
@@ -0,0 +1,5 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"dataset_name": "General Knowledge"
|
| 2 |
+
"include": "_default_arabicmmlu_template_yaml"
|
| 3 |
+
"tag": "arabicmmlu_other_tasks"
|
| 4 |
+
"task": "arabicmmlu_general_knowledge"
|
| 5 |
+
"task_alias": "General Knowledge"
|
lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_general_knowledge_middle_school.yaml
ADDED
|
@@ -0,0 +1,5 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"dataset_name": "General Knowledge (Middle School)"
|
| 2 |
+
"include": "_default_arabicmmlu_template_yaml"
|
| 3 |
+
"tag": "arabicmmlu_other_tasks"
|
| 4 |
+
"task": "arabicmmlu_general_knowledge_middle_school"
|
| 5 |
+
"task_alias": "General Knowledge (Middle School)"
|
lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_general_knowledge_primary_school.yaml
ADDED
|
@@ -0,0 +1,5 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"dataset_name": "General Knowledge (Primary School)"
|
| 2 |
+
"include": "_default_arabicmmlu_template_yaml"
|
| 3 |
+
"tag": "arabicmmlu_other_tasks"
|
| 4 |
+
"task": "arabicmmlu_general_knowledge_primary_school"
|
| 5 |
+
"task_alias": "General Knowledge (Primary School)"
|
lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_geography_high_school.yaml
ADDED
|
@@ -0,0 +1,5 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"dataset_name": "Geography (High School)"
|
| 2 |
+
"include": "_default_arabicmmlu_template_yaml"
|
| 3 |
+
"tag": "arabicmmlu_social_science_tasks"
|
| 4 |
+
"task": "arabicmmlu_geography_high_school"
|
| 5 |
+
"task_alias": "Geography (High School)"
|
lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_geography_middle_school.yaml
ADDED
|
@@ -0,0 +1,5 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"dataset_name": "Geography (Middle School)"
|
| 2 |
+
"include": "_default_arabicmmlu_template_yaml"
|
| 3 |
+
"tag": "arabicmmlu_social_science_tasks"
|
| 4 |
+
"task": "arabicmmlu_geography_middle_school"
|
| 5 |
+
"task_alias": "Geography (Middle School)"
|
lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_geography_primary_school.yaml
ADDED
|
@@ -0,0 +1,5 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"dataset_name": "Geography (Primary School)"
|
| 2 |
+
"include": "_default_arabicmmlu_template_yaml"
|
| 3 |
+
"tag": "arabicmmlu_social_science_tasks"
|
| 4 |
+
"task": "arabicmmlu_geography_primary_school"
|
| 5 |
+
"task_alias": "Geography (Primary School)"
|
lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_history_high_school.yaml
ADDED
|
@@ -0,0 +1,5 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"dataset_name": "History (High School)"
|
| 2 |
+
"include": "_default_arabicmmlu_template_yaml"
|
| 3 |
+
"tag": "arabicmmlu_humanities_tasks"
|
| 4 |
+
"task": "arabicmmlu_history_high_school"
|
| 5 |
+
"task_alias": "History (High School)"
|
lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_history_middle_school.yaml
ADDED
|
@@ -0,0 +1,5 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"dataset_name": "History (Middle School)"
|
| 2 |
+
"include": "_default_arabicmmlu_template_yaml"
|
| 3 |
+
"tag": "arabicmmlu_humanities_tasks"
|
| 4 |
+
"task": "arabicmmlu_history_middle_school"
|
| 5 |
+
"task_alias": "History (Middle School)"
|
lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_history_primary_school.yaml
ADDED
|
@@ -0,0 +1,5 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"dataset_name": "History (Primary School)"
|
| 2 |
+
"include": "_default_arabicmmlu_template_yaml"
|
| 3 |
+
"tag": "arabicmmlu_humanities_tasks"
|
| 4 |
+
"task": "arabicmmlu_history_primary_school"
|
| 5 |
+
"task_alias": "History (Primary School)"
|
lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_islamic_studies_high_school.yaml
ADDED
|
@@ -0,0 +1,5 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"dataset_name": "Islamic Studies (High School)"
|
| 2 |
+
"include": "_default_arabicmmlu_template_yaml"
|
| 3 |
+
"tag": "arabicmmlu_humanities_tasks"
|
| 4 |
+
"task": "arabicmmlu_islamic_studies_high_school"
|
| 5 |
+
"task_alias": "Islamic Studies (High School)"
|
lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_islamic_studies_primary_school.yaml
ADDED
|
@@ -0,0 +1,5 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"dataset_name": "Islamic Studies (Primary School)"
|
| 2 |
+
"include": "_default_arabicmmlu_template_yaml"
|
| 3 |
+
"tag": "arabicmmlu_humanities_tasks"
|
| 4 |
+
"task": "arabicmmlu_islamic_studies_primary_school"
|
| 5 |
+
"task_alias": "Islamic Studies (Primary School)"
|
lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_law_professional.yaml
ADDED
|
@@ -0,0 +1,5 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"dataset_name": "Law (Professional)"
|
| 2 |
+
"include": "_default_arabicmmlu_template_yaml"
|
| 3 |
+
"tag": "arabicmmlu_humanities_tasks"
|
| 4 |
+
"task": "arabicmmlu_law_professional"
|
| 5 |
+
"task_alias": "Law (Professional)"
|
lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_management_university.yaml
ADDED
|
@@ -0,0 +1,5 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"dataset_name": "Management (University)"
|
| 2 |
+
"include": "_default_arabicmmlu_template_yaml"
|
| 3 |
+
"tag": "arabicmmlu_other_tasks"
|
| 4 |
+
"task": "arabicmmlu_management_university"
|
| 5 |
+
"task_alias": "Management (University)"
|
lm-evaluation-harness/lm_eval/tasks/arabicmmlu/arabicmmlu_math_primary_school.yaml
ADDED
|
@@ -0,0 +1,5 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"dataset_name": "Math (Primary School)"
|
| 2 |
+
"include": "_default_arabicmmlu_template_yaml"
|
| 3 |
+
"tag": "arabicmmlu_stem_tasks"
|
| 4 |
+
"task": "arabicmmlu_math_primary_school"
|
| 5 |
+
"task_alias": "Math (Primary School)"
|