Add files using upload-large-folder tool
Browse filesThis view is limited to 50 files because it contains too many changes. See raw diff
- lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_fon.yaml +13 -0
- lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/utils.py +55 -0
- lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/masakhapos_ibo.yaml +14 -0
- lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/masakhapos_kin.yaml +14 -0
- lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/masakhapos_lug.yaml +14 -0
- lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/masakhapos_sna.yaml +14 -0
- lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/masakhapos_tsn.yaml +14 -0
- lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/masakhapos_wol.yaml +14 -0
- lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/masakhapos_yaml +32 -0
- lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/masakhapos_zul.yaml +14 -0
- lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/utils.py +55 -0
- lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_3/masakhapos_bbj.yaml +14 -0
- lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_3/masakhapos_ewe.yaml +14 -0
- lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_3/masakhapos_fon.yaml +14 -0
- lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_3/masakhapos_ibo.yaml +14 -0
- lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_3/masakhapos_kin.yaml +14 -0
- lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_3/masakhapos_lug.yaml +14 -0
- lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_3/masakhapos_luo.yaml +14 -0
- lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_3/masakhapos_mos.yaml +14 -0
- lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_3/masakhapos_pcm.yaml +14 -0
- lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_3/masakhapos_sna.yaml +14 -0
- lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_3/masakhapos_swa.yaml +14 -0
- lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_3/masakhapos_twi.yaml +14 -0
- lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_3/masakhapos_wol.yaml +14 -0
- lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_3/masakhapos_xho.yaml +14 -0
- lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_3/masakhapos_yaml +32 -0
- lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_3/masakhapos_yor.yaml +14 -0
- lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_3/masakhapos_zul.yaml +14 -0
- lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_3/utils.py +55 -0
- lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_4/masakhapos_bam.yaml +13 -0
- lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_4/masakhapos_ewe.yaml +13 -0
- lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_4/masakhapos_fon.yaml +13 -0
- lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_4/masakhapos_hau.yaml +13 -0
- lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_4/masakhapos_ibo.yaml +13 -0
- lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_4/masakhapos_kin.yaml +13 -0
- lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_4/masakhapos_lug.yaml +13 -0
- lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_4/masakhapos_luo.yaml +13 -0
- lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_4/masakhapos_mos.yaml +13 -0
- lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_4/masakhapos_nya.yaml +13 -0
- lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_4/masakhapos_pcm.yaml +13 -0
- lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_4/masakhapos_sna.yaml +13 -0
- lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_4/masakhapos_swa.yaml +13 -0
- lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_4/masakhapos_tsn.yaml +13 -0
- lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_4/masakhapos_twi.yaml +13 -0
- lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_4/masakhapos_wol.yaml +13 -0
- lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_4/masakhapos_xho.yaml +13 -0
- lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_4/masakhapos_yaml +32 -0
- lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_4/masakhapos_yor.yaml +13 -0
- lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_4/masakhapos_zul.yaml +13 -0
- lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_4/utils.py +55 -0
lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/masakhapos_fon.yaml
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Generated by utils.py
|
| 2 |
+
dataset_name: fon
|
| 3 |
+
doc_to_text: "Please provide the POS tags for each word in the input sentence. The\
|
| 4 |
+
\ input will be a list of words in the sentence. The output format should be a list\
|
| 5 |
+
\ of tuples, where each tuple consists of a word from the input text and its corresponding\
|
| 6 |
+
\ POS tag label from the tag label set: ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET',\
|
| 7 |
+
\ 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM', 'VERB',\
|
| 8 |
+
\ 'X']. \nYour response should include only a list of tuples, in the order that\
|
| 9 |
+
\ the words appear in the input sentence, including punctuations, with each tuple\
|
| 10 |
+
\ containing the corresponding POS tag label for a word. \n\nSentence: {{tokens}}\
|
| 11 |
+
\ \nOutput: "
|
| 12 |
+
include: masakhapos_yaml
|
| 13 |
+
task: masakhapos_fon_prompt_1
|
lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_1/utils.py
ADDED
|
@@ -0,0 +1,55 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from itertools import chain
|
| 2 |
+
|
| 3 |
+
from sklearn.metrics import accuracy_score
|
| 4 |
+
|
| 5 |
+
from lm_eval.utils import weighted_f1_score
|
| 6 |
+
|
| 7 |
+
|
| 8 |
+
def doc_to_target(doc):
|
| 9 |
+
pos_tag_map = {
|
| 10 |
+
0: "NOUN",
|
| 11 |
+
1: "PUNCT",
|
| 12 |
+
2: "ADP",
|
| 13 |
+
3: "NUM",
|
| 14 |
+
4: "SYM",
|
| 15 |
+
5: "SCONJ",
|
| 16 |
+
6: "ADJ",
|
| 17 |
+
7: "PART",
|
| 18 |
+
8: "DET",
|
| 19 |
+
9: "CCONJ",
|
| 20 |
+
10: "PROPN",
|
| 21 |
+
11: "PRON",
|
| 22 |
+
12: "X",
|
| 23 |
+
13: "_",
|
| 24 |
+
14: "ADV",
|
| 25 |
+
15: "INTJ",
|
| 26 |
+
16: "VERB",
|
| 27 |
+
17: "AUX",
|
| 28 |
+
}
|
| 29 |
+
return [pos_tag_map[tag] for tag in doc["upos"]]
|
| 30 |
+
|
| 31 |
+
|
| 32 |
+
def acc_score(items):
|
| 33 |
+
unzipped_list = list(zip(*items))
|
| 34 |
+
|
| 35 |
+
golds, preds = unzipped_list[0], unzipped_list[1]
|
| 36 |
+
|
| 37 |
+
# Flatten preds' inner lists
|
| 38 |
+
flattened_preds = [list(chain.from_iterable(p)) for p in preds]
|
| 39 |
+
|
| 40 |
+
# Calculate the accuracy for each gold-pred pair
|
| 41 |
+
accuracy_scores = []
|
| 42 |
+
for gold, pred in zip(golds, flattened_preds):
|
| 43 |
+
# Ensure both lists are of the same length, otherwise truncate to match
|
| 44 |
+
min_length = min(len(gold), len(pred))
|
| 45 |
+
gold = gold[:min_length]
|
| 46 |
+
pred = pred[:min_length]
|
| 47 |
+
|
| 48 |
+
# Calculate accuracy for the current pair and add to the list
|
| 49 |
+
accuracy = accuracy_score(gold, pred)
|
| 50 |
+
accuracy_scores.append(accuracy)
|
| 51 |
+
|
| 52 |
+
mean_accuracy = (
|
| 53 |
+
sum(accuracy_scores) / len(accuracy_scores) if accuracy_scores else 0
|
| 54 |
+
)
|
| 55 |
+
return mean_accuracy
|
lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/masakhapos_ibo.yaml
ADDED
|
@@ -0,0 +1,14 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Generated by utils.py
|
| 2 |
+
dataset_name: ibo
|
| 3 |
+
doc_to_text: "You are an expert in tagging words and sentences in Igbo with the right\
|
| 4 |
+
\ POS tag. \n\nPlease provide the POS tags for each word in the Igbo sentence. The\
|
| 5 |
+
\ input is a list of words in the sentence. POS tag label set: ['ADJ', 'ADP', 'ADV',\
|
| 6 |
+
\ 'AUX', 'CCONJ, 'DET', 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT',\
|
| 7 |
+
\ 'SCONJ', 'SYM', 'VERB', 'X']. The output format should be a list of tuples, where\
|
| 8 |
+
\ each tuple consists of a word from the input text and its corresponding POS tag\
|
| 9 |
+
\ label from the POS tag label set provided\nYour response should include only a\
|
| 10 |
+
\ list of tuples, in the order that the words appear in the input sentence, including\
|
| 11 |
+
\ punctuations, with each tuple containing the corresponding POS tag label for a\
|
| 12 |
+
\ word. \n\nSentence: {{tokens}} \nOutput: "
|
| 13 |
+
include: masakhapos_yaml
|
| 14 |
+
task: masakhapos_ibo_prompt_2
|
lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/masakhapos_kin.yaml
ADDED
|
@@ -0,0 +1,14 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Generated by utils.py
|
| 2 |
+
dataset_name: kin
|
| 3 |
+
doc_to_text: "You are an expert in tagging words and sentences in Kinyarwanda with\
|
| 4 |
+
\ the right POS tag. \n\nPlease provide the POS tags for each word in the Kinyarwanda\
|
| 5 |
+
\ sentence. The input is a list of words in the sentence. POS tag label set: ['ADJ',\
|
| 6 |
+
\ 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET', 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN',\
|
| 7 |
+
\ 'PUNCT', 'SCONJ', 'SYM', 'VERB', 'X']. The output format should be a list of tuples,\
|
| 8 |
+
\ where each tuple consists of a word from the input text and its corresponding\
|
| 9 |
+
\ POS tag label from the POS tag label set provided\nYour response should include\
|
| 10 |
+
\ only a list of tuples, in the order that the words appear in the input sentence,\
|
| 11 |
+
\ including punctuations, with each tuple containing the corresponding POS tag label\
|
| 12 |
+
\ for a word. \n\nSentence: {{tokens}} \nOutput: "
|
| 13 |
+
include: masakhapos_yaml
|
| 14 |
+
task: masakhapos_kin_prompt_2
|
lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/masakhapos_lug.yaml
ADDED
|
@@ -0,0 +1,14 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Generated by utils.py
|
| 2 |
+
dataset_name: lug
|
| 3 |
+
doc_to_text: "You are an expert in tagging words and sentences in Luganda with the\
|
| 4 |
+
\ right POS tag. \n\nPlease provide the POS tags for each word in the Luganda sentence.\
|
| 5 |
+
\ The input is a list of words in the sentence. POS tag label set: ['ADJ', 'ADP',\
|
| 6 |
+
\ 'ADV', 'AUX', 'CCONJ, 'DET', 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT',\
|
| 7 |
+
\ 'SCONJ', 'SYM', 'VERB', 'X']. The output format should be a list of tuples, where\
|
| 8 |
+
\ each tuple consists of a word from the input text and its corresponding POS tag\
|
| 9 |
+
\ label from the POS tag label set provided\nYour response should include only a\
|
| 10 |
+
\ list of tuples, in the order that the words appear in the input sentence, including\
|
| 11 |
+
\ punctuations, with each tuple containing the corresponding POS tag label for a\
|
| 12 |
+
\ word. \n\nSentence: {{tokens}} \nOutput: "
|
| 13 |
+
include: masakhapos_yaml
|
| 14 |
+
task: masakhapos_lug_prompt_2
|
lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/masakhapos_sna.yaml
ADDED
|
@@ -0,0 +1,14 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Generated by utils.py
|
| 2 |
+
dataset_name: sna
|
| 3 |
+
doc_to_text: "You are an expert in tagging words and sentences in chiShona with the\
|
| 4 |
+
\ right POS tag. \n\nPlease provide the POS tags for each word in the chiShona sentence.\
|
| 5 |
+
\ The input is a list of words in the sentence. POS tag label set: ['ADJ', 'ADP',\
|
| 6 |
+
\ 'ADV', 'AUX', 'CCONJ, 'DET', 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT',\
|
| 7 |
+
\ 'SCONJ', 'SYM', 'VERB', 'X']. The output format should be a list of tuples, where\
|
| 8 |
+
\ each tuple consists of a word from the input text and its corresponding POS tag\
|
| 9 |
+
\ label from the POS tag label set provided\nYour response should include only a\
|
| 10 |
+
\ list of tuples, in the order that the words appear in the input sentence, including\
|
| 11 |
+
\ punctuations, with each tuple containing the corresponding POS tag label for a\
|
| 12 |
+
\ word. \n\nSentence: {{tokens}} \nOutput: "
|
| 13 |
+
include: masakhapos_yaml
|
| 14 |
+
task: masakhapos_sna_prompt_2
|
lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/masakhapos_tsn.yaml
ADDED
|
@@ -0,0 +1,14 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Generated by utils.py
|
| 2 |
+
dataset_name: tsn
|
| 3 |
+
doc_to_text: "You are an expert in tagging words and sentences in Setswana with the\
|
| 4 |
+
\ right POS tag. \n\nPlease provide the POS tags for each word in the Setswana sentence.\
|
| 5 |
+
\ The input is a list of words in the sentence. POS tag label set: ['ADJ', 'ADP',\
|
| 6 |
+
\ 'ADV', 'AUX', 'CCONJ, 'DET', 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT',\
|
| 7 |
+
\ 'SCONJ', 'SYM', 'VERB', 'X']. The output format should be a list of tuples, where\
|
| 8 |
+
\ each tuple consists of a word from the input text and its corresponding POS tag\
|
| 9 |
+
\ label from the POS tag label set provided\nYour response should include only a\
|
| 10 |
+
\ list of tuples, in the order that the words appear in the input sentence, including\
|
| 11 |
+
\ punctuations, with each tuple containing the corresponding POS tag label for a\
|
| 12 |
+
\ word. \n\nSentence: {{tokens}} \nOutput: "
|
| 13 |
+
include: masakhapos_yaml
|
| 14 |
+
task: masakhapos_tsn_prompt_2
|
lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/masakhapos_wol.yaml
ADDED
|
@@ -0,0 +1,14 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Generated by utils.py
|
| 2 |
+
dataset_name: wol
|
| 3 |
+
doc_to_text: "You are an expert in tagging words and sentences in Wolof with the right\
|
| 4 |
+
\ POS tag. \n\nPlease provide the POS tags for each word in the Wolof sentence.\
|
| 5 |
+
\ The input is a list of words in the sentence. POS tag label set: ['ADJ', 'ADP',\
|
| 6 |
+
\ 'ADV', 'AUX', 'CCONJ, 'DET', 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT',\
|
| 7 |
+
\ 'SCONJ', 'SYM', 'VERB', 'X']. The output format should be a list of tuples, where\
|
| 8 |
+
\ each tuple consists of a word from the input text and its corresponding POS tag\
|
| 9 |
+
\ label from the POS tag label set provided\nYour response should include only a\
|
| 10 |
+
\ list of tuples, in the order that the words appear in the input sentence, including\
|
| 11 |
+
\ punctuations, with each tuple containing the corresponding POS tag label for a\
|
| 12 |
+
\ word. \n\nSentence: {{tokens}} \nOutput: "
|
| 13 |
+
include: masakhapos_yaml
|
| 14 |
+
task: masakhapos_wol_prompt_2
|
lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/masakhapos_yaml
ADDED
|
@@ -0,0 +1,32 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
tag:
|
| 2 |
+
- masakhapos_tasks
|
| 3 |
+
- masakhapos_prompt_2
|
| 4 |
+
dataset_path: masakhane/masakhapos
|
| 5 |
+
dataset_name: null
|
| 6 |
+
dataset_kwargs: {trust_remote_code: True}
|
| 7 |
+
output_type: generate_until
|
| 8 |
+
generation_kwargs:
|
| 9 |
+
do_sample: false
|
| 10 |
+
until:
|
| 11 |
+
- </s>
|
| 12 |
+
- <|im_end|>
|
| 13 |
+
validation_split: validation
|
| 14 |
+
test_split: test
|
| 15 |
+
fewshot_split: train
|
| 16 |
+
doc_to_target: !function utils.doc_to_target
|
| 17 |
+
should_decontaminate: true
|
| 18 |
+
doc_to_decontamination_query: "Sentence: {{token}}\nOutput:"
|
| 19 |
+
filter_list:
|
| 20 |
+
- filter:
|
| 21 |
+
- function: regex_pos
|
| 22 |
+
name: flexible-extract
|
| 23 |
+
metric_list:
|
| 24 |
+
- metric: acc
|
| 25 |
+
aggregation: !function utils.acc_score
|
| 26 |
+
higher_is_better: true
|
| 27 |
+
ignore_case: true
|
| 28 |
+
ignore_punctuation: true
|
| 29 |
+
regexes_to_ignore:
|
| 30 |
+
- ","
|
| 31 |
+
metadata:
|
| 32 |
+
version: 1.0
|
lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/masakhapos_zul.yaml
ADDED
|
@@ -0,0 +1,14 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Generated by utils.py
|
| 2 |
+
dataset_name: zul
|
| 3 |
+
doc_to_text: "You are an expert in tagging words and sentences in isiZulu with the\
|
| 4 |
+
\ right POS tag. \n\nPlease provide the POS tags for each word in the isiZulu sentence.\
|
| 5 |
+
\ The input is a list of words in the sentence. POS tag label set: ['ADJ', 'ADP',\
|
| 6 |
+
\ 'ADV', 'AUX', 'CCONJ, 'DET', 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT',\
|
| 7 |
+
\ 'SCONJ', 'SYM', 'VERB', 'X']. The output format should be a list of tuples, where\
|
| 8 |
+
\ each tuple consists of a word from the input text and its corresponding POS tag\
|
| 9 |
+
\ label from the POS tag label set provided\nYour response should include only a\
|
| 10 |
+
\ list of tuples, in the order that the words appear in the input sentence, including\
|
| 11 |
+
\ punctuations, with each tuple containing the corresponding POS tag label for a\
|
| 12 |
+
\ word. \n\nSentence: {{tokens}} \nOutput: "
|
| 13 |
+
include: masakhapos_yaml
|
| 14 |
+
task: masakhapos_zul_prompt_2
|
lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_2/utils.py
ADDED
|
@@ -0,0 +1,55 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from itertools import chain
|
| 2 |
+
|
| 3 |
+
from sklearn.metrics import accuracy_score
|
| 4 |
+
|
| 5 |
+
from lm_eval.utils import weighted_f1_score
|
| 6 |
+
|
| 7 |
+
|
| 8 |
+
def doc_to_target(doc):
|
| 9 |
+
pos_tag_map = {
|
| 10 |
+
0: "NOUN",
|
| 11 |
+
1: "PUNCT",
|
| 12 |
+
2: "ADP",
|
| 13 |
+
3: "NUM",
|
| 14 |
+
4: "SYM",
|
| 15 |
+
5: "SCONJ",
|
| 16 |
+
6: "ADJ",
|
| 17 |
+
7: "PART",
|
| 18 |
+
8: "DET",
|
| 19 |
+
9: "CCONJ",
|
| 20 |
+
10: "PROPN",
|
| 21 |
+
11: "PRON",
|
| 22 |
+
12: "X",
|
| 23 |
+
13: "_",
|
| 24 |
+
14: "ADV",
|
| 25 |
+
15: "INTJ",
|
| 26 |
+
16: "VERB",
|
| 27 |
+
17: "AUX",
|
| 28 |
+
}
|
| 29 |
+
return [pos_tag_map[tag] for tag in doc["upos"]]
|
| 30 |
+
|
| 31 |
+
|
| 32 |
+
def acc_score(items):
|
| 33 |
+
unzipped_list = list(zip(*items))
|
| 34 |
+
|
| 35 |
+
golds, preds = unzipped_list[0], unzipped_list[1]
|
| 36 |
+
|
| 37 |
+
# Flatten preds' inner lists
|
| 38 |
+
flattened_preds = [list(chain.from_iterable(p)) for p in preds]
|
| 39 |
+
|
| 40 |
+
# Calculate the accuracy for each gold-pred pair
|
| 41 |
+
accuracy_scores = []
|
| 42 |
+
for gold, pred in zip(golds, flattened_preds):
|
| 43 |
+
# Ensure both lists are of the same length, otherwise truncate to match
|
| 44 |
+
min_length = min(len(gold), len(pred))
|
| 45 |
+
gold = gold[:min_length]
|
| 46 |
+
pred = pred[:min_length]
|
| 47 |
+
|
| 48 |
+
# Calculate accuracy for the current pair and add to the list
|
| 49 |
+
accuracy = accuracy_score(gold, pred)
|
| 50 |
+
accuracy_scores.append(accuracy)
|
| 51 |
+
|
| 52 |
+
mean_accuracy = (
|
| 53 |
+
sum(accuracy_scores) / len(accuracy_scores) if accuracy_scores else 0
|
| 54 |
+
)
|
| 55 |
+
return mean_accuracy
|
lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_3/masakhapos_bbj.yaml
ADDED
|
@@ -0,0 +1,14 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Generated by utils.py
|
| 2 |
+
dataset_name: bbj
|
| 3 |
+
doc_to_text: "Acting as a Ghomala linguist and without making any corrections or changes\
|
| 4 |
+
\ to the text, perform a part of speech (POS) analysis of the sentences using the\
|
| 5 |
+
\ following POS tag label annotation ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET',\
|
| 6 |
+
\ 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM', 'VERB',\
|
| 7 |
+
\ 'X']. The input will be a list of words in the sentence. The output format should\
|
| 8 |
+
\ be a list of tuples, where each tuple consists of a word from the input text and\
|
| 9 |
+
\ its corresponding POS tag label from the POS tag label set provided\nYour response\
|
| 10 |
+
\ should include only a list of tuples, in the order that the words appear in the\
|
| 11 |
+
\ input sentence, including punctuations, with each tuple containing the corresponding\
|
| 12 |
+
\ POS tag label for a word. \n\nSentence: {{tokens}} \nOutput: "
|
| 13 |
+
include: masakhapos_yaml
|
| 14 |
+
task: masakhapos_bbj_prompt_3
|
lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_3/masakhapos_ewe.yaml
ADDED
|
@@ -0,0 +1,14 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Generated by utils.py
|
| 2 |
+
dataset_name: ewe
|
| 3 |
+
doc_to_text: "Acting as a Ewe linguist and without making any corrections or changes\
|
| 4 |
+
\ to the text, perform a part of speech (POS) analysis of the sentences using the\
|
| 5 |
+
\ following POS tag label annotation ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET',\
|
| 6 |
+
\ 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM', 'VERB',\
|
| 7 |
+
\ 'X']. The input will be a list of words in the sentence. The output format should\
|
| 8 |
+
\ be a list of tuples, where each tuple consists of a word from the input text and\
|
| 9 |
+
\ its corresponding POS tag label from the POS tag label set provided\nYour response\
|
| 10 |
+
\ should include only a list of tuples, in the order that the words appear in the\
|
| 11 |
+
\ input sentence, including punctuations, with each tuple containing the corresponding\
|
| 12 |
+
\ POS tag label for a word. \n\nSentence: {{tokens}} \nOutput: "
|
| 13 |
+
include: masakhapos_yaml
|
| 14 |
+
task: masakhapos_ewe_prompt_3
|
lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_3/masakhapos_fon.yaml
ADDED
|
@@ -0,0 +1,14 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Generated by utils.py
|
| 2 |
+
dataset_name: fon
|
| 3 |
+
doc_to_text: "Acting as a Fon linguist and without making any corrections or changes\
|
| 4 |
+
\ to the text, perform a part of speech (POS) analysis of the sentences using the\
|
| 5 |
+
\ following POS tag label annotation ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET',\
|
| 6 |
+
\ 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM', 'VERB',\
|
| 7 |
+
\ 'X']. The input will be a list of words in the sentence. The output format should\
|
| 8 |
+
\ be a list of tuples, where each tuple consists of a word from the input text and\
|
| 9 |
+
\ its corresponding POS tag label from the POS tag label set provided\nYour response\
|
| 10 |
+
\ should include only a list of tuples, in the order that the words appear in the\
|
| 11 |
+
\ input sentence, including punctuations, with each tuple containing the corresponding\
|
| 12 |
+
\ POS tag label for a word. \n\nSentence: {{tokens}} \nOutput: "
|
| 13 |
+
include: masakhapos_yaml
|
| 14 |
+
task: masakhapos_fon_prompt_3
|
lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_3/masakhapos_ibo.yaml
ADDED
|
@@ -0,0 +1,14 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Generated by utils.py
|
| 2 |
+
dataset_name: ibo
|
| 3 |
+
doc_to_text: "Acting as a Igbo linguist and without making any corrections or changes\
|
| 4 |
+
\ to the text, perform a part of speech (POS) analysis of the sentences using the\
|
| 5 |
+
\ following POS tag label annotation ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET',\
|
| 6 |
+
\ 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM', 'VERB',\
|
| 7 |
+
\ 'X']. The input will be a list of words in the sentence. The output format should\
|
| 8 |
+
\ be a list of tuples, where each tuple consists of a word from the input text and\
|
| 9 |
+
\ its corresponding POS tag label from the POS tag label set provided\nYour response\
|
| 10 |
+
\ should include only a list of tuples, in the order that the words appear in the\
|
| 11 |
+
\ input sentence, including punctuations, with each tuple containing the corresponding\
|
| 12 |
+
\ POS tag label for a word. \n\nSentence: {{tokens}} \nOutput: "
|
| 13 |
+
include: masakhapos_yaml
|
| 14 |
+
task: masakhapos_ibo_prompt_3
|
lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_3/masakhapos_kin.yaml
ADDED
|
@@ -0,0 +1,14 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Generated by utils.py
|
| 2 |
+
dataset_name: kin
|
| 3 |
+
doc_to_text: "Acting as a Kinyarwanda linguist and without making any corrections\
|
| 4 |
+
\ or changes to the text, perform a part of speech (POS) analysis of the sentences\
|
| 5 |
+
\ using the following POS tag label annotation ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ,\
|
| 6 |
+
\ 'DET', 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM',\
|
| 7 |
+
\ 'VERB', 'X']. The input will be a list of words in the sentence. The output format\
|
| 8 |
+
\ should be a list of tuples, where each tuple consists of a word from the input\
|
| 9 |
+
\ text and its corresponding POS tag label from the POS tag label set provided\n\
|
| 10 |
+
Your response should include only a list of tuples, in the order that the words\
|
| 11 |
+
\ appear in the input sentence, including punctuations, with each tuple containing\
|
| 12 |
+
\ the corresponding POS tag label for a word. \n\nSentence: {{tokens}} \nOutput: "
|
| 13 |
+
include: masakhapos_yaml
|
| 14 |
+
task: masakhapos_kin_prompt_3
|
lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_3/masakhapos_lug.yaml
ADDED
|
@@ -0,0 +1,14 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Generated by utils.py
|
| 2 |
+
dataset_name: lug
|
| 3 |
+
doc_to_text: "Acting as a Luganda linguist and without making any corrections or changes\
|
| 4 |
+
\ to the text, perform a part of speech (POS) analysis of the sentences using the\
|
| 5 |
+
\ following POS tag label annotation ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET',\
|
| 6 |
+
\ 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM', 'VERB',\
|
| 7 |
+
\ 'X']. The input will be a list of words in the sentence. The output format should\
|
| 8 |
+
\ be a list of tuples, where each tuple consists of a word from the input text and\
|
| 9 |
+
\ its corresponding POS tag label from the POS tag label set provided\nYour response\
|
| 10 |
+
\ should include only a list of tuples, in the order that the words appear in the\
|
| 11 |
+
\ input sentence, including punctuations, with each tuple containing the corresponding\
|
| 12 |
+
\ POS tag label for a word. \n\nSentence: {{tokens}} \nOutput: "
|
| 13 |
+
include: masakhapos_yaml
|
| 14 |
+
task: masakhapos_lug_prompt_3
|
lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_3/masakhapos_luo.yaml
ADDED
|
@@ -0,0 +1,14 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Generated by utils.py
|
| 2 |
+
dataset_name: luo
|
| 3 |
+
doc_to_text: "Acting as a Dholuo linguist and without making any corrections or changes\
|
| 4 |
+
\ to the text, perform a part of speech (POS) analysis of the sentences using the\
|
| 5 |
+
\ following POS tag label annotation ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET',\
|
| 6 |
+
\ 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM', 'VERB',\
|
| 7 |
+
\ 'X']. The input will be a list of words in the sentence. The output format should\
|
| 8 |
+
\ be a list of tuples, where each tuple consists of a word from the input text and\
|
| 9 |
+
\ its corresponding POS tag label from the POS tag label set provided\nYour response\
|
| 10 |
+
\ should include only a list of tuples, in the order that the words appear in the\
|
| 11 |
+
\ input sentence, including punctuations, with each tuple containing the corresponding\
|
| 12 |
+
\ POS tag label for a word. \n\nSentence: {{tokens}} \nOutput: "
|
| 13 |
+
include: masakhapos_yaml
|
| 14 |
+
task: masakhapos_luo_prompt_3
|
lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_3/masakhapos_mos.yaml
ADDED
|
@@ -0,0 +1,14 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Generated by utils.py
|
| 2 |
+
dataset_name: mos
|
| 3 |
+
doc_to_text: "Acting as a Mossi linguist and without making any corrections or changes\
|
| 4 |
+
\ to the text, perform a part of speech (POS) analysis of the sentences using the\
|
| 5 |
+
\ following POS tag label annotation ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET',\
|
| 6 |
+
\ 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM', 'VERB',\
|
| 7 |
+
\ 'X']. The input will be a list of words in the sentence. The output format should\
|
| 8 |
+
\ be a list of tuples, where each tuple consists of a word from the input text and\
|
| 9 |
+
\ its corresponding POS tag label from the POS tag label set provided\nYour response\
|
| 10 |
+
\ should include only a list of tuples, in the order that the words appear in the\
|
| 11 |
+
\ input sentence, including punctuations, with each tuple containing the corresponding\
|
| 12 |
+
\ POS tag label for a word. \n\nSentence: {{tokens}} \nOutput: "
|
| 13 |
+
include: masakhapos_yaml
|
| 14 |
+
task: masakhapos_mos_prompt_3
|
lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_3/masakhapos_pcm.yaml
ADDED
|
@@ -0,0 +1,14 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Generated by utils.py
|
| 2 |
+
dataset_name: pcm
|
| 3 |
+
doc_to_text: "Acting as a Nigerian Pidgin linguist and without making any corrections\
|
| 4 |
+
\ or changes to the text, perform a part of speech (POS) analysis of the sentences\
|
| 5 |
+
\ using the following POS tag label annotation ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ,\
|
| 6 |
+
\ 'DET', 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM',\
|
| 7 |
+
\ 'VERB', 'X']. The input will be a list of words in the sentence. The output format\
|
| 8 |
+
\ should be a list of tuples, where each tuple consists of a word from the input\
|
| 9 |
+
\ text and its corresponding POS tag label from the POS tag label set provided\n\
|
| 10 |
+
Your response should include only a list of tuples, in the order that the words\
|
| 11 |
+
\ appear in the input sentence, including punctuations, with each tuple containing\
|
| 12 |
+
\ the corresponding POS tag label for a word. \n\nSentence: {{tokens}} \nOutput: "
|
| 13 |
+
include: masakhapos_yaml
|
| 14 |
+
task: masakhapos_pcm_prompt_3
|
lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_3/masakhapos_sna.yaml
ADDED
|
@@ -0,0 +1,14 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Generated by utils.py
|
| 2 |
+
dataset_name: sna
|
| 3 |
+
doc_to_text: "Acting as a chiShona linguist and without making any corrections or\
|
| 4 |
+
\ changes to the text, perform a part of speech (POS) analysis of the sentences\
|
| 5 |
+
\ using the following POS tag label annotation ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ,\
|
| 6 |
+
\ 'DET', 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM',\
|
| 7 |
+
\ 'VERB', 'X']. The input will be a list of words in the sentence. The output format\
|
| 8 |
+
\ should be a list of tuples, where each tuple consists of a word from the input\
|
| 9 |
+
\ text and its corresponding POS tag label from the POS tag label set provided\n\
|
| 10 |
+
Your response should include only a list of tuples, in the order that the words\
|
| 11 |
+
\ appear in the input sentence, including punctuations, with each tuple containing\
|
| 12 |
+
\ the corresponding POS tag label for a word. \n\nSentence: {{tokens}} \nOutput: "
|
| 13 |
+
include: masakhapos_yaml
|
| 14 |
+
task: masakhapos_sna_prompt_3
|
lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_3/masakhapos_swa.yaml
ADDED
|
@@ -0,0 +1,14 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Generated by utils.py
|
| 2 |
+
dataset_name: swa
|
| 3 |
+
doc_to_text: "Acting as a Kiswahili linguist and without making any corrections or\
|
| 4 |
+
\ changes to the text, perform a part of speech (POS) analysis of the sentences\
|
| 5 |
+
\ using the following POS tag label annotation ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ,\
|
| 6 |
+
\ 'DET', 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM',\
|
| 7 |
+
\ 'VERB', 'X']. The input will be a list of words in the sentence. The output format\
|
| 8 |
+
\ should be a list of tuples, where each tuple consists of a word from the input\
|
| 9 |
+
\ text and its corresponding POS tag label from the POS tag label set provided\n\
|
| 10 |
+
Your response should include only a list of tuples, in the order that the words\
|
| 11 |
+
\ appear in the input sentence, including punctuations, with each tuple containing\
|
| 12 |
+
\ the corresponding POS tag label for a word. \n\nSentence: {{tokens}} \nOutput: "
|
| 13 |
+
include: masakhapos_yaml
|
| 14 |
+
task: masakhapos_swa_prompt_3
|
lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_3/masakhapos_twi.yaml
ADDED
|
@@ -0,0 +1,14 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Generated by utils.py
|
| 2 |
+
dataset_name: twi
|
| 3 |
+
doc_to_text: "Acting as a Twi linguist and without making any corrections or changes\
|
| 4 |
+
\ to the text, perform a part of speech (POS) analysis of the sentences using the\
|
| 5 |
+
\ following POS tag label annotation ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET',\
|
| 6 |
+
\ 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM', 'VERB',\
|
| 7 |
+
\ 'X']. The input will be a list of words in the sentence. The output format should\
|
| 8 |
+
\ be a list of tuples, where each tuple consists of a word from the input text and\
|
| 9 |
+
\ its corresponding POS tag label from the POS tag label set provided\nYour response\
|
| 10 |
+
\ should include only a list of tuples, in the order that the words appear in the\
|
| 11 |
+
\ input sentence, including punctuations, with each tuple containing the corresponding\
|
| 12 |
+
\ POS tag label for a word. \n\nSentence: {{tokens}} \nOutput: "
|
| 13 |
+
include: masakhapos_yaml
|
| 14 |
+
task: masakhapos_twi_prompt_3
|
lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_3/masakhapos_wol.yaml
ADDED
|
@@ -0,0 +1,14 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Generated by utils.py
|
| 2 |
+
dataset_name: wol
|
| 3 |
+
doc_to_text: "Acting as a Wolof linguist and without making any corrections or changes\
|
| 4 |
+
\ to the text, perform a part of speech (POS) analysis of the sentences using the\
|
| 5 |
+
\ following POS tag label annotation ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET',\
|
| 6 |
+
\ 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM', 'VERB',\
|
| 7 |
+
\ 'X']. The input will be a list of words in the sentence. The output format should\
|
| 8 |
+
\ be a list of tuples, where each tuple consists of a word from the input text and\
|
| 9 |
+
\ its corresponding POS tag label from the POS tag label set provided\nYour response\
|
| 10 |
+
\ should include only a list of tuples, in the order that the words appear in the\
|
| 11 |
+
\ input sentence, including punctuations, with each tuple containing the corresponding\
|
| 12 |
+
\ POS tag label for a word. \n\nSentence: {{tokens}} \nOutput: "
|
| 13 |
+
include: masakhapos_yaml
|
| 14 |
+
task: masakhapos_wol_prompt_3
|
lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_3/masakhapos_xho.yaml
ADDED
|
@@ -0,0 +1,14 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Generated by utils.py
|
| 2 |
+
dataset_name: xho
|
| 3 |
+
doc_to_text: "Acting as a isiXhosa linguist and without making any corrections or\
|
| 4 |
+
\ changes to the text, perform a part of speech (POS) analysis of the sentences\
|
| 5 |
+
\ using the following POS tag label annotation ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ,\
|
| 6 |
+
\ 'DET', 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM',\
|
| 7 |
+
\ 'VERB', 'X']. The input will be a list of words in the sentence. The output format\
|
| 8 |
+
\ should be a list of tuples, where each tuple consists of a word from the input\
|
| 9 |
+
\ text and its corresponding POS tag label from the POS tag label set provided\n\
|
| 10 |
+
Your response should include only a list of tuples, in the order that the words\
|
| 11 |
+
\ appear in the input sentence, including punctuations, with each tuple containing\
|
| 12 |
+
\ the corresponding POS tag label for a word. \n\nSentence: {{tokens}} \nOutput: "
|
| 13 |
+
include: masakhapos_yaml
|
| 14 |
+
task: masakhapos_xho_prompt_3
|
lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_3/masakhapos_yaml
ADDED
|
@@ -0,0 +1,32 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
tag:
|
| 2 |
+
- masakhapos_tasks
|
| 3 |
+
- masakhapos_prompt_3
|
| 4 |
+
dataset_path: masakhane/masakhapos
|
| 5 |
+
dataset_name: null
|
| 6 |
+
dataset_kwargs: {trust_remote_code: True}
|
| 7 |
+
output_type: generate_until
|
| 8 |
+
generation_kwargs:
|
| 9 |
+
do_sample: false
|
| 10 |
+
until:
|
| 11 |
+
- </s>
|
| 12 |
+
- <|im_end|>
|
| 13 |
+
validation_split: validation
|
| 14 |
+
test_split: test
|
| 15 |
+
fewshot_split: train
|
| 16 |
+
doc_to_target: !function utils.doc_to_target
|
| 17 |
+
should_decontaminate: true
|
| 18 |
+
doc_to_decontamination_query: "Sentence: {{token}}\nOutput:"
|
| 19 |
+
filter_list:
|
| 20 |
+
- filter:
|
| 21 |
+
- function: regex_pos
|
| 22 |
+
name: flexible-extract
|
| 23 |
+
metric_list:
|
| 24 |
+
- metric: acc
|
| 25 |
+
aggregation: !function utils.acc_score
|
| 26 |
+
higher_is_better: true
|
| 27 |
+
ignore_case: true
|
| 28 |
+
ignore_punctuation: true
|
| 29 |
+
regexes_to_ignore:
|
| 30 |
+
- ","
|
| 31 |
+
metadata:
|
| 32 |
+
version: 1.0
|
lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_3/masakhapos_yor.yaml
ADDED
|
@@ -0,0 +1,14 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Generated by utils.py
|
| 2 |
+
dataset_name: yor
|
| 3 |
+
doc_to_text: "Acting as a Yoruba linguist and without making any corrections or changes\
|
| 4 |
+
\ to the text, perform a part of speech (POS) analysis of the sentences using the\
|
| 5 |
+
\ following POS tag label annotation ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET',\
|
| 6 |
+
\ 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM', 'VERB',\
|
| 7 |
+
\ 'X']. The input will be a list of words in the sentence. The output format should\
|
| 8 |
+
\ be a list of tuples, where each tuple consists of a word from the input text and\
|
| 9 |
+
\ its corresponding POS tag label from the POS tag label set provided\nYour response\
|
| 10 |
+
\ should include only a list of tuples, in the order that the words appear in the\
|
| 11 |
+
\ input sentence, including punctuations, with each tuple containing the corresponding\
|
| 12 |
+
\ POS tag label for a word. \n\nSentence: {{tokens}} \nOutput: "
|
| 13 |
+
include: masakhapos_yaml
|
| 14 |
+
task: masakhapos_yor_prompt_3
|
lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_3/masakhapos_zul.yaml
ADDED
|
@@ -0,0 +1,14 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Generated by utils.py
|
| 2 |
+
dataset_name: zul
|
| 3 |
+
doc_to_text: "Acting as a isiZulu linguist and without making any corrections or changes\
|
| 4 |
+
\ to the text, perform a part of speech (POS) analysis of the sentences using the\
|
| 5 |
+
\ following POS tag label annotation ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET',\
|
| 6 |
+
\ 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM', 'VERB',\
|
| 7 |
+
\ 'X']. The input will be a list of words in the sentence. The output format should\
|
| 8 |
+
\ be a list of tuples, where each tuple consists of a word from the input text and\
|
| 9 |
+
\ its corresponding POS tag label from the POS tag label set provided\nYour response\
|
| 10 |
+
\ should include only a list of tuples, in the order that the words appear in the\
|
| 11 |
+
\ input sentence, including punctuations, with each tuple containing the corresponding\
|
| 12 |
+
\ POS tag label for a word. \n\nSentence: {{tokens}} \nOutput: "
|
| 13 |
+
include: masakhapos_yaml
|
| 14 |
+
task: masakhapos_zul_prompt_3
|
lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_3/utils.py
ADDED
|
@@ -0,0 +1,55 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from itertools import chain
|
| 2 |
+
|
| 3 |
+
from sklearn.metrics import accuracy_score
|
| 4 |
+
|
| 5 |
+
from lm_eval.utils import weighted_f1_score
|
| 6 |
+
|
| 7 |
+
|
| 8 |
+
def doc_to_target(doc):
|
| 9 |
+
pos_tag_map = {
|
| 10 |
+
0: "NOUN",
|
| 11 |
+
1: "PUNCT",
|
| 12 |
+
2: "ADP",
|
| 13 |
+
3: "NUM",
|
| 14 |
+
4: "SYM",
|
| 15 |
+
5: "SCONJ",
|
| 16 |
+
6: "ADJ",
|
| 17 |
+
7: "PART",
|
| 18 |
+
8: "DET",
|
| 19 |
+
9: "CCONJ",
|
| 20 |
+
10: "PROPN",
|
| 21 |
+
11: "PRON",
|
| 22 |
+
12: "X",
|
| 23 |
+
13: "_",
|
| 24 |
+
14: "ADV",
|
| 25 |
+
15: "INTJ",
|
| 26 |
+
16: "VERB",
|
| 27 |
+
17: "AUX",
|
| 28 |
+
}
|
| 29 |
+
return [pos_tag_map[tag] for tag in doc["upos"]]
|
| 30 |
+
|
| 31 |
+
|
| 32 |
+
def acc_score(items):
|
| 33 |
+
unzipped_list = list(zip(*items))
|
| 34 |
+
|
| 35 |
+
golds, preds = unzipped_list[0], unzipped_list[1]
|
| 36 |
+
|
| 37 |
+
# Flatten preds' inner lists
|
| 38 |
+
flattened_preds = [list(chain.from_iterable(p)) for p in preds]
|
| 39 |
+
|
| 40 |
+
# Calculate the accuracy for each gold-pred pair
|
| 41 |
+
accuracy_scores = []
|
| 42 |
+
for gold, pred in zip(golds, flattened_preds):
|
| 43 |
+
# Ensure both lists are of the same length, otherwise truncate to match
|
| 44 |
+
min_length = min(len(gold), len(pred))
|
| 45 |
+
gold = gold[:min_length]
|
| 46 |
+
pred = pred[:min_length]
|
| 47 |
+
|
| 48 |
+
# Calculate accuracy for the current pair and add to the list
|
| 49 |
+
accuracy = accuracy_score(gold, pred)
|
| 50 |
+
accuracy_scores.append(accuracy)
|
| 51 |
+
|
| 52 |
+
mean_accuracy = (
|
| 53 |
+
sum(accuracy_scores) / len(accuracy_scores) if accuracy_scores else 0
|
| 54 |
+
)
|
| 55 |
+
return mean_accuracy
|
lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_4/masakhapos_bam.yaml
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Generated by utils.py
|
| 2 |
+
dataset_name: bam
|
| 3 |
+
doc_to_text: "Annotate each word in the provided sentence with the appropriate POS\
|
| 4 |
+
\ tag. The annotation list is given as: ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET',\
|
| 5 |
+
\ 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM', 'VERB',\
|
| 6 |
+
\ 'X']. The input sentence will be a list of words in the sentence. The output format\
|
| 7 |
+
\ should be a list of tuples, where each tuple consists of a word from the input\
|
| 8 |
+
\ text and its corresponding POS tag label from the POS tag label set provided\n\
|
| 9 |
+
Your response should include only a list of tuples, in the order that the words\
|
| 10 |
+
\ appear in the input sentence, including punctuations, with each tuple containing\
|
| 11 |
+
\ the corresponding POS tag label for a word. \n\nSentence: {{tokens}} \nOutput: "
|
| 12 |
+
include: masakhapos_yaml
|
| 13 |
+
task: masakhapos_bam_prompt_4
|
lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_4/masakhapos_ewe.yaml
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Generated by utils.py
|
| 2 |
+
dataset_name: ewe
|
| 3 |
+
doc_to_text: "Annotate each word in the provided sentence with the appropriate POS\
|
| 4 |
+
\ tag. The annotation list is given as: ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET',\
|
| 5 |
+
\ 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM', 'VERB',\
|
| 6 |
+
\ 'X']. The input sentence will be a list of words in the sentence. The output format\
|
| 7 |
+
\ should be a list of tuples, where each tuple consists of a word from the input\
|
| 8 |
+
\ text and its corresponding POS tag label from the POS tag label set provided\n\
|
| 9 |
+
Your response should include only a list of tuples, in the order that the words\
|
| 10 |
+
\ appear in the input sentence, including punctuations, with each tuple containing\
|
| 11 |
+
\ the corresponding POS tag label for a word. \n\nSentence: {{tokens}} \nOutput: "
|
| 12 |
+
include: masakhapos_yaml
|
| 13 |
+
task: masakhapos_ewe_prompt_4
|
lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_4/masakhapos_fon.yaml
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Generated by utils.py
|
| 2 |
+
dataset_name: fon
|
| 3 |
+
doc_to_text: "Annotate each word in the provided sentence with the appropriate POS\
|
| 4 |
+
\ tag. The annotation list is given as: ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET',\
|
| 5 |
+
\ 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM', 'VERB',\
|
| 6 |
+
\ 'X']. The input sentence will be a list of words in the sentence. The output format\
|
| 7 |
+
\ should be a list of tuples, where each tuple consists of a word from the input\
|
| 8 |
+
\ text and its corresponding POS tag label from the POS tag label set provided\n\
|
| 9 |
+
Your response should include only a list of tuples, in the order that the words\
|
| 10 |
+
\ appear in the input sentence, including punctuations, with each tuple containing\
|
| 11 |
+
\ the corresponding POS tag label for a word. \n\nSentence: {{tokens}} \nOutput: "
|
| 12 |
+
include: masakhapos_yaml
|
| 13 |
+
task: masakhapos_fon_prompt_4
|
lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_4/masakhapos_hau.yaml
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Generated by utils.py
|
| 2 |
+
dataset_name: hau
|
| 3 |
+
doc_to_text: "Annotate each word in the provided sentence with the appropriate POS\
|
| 4 |
+
\ tag. The annotation list is given as: ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET',\
|
| 5 |
+
\ 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM', 'VERB',\
|
| 6 |
+
\ 'X']. The input sentence will be a list of words in the sentence. The output format\
|
| 7 |
+
\ should be a list of tuples, where each tuple consists of a word from the input\
|
| 8 |
+
\ text and its corresponding POS tag label from the POS tag label set provided\n\
|
| 9 |
+
Your response should include only a list of tuples, in the order that the words\
|
| 10 |
+
\ appear in the input sentence, including punctuations, with each tuple containing\
|
| 11 |
+
\ the corresponding POS tag label for a word. \n\nSentence: {{tokens}} \nOutput: "
|
| 12 |
+
include: masakhapos_yaml
|
| 13 |
+
task: masakhapos_hau_prompt_4
|
lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_4/masakhapos_ibo.yaml
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Generated by utils.py
|
| 2 |
+
dataset_name: ibo
|
| 3 |
+
doc_to_text: "Annotate each word in the provided sentence with the appropriate POS\
|
| 4 |
+
\ tag. The annotation list is given as: ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET',\
|
| 5 |
+
\ 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM', 'VERB',\
|
| 6 |
+
\ 'X']. The input sentence will be a list of words in the sentence. The output format\
|
| 7 |
+
\ should be a list of tuples, where each tuple consists of a word from the input\
|
| 8 |
+
\ text and its corresponding POS tag label from the POS tag label set provided\n\
|
| 9 |
+
Your response should include only a list of tuples, in the order that the words\
|
| 10 |
+
\ appear in the input sentence, including punctuations, with each tuple containing\
|
| 11 |
+
\ the corresponding POS tag label for a word. \n\nSentence: {{tokens}} \nOutput: "
|
| 12 |
+
include: masakhapos_yaml
|
| 13 |
+
task: masakhapos_ibo_prompt_4
|
lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_4/masakhapos_kin.yaml
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Generated by utils.py
|
| 2 |
+
dataset_name: kin
|
| 3 |
+
doc_to_text: "Annotate each word in the provided sentence with the appropriate POS\
|
| 4 |
+
\ tag. The annotation list is given as: ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET',\
|
| 5 |
+
\ 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM', 'VERB',\
|
| 6 |
+
\ 'X']. The input sentence will be a list of words in the sentence. The output format\
|
| 7 |
+
\ should be a list of tuples, where each tuple consists of a word from the input\
|
| 8 |
+
\ text and its corresponding POS tag label from the POS tag label set provided\n\
|
| 9 |
+
Your response should include only a list of tuples, in the order that the words\
|
| 10 |
+
\ appear in the input sentence, including punctuations, with each tuple containing\
|
| 11 |
+
\ the corresponding POS tag label for a word. \n\nSentence: {{tokens}} \nOutput: "
|
| 12 |
+
include: masakhapos_yaml
|
| 13 |
+
task: masakhapos_kin_prompt_4
|
lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_4/masakhapos_lug.yaml
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Generated by utils.py
|
| 2 |
+
dataset_name: lug
|
| 3 |
+
doc_to_text: "Annotate each word in the provided sentence with the appropriate POS\
|
| 4 |
+
\ tag. The annotation list is given as: ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET',\
|
| 5 |
+
\ 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM', 'VERB',\
|
| 6 |
+
\ 'X']. The input sentence will be a list of words in the sentence. The output format\
|
| 7 |
+
\ should be a list of tuples, where each tuple consists of a word from the input\
|
| 8 |
+
\ text and its corresponding POS tag label from the POS tag label set provided\n\
|
| 9 |
+
Your response should include only a list of tuples, in the order that the words\
|
| 10 |
+
\ appear in the input sentence, including punctuations, with each tuple containing\
|
| 11 |
+
\ the corresponding POS tag label for a word. \n\nSentence: {{tokens}} \nOutput: "
|
| 12 |
+
include: masakhapos_yaml
|
| 13 |
+
task: masakhapos_lug_prompt_4
|
lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_4/masakhapos_luo.yaml
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Generated by utils.py
|
| 2 |
+
dataset_name: luo
|
| 3 |
+
doc_to_text: "Annotate each word in the provided sentence with the appropriate POS\
|
| 4 |
+
\ tag. The annotation list is given as: ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET',\
|
| 5 |
+
\ 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM', 'VERB',\
|
| 6 |
+
\ 'X']. The input sentence will be a list of words in the sentence. The output format\
|
| 7 |
+
\ should be a list of tuples, where each tuple consists of a word from the input\
|
| 8 |
+
\ text and its corresponding POS tag label from the POS tag label set provided\n\
|
| 9 |
+
Your response should include only a list of tuples, in the order that the words\
|
| 10 |
+
\ appear in the input sentence, including punctuations, with each tuple containing\
|
| 11 |
+
\ the corresponding POS tag label for a word. \n\nSentence: {{tokens}} \nOutput: "
|
| 12 |
+
include: masakhapos_yaml
|
| 13 |
+
task: masakhapos_luo_prompt_4
|
lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_4/masakhapos_mos.yaml
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Generated by utils.py
|
| 2 |
+
dataset_name: mos
|
| 3 |
+
doc_to_text: "Annotate each word in the provided sentence with the appropriate POS\
|
| 4 |
+
\ tag. The annotation list is given as: ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET',\
|
| 5 |
+
\ 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM', 'VERB',\
|
| 6 |
+
\ 'X']. The input sentence will be a list of words in the sentence. The output format\
|
| 7 |
+
\ should be a list of tuples, where each tuple consists of a word from the input\
|
| 8 |
+
\ text and its corresponding POS tag label from the POS tag label set provided\n\
|
| 9 |
+
Your response should include only a list of tuples, in the order that the words\
|
| 10 |
+
\ appear in the input sentence, including punctuations, with each tuple containing\
|
| 11 |
+
\ the corresponding POS tag label for a word. \n\nSentence: {{tokens}} \nOutput: "
|
| 12 |
+
include: masakhapos_yaml
|
| 13 |
+
task: masakhapos_mos_prompt_4
|
lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_4/masakhapos_nya.yaml
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Generated by utils.py
|
| 2 |
+
dataset_name: nya
|
| 3 |
+
doc_to_text: "Annotate each word in the provided sentence with the appropriate POS\
|
| 4 |
+
\ tag. The annotation list is given as: ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET',\
|
| 5 |
+
\ 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM', 'VERB',\
|
| 6 |
+
\ 'X']. The input sentence will be a list of words in the sentence. The output format\
|
| 7 |
+
\ should be a list of tuples, where each tuple consists of a word from the input\
|
| 8 |
+
\ text and its corresponding POS tag label from the POS tag label set provided\n\
|
| 9 |
+
Your response should include only a list of tuples, in the order that the words\
|
| 10 |
+
\ appear in the input sentence, including punctuations, with each tuple containing\
|
| 11 |
+
\ the corresponding POS tag label for a word. \n\nSentence: {{tokens}} \nOutput: "
|
| 12 |
+
include: masakhapos_yaml
|
| 13 |
+
task: masakhapos_nya_prompt_4
|
lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_4/masakhapos_pcm.yaml
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Generated by utils.py
|
| 2 |
+
dataset_name: pcm
|
| 3 |
+
doc_to_text: "Annotate each word in the provided sentence with the appropriate POS\
|
| 4 |
+
\ tag. The annotation list is given as: ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET',\
|
| 5 |
+
\ 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM', 'VERB',\
|
| 6 |
+
\ 'X']. The input sentence will be a list of words in the sentence. The output format\
|
| 7 |
+
\ should be a list of tuples, where each tuple consists of a word from the input\
|
| 8 |
+
\ text and its corresponding POS tag label from the POS tag label set provided\n\
|
| 9 |
+
Your response should include only a list of tuples, in the order that the words\
|
| 10 |
+
\ appear in the input sentence, including punctuations, with each tuple containing\
|
| 11 |
+
\ the corresponding POS tag label for a word. \n\nSentence: {{tokens}} \nOutput: "
|
| 12 |
+
include: masakhapos_yaml
|
| 13 |
+
task: masakhapos_pcm_prompt_4
|
lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_4/masakhapos_sna.yaml
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Generated by utils.py
|
| 2 |
+
dataset_name: sna
|
| 3 |
+
doc_to_text: "Annotate each word in the provided sentence with the appropriate POS\
|
| 4 |
+
\ tag. The annotation list is given as: ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET',\
|
| 5 |
+
\ 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM', 'VERB',\
|
| 6 |
+
\ 'X']. The input sentence will be a list of words in the sentence. The output format\
|
| 7 |
+
\ should be a list of tuples, where each tuple consists of a word from the input\
|
| 8 |
+
\ text and its corresponding POS tag label from the POS tag label set provided\n\
|
| 9 |
+
Your response should include only a list of tuples, in the order that the words\
|
| 10 |
+
\ appear in the input sentence, including punctuations, with each tuple containing\
|
| 11 |
+
\ the corresponding POS tag label for a word. \n\nSentence: {{tokens}} \nOutput: "
|
| 12 |
+
include: masakhapos_yaml
|
| 13 |
+
task: masakhapos_sna_prompt_4
|
lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_4/masakhapos_swa.yaml
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Generated by utils.py
|
| 2 |
+
dataset_name: swa
|
| 3 |
+
doc_to_text: "Annotate each word in the provided sentence with the appropriate POS\
|
| 4 |
+
\ tag. The annotation list is given as: ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET',\
|
| 5 |
+
\ 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM', 'VERB',\
|
| 6 |
+
\ 'X']. The input sentence will be a list of words in the sentence. The output format\
|
| 7 |
+
\ should be a list of tuples, where each tuple consists of a word from the input\
|
| 8 |
+
\ text and its corresponding POS tag label from the POS tag label set provided\n\
|
| 9 |
+
Your response should include only a list of tuples, in the order that the words\
|
| 10 |
+
\ appear in the input sentence, including punctuations, with each tuple containing\
|
| 11 |
+
\ the corresponding POS tag label for a word. \n\nSentence: {{tokens}} \nOutput: "
|
| 12 |
+
include: masakhapos_yaml
|
| 13 |
+
task: masakhapos_swa_prompt_4
|
lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_4/masakhapos_tsn.yaml
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Generated by utils.py
|
| 2 |
+
dataset_name: tsn
|
| 3 |
+
doc_to_text: "Annotate each word in the provided sentence with the appropriate POS\
|
| 4 |
+
\ tag. The annotation list is given as: ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET',\
|
| 5 |
+
\ 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM', 'VERB',\
|
| 6 |
+
\ 'X']. The input sentence will be a list of words in the sentence. The output format\
|
| 7 |
+
\ should be a list of tuples, where each tuple consists of a word from the input\
|
| 8 |
+
\ text and its corresponding POS tag label from the POS tag label set provided\n\
|
| 9 |
+
Your response should include only a list of tuples, in the order that the words\
|
| 10 |
+
\ appear in the input sentence, including punctuations, with each tuple containing\
|
| 11 |
+
\ the corresponding POS tag label for a word. \n\nSentence: {{tokens}} \nOutput: "
|
| 12 |
+
include: masakhapos_yaml
|
| 13 |
+
task: masakhapos_tsn_prompt_4
|
lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_4/masakhapos_twi.yaml
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Generated by utils.py
|
| 2 |
+
dataset_name: twi
|
| 3 |
+
doc_to_text: "Annotate each word in the provided sentence with the appropriate POS\
|
| 4 |
+
\ tag. The annotation list is given as: ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET',\
|
| 5 |
+
\ 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM', 'VERB',\
|
| 6 |
+
\ 'X']. The input sentence will be a list of words in the sentence. The output format\
|
| 7 |
+
\ should be a list of tuples, where each tuple consists of a word from the input\
|
| 8 |
+
\ text and its corresponding POS tag label from the POS tag label set provided\n\
|
| 9 |
+
Your response should include only a list of tuples, in the order that the words\
|
| 10 |
+
\ appear in the input sentence, including punctuations, with each tuple containing\
|
| 11 |
+
\ the corresponding POS tag label for a word. \n\nSentence: {{tokens}} \nOutput: "
|
| 12 |
+
include: masakhapos_yaml
|
| 13 |
+
task: masakhapos_twi_prompt_4
|
lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_4/masakhapos_wol.yaml
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Generated by utils.py
|
| 2 |
+
dataset_name: wol
|
| 3 |
+
doc_to_text: "Annotate each word in the provided sentence with the appropriate POS\
|
| 4 |
+
\ tag. The annotation list is given as: ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET',\
|
| 5 |
+
\ 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM', 'VERB',\
|
| 6 |
+
\ 'X']. The input sentence will be a list of words in the sentence. The output format\
|
| 7 |
+
\ should be a list of tuples, where each tuple consists of a word from the input\
|
| 8 |
+
\ text and its corresponding POS tag label from the POS tag label set provided\n\
|
| 9 |
+
Your response should include only a list of tuples, in the order that the words\
|
| 10 |
+
\ appear in the input sentence, including punctuations, with each tuple containing\
|
| 11 |
+
\ the corresponding POS tag label for a word. \n\nSentence: {{tokens}} \nOutput: "
|
| 12 |
+
include: masakhapos_yaml
|
| 13 |
+
task: masakhapos_wol_prompt_4
|
lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_4/masakhapos_xho.yaml
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Generated by utils.py
|
| 2 |
+
dataset_name: xho
|
| 3 |
+
doc_to_text: "Annotate each word in the provided sentence with the appropriate POS\
|
| 4 |
+
\ tag. The annotation list is given as: ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET',\
|
| 5 |
+
\ 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM', 'VERB',\
|
| 6 |
+
\ 'X']. The input sentence will be a list of words in the sentence. The output format\
|
| 7 |
+
\ should be a list of tuples, where each tuple consists of a word from the input\
|
| 8 |
+
\ text and its corresponding POS tag label from the POS tag label set provided\n\
|
| 9 |
+
Your response should include only a list of tuples, in the order that the words\
|
| 10 |
+
\ appear in the input sentence, including punctuations, with each tuple containing\
|
| 11 |
+
\ the corresponding POS tag label for a word. \n\nSentence: {{tokens}} \nOutput: "
|
| 12 |
+
include: masakhapos_yaml
|
| 13 |
+
task: masakhapos_xho_prompt_4
|
lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_4/masakhapos_yaml
ADDED
|
@@ -0,0 +1,32 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
tag:
|
| 2 |
+
- masakhapos_tasks
|
| 3 |
+
- masakhapos_prompt_4
|
| 4 |
+
dataset_path: masakhane/masakhapos
|
| 5 |
+
dataset_name: null
|
| 6 |
+
dataset_kwargs: {trust_remote_code: True}
|
| 7 |
+
output_type: generate_until
|
| 8 |
+
generation_kwargs:
|
| 9 |
+
do_sample: false
|
| 10 |
+
until:
|
| 11 |
+
- </s>
|
| 12 |
+
- <|im_end|>
|
| 13 |
+
validation_split: validation
|
| 14 |
+
test_split: test
|
| 15 |
+
fewshot_split: train
|
| 16 |
+
doc_to_target: !function utils.doc_to_target
|
| 17 |
+
should_decontaminate: true
|
| 18 |
+
doc_to_decontamination_query: "Sentence: {{token}}\nOutput:"
|
| 19 |
+
filter_list:
|
| 20 |
+
- filter:
|
| 21 |
+
- function: regex_pos
|
| 22 |
+
name: flexible-extract
|
| 23 |
+
metric_list:
|
| 24 |
+
- metric: acc
|
| 25 |
+
aggregation: !function utils.acc_score
|
| 26 |
+
higher_is_better: true
|
| 27 |
+
ignore_case: true
|
| 28 |
+
ignore_punctuation: true
|
| 29 |
+
regexes_to_ignore:
|
| 30 |
+
- ","
|
| 31 |
+
metadata:
|
| 32 |
+
version: 1.0
|
lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_4/masakhapos_yor.yaml
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Generated by utils.py
|
| 2 |
+
dataset_name: yor
|
| 3 |
+
doc_to_text: "Annotate each word in the provided sentence with the appropriate POS\
|
| 4 |
+
\ tag. The annotation list is given as: ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET',\
|
| 5 |
+
\ 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM', 'VERB',\
|
| 6 |
+
\ 'X']. The input sentence will be a list of words in the sentence. The output format\
|
| 7 |
+
\ should be a list of tuples, where each tuple consists of a word from the input\
|
| 8 |
+
\ text and its corresponding POS tag label from the POS tag label set provided\n\
|
| 9 |
+
Your response should include only a list of tuples, in the order that the words\
|
| 10 |
+
\ appear in the input sentence, including punctuations, with each tuple containing\
|
| 11 |
+
\ the corresponding POS tag label for a word. \n\nSentence: {{tokens}} \nOutput: "
|
| 12 |
+
include: masakhapos_yaml
|
| 13 |
+
task: masakhapos_yor_prompt_4
|
lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_4/masakhapos_zul.yaml
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Generated by utils.py
|
| 2 |
+
dataset_name: zul
|
| 3 |
+
doc_to_text: "Annotate each word in the provided sentence with the appropriate POS\
|
| 4 |
+
\ tag. The annotation list is given as: ['ADJ', 'ADP', 'ADV', 'AUX', 'CCONJ, 'DET',\
|
| 5 |
+
\ 'INTJ', 'NOUN', 'NUM', 'PART', 'PRON', 'PROPN', 'PUNCT', 'SCONJ', 'SYM', 'VERB',\
|
| 6 |
+
\ 'X']. The input sentence will be a list of words in the sentence. The output format\
|
| 7 |
+
\ should be a list of tuples, where each tuple consists of a word from the input\
|
| 8 |
+
\ text and its corresponding POS tag label from the POS tag label set provided\n\
|
| 9 |
+
Your response should include only a list of tuples, in the order that the words\
|
| 10 |
+
\ appear in the input sentence, including punctuations, with each tuple containing\
|
| 11 |
+
\ the corresponding POS tag label for a word. \n\nSentence: {{tokens}} \nOutput: "
|
| 12 |
+
include: masakhapos_yaml
|
| 13 |
+
task: masakhapos_zul_prompt_4
|
lm-evaluation-harness/lm_eval/tasks/afrobench/masakhapos/prompt_4/utils.py
ADDED
|
@@ -0,0 +1,55 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from itertools import chain
|
| 2 |
+
|
| 3 |
+
from sklearn.metrics import accuracy_score
|
| 4 |
+
|
| 5 |
+
from lm_eval.utils import weighted_f1_score
|
| 6 |
+
|
| 7 |
+
|
| 8 |
+
def doc_to_target(doc):
|
| 9 |
+
pos_tag_map = {
|
| 10 |
+
0: "NOUN",
|
| 11 |
+
1: "PUNCT",
|
| 12 |
+
2: "ADP",
|
| 13 |
+
3: "NUM",
|
| 14 |
+
4: "SYM",
|
| 15 |
+
5: "SCONJ",
|
| 16 |
+
6: "ADJ",
|
| 17 |
+
7: "PART",
|
| 18 |
+
8: "DET",
|
| 19 |
+
9: "CCONJ",
|
| 20 |
+
10: "PROPN",
|
| 21 |
+
11: "PRON",
|
| 22 |
+
12: "X",
|
| 23 |
+
13: "_",
|
| 24 |
+
14: "ADV",
|
| 25 |
+
15: "INTJ",
|
| 26 |
+
16: "VERB",
|
| 27 |
+
17: "AUX",
|
| 28 |
+
}
|
| 29 |
+
return [pos_tag_map[tag] for tag in doc["upos"]]
|
| 30 |
+
|
| 31 |
+
|
| 32 |
+
def acc_score(items):
|
| 33 |
+
unzipped_list = list(zip(*items))
|
| 34 |
+
|
| 35 |
+
golds, preds = unzipped_list[0], unzipped_list[1]
|
| 36 |
+
|
| 37 |
+
# Flatten preds' inner lists
|
| 38 |
+
flattened_preds = [list(chain.from_iterable(p)) for p in preds]
|
| 39 |
+
|
| 40 |
+
# Calculate the accuracy for each gold-pred pair
|
| 41 |
+
accuracy_scores = []
|
| 42 |
+
for gold, pred in zip(golds, flattened_preds):
|
| 43 |
+
# Ensure both lists are of the same length, otherwise truncate to match
|
| 44 |
+
min_length = min(len(gold), len(pred))
|
| 45 |
+
gold = gold[:min_length]
|
| 46 |
+
pred = pred[:min_length]
|
| 47 |
+
|
| 48 |
+
# Calculate accuracy for the current pair and add to the list
|
| 49 |
+
accuracy = accuracy_score(gold, pred)
|
| 50 |
+
accuracy_scores.append(accuracy)
|
| 51 |
+
|
| 52 |
+
mean_accuracy = (
|
| 53 |
+
sum(accuracy_scores) / len(accuracy_scores) if accuracy_scores else 0
|
| 54 |
+
)
|
| 55 |
+
return mean_accuracy
|