| { |
| "results": { |
| "blimp": { |
| "acc,none": 0.7272835820895525, |
| "acc_stderr,none": 0.19867610993066823, |
| "alias": "blimp" |
| }, |
| "blimp_adjunct_island": { |
| "acc,none": 0.753, |
| "acc_stderr,none": 0.013644675781314128, |
| "alias": " - blimp_adjunct_island" |
| }, |
| "blimp_anaphor_gender_agreement": { |
| "acc,none": 0.803, |
| "acc_stderr,none": 0.012583693787968118, |
| "alias": " - blimp_anaphor_gender_agreement" |
| }, |
| "blimp_anaphor_number_agreement": { |
| "acc,none": 0.939, |
| "acc_stderr,none": 0.007572076091557415, |
| "alias": " - blimp_anaphor_number_agreement" |
| }, |
| "blimp_animate_subject_passive": { |
| "acc,none": 0.782, |
| "acc_stderr,none": 0.013063179040595277, |
| "alias": " - blimp_animate_subject_passive" |
| }, |
| "blimp_animate_subject_trans": { |
| "acc,none": 0.816, |
| "acc_stderr,none": 0.012259457340938595, |
| "alias": " - blimp_animate_subject_trans" |
| }, |
| "blimp_causative": { |
| "acc,none": 0.705, |
| "acc_stderr,none": 0.014428554438445514, |
| "alias": " - blimp_causative" |
| }, |
| "blimp_complex_NP_island": { |
| "acc,none": 0.556, |
| "acc_stderr,none": 0.01571976816340209, |
| "alias": " - blimp_complex_NP_island" |
| }, |
| "blimp_coordinate_structure_constraint_complex_left_branch": { |
| "acc,none": 0.487, |
| "acc_stderr,none": 0.01581395210189663, |
| "alias": " - blimp_coordinate_structure_constraint_complex_left_branch" |
| }, |
| "blimp_coordinate_structure_constraint_object_extraction": { |
| "acc,none": 0.695, |
| "acc_stderr,none": 0.014566646394664384, |
| "alias": " - blimp_coordinate_structure_constraint_object_extraction" |
| }, |
| "blimp_determiner_noun_agreement_1": { |
| "acc,none": 0.99, |
| "acc_stderr,none": 0.003148000938676773, |
| "alias": " - blimp_determiner_noun_agreement_1" |
| }, |
| "blimp_determiner_noun_agreement_2": { |
| "acc,none": 0.986, |
| "acc_stderr,none": 0.0037172325482565955, |
| "alias": " - blimp_determiner_noun_agreement_2" |
| }, |
| "blimp_determiner_noun_agreement_irregular_1": { |
| "acc,none": 0.896, |
| "acc_stderr,none": 0.0096580162185243, |
| "alias": " - blimp_determiner_noun_agreement_irregular_1" |
| }, |
| "blimp_determiner_noun_agreement_irregular_2": { |
| "acc,none": 0.942, |
| "acc_stderr,none": 0.007395315455792957, |
| "alias": " - blimp_determiner_noun_agreement_irregular_2" |
| }, |
| "blimp_determiner_noun_agreement_with_adj_2": { |
| "acc,none": 0.946, |
| "acc_stderr,none": 0.0071508835212954446, |
| "alias": " - blimp_determiner_noun_agreement_with_adj_2" |
| }, |
| "blimp_determiner_noun_agreement_with_adj_irregular_1": { |
| "acc,none": 0.852, |
| "acc_stderr,none": 0.011234866364235261, |
| "alias": " - blimp_determiner_noun_agreement_with_adj_irregular_1" |
| }, |
| "blimp_determiner_noun_agreement_with_adj_irregular_2": { |
| "acc,none": 0.902, |
| "acc_stderr,none": 0.009406619184621231, |
| "alias": " - blimp_determiner_noun_agreement_with_adj_irregular_2" |
| }, |
| "blimp_determiner_noun_agreement_with_adjective_1": { |
| "acc,none": 0.953, |
| "acc_stderr,none": 0.006695956678163046, |
| "alias": " - blimp_determiner_noun_agreement_with_adjective_1" |
| }, |
| "blimp_distractor_agreement_relational_noun": { |
| "acc,none": 0.571, |
| "acc_stderr,none": 0.015658997547870247, |
| "alias": " - blimp_distractor_agreement_relational_noun" |
| }, |
| "blimp_distractor_agreement_relative_clause": { |
| "acc,none": 0.51, |
| "acc_stderr,none": 0.015816135752773207, |
| "alias": " - blimp_distractor_agreement_relative_clause" |
| }, |
| "blimp_drop_argument": { |
| "acc,none": 0.78, |
| "acc_stderr,none": 0.013106173040661768, |
| "alias": " - blimp_drop_argument" |
| }, |
| "blimp_ellipsis_n_bar_1": { |
| "acc,none": 0.7, |
| "acc_stderr,none": 0.014498627873361432, |
| "alias": " - blimp_ellipsis_n_bar_1" |
| }, |
| "blimp_ellipsis_n_bar_2": { |
| "acc,none": 0.645, |
| "acc_stderr,none": 0.01513949154378053, |
| "alias": " - blimp_ellipsis_n_bar_2" |
| }, |
| "blimp_existential_there_object_raising": { |
| "acc,none": 0.66, |
| "acc_stderr,none": 0.014987482264363935, |
| "alias": " - blimp_existential_there_object_raising" |
| }, |
| "blimp_existential_there_quantifiers_1": { |
| "acc,none": 0.939, |
| "acc_stderr,none": 0.007572076091557425, |
| "alias": " - blimp_existential_there_quantifiers_1" |
| }, |
| "blimp_existential_there_quantifiers_2": { |
| "acc,none": 0.545, |
| "acc_stderr,none": 0.01575510149834709, |
| "alias": " - blimp_existential_there_quantifiers_2" |
| }, |
| "blimp_existential_there_subject_raising": { |
| "acc,none": 0.836, |
| "acc_stderr,none": 0.011715000693181318, |
| "alias": " - blimp_existential_there_subject_raising" |
| }, |
| "blimp_expletive_it_object_raising": { |
| "acc,none": 0.714, |
| "acc_stderr,none": 0.01429714686251791, |
| "alias": " - blimp_expletive_it_object_raising" |
| }, |
| "blimp_inchoative": { |
| "acc,none": 0.593, |
| "acc_stderr,none": 0.015543249100255545, |
| "alias": " - blimp_inchoative" |
| }, |
| "blimp_intransitive": { |
| "acc,none": 0.752, |
| "acc_stderr,none": 0.013663187134877639, |
| "alias": " - blimp_intransitive" |
| }, |
| "blimp_irregular_past_participle_adjectives": { |
| "acc,none": 0.962, |
| "acc_stderr,none": 0.0060491811505849384, |
| "alias": " - blimp_irregular_past_participle_adjectives" |
| }, |
| "blimp_irregular_past_participle_verbs": { |
| "acc,none": 0.85, |
| "acc_stderr,none": 0.0112972398234093, |
| "alias": " - blimp_irregular_past_participle_verbs" |
| }, |
| "blimp_irregular_plural_subject_verb_agreement_1": { |
| "acc,none": 0.808, |
| "acc_stderr,none": 0.012461592646659976, |
| "alias": " - blimp_irregular_plural_subject_verb_agreement_1" |
| }, |
| "blimp_irregular_plural_subject_verb_agreement_2": { |
| "acc,none": 0.766, |
| "acc_stderr,none": 0.01339490288966001, |
| "alias": " - blimp_irregular_plural_subject_verb_agreement_2" |
| }, |
| "blimp_left_branch_island_echo_question": { |
| "acc,none": 0.635, |
| "acc_stderr,none": 0.015231776226264905, |
| "alias": " - blimp_left_branch_island_echo_question" |
| }, |
| "blimp_left_branch_island_simple_question": { |
| "acc,none": 0.543, |
| "acc_stderr,none": 0.015760691590136378, |
| "alias": " - blimp_left_branch_island_simple_question" |
| }, |
| "blimp_matrix_question_npi_licensor_present": { |
| "acc,none": 0.103, |
| "acc_stderr,none": 0.009616833339695796, |
| "alias": " - blimp_matrix_question_npi_licensor_present" |
| }, |
| "blimp_npi_present_1": { |
| "acc,none": 0.545, |
| "acc_stderr,none": 0.01575510149834709, |
| "alias": " - blimp_npi_present_1" |
| }, |
| "blimp_npi_present_2": { |
| "acc,none": 0.572, |
| "acc_stderr,none": 0.01565442624502928, |
| "alias": " - blimp_npi_present_2" |
| }, |
| "blimp_only_npi_licensor_present": { |
| "acc,none": 0.979, |
| "acc_stderr,none": 0.004536472151306522, |
| "alias": " - blimp_only_npi_licensor_present" |
| }, |
| "blimp_only_npi_scope": { |
| "acc,none": 0.482, |
| "acc_stderr,none": 0.015809045699406728, |
| "alias": " - blimp_only_npi_scope" |
| }, |
| "blimp_passive_1": { |
| "acc,none": 0.848, |
| "acc_stderr,none": 0.01135891830347528, |
| "alias": " - blimp_passive_1" |
| }, |
| "blimp_passive_2": { |
| "acc,none": 0.87, |
| "acc_stderr,none": 0.010640169792499364, |
| "alias": " - blimp_passive_2" |
| }, |
| "blimp_principle_A_c_command": { |
| "acc,none": 0.563, |
| "acc_stderr,none": 0.01569322392873038, |
| "alias": " - blimp_principle_A_c_command" |
| }, |
| "blimp_principle_A_case_1": { |
| "acc,none": 1.0, |
| "acc_stderr,none": 0.0, |
| "alias": " - blimp_principle_A_case_1" |
| }, |
| "blimp_principle_A_case_2": { |
| "acc,none": 0.908, |
| "acc_stderr,none": 0.009144376393151103, |
| "alias": " - blimp_principle_A_case_2" |
| }, |
| "blimp_principle_A_domain_1": { |
| "acc,none": 0.986, |
| "acc_stderr,none": 0.0037172325482565964, |
| "alias": " - blimp_principle_A_domain_1" |
| }, |
| "blimp_principle_A_domain_2": { |
| "acc,none": 0.758, |
| "acc_stderr,none": 0.01355063170555596, |
| "alias": " - blimp_principle_A_domain_2" |
| }, |
| "blimp_principle_A_domain_3": { |
| "acc,none": 0.559, |
| "acc_stderr,none": 0.01570877989424268, |
| "alias": " - blimp_principle_A_domain_3" |
| }, |
| "blimp_principle_A_reconstruction": { |
| "acc,none": 0.218, |
| "acc_stderr,none": 0.013063179040595283, |
| "alias": " - blimp_principle_A_reconstruction" |
| }, |
| "blimp_regular_plural_subject_verb_agreement_1": { |
| "acc,none": 0.871, |
| "acc_stderr,none": 0.010605256784796594, |
| "alias": " - blimp_regular_plural_subject_verb_agreement_1" |
| }, |
| "blimp_regular_plural_subject_verb_agreement_2": { |
| "acc,none": 0.755, |
| "acc_stderr,none": 0.01360735683959812, |
| "alias": " - blimp_regular_plural_subject_verb_agreement_2" |
| }, |
| "blimp_sentential_negation_npi_licensor_present": { |
| "acc,none": 0.982, |
| "acc_stderr,none": 0.004206387249611467, |
| "alias": " - blimp_sentential_negation_npi_licensor_present" |
| }, |
| "blimp_sentential_negation_npi_scope": { |
| "acc,none": 0.448, |
| "acc_stderr,none": 0.015733516566347833, |
| "alias": " - blimp_sentential_negation_npi_scope" |
| }, |
| "blimp_sentential_subject_island": { |
| "acc,none": 0.339, |
| "acc_stderr,none": 0.014976758771620342, |
| "alias": " - blimp_sentential_subject_island" |
| }, |
| "blimp_superlative_quantifiers_1": { |
| "acc,none": 0.616, |
| "acc_stderr,none": 0.015387682761897066, |
| "alias": " - blimp_superlative_quantifiers_1" |
| }, |
| "blimp_superlative_quantifiers_2": { |
| "acc,none": 0.773, |
| "acc_stderr,none": 0.01325317496476391, |
| "alias": " - blimp_superlative_quantifiers_2" |
| }, |
| "blimp_tough_vs_raising_1": { |
| "acc,none": 0.551, |
| "acc_stderr,none": 0.015736792768752023, |
| "alias": " - blimp_tough_vs_raising_1" |
| }, |
| "blimp_tough_vs_raising_2": { |
| "acc,none": 0.785, |
| "acc_stderr,none": 0.01299784381903181, |
| "alias": " - blimp_tough_vs_raising_2" |
| }, |
| "blimp_transitive": { |
| "acc,none": 0.792, |
| "acc_stderr,none": 0.01284137457209693, |
| "alias": " - blimp_transitive" |
| }, |
| "blimp_wh_island": { |
| "acc,none": 0.619, |
| "acc_stderr,none": 0.015364734787007436, |
| "alias": " - blimp_wh_island" |
| }, |
| "blimp_wh_questions_object_gap": { |
| "acc,none": 0.734, |
| "acc_stderr,none": 0.013979965645145151, |
| "alias": " - blimp_wh_questions_object_gap" |
| }, |
| "blimp_wh_questions_subject_gap": { |
| "acc,none": 0.917, |
| "acc_stderr,none": 0.008728527206074794, |
| "alias": " - blimp_wh_questions_subject_gap" |
| }, |
| "blimp_wh_questions_subject_gap_long_distance": { |
| "acc,none": 0.896, |
| "acc_stderr,none": 0.009658016218524298, |
| "alias": " - blimp_wh_questions_subject_gap_long_distance" |
| }, |
| "blimp_wh_vs_that_no_gap": { |
| "acc,none": 0.963, |
| "acc_stderr,none": 0.005972157622389644, |
| "alias": " - blimp_wh_vs_that_no_gap" |
| }, |
| "blimp_wh_vs_that_no_gap_long_distance": { |
| "acc,none": 0.982, |
| "acc_stderr,none": 0.004206387249611444, |
| "alias": " - blimp_wh_vs_that_no_gap_long_distance" |
| }, |
| "blimp_wh_vs_that_with_gap": { |
| "acc,none": 0.418, |
| "acc_stderr,none": 0.015605111967541944, |
| "alias": " - blimp_wh_vs_that_with_gap" |
| }, |
| "blimp_wh_vs_that_with_gap_long_distance": { |
| "acc,none": 0.084, |
| "acc_stderr,none": 0.008776162089491134, |
| "alias": " - blimp_wh_vs_that_with_gap_long_distance" |
| } |
| }, |
| "groups": { |
| "blimp": { |
| "acc,none": 0.7272835820895525, |
| "acc_stderr,none": 0.19867610993066823, |
| "alias": "blimp" |
| } |
| }, |
| "configs": { |
| "blimp_adjunct_island": { |
| "task": "blimp_adjunct_island", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "adjunct_island", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_anaphor_gender_agreement": { |
| "task": "blimp_anaphor_gender_agreement", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "anaphor_gender_agreement", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_anaphor_number_agreement": { |
| "task": "blimp_anaphor_number_agreement", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "anaphor_number_agreement", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_animate_subject_passive": { |
| "task": "blimp_animate_subject_passive", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "animate_subject_passive", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_animate_subject_trans": { |
| "task": "blimp_animate_subject_trans", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "animate_subject_trans", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_causative": { |
| "task": "blimp_causative", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "causative", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_complex_NP_island": { |
| "task": "blimp_complex_NP_island", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "complex_NP_island", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_coordinate_structure_constraint_complex_left_branch": { |
| "task": "blimp_coordinate_structure_constraint_complex_left_branch", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "coordinate_structure_constraint_complex_left_branch", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_coordinate_structure_constraint_object_extraction": { |
| "task": "blimp_coordinate_structure_constraint_object_extraction", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "coordinate_structure_constraint_object_extraction", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_determiner_noun_agreement_1": { |
| "task": "blimp_determiner_noun_agreement_1", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "determiner_noun_agreement_1", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_determiner_noun_agreement_2": { |
| "task": "blimp_determiner_noun_agreement_2", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "determiner_noun_agreement_2", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_determiner_noun_agreement_irregular_1": { |
| "task": "blimp_determiner_noun_agreement_irregular_1", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "determiner_noun_agreement_irregular_1", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_determiner_noun_agreement_irregular_2": { |
| "task": "blimp_determiner_noun_agreement_irregular_2", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "determiner_noun_agreement_irregular_2", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_determiner_noun_agreement_with_adj_2": { |
| "task": "blimp_determiner_noun_agreement_with_adj_2", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "determiner_noun_agreement_with_adj_2", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_determiner_noun_agreement_with_adj_irregular_1": { |
| "task": "blimp_determiner_noun_agreement_with_adj_irregular_1", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "determiner_noun_agreement_with_adj_irregular_1", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_determiner_noun_agreement_with_adj_irregular_2": { |
| "task": "blimp_determiner_noun_agreement_with_adj_irregular_2", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "determiner_noun_agreement_with_adj_irregular_2", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_determiner_noun_agreement_with_adjective_1": { |
| "task": "blimp_determiner_noun_agreement_with_adjective_1", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "determiner_noun_agreement_with_adjective_1", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_distractor_agreement_relational_noun": { |
| "task": "blimp_distractor_agreement_relational_noun", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "distractor_agreement_relational_noun", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_distractor_agreement_relative_clause": { |
| "task": "blimp_distractor_agreement_relative_clause", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "distractor_agreement_relative_clause", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_drop_argument": { |
| "task": "blimp_drop_argument", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "drop_argument", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_ellipsis_n_bar_1": { |
| "task": "blimp_ellipsis_n_bar_1", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "ellipsis_n_bar_1", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_ellipsis_n_bar_2": { |
| "task": "blimp_ellipsis_n_bar_2", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "ellipsis_n_bar_2", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_existential_there_object_raising": { |
| "task": "blimp_existential_there_object_raising", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "existential_there_object_raising", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_existential_there_quantifiers_1": { |
| "task": "blimp_existential_there_quantifiers_1", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "existential_there_quantifiers_1", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_existential_there_quantifiers_2": { |
| "task": "blimp_existential_there_quantifiers_2", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "existential_there_quantifiers_2", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_existential_there_subject_raising": { |
| "task": "blimp_existential_there_subject_raising", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "existential_there_subject_raising", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_expletive_it_object_raising": { |
| "task": "blimp_expletive_it_object_raising", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "expletive_it_object_raising", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_inchoative": { |
| "task": "blimp_inchoative", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "inchoative", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_intransitive": { |
| "task": "blimp_intransitive", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "intransitive", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_irregular_past_participle_adjectives": { |
| "task": "blimp_irregular_past_participle_adjectives", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "irregular_past_participle_adjectives", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_irregular_past_participle_verbs": { |
| "task": "blimp_irregular_past_participle_verbs", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "irregular_past_participle_verbs", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_irregular_plural_subject_verb_agreement_1": { |
| "task": "blimp_irregular_plural_subject_verb_agreement_1", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "irregular_plural_subject_verb_agreement_1", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_irregular_plural_subject_verb_agreement_2": { |
| "task": "blimp_irregular_plural_subject_verb_agreement_2", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "irregular_plural_subject_verb_agreement_2", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_left_branch_island_echo_question": { |
| "task": "blimp_left_branch_island_echo_question", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "left_branch_island_echo_question", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_left_branch_island_simple_question": { |
| "task": "blimp_left_branch_island_simple_question", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "left_branch_island_simple_question", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_matrix_question_npi_licensor_present": { |
| "task": "blimp_matrix_question_npi_licensor_present", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "matrix_question_npi_licensor_present", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_npi_present_1": { |
| "task": "blimp_npi_present_1", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "npi_present_1", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_npi_present_2": { |
| "task": "blimp_npi_present_2", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "npi_present_2", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_only_npi_licensor_present": { |
| "task": "blimp_only_npi_licensor_present", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "only_npi_licensor_present", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_only_npi_scope": { |
| "task": "blimp_only_npi_scope", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "only_npi_scope", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_passive_1": { |
| "task": "blimp_passive_1", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "passive_1", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_passive_2": { |
| "task": "blimp_passive_2", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "passive_2", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_principle_A_c_command": { |
| "task": "blimp_principle_A_c_command", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "principle_A_c_command", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_principle_A_case_1": { |
| "task": "blimp_principle_A_case_1", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "principle_A_case_1", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_principle_A_case_2": { |
| "task": "blimp_principle_A_case_2", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "principle_A_case_2", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_principle_A_domain_1": { |
| "task": "blimp_principle_A_domain_1", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "principle_A_domain_1", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_principle_A_domain_2": { |
| "task": "blimp_principle_A_domain_2", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "principle_A_domain_2", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_principle_A_domain_3": { |
| "task": "blimp_principle_A_domain_3", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "principle_A_domain_3", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_principle_A_reconstruction": { |
| "task": "blimp_principle_A_reconstruction", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "principle_A_reconstruction", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_regular_plural_subject_verb_agreement_1": { |
| "task": "blimp_regular_plural_subject_verb_agreement_1", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "regular_plural_subject_verb_agreement_1", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_regular_plural_subject_verb_agreement_2": { |
| "task": "blimp_regular_plural_subject_verb_agreement_2", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "regular_plural_subject_verb_agreement_2", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_sentential_negation_npi_licensor_present": { |
| "task": "blimp_sentential_negation_npi_licensor_present", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "sentential_negation_npi_licensor_present", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_sentential_negation_npi_scope": { |
| "task": "blimp_sentential_negation_npi_scope", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "sentential_negation_npi_scope", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_sentential_subject_island": { |
| "task": "blimp_sentential_subject_island", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "sentential_subject_island", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_superlative_quantifiers_1": { |
| "task": "blimp_superlative_quantifiers_1", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "superlative_quantifiers_1", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_superlative_quantifiers_2": { |
| "task": "blimp_superlative_quantifiers_2", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "superlative_quantifiers_2", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_tough_vs_raising_1": { |
| "task": "blimp_tough_vs_raising_1", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "tough_vs_raising_1", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_tough_vs_raising_2": { |
| "task": "blimp_tough_vs_raising_2", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "tough_vs_raising_2", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_transitive": { |
| "task": "blimp_transitive", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "transitive", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_wh_island": { |
| "task": "blimp_wh_island", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "wh_island", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_wh_questions_object_gap": { |
| "task": "blimp_wh_questions_object_gap", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "wh_questions_object_gap", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_wh_questions_subject_gap": { |
| "task": "blimp_wh_questions_subject_gap", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "wh_questions_subject_gap", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_wh_questions_subject_gap_long_distance": { |
| "task": "blimp_wh_questions_subject_gap_long_distance", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "wh_questions_subject_gap_long_distance", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_wh_vs_that_no_gap": { |
| "task": "blimp_wh_vs_that_no_gap", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "wh_vs_that_no_gap", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_wh_vs_that_no_gap_long_distance": { |
| "task": "blimp_wh_vs_that_no_gap_long_distance", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "wh_vs_that_no_gap_long_distance", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_wh_vs_that_with_gap": { |
| "task": "blimp_wh_vs_that_with_gap", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "wh_vs_that_with_gap", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_wh_vs_that_with_gap_long_distance": { |
| "task": "blimp_wh_vs_that_with_gap_long_distance", |
| "group": "blimp", |
| "dataset_path": "blimp", |
| "dataset_name": "wh_vs_that_with_gap_long_distance", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc" |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| } |
| }, |
| "versions": { |
| "blimp": "N/A", |
| "blimp_adjunct_island": 1.0, |
| "blimp_anaphor_gender_agreement": 1.0, |
| "blimp_anaphor_number_agreement": 1.0, |
| "blimp_animate_subject_passive": 1.0, |
| "blimp_animate_subject_trans": 1.0, |
| "blimp_causative": 1.0, |
| "blimp_complex_NP_island": 1.0, |
| "blimp_coordinate_structure_constraint_complex_left_branch": 1.0, |
| "blimp_coordinate_structure_constraint_object_extraction": 1.0, |
| "blimp_determiner_noun_agreement_1": 1.0, |
| "blimp_determiner_noun_agreement_2": 1.0, |
| "blimp_determiner_noun_agreement_irregular_1": 1.0, |
| "blimp_determiner_noun_agreement_irregular_2": 1.0, |
| "blimp_determiner_noun_agreement_with_adj_2": 1.0, |
| "blimp_determiner_noun_agreement_with_adj_irregular_1": 1.0, |
| "blimp_determiner_noun_agreement_with_adj_irregular_2": 1.0, |
| "blimp_determiner_noun_agreement_with_adjective_1": 1.0, |
| "blimp_distractor_agreement_relational_noun": 1.0, |
| "blimp_distractor_agreement_relative_clause": 1.0, |
| "blimp_drop_argument": 1.0, |
| "blimp_ellipsis_n_bar_1": 1.0, |
| "blimp_ellipsis_n_bar_2": 1.0, |
| "blimp_existential_there_object_raising": 1.0, |
| "blimp_existential_there_quantifiers_1": 1.0, |
| "blimp_existential_there_quantifiers_2": 1.0, |
| "blimp_existential_there_subject_raising": 1.0, |
| "blimp_expletive_it_object_raising": 1.0, |
| "blimp_inchoative": 1.0, |
| "blimp_intransitive": 1.0, |
| "blimp_irregular_past_participle_adjectives": 1.0, |
| "blimp_irregular_past_participle_verbs": 1.0, |
| "blimp_irregular_plural_subject_verb_agreement_1": 1.0, |
| "blimp_irregular_plural_subject_verb_agreement_2": 1.0, |
| "blimp_left_branch_island_echo_question": 1.0, |
| "blimp_left_branch_island_simple_question": 1.0, |
| "blimp_matrix_question_npi_licensor_present": 1.0, |
| "blimp_npi_present_1": 1.0, |
| "blimp_npi_present_2": 1.0, |
| "blimp_only_npi_licensor_present": 1.0, |
| "blimp_only_npi_scope": 1.0, |
| "blimp_passive_1": 1.0, |
| "blimp_passive_2": 1.0, |
| "blimp_principle_A_c_command": 1.0, |
| "blimp_principle_A_case_1": 1.0, |
| "blimp_principle_A_case_2": 1.0, |
| "blimp_principle_A_domain_1": 1.0, |
| "blimp_principle_A_domain_2": 1.0, |
| "blimp_principle_A_domain_3": 1.0, |
| "blimp_principle_A_reconstruction": 1.0, |
| "blimp_regular_plural_subject_verb_agreement_1": 1.0, |
| "blimp_regular_plural_subject_verb_agreement_2": 1.0, |
| "blimp_sentential_negation_npi_licensor_present": 1.0, |
| "blimp_sentential_negation_npi_scope": 1.0, |
| "blimp_sentential_subject_island": 1.0, |
| "blimp_superlative_quantifiers_1": 1.0, |
| "blimp_superlative_quantifiers_2": 1.0, |
| "blimp_tough_vs_raising_1": 1.0, |
| "blimp_tough_vs_raising_2": 1.0, |
| "blimp_transitive": 1.0, |
| "blimp_wh_island": 1.0, |
| "blimp_wh_questions_object_gap": 1.0, |
| "blimp_wh_questions_subject_gap": 1.0, |
| "blimp_wh_questions_subject_gap_long_distance": 1.0, |
| "blimp_wh_vs_that_no_gap": 1.0, |
| "blimp_wh_vs_that_no_gap_long_distance": 1.0, |
| "blimp_wh_vs_that_with_gap": 1.0, |
| "blimp_wh_vs_that_with_gap_long_distance": 1.0 |
| }, |
| "n-shot": { |
| "blimp": 0, |
| "blimp_adjunct_island": 0, |
| "blimp_anaphor_gender_agreement": 0, |
| "blimp_anaphor_number_agreement": 0, |
| "blimp_animate_subject_passive": 0, |
| "blimp_animate_subject_trans": 0, |
| "blimp_causative": 0, |
| "blimp_complex_NP_island": 0, |
| "blimp_coordinate_structure_constraint_complex_left_branch": 0, |
| "blimp_coordinate_structure_constraint_object_extraction": 0, |
| "blimp_determiner_noun_agreement_1": 0, |
| "blimp_determiner_noun_agreement_2": 0, |
| "blimp_determiner_noun_agreement_irregular_1": 0, |
| "blimp_determiner_noun_agreement_irregular_2": 0, |
| "blimp_determiner_noun_agreement_with_adj_2": 0, |
| "blimp_determiner_noun_agreement_with_adj_irregular_1": 0, |
| "blimp_determiner_noun_agreement_with_adj_irregular_2": 0, |
| "blimp_determiner_noun_agreement_with_adjective_1": 0, |
| "blimp_distractor_agreement_relational_noun": 0, |
| "blimp_distractor_agreement_relative_clause": 0, |
| "blimp_drop_argument": 0, |
| "blimp_ellipsis_n_bar_1": 0, |
| "blimp_ellipsis_n_bar_2": 0, |
| "blimp_existential_there_object_raising": 0, |
| "blimp_existential_there_quantifiers_1": 0, |
| "blimp_existential_there_quantifiers_2": 0, |
| "blimp_existential_there_subject_raising": 0, |
| "blimp_expletive_it_object_raising": 0, |
| "blimp_inchoative": 0, |
| "blimp_intransitive": 0, |
| "blimp_irregular_past_participle_adjectives": 0, |
| "blimp_irregular_past_participle_verbs": 0, |
| "blimp_irregular_plural_subject_verb_agreement_1": 0, |
| "blimp_irregular_plural_subject_verb_agreement_2": 0, |
| "blimp_left_branch_island_echo_question": 0, |
| "blimp_left_branch_island_simple_question": 0, |
| "blimp_matrix_question_npi_licensor_present": 0, |
| "blimp_npi_present_1": 0, |
| "blimp_npi_present_2": 0, |
| "blimp_only_npi_licensor_present": 0, |
| "blimp_only_npi_scope": 0, |
| "blimp_passive_1": 0, |
| "blimp_passive_2": 0, |
| "blimp_principle_A_c_command": 0, |
| "blimp_principle_A_case_1": 0, |
| "blimp_principle_A_case_2": 0, |
| "blimp_principle_A_domain_1": 0, |
| "blimp_principle_A_domain_2": 0, |
| "blimp_principle_A_domain_3": 0, |
| "blimp_principle_A_reconstruction": 0, |
| "blimp_regular_plural_subject_verb_agreement_1": 0, |
| "blimp_regular_plural_subject_verb_agreement_2": 0, |
| "blimp_sentential_negation_npi_licensor_present": 0, |
| "blimp_sentential_negation_npi_scope": 0, |
| "blimp_sentential_subject_island": 0, |
| "blimp_superlative_quantifiers_1": 0, |
| "blimp_superlative_quantifiers_2": 0, |
| "blimp_tough_vs_raising_1": 0, |
| "blimp_tough_vs_raising_2": 0, |
| "blimp_transitive": 0, |
| "blimp_wh_island": 0, |
| "blimp_wh_questions_object_gap": 0, |
| "blimp_wh_questions_subject_gap": 0, |
| "blimp_wh_questions_subject_gap_long_distance": 0, |
| "blimp_wh_vs_that_no_gap": 0, |
| "blimp_wh_vs_that_no_gap_long_distance": 0, |
| "blimp_wh_vs_that_with_gap": 0, |
| "blimp_wh_vs_that_with_gap_long_distance": 0 |
| }, |
| "config": { |
| "model": "hf", |
| "model_args": "pretrained=/home/bastian/Dokumente/tweenie_llamas/models/final_20", |
| "batch_size": 1, |
| "batch_sizes": [], |
| "device": "cuda", |
| "use_cache": null, |
| "limit": null, |
| "bootstrap_iters": 100000, |
| "gen_kwargs": null |
| }, |
| "git_hash": null |
| } |