| { |
| "results": { |
| "blimp": { |
| "acc,none": 0.7812835820895526, |
| "acc_stderr,none": 0.001417250797428721, |
| "alias": "blimp" |
| }, |
| "blimp_adjunct_island": { |
| "alias": " - blimp_adjunct_island", |
| "acc,none": 0.759, |
| "acc_stderr,none": 0.01353152253451543 |
| }, |
| "blimp_anaphor_gender_agreement": { |
| "alias": " - blimp_anaphor_gender_agreement", |
| "acc,none": 0.895, |
| "acc_stderr,none": 0.009698921026024944 |
| }, |
| "blimp_anaphor_number_agreement": { |
| "alias": " - blimp_anaphor_number_agreement", |
| "acc,none": 0.982, |
| "acc_stderr,none": 0.004206387249611461 |
| }, |
| "blimp_animate_subject_passive": { |
| "alias": " - blimp_animate_subject_passive", |
| "acc,none": 0.78, |
| "acc_stderr,none": 0.013106173040661782 |
| }, |
| "blimp_animate_subject_trans": { |
| "alias": " - blimp_animate_subject_trans", |
| "acc,none": 0.878, |
| "acc_stderr,none": 0.010354864712936722 |
| }, |
| "blimp_causative": { |
| "alias": " - blimp_causative", |
| "acc,none": 0.754, |
| "acc_stderr,none": 0.01362606581775064 |
| }, |
| "blimp_complex_NP_island": { |
| "alias": " - blimp_complex_NP_island", |
| "acc,none": 0.483, |
| "acc_stderr,none": 0.015810153729833427 |
| }, |
| "blimp_coordinate_structure_constraint_complex_left_branch": { |
| "alias": " - blimp_coordinate_structure_constraint_complex_left_branch", |
| "acc,none": 0.537, |
| "acc_stderr,none": 0.01577592722726242 |
| }, |
| "blimp_coordinate_structure_constraint_object_extraction": { |
| "alias": " - blimp_coordinate_structure_constraint_object_extraction", |
| "acc,none": 0.803, |
| "acc_stderr,none": 0.012583693787968133 |
| }, |
| "blimp_determiner_noun_agreement_1": { |
| "alias": " - blimp_determiner_noun_agreement_1", |
| "acc,none": 0.987, |
| "acc_stderr,none": 0.0035838308894036415 |
| }, |
| "blimp_determiner_noun_agreement_2": { |
| "alias": " - blimp_determiner_noun_agreement_2", |
| "acc,none": 0.988, |
| "acc_stderr,none": 0.0034449771940998413 |
| }, |
| "blimp_determiner_noun_agreement_irregular_1": { |
| "alias": " - blimp_determiner_noun_agreement_irregular_1", |
| "acc,none": 0.932, |
| "acc_stderr,none": 0.007964887911291603 |
| }, |
| "blimp_determiner_noun_agreement_irregular_2": { |
| "alias": " - blimp_determiner_noun_agreement_irregular_2", |
| "acc,none": 0.968, |
| "acc_stderr,none": 0.005568393575081352 |
| }, |
| "blimp_determiner_noun_agreement_with_adj_2": { |
| "alias": " - blimp_determiner_noun_agreement_with_adj_2", |
| "acc,none": 0.942, |
| "acc_stderr,none": 0.007395315455792946 |
| }, |
| "blimp_determiner_noun_agreement_with_adj_irregular_1": { |
| "alias": " - blimp_determiner_noun_agreement_with_adj_irregular_1", |
| "acc,none": 0.917, |
| "acc_stderr,none": 0.008728527206074792 |
| }, |
| "blimp_determiner_noun_agreement_with_adj_irregular_2": { |
| "alias": " - blimp_determiner_noun_agreement_with_adj_irregular_2", |
| "acc,none": 0.915, |
| "acc_stderr,none": 0.008823426366942317 |
| }, |
| "blimp_determiner_noun_agreement_with_adjective_1": { |
| "alias": " - blimp_determiner_noun_agreement_with_adjective_1", |
| "acc,none": 0.955, |
| "acc_stderr,none": 0.00655881224140609 |
| }, |
| "blimp_distractor_agreement_relational_noun": { |
| "alias": " - blimp_distractor_agreement_relational_noun", |
| "acc,none": 0.856, |
| "acc_stderr,none": 0.01110798754893915 |
| }, |
| "blimp_distractor_agreement_relative_clause": { |
| "alias": " - blimp_distractor_agreement_relative_clause", |
| "acc,none": 0.768, |
| "acc_stderr,none": 0.013354937452281576 |
| }, |
| "blimp_drop_argument": { |
| "alias": " - blimp_drop_argument", |
| "acc,none": 0.778, |
| "acc_stderr,none": 0.013148721948877364 |
| }, |
| "blimp_ellipsis_n_bar_1": { |
| "alias": " - blimp_ellipsis_n_bar_1", |
| "acc,none": 0.822, |
| "acc_stderr,none": 0.012102167676183589 |
| }, |
| "blimp_ellipsis_n_bar_2": { |
| "alias": " - blimp_ellipsis_n_bar_2", |
| "acc,none": 0.907, |
| "acc_stderr,none": 0.009188875634996685 |
| }, |
| "blimp_existential_there_object_raising": { |
| "alias": " - blimp_existential_there_object_raising", |
| "acc,none": 0.814, |
| "acc_stderr,none": 0.012310790208412796 |
| }, |
| "blimp_existential_there_quantifiers_1": { |
| "alias": " - blimp_existential_there_quantifiers_1", |
| "acc,none": 0.954, |
| "acc_stderr,none": 0.006627814717380709 |
| }, |
| "blimp_existential_there_quantifiers_2": { |
| "alias": " - blimp_existential_there_quantifiers_2", |
| "acc,none": 0.234, |
| "acc_stderr,none": 0.013394902889660007 |
| }, |
| "blimp_existential_there_subject_raising": { |
| "alias": " - blimp_existential_there_subject_raising", |
| "acc,none": 0.893, |
| "acc_stderr,none": 0.009779910359847169 |
| }, |
| "blimp_expletive_it_object_raising": { |
| "alias": " - blimp_expletive_it_object_raising", |
| "acc,none": 0.788, |
| "acc_stderr,none": 0.012931481864938046 |
| }, |
| "blimp_inchoative": { |
| "alias": " - blimp_inchoative", |
| "acc,none": 0.672, |
| "acc_stderr,none": 0.014853842487270336 |
| }, |
| "blimp_intransitive": { |
| "alias": " - blimp_intransitive", |
| "acc,none": 0.793, |
| "acc_stderr,none": 0.012818553557843967 |
| }, |
| "blimp_irregular_past_participle_adjectives": { |
| "alias": " - blimp_irregular_past_participle_adjectives", |
| "acc,none": 0.968, |
| "acc_stderr,none": 0.005568393575081358 |
| }, |
| "blimp_irregular_past_participle_verbs": { |
| "alias": " - blimp_irregular_past_participle_verbs", |
| "acc,none": 0.875, |
| "acc_stderr,none": 0.010463483381956722 |
| }, |
| "blimp_irregular_plural_subject_verb_agreement_1": { |
| "alias": " - blimp_irregular_plural_subject_verb_agreement_1", |
| "acc,none": 0.891, |
| "acc_stderr,none": 0.009859828407037188 |
| }, |
| "blimp_irregular_plural_subject_verb_agreement_2": { |
| "alias": " - blimp_irregular_plural_subject_verb_agreement_2", |
| "acc,none": 0.928, |
| "acc_stderr,none": 0.008178195576218681 |
| }, |
| "blimp_left_branch_island_echo_question": { |
| "alias": " - blimp_left_branch_island_echo_question", |
| "acc,none": 0.532, |
| "acc_stderr,none": 0.015786868759359016 |
| }, |
| "blimp_left_branch_island_simple_question": { |
| "alias": " - blimp_left_branch_island_simple_question", |
| "acc,none": 0.598, |
| "acc_stderr,none": 0.015512467135715077 |
| }, |
| "blimp_matrix_question_npi_licensor_present": { |
| "alias": " - blimp_matrix_question_npi_licensor_present", |
| "acc,none": 0.567, |
| "acc_stderr,none": 0.01567663091218133 |
| }, |
| "blimp_npi_present_1": { |
| "alias": " - blimp_npi_present_1", |
| "acc,none": 0.673, |
| "acc_stderr,none": 0.01484221315341124 |
| }, |
| "blimp_npi_present_2": { |
| "alias": " - blimp_npi_present_2", |
| "acc,none": 0.681, |
| "acc_stderr,none": 0.014746404865473487 |
| }, |
| "blimp_only_npi_licensor_present": { |
| "alias": " - blimp_only_npi_licensor_present", |
| "acc,none": 0.692, |
| "acc_stderr,none": 0.01460648312734276 |
| }, |
| "blimp_only_npi_scope": { |
| "alias": " - blimp_only_npi_scope", |
| "acc,none": 0.28, |
| "acc_stderr,none": 0.014205696104091508 |
| }, |
| "blimp_passive_1": { |
| "alias": " - blimp_passive_1", |
| "acc,none": 0.888, |
| "acc_stderr,none": 0.009977753031397238 |
| }, |
| "blimp_passive_2": { |
| "alias": " - blimp_passive_2", |
| "acc,none": 0.895, |
| "acc_stderr,none": 0.009698921026024957 |
| }, |
| "blimp_principle_A_c_command": { |
| "alias": " - blimp_principle_A_c_command", |
| "acc,none": 0.707, |
| "acc_stderr,none": 0.014399942998441271 |
| }, |
| "blimp_principle_A_case_1": { |
| "alias": " - blimp_principle_A_case_1", |
| "acc,none": 1.0, |
| "acc_stderr,none": 0.0 |
| }, |
| "blimp_principle_A_case_2": { |
| "alias": " - blimp_principle_A_case_2", |
| "acc,none": 0.965, |
| "acc_stderr,none": 0.005814534272734965 |
| }, |
| "blimp_principle_A_domain_1": { |
| "alias": " - blimp_principle_A_domain_1", |
| "acc,none": 0.945, |
| "acc_stderr,none": 0.007212976294639241 |
| }, |
| "blimp_principle_A_domain_2": { |
| "alias": " - blimp_principle_A_domain_2", |
| "acc,none": 0.844, |
| "acc_stderr,none": 0.011480235006122368 |
| }, |
| "blimp_principle_A_domain_3": { |
| "alias": " - blimp_principle_A_domain_3", |
| "acc,none": 0.681, |
| "acc_stderr,none": 0.014746404865473493 |
| }, |
| "blimp_principle_A_reconstruction": { |
| "alias": " - blimp_principle_A_reconstruction", |
| "acc,none": 0.374, |
| "acc_stderr,none": 0.015308767369006363 |
| }, |
| "blimp_regular_plural_subject_verb_agreement_1": { |
| "alias": " - blimp_regular_plural_subject_verb_agreement_1", |
| "acc,none": 0.934, |
| "acc_stderr,none": 0.007855297938697608 |
| }, |
| "blimp_regular_plural_subject_verb_agreement_2": { |
| "alias": " - blimp_regular_plural_subject_verb_agreement_2", |
| "acc,none": 0.895, |
| "acc_stderr,none": 0.009698921026024971 |
| }, |
| "blimp_sentential_negation_npi_licensor_present": { |
| "alias": " - blimp_sentential_negation_npi_licensor_present", |
| "acc,none": 0.965, |
| "acc_stderr,none": 0.00581453427273496 |
| }, |
| "blimp_sentential_negation_npi_scope": { |
| "alias": " - blimp_sentential_negation_npi_scope", |
| "acc,none": 0.707, |
| "acc_stderr,none": 0.014399942998441273 |
| }, |
| "blimp_sentential_subject_island": { |
| "alias": " - blimp_sentential_subject_island", |
| "acc,none": 0.375, |
| "acc_stderr,none": 0.015316971293620996 |
| }, |
| "blimp_superlative_quantifiers_1": { |
| "alias": " - blimp_superlative_quantifiers_1", |
| "acc,none": 0.828, |
| "acc_stderr,none": 0.011939788882495321 |
| }, |
| "blimp_superlative_quantifiers_2": { |
| "alias": " - blimp_superlative_quantifiers_2", |
| "acc,none": 0.574, |
| "acc_stderr,none": 0.01564508768811381 |
| }, |
| "blimp_tough_vs_raising_1": { |
| "alias": " - blimp_tough_vs_raising_1", |
| "acc,none": 0.562, |
| "acc_stderr,none": 0.01569721001969469 |
| }, |
| "blimp_tough_vs_raising_2": { |
| "alias": " - blimp_tough_vs_raising_2", |
| "acc,none": 0.906, |
| "acc_stderr,none": 0.009233052000787731 |
| }, |
| "blimp_transitive": { |
| "alias": " - blimp_transitive", |
| "acc,none": 0.851, |
| "acc_stderr,none": 0.011266140684632176 |
| }, |
| "blimp_wh_island": { |
| "alias": " - blimp_wh_island", |
| "acc,none": 0.719, |
| "acc_stderr,none": 0.014221154708434946 |
| }, |
| "blimp_wh_questions_object_gap": { |
| "alias": " - blimp_wh_questions_object_gap", |
| "acc,none": 0.763, |
| "acc_stderr,none": 0.013454070462577938 |
| }, |
| "blimp_wh_questions_subject_gap": { |
| "alias": " - blimp_wh_questions_subject_gap", |
| "acc,none": 0.944, |
| "acc_stderr,none": 0.007274401481697063 |
| }, |
| "blimp_wh_questions_subject_gap_long_distance": { |
| "alias": " - blimp_wh_questions_subject_gap_long_distance", |
| "acc,none": 0.844, |
| "acc_stderr,none": 0.011480235006122372 |
| }, |
| "blimp_wh_vs_that_no_gap": { |
| "alias": " - blimp_wh_vs_that_no_gap", |
| "acc,none": 0.984, |
| "acc_stderr,none": 0.00396985639031941 |
| }, |
| "blimp_wh_vs_that_no_gap_long_distance": { |
| "alias": " - blimp_wh_vs_that_no_gap_long_distance", |
| "acc,none": 0.975, |
| "acc_stderr,none": 0.004939574819698458 |
| }, |
| "blimp_wh_vs_that_with_gap": { |
| "alias": " - blimp_wh_vs_that_with_gap", |
| "acc,none": 0.503, |
| "acc_stderr,none": 0.015819015179246724 |
| }, |
| "blimp_wh_vs_that_with_gap_long_distance": { |
| "alias": " - blimp_wh_vs_that_with_gap_long_distance", |
| "acc,none": 0.279, |
| "acc_stderr,none": 0.014190150117612028 |
| } |
| }, |
| "groups": { |
| "blimp": { |
| "acc,none": 0.7812835820895526, |
| "acc_stderr,none": 0.001417250797428721, |
| "alias": "blimp" |
| } |
| }, |
| "group_subtasks": { |
| "blimp": [ |
| "blimp_adjunct_island", |
| "blimp_anaphor_gender_agreement", |
| "blimp_anaphor_number_agreement", |
| "blimp_animate_subject_passive", |
| "blimp_animate_subject_trans", |
| "blimp_causative", |
| "blimp_complex_NP_island", |
| "blimp_coordinate_structure_constraint_complex_left_branch", |
| "blimp_coordinate_structure_constraint_object_extraction", |
| "blimp_determiner_noun_agreement_1", |
| "blimp_determiner_noun_agreement_2", |
| "blimp_determiner_noun_agreement_irregular_1", |
| "blimp_determiner_noun_agreement_irregular_2", |
| "blimp_determiner_noun_agreement_with_adj_2", |
| "blimp_determiner_noun_agreement_with_adj_irregular_1", |
| "blimp_determiner_noun_agreement_with_adj_irregular_2", |
| "blimp_determiner_noun_agreement_with_adjective_1", |
| "blimp_distractor_agreement_relational_noun", |
| "blimp_distractor_agreement_relative_clause", |
| "blimp_drop_argument", |
| "blimp_ellipsis_n_bar_1", |
| "blimp_ellipsis_n_bar_2", |
| "blimp_existential_there_object_raising", |
| "blimp_existential_there_quantifiers_1", |
| "blimp_existential_there_quantifiers_2", |
| "blimp_existential_there_subject_raising", |
| "blimp_expletive_it_object_raising", |
| "blimp_inchoative", |
| "blimp_intransitive", |
| "blimp_irregular_past_participle_adjectives", |
| "blimp_irregular_past_participle_verbs", |
| "blimp_irregular_plural_subject_verb_agreement_1", |
| "blimp_irregular_plural_subject_verb_agreement_2", |
| "blimp_left_branch_island_echo_question", |
| "blimp_left_branch_island_simple_question", |
| "blimp_matrix_question_npi_licensor_present", |
| "blimp_npi_present_1", |
| "blimp_npi_present_2", |
| "blimp_only_npi_licensor_present", |
| "blimp_only_npi_scope", |
| "blimp_passive_1", |
| "blimp_passive_2", |
| "blimp_principle_A_c_command", |
| "blimp_principle_A_case_1", |
| "blimp_principle_A_case_2", |
| "blimp_principle_A_domain_1", |
| "blimp_principle_A_domain_2", |
| "blimp_principle_A_domain_3", |
| "blimp_principle_A_reconstruction", |
| "blimp_regular_plural_subject_verb_agreement_1", |
| "blimp_regular_plural_subject_verb_agreement_2", |
| "blimp_sentential_negation_npi_licensor_present", |
| "blimp_sentential_negation_npi_scope", |
| "blimp_sentential_subject_island", |
| "blimp_superlative_quantifiers_1", |
| "blimp_superlative_quantifiers_2", |
| "blimp_tough_vs_raising_1", |
| "blimp_tough_vs_raising_2", |
| "blimp_transitive", |
| "blimp_wh_island", |
| "blimp_wh_questions_object_gap", |
| "blimp_wh_questions_subject_gap", |
| "blimp_wh_questions_subject_gap_long_distance", |
| "blimp_wh_vs_that_no_gap", |
| "blimp_wh_vs_that_no_gap_long_distance", |
| "blimp_wh_vs_that_with_gap", |
| "blimp_wh_vs_that_with_gap_long_distance" |
| ] |
| }, |
| "configs": { |
| "blimp_adjunct_island": { |
| "task": "blimp_adjunct_island", |
| "dataset_path": "blimp", |
| "dataset_name": "adjunct_island", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_anaphor_gender_agreement": { |
| "task": "blimp_anaphor_gender_agreement", |
| "dataset_path": "blimp", |
| "dataset_name": "anaphor_gender_agreement", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_anaphor_number_agreement": { |
| "task": "blimp_anaphor_number_agreement", |
| "dataset_path": "blimp", |
| "dataset_name": "anaphor_number_agreement", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_animate_subject_passive": { |
| "task": "blimp_animate_subject_passive", |
| "dataset_path": "blimp", |
| "dataset_name": "animate_subject_passive", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_animate_subject_trans": { |
| "task": "blimp_animate_subject_trans", |
| "dataset_path": "blimp", |
| "dataset_name": "animate_subject_trans", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_causative": { |
| "task": "blimp_causative", |
| "dataset_path": "blimp", |
| "dataset_name": "causative", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_complex_NP_island": { |
| "task": "blimp_complex_NP_island", |
| "dataset_path": "blimp", |
| "dataset_name": "complex_NP_island", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_coordinate_structure_constraint_complex_left_branch": { |
| "task": "blimp_coordinate_structure_constraint_complex_left_branch", |
| "dataset_path": "blimp", |
| "dataset_name": "coordinate_structure_constraint_complex_left_branch", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_coordinate_structure_constraint_object_extraction": { |
| "task": "blimp_coordinate_structure_constraint_object_extraction", |
| "dataset_path": "blimp", |
| "dataset_name": "coordinate_structure_constraint_object_extraction", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_determiner_noun_agreement_1": { |
| "task": "blimp_determiner_noun_agreement_1", |
| "dataset_path": "blimp", |
| "dataset_name": "determiner_noun_agreement_1", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_determiner_noun_agreement_2": { |
| "task": "blimp_determiner_noun_agreement_2", |
| "dataset_path": "blimp", |
| "dataset_name": "determiner_noun_agreement_2", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_determiner_noun_agreement_irregular_1": { |
| "task": "blimp_determiner_noun_agreement_irregular_1", |
| "dataset_path": "blimp", |
| "dataset_name": "determiner_noun_agreement_irregular_1", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_determiner_noun_agreement_irregular_2": { |
| "task": "blimp_determiner_noun_agreement_irregular_2", |
| "dataset_path": "blimp", |
| "dataset_name": "determiner_noun_agreement_irregular_2", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_determiner_noun_agreement_with_adj_2": { |
| "task": "blimp_determiner_noun_agreement_with_adj_2", |
| "dataset_path": "blimp", |
| "dataset_name": "determiner_noun_agreement_with_adj_2", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_determiner_noun_agreement_with_adj_irregular_1": { |
| "task": "blimp_determiner_noun_agreement_with_adj_irregular_1", |
| "dataset_path": "blimp", |
| "dataset_name": "determiner_noun_agreement_with_adj_irregular_1", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_determiner_noun_agreement_with_adj_irregular_2": { |
| "task": "blimp_determiner_noun_agreement_with_adj_irregular_2", |
| "dataset_path": "blimp", |
| "dataset_name": "determiner_noun_agreement_with_adj_irregular_2", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_determiner_noun_agreement_with_adjective_1": { |
| "task": "blimp_determiner_noun_agreement_with_adjective_1", |
| "dataset_path": "blimp", |
| "dataset_name": "determiner_noun_agreement_with_adjective_1", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_distractor_agreement_relational_noun": { |
| "task": "blimp_distractor_agreement_relational_noun", |
| "dataset_path": "blimp", |
| "dataset_name": "distractor_agreement_relational_noun", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_distractor_agreement_relative_clause": { |
| "task": "blimp_distractor_agreement_relative_clause", |
| "dataset_path": "blimp", |
| "dataset_name": "distractor_agreement_relative_clause", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_drop_argument": { |
| "task": "blimp_drop_argument", |
| "dataset_path": "blimp", |
| "dataset_name": "drop_argument", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_ellipsis_n_bar_1": { |
| "task": "blimp_ellipsis_n_bar_1", |
| "dataset_path": "blimp", |
| "dataset_name": "ellipsis_n_bar_1", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_ellipsis_n_bar_2": { |
| "task": "blimp_ellipsis_n_bar_2", |
| "dataset_path": "blimp", |
| "dataset_name": "ellipsis_n_bar_2", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_existential_there_object_raising": { |
| "task": "blimp_existential_there_object_raising", |
| "dataset_path": "blimp", |
| "dataset_name": "existential_there_object_raising", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_existential_there_quantifiers_1": { |
| "task": "blimp_existential_there_quantifiers_1", |
| "dataset_path": "blimp", |
| "dataset_name": "existential_there_quantifiers_1", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_existential_there_quantifiers_2": { |
| "task": "blimp_existential_there_quantifiers_2", |
| "dataset_path": "blimp", |
| "dataset_name": "existential_there_quantifiers_2", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_existential_there_subject_raising": { |
| "task": "blimp_existential_there_subject_raising", |
| "dataset_path": "blimp", |
| "dataset_name": "existential_there_subject_raising", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_expletive_it_object_raising": { |
| "task": "blimp_expletive_it_object_raising", |
| "dataset_path": "blimp", |
| "dataset_name": "expletive_it_object_raising", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_inchoative": { |
| "task": "blimp_inchoative", |
| "dataset_path": "blimp", |
| "dataset_name": "inchoative", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_intransitive": { |
| "task": "blimp_intransitive", |
| "dataset_path": "blimp", |
| "dataset_name": "intransitive", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_irregular_past_participle_adjectives": { |
| "task": "blimp_irregular_past_participle_adjectives", |
| "dataset_path": "blimp", |
| "dataset_name": "irregular_past_participle_adjectives", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_irregular_past_participle_verbs": { |
| "task": "blimp_irregular_past_participle_verbs", |
| "dataset_path": "blimp", |
| "dataset_name": "irregular_past_participle_verbs", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_irregular_plural_subject_verb_agreement_1": { |
| "task": "blimp_irregular_plural_subject_verb_agreement_1", |
| "dataset_path": "blimp", |
| "dataset_name": "irregular_plural_subject_verb_agreement_1", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_irregular_plural_subject_verb_agreement_2": { |
| "task": "blimp_irregular_plural_subject_verb_agreement_2", |
| "dataset_path": "blimp", |
| "dataset_name": "irregular_plural_subject_verb_agreement_2", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_left_branch_island_echo_question": { |
| "task": "blimp_left_branch_island_echo_question", |
| "dataset_path": "blimp", |
| "dataset_name": "left_branch_island_echo_question", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_left_branch_island_simple_question": { |
| "task": "blimp_left_branch_island_simple_question", |
| "dataset_path": "blimp", |
| "dataset_name": "left_branch_island_simple_question", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_matrix_question_npi_licensor_present": { |
| "task": "blimp_matrix_question_npi_licensor_present", |
| "dataset_path": "blimp", |
| "dataset_name": "matrix_question_npi_licensor_present", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_npi_present_1": { |
| "task": "blimp_npi_present_1", |
| "dataset_path": "blimp", |
| "dataset_name": "npi_present_1", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_npi_present_2": { |
| "task": "blimp_npi_present_2", |
| "dataset_path": "blimp", |
| "dataset_name": "npi_present_2", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_only_npi_licensor_present": { |
| "task": "blimp_only_npi_licensor_present", |
| "dataset_path": "blimp", |
| "dataset_name": "only_npi_licensor_present", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_only_npi_scope": { |
| "task": "blimp_only_npi_scope", |
| "dataset_path": "blimp", |
| "dataset_name": "only_npi_scope", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_passive_1": { |
| "task": "blimp_passive_1", |
| "dataset_path": "blimp", |
| "dataset_name": "passive_1", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_passive_2": { |
| "task": "blimp_passive_2", |
| "dataset_path": "blimp", |
| "dataset_name": "passive_2", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_principle_A_c_command": { |
| "task": "blimp_principle_A_c_command", |
| "dataset_path": "blimp", |
| "dataset_name": "principle_A_c_command", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_principle_A_case_1": { |
| "task": "blimp_principle_A_case_1", |
| "dataset_path": "blimp", |
| "dataset_name": "principle_A_case_1", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_principle_A_case_2": { |
| "task": "blimp_principle_A_case_2", |
| "dataset_path": "blimp", |
| "dataset_name": "principle_A_case_2", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_principle_A_domain_1": { |
| "task": "blimp_principle_A_domain_1", |
| "dataset_path": "blimp", |
| "dataset_name": "principle_A_domain_1", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_principle_A_domain_2": { |
| "task": "blimp_principle_A_domain_2", |
| "dataset_path": "blimp", |
| "dataset_name": "principle_A_domain_2", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_principle_A_domain_3": { |
| "task": "blimp_principle_A_domain_3", |
| "dataset_path": "blimp", |
| "dataset_name": "principle_A_domain_3", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_principle_A_reconstruction": { |
| "task": "blimp_principle_A_reconstruction", |
| "dataset_path": "blimp", |
| "dataset_name": "principle_A_reconstruction", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_regular_plural_subject_verb_agreement_1": { |
| "task": "blimp_regular_plural_subject_verb_agreement_1", |
| "dataset_path": "blimp", |
| "dataset_name": "regular_plural_subject_verb_agreement_1", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_regular_plural_subject_verb_agreement_2": { |
| "task": "blimp_regular_plural_subject_verb_agreement_2", |
| "dataset_path": "blimp", |
| "dataset_name": "regular_plural_subject_verb_agreement_2", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_sentential_negation_npi_licensor_present": { |
| "task": "blimp_sentential_negation_npi_licensor_present", |
| "dataset_path": "blimp", |
| "dataset_name": "sentential_negation_npi_licensor_present", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_sentential_negation_npi_scope": { |
| "task": "blimp_sentential_negation_npi_scope", |
| "dataset_path": "blimp", |
| "dataset_name": "sentential_negation_npi_scope", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_sentential_subject_island": { |
| "task": "blimp_sentential_subject_island", |
| "dataset_path": "blimp", |
| "dataset_name": "sentential_subject_island", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_superlative_quantifiers_1": { |
| "task": "blimp_superlative_quantifiers_1", |
| "dataset_path": "blimp", |
| "dataset_name": "superlative_quantifiers_1", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_superlative_quantifiers_2": { |
| "task": "blimp_superlative_quantifiers_2", |
| "dataset_path": "blimp", |
| "dataset_name": "superlative_quantifiers_2", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_tough_vs_raising_1": { |
| "task": "blimp_tough_vs_raising_1", |
| "dataset_path": "blimp", |
| "dataset_name": "tough_vs_raising_1", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_tough_vs_raising_2": { |
| "task": "blimp_tough_vs_raising_2", |
| "dataset_path": "blimp", |
| "dataset_name": "tough_vs_raising_2", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_transitive": { |
| "task": "blimp_transitive", |
| "dataset_path": "blimp", |
| "dataset_name": "transitive", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_wh_island": { |
| "task": "blimp_wh_island", |
| "dataset_path": "blimp", |
| "dataset_name": "wh_island", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_wh_questions_object_gap": { |
| "task": "blimp_wh_questions_object_gap", |
| "dataset_path": "blimp", |
| "dataset_name": "wh_questions_object_gap", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_wh_questions_subject_gap": { |
| "task": "blimp_wh_questions_subject_gap", |
| "dataset_path": "blimp", |
| "dataset_name": "wh_questions_subject_gap", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_wh_questions_subject_gap_long_distance": { |
| "task": "blimp_wh_questions_subject_gap_long_distance", |
| "dataset_path": "blimp", |
| "dataset_name": "wh_questions_subject_gap_long_distance", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_wh_vs_that_no_gap": { |
| "task": "blimp_wh_vs_that_no_gap", |
| "dataset_path": "blimp", |
| "dataset_name": "wh_vs_that_no_gap", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_wh_vs_that_no_gap_long_distance": { |
| "task": "blimp_wh_vs_that_no_gap_long_distance", |
| "dataset_path": "blimp", |
| "dataset_name": "wh_vs_that_no_gap_long_distance", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_wh_vs_that_with_gap": { |
| "task": "blimp_wh_vs_that_with_gap", |
| "dataset_path": "blimp", |
| "dataset_name": "wh_vs_that_with_gap", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_wh_vs_that_with_gap_long_distance": { |
| "task": "blimp_wh_vs_that_with_gap_long_distance", |
| "dataset_path": "blimp", |
| "dataset_name": "wh_vs_that_with_gap_long_distance", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| } |
| }, |
| "versions": { |
| "blimp": 2.0, |
| "blimp_adjunct_island": 1.0, |
| "blimp_anaphor_gender_agreement": 1.0, |
| "blimp_anaphor_number_agreement": 1.0, |
| "blimp_animate_subject_passive": 1.0, |
| "blimp_animate_subject_trans": 1.0, |
| "blimp_causative": 1.0, |
| "blimp_complex_NP_island": 1.0, |
| "blimp_coordinate_structure_constraint_complex_left_branch": 1.0, |
| "blimp_coordinate_structure_constraint_object_extraction": 1.0, |
| "blimp_determiner_noun_agreement_1": 1.0, |
| "blimp_determiner_noun_agreement_2": 1.0, |
| "blimp_determiner_noun_agreement_irregular_1": 1.0, |
| "blimp_determiner_noun_agreement_irregular_2": 1.0, |
| "blimp_determiner_noun_agreement_with_adj_2": 1.0, |
| "blimp_determiner_noun_agreement_with_adj_irregular_1": 1.0, |
| "blimp_determiner_noun_agreement_with_adj_irregular_2": 1.0, |
| "blimp_determiner_noun_agreement_with_adjective_1": 1.0, |
| "blimp_distractor_agreement_relational_noun": 1.0, |
| "blimp_distractor_agreement_relative_clause": 1.0, |
| "blimp_drop_argument": 1.0, |
| "blimp_ellipsis_n_bar_1": 1.0, |
| "blimp_ellipsis_n_bar_2": 1.0, |
| "blimp_existential_there_object_raising": 1.0, |
| "blimp_existential_there_quantifiers_1": 1.0, |
| "blimp_existential_there_quantifiers_2": 1.0, |
| "blimp_existential_there_subject_raising": 1.0, |
| "blimp_expletive_it_object_raising": 1.0, |
| "blimp_inchoative": 1.0, |
| "blimp_intransitive": 1.0, |
| "blimp_irregular_past_participle_adjectives": 1.0, |
| "blimp_irregular_past_participle_verbs": 1.0, |
| "blimp_irregular_plural_subject_verb_agreement_1": 1.0, |
| "blimp_irregular_plural_subject_verb_agreement_2": 1.0, |
| "blimp_left_branch_island_echo_question": 1.0, |
| "blimp_left_branch_island_simple_question": 1.0, |
| "blimp_matrix_question_npi_licensor_present": 1.0, |
| "blimp_npi_present_1": 1.0, |
| "blimp_npi_present_2": 1.0, |
| "blimp_only_npi_licensor_present": 1.0, |
| "blimp_only_npi_scope": 1.0, |
| "blimp_passive_1": 1.0, |
| "blimp_passive_2": 1.0, |
| "blimp_principle_A_c_command": 1.0, |
| "blimp_principle_A_case_1": 1.0, |
| "blimp_principle_A_case_2": 1.0, |
| "blimp_principle_A_domain_1": 1.0, |
| "blimp_principle_A_domain_2": 1.0, |
| "blimp_principle_A_domain_3": 1.0, |
| "blimp_principle_A_reconstruction": 1.0, |
| "blimp_regular_plural_subject_verb_agreement_1": 1.0, |
| "blimp_regular_plural_subject_verb_agreement_2": 1.0, |
| "blimp_sentential_negation_npi_licensor_present": 1.0, |
| "blimp_sentential_negation_npi_scope": 1.0, |
| "blimp_sentential_subject_island": 1.0, |
| "blimp_superlative_quantifiers_1": 1.0, |
| "blimp_superlative_quantifiers_2": 1.0, |
| "blimp_tough_vs_raising_1": 1.0, |
| "blimp_tough_vs_raising_2": 1.0, |
| "blimp_transitive": 1.0, |
| "blimp_wh_island": 1.0, |
| "blimp_wh_questions_object_gap": 1.0, |
| "blimp_wh_questions_subject_gap": 1.0, |
| "blimp_wh_questions_subject_gap_long_distance": 1.0, |
| "blimp_wh_vs_that_no_gap": 1.0, |
| "blimp_wh_vs_that_no_gap_long_distance": 1.0, |
| "blimp_wh_vs_that_with_gap": 1.0, |
| "blimp_wh_vs_that_with_gap_long_distance": 1.0 |
| }, |
| "n-shot": { |
| "blimp_adjunct_island": 0, |
| "blimp_anaphor_gender_agreement": 0, |
| "blimp_anaphor_number_agreement": 0, |
| "blimp_animate_subject_passive": 0, |
| "blimp_animate_subject_trans": 0, |
| "blimp_causative": 0, |
| "blimp_complex_NP_island": 0, |
| "blimp_coordinate_structure_constraint_complex_left_branch": 0, |
| "blimp_coordinate_structure_constraint_object_extraction": 0, |
| "blimp_determiner_noun_agreement_1": 0, |
| "blimp_determiner_noun_agreement_2": 0, |
| "blimp_determiner_noun_agreement_irregular_1": 0, |
| "blimp_determiner_noun_agreement_irregular_2": 0, |
| "blimp_determiner_noun_agreement_with_adj_2": 0, |
| "blimp_determiner_noun_agreement_with_adj_irregular_1": 0, |
| "blimp_determiner_noun_agreement_with_adj_irregular_2": 0, |
| "blimp_determiner_noun_agreement_with_adjective_1": 0, |
| "blimp_distractor_agreement_relational_noun": 0, |
| "blimp_distractor_agreement_relative_clause": 0, |
| "blimp_drop_argument": 0, |
| "blimp_ellipsis_n_bar_1": 0, |
| "blimp_ellipsis_n_bar_2": 0, |
| "blimp_existential_there_object_raising": 0, |
| "blimp_existential_there_quantifiers_1": 0, |
| "blimp_existential_there_quantifiers_2": 0, |
| "blimp_existential_there_subject_raising": 0, |
| "blimp_expletive_it_object_raising": 0, |
| "blimp_inchoative": 0, |
| "blimp_intransitive": 0, |
| "blimp_irregular_past_participle_adjectives": 0, |
| "blimp_irregular_past_participle_verbs": 0, |
| "blimp_irregular_plural_subject_verb_agreement_1": 0, |
| "blimp_irregular_plural_subject_verb_agreement_2": 0, |
| "blimp_left_branch_island_echo_question": 0, |
| "blimp_left_branch_island_simple_question": 0, |
| "blimp_matrix_question_npi_licensor_present": 0, |
| "blimp_npi_present_1": 0, |
| "blimp_npi_present_2": 0, |
| "blimp_only_npi_licensor_present": 0, |
| "blimp_only_npi_scope": 0, |
| "blimp_passive_1": 0, |
| "blimp_passive_2": 0, |
| "blimp_principle_A_c_command": 0, |
| "blimp_principle_A_case_1": 0, |
| "blimp_principle_A_case_2": 0, |
| "blimp_principle_A_domain_1": 0, |
| "blimp_principle_A_domain_2": 0, |
| "blimp_principle_A_domain_3": 0, |
| "blimp_principle_A_reconstruction": 0, |
| "blimp_regular_plural_subject_verb_agreement_1": 0, |
| "blimp_regular_plural_subject_verb_agreement_2": 0, |
| "blimp_sentential_negation_npi_licensor_present": 0, |
| "blimp_sentential_negation_npi_scope": 0, |
| "blimp_sentential_subject_island": 0, |
| "blimp_superlative_quantifiers_1": 0, |
| "blimp_superlative_quantifiers_2": 0, |
| "blimp_tough_vs_raising_1": 0, |
| "blimp_tough_vs_raising_2": 0, |
| "blimp_transitive": 0, |
| "blimp_wh_island": 0, |
| "blimp_wh_questions_object_gap": 0, |
| "blimp_wh_questions_subject_gap": 0, |
| "blimp_wh_questions_subject_gap_long_distance": 0, |
| "blimp_wh_vs_that_no_gap": 0, |
| "blimp_wh_vs_that_no_gap_long_distance": 0, |
| "blimp_wh_vs_that_with_gap": 0, |
| "blimp_wh_vs_that_with_gap_long_distance": 0 |
| }, |
| "higher_is_better": { |
| "blimp": { |
| "acc": true |
| }, |
| "blimp_adjunct_island": { |
| "acc": true |
| }, |
| "blimp_anaphor_gender_agreement": { |
| "acc": true |
| }, |
| "blimp_anaphor_number_agreement": { |
| "acc": true |
| }, |
| "blimp_animate_subject_passive": { |
| "acc": true |
| }, |
| "blimp_animate_subject_trans": { |
| "acc": true |
| }, |
| "blimp_causative": { |
| "acc": true |
| }, |
| "blimp_complex_NP_island": { |
| "acc": true |
| }, |
| "blimp_coordinate_structure_constraint_complex_left_branch": { |
| "acc": true |
| }, |
| "blimp_coordinate_structure_constraint_object_extraction": { |
| "acc": true |
| }, |
| "blimp_determiner_noun_agreement_1": { |
| "acc": true |
| }, |
| "blimp_determiner_noun_agreement_2": { |
| "acc": true |
| }, |
| "blimp_determiner_noun_agreement_irregular_1": { |
| "acc": true |
| }, |
| "blimp_determiner_noun_agreement_irregular_2": { |
| "acc": true |
| }, |
| "blimp_determiner_noun_agreement_with_adj_2": { |
| "acc": true |
| }, |
| "blimp_determiner_noun_agreement_with_adj_irregular_1": { |
| "acc": true |
| }, |
| "blimp_determiner_noun_agreement_with_adj_irregular_2": { |
| "acc": true |
| }, |
| "blimp_determiner_noun_agreement_with_adjective_1": { |
| "acc": true |
| }, |
| "blimp_distractor_agreement_relational_noun": { |
| "acc": true |
| }, |
| "blimp_distractor_agreement_relative_clause": { |
| "acc": true |
| }, |
| "blimp_drop_argument": { |
| "acc": true |
| }, |
| "blimp_ellipsis_n_bar_1": { |
| "acc": true |
| }, |
| "blimp_ellipsis_n_bar_2": { |
| "acc": true |
| }, |
| "blimp_existential_there_object_raising": { |
| "acc": true |
| }, |
| "blimp_existential_there_quantifiers_1": { |
| "acc": true |
| }, |
| "blimp_existential_there_quantifiers_2": { |
| "acc": true |
| }, |
| "blimp_existential_there_subject_raising": { |
| "acc": true |
| }, |
| "blimp_expletive_it_object_raising": { |
| "acc": true |
| }, |
| "blimp_inchoative": { |
| "acc": true |
| }, |
| "blimp_intransitive": { |
| "acc": true |
| }, |
| "blimp_irregular_past_participle_adjectives": { |
| "acc": true |
| }, |
| "blimp_irregular_past_participle_verbs": { |
| "acc": true |
| }, |
| "blimp_irregular_plural_subject_verb_agreement_1": { |
| "acc": true |
| }, |
| "blimp_irregular_plural_subject_verb_agreement_2": { |
| "acc": true |
| }, |
| "blimp_left_branch_island_echo_question": { |
| "acc": true |
| }, |
| "blimp_left_branch_island_simple_question": { |
| "acc": true |
| }, |
| "blimp_matrix_question_npi_licensor_present": { |
| "acc": true |
| }, |
| "blimp_npi_present_1": { |
| "acc": true |
| }, |
| "blimp_npi_present_2": { |
| "acc": true |
| }, |
| "blimp_only_npi_licensor_present": { |
| "acc": true |
| }, |
| "blimp_only_npi_scope": { |
| "acc": true |
| }, |
| "blimp_passive_1": { |
| "acc": true |
| }, |
| "blimp_passive_2": { |
| "acc": true |
| }, |
| "blimp_principle_A_c_command": { |
| "acc": true |
| }, |
| "blimp_principle_A_case_1": { |
| "acc": true |
| }, |
| "blimp_principle_A_case_2": { |
| "acc": true |
| }, |
| "blimp_principle_A_domain_1": { |
| "acc": true |
| }, |
| "blimp_principle_A_domain_2": { |
| "acc": true |
| }, |
| "blimp_principle_A_domain_3": { |
| "acc": true |
| }, |
| "blimp_principle_A_reconstruction": { |
| "acc": true |
| }, |
| "blimp_regular_plural_subject_verb_agreement_1": { |
| "acc": true |
| }, |
| "blimp_regular_plural_subject_verb_agreement_2": { |
| "acc": true |
| }, |
| "blimp_sentential_negation_npi_licensor_present": { |
| "acc": true |
| }, |
| "blimp_sentential_negation_npi_scope": { |
| "acc": true |
| }, |
| "blimp_sentential_subject_island": { |
| "acc": true |
| }, |
| "blimp_superlative_quantifiers_1": { |
| "acc": true |
| }, |
| "blimp_superlative_quantifiers_2": { |
| "acc": true |
| }, |
| "blimp_tough_vs_raising_1": { |
| "acc": true |
| }, |
| "blimp_tough_vs_raising_2": { |
| "acc": true |
| }, |
| "blimp_transitive": { |
| "acc": true |
| }, |
| "blimp_wh_island": { |
| "acc": true |
| }, |
| "blimp_wh_questions_object_gap": { |
| "acc": true |
| }, |
| "blimp_wh_questions_subject_gap": { |
| "acc": true |
| }, |
| "blimp_wh_questions_subject_gap_long_distance": { |
| "acc": true |
| }, |
| "blimp_wh_vs_that_no_gap": { |
| "acc": true |
| }, |
| "blimp_wh_vs_that_no_gap_long_distance": { |
| "acc": true |
| }, |
| "blimp_wh_vs_that_with_gap": { |
| "acc": true |
| }, |
| "blimp_wh_vs_that_with_gap_long_distance": { |
| "acc": true |
| } |
| }, |
| "n-samples": { |
| "blimp_adjunct_island": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_anaphor_gender_agreement": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_anaphor_number_agreement": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_animate_subject_passive": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_animate_subject_trans": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_causative": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_complex_NP_island": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_coordinate_structure_constraint_complex_left_branch": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_coordinate_structure_constraint_object_extraction": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_determiner_noun_agreement_1": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_determiner_noun_agreement_2": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_determiner_noun_agreement_irregular_1": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_determiner_noun_agreement_irregular_2": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_determiner_noun_agreement_with_adj_2": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_determiner_noun_agreement_with_adj_irregular_1": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_determiner_noun_agreement_with_adj_irregular_2": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_determiner_noun_agreement_with_adjective_1": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_distractor_agreement_relational_noun": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_distractor_agreement_relative_clause": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_drop_argument": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_ellipsis_n_bar_1": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_ellipsis_n_bar_2": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_existential_there_object_raising": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_existential_there_quantifiers_1": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_existential_there_quantifiers_2": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_existential_there_subject_raising": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_expletive_it_object_raising": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_inchoative": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_intransitive": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_irregular_past_participle_adjectives": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_irregular_past_participle_verbs": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_irregular_plural_subject_verb_agreement_1": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_irregular_plural_subject_verb_agreement_2": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_left_branch_island_echo_question": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_left_branch_island_simple_question": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_matrix_question_npi_licensor_present": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_npi_present_1": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_npi_present_2": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_only_npi_licensor_present": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_only_npi_scope": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_passive_1": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_passive_2": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_principle_A_c_command": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_principle_A_case_1": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_principle_A_case_2": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_principle_A_domain_1": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_principle_A_domain_2": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_principle_A_domain_3": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_principle_A_reconstruction": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_regular_plural_subject_verb_agreement_1": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_regular_plural_subject_verb_agreement_2": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_sentential_negation_npi_licensor_present": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_sentential_negation_npi_scope": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_sentential_subject_island": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_superlative_quantifiers_1": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_superlative_quantifiers_2": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_tough_vs_raising_1": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_tough_vs_raising_2": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_transitive": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_wh_island": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_wh_questions_object_gap": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_wh_questions_subject_gap": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_wh_questions_subject_gap_long_distance": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_wh_vs_that_no_gap": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_wh_vs_that_no_gap_long_distance": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_wh_vs_that_with_gap": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_wh_vs_that_with_gap_long_distance": { |
| "original": 1000, |
| "effective": 1000 |
| } |
| }, |
| "config": { |
| "model": "hf", |
| "model_args": "pretrained=outputs/fw57M-tied/42/fw57M_Surprisal_bytespanP1-0T30_64000/.cache/eval_model", |
| "model_num_parameters": 105785088, |
| "model_dtype": "torch.bfloat16", |
| "model_revision": "main", |
| "model_sha": "", |
| "batch_size": 1, |
| "batch_sizes": [], |
| "device": null, |
| "use_cache": null, |
| "limit": null, |
| "bootstrap_iters": 100000, |
| "gen_kwargs": null, |
| "random_seed": 0, |
| "numpy_seed": 1234, |
| "torch_seed": 1234, |
| "fewshot_seed": 1234 |
| }, |
| "git_hash": "778f288", |
| "date": 1749638040.021471, |
| "pretty_env_info": "'NoneType' object has no attribute 'splitlines'", |
| "transformers_version": "4.52.4", |
| "upper_git_hash": null, |
| "tokenizer_pad_token": [ |
| "<|padding|>", |
| "0" |
| ], |
| "tokenizer_eos_token": [ |
| "<|endoftext|>", |
| "1" |
| ], |
| "tokenizer_bos_token": [ |
| "<|endoftext|>", |
| "1" |
| ], |
| "eot_token_id": 1, |
| "max_length": 2048, |
| "task_hashes": {}, |
| "model_source": "hf", |
| "model_name": "outputs/fw57M-tied/42/fw57M_Surprisal_bytespanP1-0T30_64000/.cache/eval_model", |
| "model_name_sanitized": "outputs__fw57M-tied__42__fw57M_Surprisal_bytespanP1-0T30_64000__.cache__eval_model", |
| "system_instruction": null, |
| "system_instruction_sha": null, |
| "fewshot_as_multiturn": false, |
| "chat_template": null, |
| "chat_template_sha": null, |
| "start_time": 211176.158305025, |
| "end_time": 212750.291630051, |
| "total_evaluation_time_seconds": "1574.1333250260213" |
| } |