| { |
| "results": { |
| "blimp": { |
| "acc,none": 0.7874626865671641, |
| "acc_stderr,none": 0.0014187246183603329, |
| "alias": "blimp" |
| }, |
| "blimp_adjunct_island": { |
| "alias": " - blimp_adjunct_island", |
| "acc,none": 0.873, |
| "acc_stderr,none": 0.010534798620855757 |
| }, |
| "blimp_anaphor_gender_agreement": { |
| "alias": " - blimp_anaphor_gender_agreement", |
| "acc,none": 0.909, |
| "acc_stderr,none": 0.009099549538400219 |
| }, |
| "blimp_anaphor_number_agreement": { |
| "alias": " - blimp_anaphor_number_agreement", |
| "acc,none": 0.983, |
| "acc_stderr,none": 0.004089954489689082 |
| }, |
| "blimp_animate_subject_passive": { |
| "alias": " - blimp_animate_subject_passive", |
| "acc,none": 0.734, |
| "acc_stderr,none": 0.013979965645145155 |
| }, |
| "blimp_animate_subject_trans": { |
| "alias": " - blimp_animate_subject_trans", |
| "acc,none": 0.874, |
| "acc_stderr,none": 0.010499249222408018 |
| }, |
| "blimp_causative": { |
| "alias": " - blimp_causative", |
| "acc,none": 0.727, |
| "acc_stderr,none": 0.014095022868717583 |
| }, |
| "blimp_complex_NP_island": { |
| "alias": " - blimp_complex_NP_island", |
| "acc,none": 0.547, |
| "acc_stderr,none": 0.015749255189977586 |
| }, |
| "blimp_coordinate_structure_constraint_complex_left_branch": { |
| "alias": " - blimp_coordinate_structure_constraint_complex_left_branch", |
| "acc,none": 0.601, |
| "acc_stderr,none": 0.015493193313162906 |
| }, |
| "blimp_coordinate_structure_constraint_object_extraction": { |
| "alias": " - blimp_coordinate_structure_constraint_object_extraction", |
| "acc,none": 0.841, |
| "acc_stderr,none": 0.0115694793682713 |
| }, |
| "blimp_determiner_noun_agreement_1": { |
| "alias": " - blimp_determiner_noun_agreement_1", |
| "acc,none": 0.981, |
| "acc_stderr,none": 0.00431945108291065 |
| }, |
| "blimp_determiner_noun_agreement_2": { |
| "alias": " - blimp_determiner_noun_agreement_2", |
| "acc,none": 0.954, |
| "acc_stderr,none": 0.006627814717380719 |
| }, |
| "blimp_determiner_noun_agreement_irregular_1": { |
| "alias": " - blimp_determiner_noun_agreement_irregular_1", |
| "acc,none": 0.92, |
| "acc_stderr,none": 0.00858333697775365 |
| }, |
| "blimp_determiner_noun_agreement_irregular_2": { |
| "alias": " - blimp_determiner_noun_agreement_irregular_2", |
| "acc,none": 0.935, |
| "acc_stderr,none": 0.0077997330618320105 |
| }, |
| "blimp_determiner_noun_agreement_with_adj_2": { |
| "alias": " - blimp_determiner_noun_agreement_with_adj_2", |
| "acc,none": 0.927, |
| "acc_stderr,none": 0.008230354715244073 |
| }, |
| "blimp_determiner_noun_agreement_with_adj_irregular_1": { |
| "alias": " - blimp_determiner_noun_agreement_with_adj_irregular_1", |
| "acc,none": 0.885, |
| "acc_stderr,none": 0.010093407594904635 |
| }, |
| "blimp_determiner_noun_agreement_with_adj_irregular_2": { |
| "alias": " - blimp_determiner_noun_agreement_with_adj_irregular_2", |
| "acc,none": 0.925, |
| "acc_stderr,none": 0.00833333333333334 |
| }, |
| "blimp_determiner_noun_agreement_with_adjective_1": { |
| "alias": " - blimp_determiner_noun_agreement_with_adjective_1", |
| "acc,none": 0.949, |
| "acc_stderr,none": 0.006960420062571412 |
| }, |
| "blimp_distractor_agreement_relational_noun": { |
| "alias": " - blimp_distractor_agreement_relational_noun", |
| "acc,none": 0.874, |
| "acc_stderr,none": 0.010499249222408023 |
| }, |
| "blimp_distractor_agreement_relative_clause": { |
| "alias": " - blimp_distractor_agreement_relative_clause", |
| "acc,none": 0.765, |
| "acc_stderr,none": 0.01341472903024713 |
| }, |
| "blimp_drop_argument": { |
| "alias": " - blimp_drop_argument", |
| "acc,none": 0.776, |
| "acc_stderr,none": 0.013190830072364457 |
| }, |
| "blimp_ellipsis_n_bar_1": { |
| "alias": " - blimp_ellipsis_n_bar_1", |
| "acc,none": 0.803, |
| "acc_stderr,none": 0.012583693787968118 |
| }, |
| "blimp_ellipsis_n_bar_2": { |
| "alias": " - blimp_ellipsis_n_bar_2", |
| "acc,none": 0.83, |
| "acc_stderr,none": 0.011884495834541656 |
| }, |
| "blimp_existential_there_object_raising": { |
| "alias": " - blimp_existential_there_object_raising", |
| "acc,none": 0.819, |
| "acc_stderr,none": 0.012181436179177904 |
| }, |
| "blimp_existential_there_quantifiers_1": { |
| "alias": " - blimp_existential_there_quantifiers_1", |
| "acc,none": 0.979, |
| "acc_stderr,none": 0.004536472151306512 |
| }, |
| "blimp_existential_there_quantifiers_2": { |
| "alias": " - blimp_existential_there_quantifiers_2", |
| "acc,none": 0.263, |
| "acc_stderr,none": 0.013929286594259734 |
| }, |
| "blimp_existential_there_subject_raising": { |
| "alias": " - blimp_existential_there_subject_raising", |
| "acc,none": 0.867, |
| "acc_stderr,none": 0.010743669132397349 |
| }, |
| "blimp_expletive_it_object_raising": { |
| "alias": " - blimp_expletive_it_object_raising", |
| "acc,none": 0.75, |
| "acc_stderr,none": 0.013699915608779773 |
| }, |
| "blimp_inchoative": { |
| "alias": " - blimp_inchoative", |
| "acc,none": 0.643, |
| "acc_stderr,none": 0.015158521721486767 |
| }, |
| "blimp_intransitive": { |
| "alias": " - blimp_intransitive", |
| "acc,none": 0.777, |
| "acc_stderr,none": 0.013169830843425689 |
| }, |
| "blimp_irregular_past_participle_adjectives": { |
| "alias": " - blimp_irregular_past_participle_adjectives", |
| "acc,none": 0.847, |
| "acc_stderr,none": 0.01138950045966553 |
| }, |
| "blimp_irregular_past_participle_verbs": { |
| "alias": " - blimp_irregular_past_participle_verbs", |
| "acc,none": 0.839, |
| "acc_stderr,none": 0.01162816469672718 |
| }, |
| "blimp_irregular_plural_subject_verb_agreement_1": { |
| "alias": " - blimp_irregular_plural_subject_verb_agreement_1", |
| "acc,none": 0.898, |
| "acc_stderr,none": 0.009575368801653885 |
| }, |
| "blimp_irregular_plural_subject_verb_agreement_2": { |
| "alias": " - blimp_irregular_plural_subject_verb_agreement_2", |
| "acc,none": 0.899, |
| "acc_stderr,none": 0.009533618929340971 |
| }, |
| "blimp_left_branch_island_echo_question": { |
| "alias": " - blimp_left_branch_island_echo_question", |
| "acc,none": 0.568, |
| "acc_stderr,none": 0.015672320237336203 |
| }, |
| "blimp_left_branch_island_simple_question": { |
| "alias": " - blimp_left_branch_island_simple_question", |
| "acc,none": 0.649, |
| "acc_stderr,none": 0.015100563798316407 |
| }, |
| "blimp_matrix_question_npi_licensor_present": { |
| "alias": " - blimp_matrix_question_npi_licensor_present", |
| "acc,none": 0.313, |
| "acc_stderr,none": 0.014671272822977888 |
| }, |
| "blimp_npi_present_1": { |
| "alias": " - blimp_npi_present_1", |
| "acc,none": 0.591, |
| "acc_stderr,none": 0.015555094373257942 |
| }, |
| "blimp_npi_present_2": { |
| "alias": " - blimp_npi_present_2", |
| "acc,none": 0.696, |
| "acc_stderr,none": 0.014553205687950436 |
| }, |
| "blimp_only_npi_licensor_present": { |
| "alias": " - blimp_only_npi_licensor_present", |
| "acc,none": 0.961, |
| "acc_stderr,none": 0.006125072776426101 |
| }, |
| "blimp_only_npi_scope": { |
| "alias": " - blimp_only_npi_scope", |
| "acc,none": 0.522, |
| "acc_stderr,none": 0.015803979428161946 |
| }, |
| "blimp_passive_1": { |
| "alias": " - blimp_passive_1", |
| "acc,none": 0.9, |
| "acc_stderr,none": 0.009491579957525038 |
| }, |
| "blimp_passive_2": { |
| "alias": " - blimp_passive_2", |
| "acc,none": 0.89, |
| "acc_stderr,none": 0.009899393819724439 |
| }, |
| "blimp_principle_A_c_command": { |
| "alias": " - blimp_principle_A_c_command", |
| "acc,none": 0.669, |
| "acc_stderr,none": 0.014888272588203936 |
| }, |
| "blimp_principle_A_case_1": { |
| "alias": " - blimp_principle_A_case_1", |
| "acc,none": 1.0, |
| "acc_stderr,none": 0.0 |
| }, |
| "blimp_principle_A_case_2": { |
| "alias": " - blimp_principle_A_case_2", |
| "acc,none": 0.951, |
| "acc_stderr,none": 0.0068297617561409165 |
| }, |
| "blimp_principle_A_domain_1": { |
| "alias": " - blimp_principle_A_domain_1", |
| "acc,none": 0.945, |
| "acc_stderr,none": 0.007212976294639237 |
| }, |
| "blimp_principle_A_domain_2": { |
| "alias": " - blimp_principle_A_domain_2", |
| "acc,none": 0.832, |
| "acc_stderr,none": 0.011828605831454276 |
| }, |
| "blimp_principle_A_domain_3": { |
| "alias": " - blimp_principle_A_domain_3", |
| "acc,none": 0.646, |
| "acc_stderr,none": 0.015129868238451772 |
| }, |
| "blimp_principle_A_reconstruction": { |
| "alias": " - blimp_principle_A_reconstruction", |
| "acc,none": 0.391, |
| "acc_stderr,none": 0.015438826294681783 |
| }, |
| "blimp_regular_plural_subject_verb_agreement_1": { |
| "alias": " - blimp_regular_plural_subject_verb_agreement_1", |
| "acc,none": 0.93, |
| "acc_stderr,none": 0.008072494358323494 |
| }, |
| "blimp_regular_plural_subject_verb_agreement_2": { |
| "alias": " - blimp_regular_plural_subject_verb_agreement_2", |
| "acc,none": 0.888, |
| "acc_stderr,none": 0.00997775303139725 |
| }, |
| "blimp_sentential_negation_npi_licensor_present": { |
| "alias": " - blimp_sentential_negation_npi_licensor_present", |
| "acc,none": 0.97, |
| "acc_stderr,none": 0.005397140829099205 |
| }, |
| "blimp_sentential_negation_npi_scope": { |
| "alias": " - blimp_sentential_negation_npi_scope", |
| "acc,none": 0.662, |
| "acc_stderr,none": 0.014965960710224485 |
| }, |
| "blimp_sentential_subject_island": { |
| "alias": " - blimp_sentential_subject_island", |
| "acc,none": 0.511, |
| "acc_stderr,none": 0.01581547119529269 |
| }, |
| "blimp_superlative_quantifiers_1": { |
| "alias": " - blimp_superlative_quantifiers_1", |
| "acc,none": 0.84, |
| "acc_stderr,none": 0.011598902298689009 |
| }, |
| "blimp_superlative_quantifiers_2": { |
| "alias": " - blimp_superlative_quantifiers_2", |
| "acc,none": 0.705, |
| "acc_stderr,none": 0.01442855443844551 |
| }, |
| "blimp_tough_vs_raising_1": { |
| "alias": " - blimp_tough_vs_raising_1", |
| "acc,none": 0.655, |
| "acc_stderr,none": 0.015039986742055237 |
| }, |
| "blimp_tough_vs_raising_2": { |
| "alias": " - blimp_tough_vs_raising_2", |
| "acc,none": 0.882, |
| "acc_stderr,none": 0.010206869264381793 |
| }, |
| "blimp_transitive": { |
| "alias": " - blimp_transitive", |
| "acc,none": 0.838, |
| "acc_stderr,none": 0.01165726777130441 |
| }, |
| "blimp_wh_island": { |
| "alias": " - blimp_wh_island", |
| "acc,none": 0.763, |
| "acc_stderr,none": 0.01345407046257795 |
| }, |
| "blimp_wh_questions_object_gap": { |
| "alias": " - blimp_wh_questions_object_gap", |
| "acc,none": 0.825, |
| "acc_stderr,none": 0.012021627157731972 |
| }, |
| "blimp_wh_questions_subject_gap": { |
| "alias": " - blimp_wh_questions_subject_gap", |
| "acc,none": 0.955, |
| "acc_stderr,none": 0.00655881224140612 |
| }, |
| "blimp_wh_questions_subject_gap_long_distance": { |
| "alias": " - blimp_wh_questions_subject_gap_long_distance", |
| "acc,none": 0.916, |
| "acc_stderr,none": 0.008776162089491122 |
| }, |
| "blimp_wh_vs_that_no_gap": { |
| "alias": " - blimp_wh_vs_that_no_gap", |
| "acc,none": 0.983, |
| "acc_stderr,none": 0.004089954489689092 |
| }, |
| "blimp_wh_vs_that_no_gap_long_distance": { |
| "alias": " - blimp_wh_vs_that_no_gap_long_distance", |
| "acc,none": 0.987, |
| "acc_stderr,none": 0.0035838308894036368 |
| }, |
| "blimp_wh_vs_that_with_gap": { |
| "alias": " - blimp_wh_vs_that_with_gap", |
| "acc,none": 0.486, |
| "acc_stderr,none": 0.015813097547730987 |
| }, |
| "blimp_wh_vs_that_with_gap_long_distance": { |
| "alias": " - blimp_wh_vs_that_with_gap_long_distance", |
| "acc,none": 0.246, |
| "acc_stderr,none": 0.013626065817750634 |
| } |
| }, |
| "groups": { |
| "blimp": { |
| "acc,none": 0.7874626865671641, |
| "acc_stderr,none": 0.0014187246183603329, |
| "alias": "blimp" |
| } |
| }, |
| "group_subtasks": { |
| "blimp": [ |
| "blimp_adjunct_island", |
| "blimp_anaphor_gender_agreement", |
| "blimp_anaphor_number_agreement", |
| "blimp_animate_subject_passive", |
| "blimp_animate_subject_trans", |
| "blimp_causative", |
| "blimp_complex_NP_island", |
| "blimp_coordinate_structure_constraint_complex_left_branch", |
| "blimp_coordinate_structure_constraint_object_extraction", |
| "blimp_determiner_noun_agreement_1", |
| "blimp_determiner_noun_agreement_2", |
| "blimp_determiner_noun_agreement_irregular_1", |
| "blimp_determiner_noun_agreement_irregular_2", |
| "blimp_determiner_noun_agreement_with_adj_2", |
| "blimp_determiner_noun_agreement_with_adj_irregular_1", |
| "blimp_determiner_noun_agreement_with_adj_irregular_2", |
| "blimp_determiner_noun_agreement_with_adjective_1", |
| "blimp_distractor_agreement_relational_noun", |
| "blimp_distractor_agreement_relative_clause", |
| "blimp_drop_argument", |
| "blimp_ellipsis_n_bar_1", |
| "blimp_ellipsis_n_bar_2", |
| "blimp_existential_there_object_raising", |
| "blimp_existential_there_quantifiers_1", |
| "blimp_existential_there_quantifiers_2", |
| "blimp_existential_there_subject_raising", |
| "blimp_expletive_it_object_raising", |
| "blimp_inchoative", |
| "blimp_intransitive", |
| "blimp_irregular_past_participle_adjectives", |
| "blimp_irregular_past_participle_verbs", |
| "blimp_irregular_plural_subject_verb_agreement_1", |
| "blimp_irregular_plural_subject_verb_agreement_2", |
| "blimp_left_branch_island_echo_question", |
| "blimp_left_branch_island_simple_question", |
| "blimp_matrix_question_npi_licensor_present", |
| "blimp_npi_present_1", |
| "blimp_npi_present_2", |
| "blimp_only_npi_licensor_present", |
| "blimp_only_npi_scope", |
| "blimp_passive_1", |
| "blimp_passive_2", |
| "blimp_principle_A_c_command", |
| "blimp_principle_A_case_1", |
| "blimp_principle_A_case_2", |
| "blimp_principle_A_domain_1", |
| "blimp_principle_A_domain_2", |
| "blimp_principle_A_domain_3", |
| "blimp_principle_A_reconstruction", |
| "blimp_regular_plural_subject_verb_agreement_1", |
| "blimp_regular_plural_subject_verb_agreement_2", |
| "blimp_sentential_negation_npi_licensor_present", |
| "blimp_sentential_negation_npi_scope", |
| "blimp_sentential_subject_island", |
| "blimp_superlative_quantifiers_1", |
| "blimp_superlative_quantifiers_2", |
| "blimp_tough_vs_raising_1", |
| "blimp_tough_vs_raising_2", |
| "blimp_transitive", |
| "blimp_wh_island", |
| "blimp_wh_questions_object_gap", |
| "blimp_wh_questions_subject_gap", |
| "blimp_wh_questions_subject_gap_long_distance", |
| "blimp_wh_vs_that_no_gap", |
| "blimp_wh_vs_that_no_gap_long_distance", |
| "blimp_wh_vs_that_with_gap", |
| "blimp_wh_vs_that_with_gap_long_distance" |
| ] |
| }, |
| "configs": { |
| "blimp_adjunct_island": { |
| "task": "blimp_adjunct_island", |
| "dataset_path": "blimp", |
| "dataset_name": "adjunct_island", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_anaphor_gender_agreement": { |
| "task": "blimp_anaphor_gender_agreement", |
| "dataset_path": "blimp", |
| "dataset_name": "anaphor_gender_agreement", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_anaphor_number_agreement": { |
| "task": "blimp_anaphor_number_agreement", |
| "dataset_path": "blimp", |
| "dataset_name": "anaphor_number_agreement", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_animate_subject_passive": { |
| "task": "blimp_animate_subject_passive", |
| "dataset_path": "blimp", |
| "dataset_name": "animate_subject_passive", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_animate_subject_trans": { |
| "task": "blimp_animate_subject_trans", |
| "dataset_path": "blimp", |
| "dataset_name": "animate_subject_trans", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_causative": { |
| "task": "blimp_causative", |
| "dataset_path": "blimp", |
| "dataset_name": "causative", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_complex_NP_island": { |
| "task": "blimp_complex_NP_island", |
| "dataset_path": "blimp", |
| "dataset_name": "complex_NP_island", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_coordinate_structure_constraint_complex_left_branch": { |
| "task": "blimp_coordinate_structure_constraint_complex_left_branch", |
| "dataset_path": "blimp", |
| "dataset_name": "coordinate_structure_constraint_complex_left_branch", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_coordinate_structure_constraint_object_extraction": { |
| "task": "blimp_coordinate_structure_constraint_object_extraction", |
| "dataset_path": "blimp", |
| "dataset_name": "coordinate_structure_constraint_object_extraction", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_determiner_noun_agreement_1": { |
| "task": "blimp_determiner_noun_agreement_1", |
| "dataset_path": "blimp", |
| "dataset_name": "determiner_noun_agreement_1", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_determiner_noun_agreement_2": { |
| "task": "blimp_determiner_noun_agreement_2", |
| "dataset_path": "blimp", |
| "dataset_name": "determiner_noun_agreement_2", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_determiner_noun_agreement_irregular_1": { |
| "task": "blimp_determiner_noun_agreement_irregular_1", |
| "dataset_path": "blimp", |
| "dataset_name": "determiner_noun_agreement_irregular_1", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_determiner_noun_agreement_irregular_2": { |
| "task": "blimp_determiner_noun_agreement_irregular_2", |
| "dataset_path": "blimp", |
| "dataset_name": "determiner_noun_agreement_irregular_2", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_determiner_noun_agreement_with_adj_2": { |
| "task": "blimp_determiner_noun_agreement_with_adj_2", |
| "dataset_path": "blimp", |
| "dataset_name": "determiner_noun_agreement_with_adj_2", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_determiner_noun_agreement_with_adj_irregular_1": { |
| "task": "blimp_determiner_noun_agreement_with_adj_irregular_1", |
| "dataset_path": "blimp", |
| "dataset_name": "determiner_noun_agreement_with_adj_irregular_1", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_determiner_noun_agreement_with_adj_irregular_2": { |
| "task": "blimp_determiner_noun_agreement_with_adj_irregular_2", |
| "dataset_path": "blimp", |
| "dataset_name": "determiner_noun_agreement_with_adj_irregular_2", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_determiner_noun_agreement_with_adjective_1": { |
| "task": "blimp_determiner_noun_agreement_with_adjective_1", |
| "dataset_path": "blimp", |
| "dataset_name": "determiner_noun_agreement_with_adjective_1", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_distractor_agreement_relational_noun": { |
| "task": "blimp_distractor_agreement_relational_noun", |
| "dataset_path": "blimp", |
| "dataset_name": "distractor_agreement_relational_noun", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_distractor_agreement_relative_clause": { |
| "task": "blimp_distractor_agreement_relative_clause", |
| "dataset_path": "blimp", |
| "dataset_name": "distractor_agreement_relative_clause", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_drop_argument": { |
| "task": "blimp_drop_argument", |
| "dataset_path": "blimp", |
| "dataset_name": "drop_argument", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_ellipsis_n_bar_1": { |
| "task": "blimp_ellipsis_n_bar_1", |
| "dataset_path": "blimp", |
| "dataset_name": "ellipsis_n_bar_1", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_ellipsis_n_bar_2": { |
| "task": "blimp_ellipsis_n_bar_2", |
| "dataset_path": "blimp", |
| "dataset_name": "ellipsis_n_bar_2", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_existential_there_object_raising": { |
| "task": "blimp_existential_there_object_raising", |
| "dataset_path": "blimp", |
| "dataset_name": "existential_there_object_raising", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_existential_there_quantifiers_1": { |
| "task": "blimp_existential_there_quantifiers_1", |
| "dataset_path": "blimp", |
| "dataset_name": "existential_there_quantifiers_1", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_existential_there_quantifiers_2": { |
| "task": "blimp_existential_there_quantifiers_2", |
| "dataset_path": "blimp", |
| "dataset_name": "existential_there_quantifiers_2", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_existential_there_subject_raising": { |
| "task": "blimp_existential_there_subject_raising", |
| "dataset_path": "blimp", |
| "dataset_name": "existential_there_subject_raising", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_expletive_it_object_raising": { |
| "task": "blimp_expletive_it_object_raising", |
| "dataset_path": "blimp", |
| "dataset_name": "expletive_it_object_raising", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_inchoative": { |
| "task": "blimp_inchoative", |
| "dataset_path": "blimp", |
| "dataset_name": "inchoative", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_intransitive": { |
| "task": "blimp_intransitive", |
| "dataset_path": "blimp", |
| "dataset_name": "intransitive", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_irregular_past_participle_adjectives": { |
| "task": "blimp_irregular_past_participle_adjectives", |
| "dataset_path": "blimp", |
| "dataset_name": "irregular_past_participle_adjectives", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_irregular_past_participle_verbs": { |
| "task": "blimp_irregular_past_participle_verbs", |
| "dataset_path": "blimp", |
| "dataset_name": "irregular_past_participle_verbs", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_irregular_plural_subject_verb_agreement_1": { |
| "task": "blimp_irregular_plural_subject_verb_agreement_1", |
| "dataset_path": "blimp", |
| "dataset_name": "irregular_plural_subject_verb_agreement_1", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_irregular_plural_subject_verb_agreement_2": { |
| "task": "blimp_irregular_plural_subject_verb_agreement_2", |
| "dataset_path": "blimp", |
| "dataset_name": "irregular_plural_subject_verb_agreement_2", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_left_branch_island_echo_question": { |
| "task": "blimp_left_branch_island_echo_question", |
| "dataset_path": "blimp", |
| "dataset_name": "left_branch_island_echo_question", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_left_branch_island_simple_question": { |
| "task": "blimp_left_branch_island_simple_question", |
| "dataset_path": "blimp", |
| "dataset_name": "left_branch_island_simple_question", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_matrix_question_npi_licensor_present": { |
| "task": "blimp_matrix_question_npi_licensor_present", |
| "dataset_path": "blimp", |
| "dataset_name": "matrix_question_npi_licensor_present", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_npi_present_1": { |
| "task": "blimp_npi_present_1", |
| "dataset_path": "blimp", |
| "dataset_name": "npi_present_1", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_npi_present_2": { |
| "task": "blimp_npi_present_2", |
| "dataset_path": "blimp", |
| "dataset_name": "npi_present_2", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_only_npi_licensor_present": { |
| "task": "blimp_only_npi_licensor_present", |
| "dataset_path": "blimp", |
| "dataset_name": "only_npi_licensor_present", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_only_npi_scope": { |
| "task": "blimp_only_npi_scope", |
| "dataset_path": "blimp", |
| "dataset_name": "only_npi_scope", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_passive_1": { |
| "task": "blimp_passive_1", |
| "dataset_path": "blimp", |
| "dataset_name": "passive_1", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_passive_2": { |
| "task": "blimp_passive_2", |
| "dataset_path": "blimp", |
| "dataset_name": "passive_2", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_principle_A_c_command": { |
| "task": "blimp_principle_A_c_command", |
| "dataset_path": "blimp", |
| "dataset_name": "principle_A_c_command", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_principle_A_case_1": { |
| "task": "blimp_principle_A_case_1", |
| "dataset_path": "blimp", |
| "dataset_name": "principle_A_case_1", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_principle_A_case_2": { |
| "task": "blimp_principle_A_case_2", |
| "dataset_path": "blimp", |
| "dataset_name": "principle_A_case_2", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_principle_A_domain_1": { |
| "task": "blimp_principle_A_domain_1", |
| "dataset_path": "blimp", |
| "dataset_name": "principle_A_domain_1", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_principle_A_domain_2": { |
| "task": "blimp_principle_A_domain_2", |
| "dataset_path": "blimp", |
| "dataset_name": "principle_A_domain_2", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_principle_A_domain_3": { |
| "task": "blimp_principle_A_domain_3", |
| "dataset_path": "blimp", |
| "dataset_name": "principle_A_domain_3", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_principle_A_reconstruction": { |
| "task": "blimp_principle_A_reconstruction", |
| "dataset_path": "blimp", |
| "dataset_name": "principle_A_reconstruction", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_regular_plural_subject_verb_agreement_1": { |
| "task": "blimp_regular_plural_subject_verb_agreement_1", |
| "dataset_path": "blimp", |
| "dataset_name": "regular_plural_subject_verb_agreement_1", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_regular_plural_subject_verb_agreement_2": { |
| "task": "blimp_regular_plural_subject_verb_agreement_2", |
| "dataset_path": "blimp", |
| "dataset_name": "regular_plural_subject_verb_agreement_2", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_sentential_negation_npi_licensor_present": { |
| "task": "blimp_sentential_negation_npi_licensor_present", |
| "dataset_path": "blimp", |
| "dataset_name": "sentential_negation_npi_licensor_present", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_sentential_negation_npi_scope": { |
| "task": "blimp_sentential_negation_npi_scope", |
| "dataset_path": "blimp", |
| "dataset_name": "sentential_negation_npi_scope", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_sentential_subject_island": { |
| "task": "blimp_sentential_subject_island", |
| "dataset_path": "blimp", |
| "dataset_name": "sentential_subject_island", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_superlative_quantifiers_1": { |
| "task": "blimp_superlative_quantifiers_1", |
| "dataset_path": "blimp", |
| "dataset_name": "superlative_quantifiers_1", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_superlative_quantifiers_2": { |
| "task": "blimp_superlative_quantifiers_2", |
| "dataset_path": "blimp", |
| "dataset_name": "superlative_quantifiers_2", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_tough_vs_raising_1": { |
| "task": "blimp_tough_vs_raising_1", |
| "dataset_path": "blimp", |
| "dataset_name": "tough_vs_raising_1", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_tough_vs_raising_2": { |
| "task": "blimp_tough_vs_raising_2", |
| "dataset_path": "blimp", |
| "dataset_name": "tough_vs_raising_2", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_transitive": { |
| "task": "blimp_transitive", |
| "dataset_path": "blimp", |
| "dataset_name": "transitive", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_wh_island": { |
| "task": "blimp_wh_island", |
| "dataset_path": "blimp", |
| "dataset_name": "wh_island", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_wh_questions_object_gap": { |
| "task": "blimp_wh_questions_object_gap", |
| "dataset_path": "blimp", |
| "dataset_name": "wh_questions_object_gap", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_wh_questions_subject_gap": { |
| "task": "blimp_wh_questions_subject_gap", |
| "dataset_path": "blimp", |
| "dataset_name": "wh_questions_subject_gap", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_wh_questions_subject_gap_long_distance": { |
| "task": "blimp_wh_questions_subject_gap_long_distance", |
| "dataset_path": "blimp", |
| "dataset_name": "wh_questions_subject_gap_long_distance", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_wh_vs_that_no_gap": { |
| "task": "blimp_wh_vs_that_no_gap", |
| "dataset_path": "blimp", |
| "dataset_name": "wh_vs_that_no_gap", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_wh_vs_that_no_gap_long_distance": { |
| "task": "blimp_wh_vs_that_no_gap_long_distance", |
| "dataset_path": "blimp", |
| "dataset_name": "wh_vs_that_no_gap_long_distance", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_wh_vs_that_with_gap": { |
| "task": "blimp_wh_vs_that_with_gap", |
| "dataset_path": "blimp", |
| "dataset_name": "wh_vs_that_with_gap", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| }, |
| "blimp_wh_vs_that_with_gap_long_distance": { |
| "task": "blimp_wh_vs_that_with_gap_long_distance", |
| "dataset_path": "blimp", |
| "dataset_name": "wh_vs_that_with_gap_long_distance", |
| "validation_split": "train", |
| "doc_to_text": "", |
| "doc_to_target": 0, |
| "unsafe_code": false, |
| "doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
| "description": "", |
| "target_delimiter": " ", |
| "fewshot_delimiter": "\n\n", |
| "num_fewshot": 0, |
| "metric_list": [ |
| { |
| "metric": "acc", |
| "aggregation": "mean", |
| "higher_is_better": true |
| } |
| ], |
| "output_type": "multiple_choice", |
| "repeats": 1, |
| "should_decontaminate": true, |
| "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
| "metadata": { |
| "version": 1.0 |
| } |
| } |
| }, |
| "versions": { |
| "blimp": 2.0, |
| "blimp_adjunct_island": 1.0, |
| "blimp_anaphor_gender_agreement": 1.0, |
| "blimp_anaphor_number_agreement": 1.0, |
| "blimp_animate_subject_passive": 1.0, |
| "blimp_animate_subject_trans": 1.0, |
| "blimp_causative": 1.0, |
| "blimp_complex_NP_island": 1.0, |
| "blimp_coordinate_structure_constraint_complex_left_branch": 1.0, |
| "blimp_coordinate_structure_constraint_object_extraction": 1.0, |
| "blimp_determiner_noun_agreement_1": 1.0, |
| "blimp_determiner_noun_agreement_2": 1.0, |
| "blimp_determiner_noun_agreement_irregular_1": 1.0, |
| "blimp_determiner_noun_agreement_irregular_2": 1.0, |
| "blimp_determiner_noun_agreement_with_adj_2": 1.0, |
| "blimp_determiner_noun_agreement_with_adj_irregular_1": 1.0, |
| "blimp_determiner_noun_agreement_with_adj_irregular_2": 1.0, |
| "blimp_determiner_noun_agreement_with_adjective_1": 1.0, |
| "blimp_distractor_agreement_relational_noun": 1.0, |
| "blimp_distractor_agreement_relative_clause": 1.0, |
| "blimp_drop_argument": 1.0, |
| "blimp_ellipsis_n_bar_1": 1.0, |
| "blimp_ellipsis_n_bar_2": 1.0, |
| "blimp_existential_there_object_raising": 1.0, |
| "blimp_existential_there_quantifiers_1": 1.0, |
| "blimp_existential_there_quantifiers_2": 1.0, |
| "blimp_existential_there_subject_raising": 1.0, |
| "blimp_expletive_it_object_raising": 1.0, |
| "blimp_inchoative": 1.0, |
| "blimp_intransitive": 1.0, |
| "blimp_irregular_past_participle_adjectives": 1.0, |
| "blimp_irregular_past_participle_verbs": 1.0, |
| "blimp_irregular_plural_subject_verb_agreement_1": 1.0, |
| "blimp_irregular_plural_subject_verb_agreement_2": 1.0, |
| "blimp_left_branch_island_echo_question": 1.0, |
| "blimp_left_branch_island_simple_question": 1.0, |
| "blimp_matrix_question_npi_licensor_present": 1.0, |
| "blimp_npi_present_1": 1.0, |
| "blimp_npi_present_2": 1.0, |
| "blimp_only_npi_licensor_present": 1.0, |
| "blimp_only_npi_scope": 1.0, |
| "blimp_passive_1": 1.0, |
| "blimp_passive_2": 1.0, |
| "blimp_principle_A_c_command": 1.0, |
| "blimp_principle_A_case_1": 1.0, |
| "blimp_principle_A_case_2": 1.0, |
| "blimp_principle_A_domain_1": 1.0, |
| "blimp_principle_A_domain_2": 1.0, |
| "blimp_principle_A_domain_3": 1.0, |
| "blimp_principle_A_reconstruction": 1.0, |
| "blimp_regular_plural_subject_verb_agreement_1": 1.0, |
| "blimp_regular_plural_subject_verb_agreement_2": 1.0, |
| "blimp_sentential_negation_npi_licensor_present": 1.0, |
| "blimp_sentential_negation_npi_scope": 1.0, |
| "blimp_sentential_subject_island": 1.0, |
| "blimp_superlative_quantifiers_1": 1.0, |
| "blimp_superlative_quantifiers_2": 1.0, |
| "blimp_tough_vs_raising_1": 1.0, |
| "blimp_tough_vs_raising_2": 1.0, |
| "blimp_transitive": 1.0, |
| "blimp_wh_island": 1.0, |
| "blimp_wh_questions_object_gap": 1.0, |
| "blimp_wh_questions_subject_gap": 1.0, |
| "blimp_wh_questions_subject_gap_long_distance": 1.0, |
| "blimp_wh_vs_that_no_gap": 1.0, |
| "blimp_wh_vs_that_no_gap_long_distance": 1.0, |
| "blimp_wh_vs_that_with_gap": 1.0, |
| "blimp_wh_vs_that_with_gap_long_distance": 1.0 |
| }, |
| "n-shot": { |
| "blimp_adjunct_island": 0, |
| "blimp_anaphor_gender_agreement": 0, |
| "blimp_anaphor_number_agreement": 0, |
| "blimp_animate_subject_passive": 0, |
| "blimp_animate_subject_trans": 0, |
| "blimp_causative": 0, |
| "blimp_complex_NP_island": 0, |
| "blimp_coordinate_structure_constraint_complex_left_branch": 0, |
| "blimp_coordinate_structure_constraint_object_extraction": 0, |
| "blimp_determiner_noun_agreement_1": 0, |
| "blimp_determiner_noun_agreement_2": 0, |
| "blimp_determiner_noun_agreement_irregular_1": 0, |
| "blimp_determiner_noun_agreement_irregular_2": 0, |
| "blimp_determiner_noun_agreement_with_adj_2": 0, |
| "blimp_determiner_noun_agreement_with_adj_irregular_1": 0, |
| "blimp_determiner_noun_agreement_with_adj_irregular_2": 0, |
| "blimp_determiner_noun_agreement_with_adjective_1": 0, |
| "blimp_distractor_agreement_relational_noun": 0, |
| "blimp_distractor_agreement_relative_clause": 0, |
| "blimp_drop_argument": 0, |
| "blimp_ellipsis_n_bar_1": 0, |
| "blimp_ellipsis_n_bar_2": 0, |
| "blimp_existential_there_object_raising": 0, |
| "blimp_existential_there_quantifiers_1": 0, |
| "blimp_existential_there_quantifiers_2": 0, |
| "blimp_existential_there_subject_raising": 0, |
| "blimp_expletive_it_object_raising": 0, |
| "blimp_inchoative": 0, |
| "blimp_intransitive": 0, |
| "blimp_irregular_past_participle_adjectives": 0, |
| "blimp_irregular_past_participle_verbs": 0, |
| "blimp_irregular_plural_subject_verb_agreement_1": 0, |
| "blimp_irregular_plural_subject_verb_agreement_2": 0, |
| "blimp_left_branch_island_echo_question": 0, |
| "blimp_left_branch_island_simple_question": 0, |
| "blimp_matrix_question_npi_licensor_present": 0, |
| "blimp_npi_present_1": 0, |
| "blimp_npi_present_2": 0, |
| "blimp_only_npi_licensor_present": 0, |
| "blimp_only_npi_scope": 0, |
| "blimp_passive_1": 0, |
| "blimp_passive_2": 0, |
| "blimp_principle_A_c_command": 0, |
| "blimp_principle_A_case_1": 0, |
| "blimp_principle_A_case_2": 0, |
| "blimp_principle_A_domain_1": 0, |
| "blimp_principle_A_domain_2": 0, |
| "blimp_principle_A_domain_3": 0, |
| "blimp_principle_A_reconstruction": 0, |
| "blimp_regular_plural_subject_verb_agreement_1": 0, |
| "blimp_regular_plural_subject_verb_agreement_2": 0, |
| "blimp_sentential_negation_npi_licensor_present": 0, |
| "blimp_sentential_negation_npi_scope": 0, |
| "blimp_sentential_subject_island": 0, |
| "blimp_superlative_quantifiers_1": 0, |
| "blimp_superlative_quantifiers_2": 0, |
| "blimp_tough_vs_raising_1": 0, |
| "blimp_tough_vs_raising_2": 0, |
| "blimp_transitive": 0, |
| "blimp_wh_island": 0, |
| "blimp_wh_questions_object_gap": 0, |
| "blimp_wh_questions_subject_gap": 0, |
| "blimp_wh_questions_subject_gap_long_distance": 0, |
| "blimp_wh_vs_that_no_gap": 0, |
| "blimp_wh_vs_that_no_gap_long_distance": 0, |
| "blimp_wh_vs_that_with_gap": 0, |
| "blimp_wh_vs_that_with_gap_long_distance": 0 |
| }, |
| "higher_is_better": { |
| "blimp": { |
| "acc": true |
| }, |
| "blimp_adjunct_island": { |
| "acc": true |
| }, |
| "blimp_anaphor_gender_agreement": { |
| "acc": true |
| }, |
| "blimp_anaphor_number_agreement": { |
| "acc": true |
| }, |
| "blimp_animate_subject_passive": { |
| "acc": true |
| }, |
| "blimp_animate_subject_trans": { |
| "acc": true |
| }, |
| "blimp_causative": { |
| "acc": true |
| }, |
| "blimp_complex_NP_island": { |
| "acc": true |
| }, |
| "blimp_coordinate_structure_constraint_complex_left_branch": { |
| "acc": true |
| }, |
| "blimp_coordinate_structure_constraint_object_extraction": { |
| "acc": true |
| }, |
| "blimp_determiner_noun_agreement_1": { |
| "acc": true |
| }, |
| "blimp_determiner_noun_agreement_2": { |
| "acc": true |
| }, |
| "blimp_determiner_noun_agreement_irregular_1": { |
| "acc": true |
| }, |
| "blimp_determiner_noun_agreement_irregular_2": { |
| "acc": true |
| }, |
| "blimp_determiner_noun_agreement_with_adj_2": { |
| "acc": true |
| }, |
| "blimp_determiner_noun_agreement_with_adj_irregular_1": { |
| "acc": true |
| }, |
| "blimp_determiner_noun_agreement_with_adj_irregular_2": { |
| "acc": true |
| }, |
| "blimp_determiner_noun_agreement_with_adjective_1": { |
| "acc": true |
| }, |
| "blimp_distractor_agreement_relational_noun": { |
| "acc": true |
| }, |
| "blimp_distractor_agreement_relative_clause": { |
| "acc": true |
| }, |
| "blimp_drop_argument": { |
| "acc": true |
| }, |
| "blimp_ellipsis_n_bar_1": { |
| "acc": true |
| }, |
| "blimp_ellipsis_n_bar_2": { |
| "acc": true |
| }, |
| "blimp_existential_there_object_raising": { |
| "acc": true |
| }, |
| "blimp_existential_there_quantifiers_1": { |
| "acc": true |
| }, |
| "blimp_existential_there_quantifiers_2": { |
| "acc": true |
| }, |
| "blimp_existential_there_subject_raising": { |
| "acc": true |
| }, |
| "blimp_expletive_it_object_raising": { |
| "acc": true |
| }, |
| "blimp_inchoative": { |
| "acc": true |
| }, |
| "blimp_intransitive": { |
| "acc": true |
| }, |
| "blimp_irregular_past_participle_adjectives": { |
| "acc": true |
| }, |
| "blimp_irregular_past_participle_verbs": { |
| "acc": true |
| }, |
| "blimp_irregular_plural_subject_verb_agreement_1": { |
| "acc": true |
| }, |
| "blimp_irregular_plural_subject_verb_agreement_2": { |
| "acc": true |
| }, |
| "blimp_left_branch_island_echo_question": { |
| "acc": true |
| }, |
| "blimp_left_branch_island_simple_question": { |
| "acc": true |
| }, |
| "blimp_matrix_question_npi_licensor_present": { |
| "acc": true |
| }, |
| "blimp_npi_present_1": { |
| "acc": true |
| }, |
| "blimp_npi_present_2": { |
| "acc": true |
| }, |
| "blimp_only_npi_licensor_present": { |
| "acc": true |
| }, |
| "blimp_only_npi_scope": { |
| "acc": true |
| }, |
| "blimp_passive_1": { |
| "acc": true |
| }, |
| "blimp_passive_2": { |
| "acc": true |
| }, |
| "blimp_principle_A_c_command": { |
| "acc": true |
| }, |
| "blimp_principle_A_case_1": { |
| "acc": true |
| }, |
| "blimp_principle_A_case_2": { |
| "acc": true |
| }, |
| "blimp_principle_A_domain_1": { |
| "acc": true |
| }, |
| "blimp_principle_A_domain_2": { |
| "acc": true |
| }, |
| "blimp_principle_A_domain_3": { |
| "acc": true |
| }, |
| "blimp_principle_A_reconstruction": { |
| "acc": true |
| }, |
| "blimp_regular_plural_subject_verb_agreement_1": { |
| "acc": true |
| }, |
| "blimp_regular_plural_subject_verb_agreement_2": { |
| "acc": true |
| }, |
| "blimp_sentential_negation_npi_licensor_present": { |
| "acc": true |
| }, |
| "blimp_sentential_negation_npi_scope": { |
| "acc": true |
| }, |
| "blimp_sentential_subject_island": { |
| "acc": true |
| }, |
| "blimp_superlative_quantifiers_1": { |
| "acc": true |
| }, |
| "blimp_superlative_quantifiers_2": { |
| "acc": true |
| }, |
| "blimp_tough_vs_raising_1": { |
| "acc": true |
| }, |
| "blimp_tough_vs_raising_2": { |
| "acc": true |
| }, |
| "blimp_transitive": { |
| "acc": true |
| }, |
| "blimp_wh_island": { |
| "acc": true |
| }, |
| "blimp_wh_questions_object_gap": { |
| "acc": true |
| }, |
| "blimp_wh_questions_subject_gap": { |
| "acc": true |
| }, |
| "blimp_wh_questions_subject_gap_long_distance": { |
| "acc": true |
| }, |
| "blimp_wh_vs_that_no_gap": { |
| "acc": true |
| }, |
| "blimp_wh_vs_that_no_gap_long_distance": { |
| "acc": true |
| }, |
| "blimp_wh_vs_that_with_gap": { |
| "acc": true |
| }, |
| "blimp_wh_vs_that_with_gap_long_distance": { |
| "acc": true |
| } |
| }, |
| "n-samples": { |
| "blimp_adjunct_island": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_anaphor_gender_agreement": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_anaphor_number_agreement": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_animate_subject_passive": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_animate_subject_trans": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_causative": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_complex_NP_island": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_coordinate_structure_constraint_complex_left_branch": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_coordinate_structure_constraint_object_extraction": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_determiner_noun_agreement_1": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_determiner_noun_agreement_2": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_determiner_noun_agreement_irregular_1": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_determiner_noun_agreement_irregular_2": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_determiner_noun_agreement_with_adj_2": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_determiner_noun_agreement_with_adj_irregular_1": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_determiner_noun_agreement_with_adj_irregular_2": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_determiner_noun_agreement_with_adjective_1": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_distractor_agreement_relational_noun": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_distractor_agreement_relative_clause": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_drop_argument": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_ellipsis_n_bar_1": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_ellipsis_n_bar_2": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_existential_there_object_raising": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_existential_there_quantifiers_1": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_existential_there_quantifiers_2": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_existential_there_subject_raising": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_expletive_it_object_raising": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_inchoative": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_intransitive": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_irregular_past_participle_adjectives": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_irregular_past_participle_verbs": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_irregular_plural_subject_verb_agreement_1": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_irregular_plural_subject_verb_agreement_2": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_left_branch_island_echo_question": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_left_branch_island_simple_question": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_matrix_question_npi_licensor_present": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_npi_present_1": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_npi_present_2": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_only_npi_licensor_present": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_only_npi_scope": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_passive_1": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_passive_2": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_principle_A_c_command": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_principle_A_case_1": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_principle_A_case_2": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_principle_A_domain_1": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_principle_A_domain_2": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_principle_A_domain_3": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_principle_A_reconstruction": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_regular_plural_subject_verb_agreement_1": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_regular_plural_subject_verb_agreement_2": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_sentential_negation_npi_licensor_present": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_sentential_negation_npi_scope": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_sentential_subject_island": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_superlative_quantifiers_1": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_superlative_quantifiers_2": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_tough_vs_raising_1": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_tough_vs_raising_2": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_transitive": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_wh_island": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_wh_questions_object_gap": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_wh_questions_subject_gap": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_wh_questions_subject_gap_long_distance": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_wh_vs_that_no_gap": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_wh_vs_that_no_gap_long_distance": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_wh_vs_that_with_gap": { |
| "original": 1000, |
| "effective": 1000 |
| }, |
| "blimp_wh_vs_that_with_gap_long_distance": { |
| "original": 1000, |
| "effective": 1000 |
| } |
| }, |
| "config": { |
| "model": "hf", |
| "model_args": "pretrained=outputs/fw57M-tied/42/frequency_64000/.cache/eval_model", |
| "model_num_parameters": 105785088, |
| "model_dtype": "torch.bfloat16", |
| "model_revision": "main", |
| "model_sha": "", |
| "batch_size": 1, |
| "batch_sizes": [], |
| "device": null, |
| "use_cache": null, |
| "limit": null, |
| "bootstrap_iters": 100000, |
| "gen_kwargs": null, |
| "random_seed": 0, |
| "numpy_seed": 1234, |
| "torch_seed": 1234, |
| "fewshot_seed": 1234 |
| }, |
| "git_hash": "778f288", |
| "date": 1749552653.540547, |
| "pretty_env_info": "'NoneType' object has no attribute 'splitlines'", |
| "transformers_version": "4.52.4", |
| "upper_git_hash": null, |
| "tokenizer_pad_token": [ |
| "<|padding|>", |
| "0" |
| ], |
| "tokenizer_eos_token": [ |
| "<|endoftext|>", |
| "1" |
| ], |
| "tokenizer_bos_token": [ |
| null, |
| "None" |
| ], |
| "eot_token_id": 1, |
| "max_length": 2048, |
| "task_hashes": {}, |
| "model_source": "hf", |
| "model_name": "outputs/fw57M-tied/42/frequency_64000/.cache/eval_model", |
| "model_name_sanitized": "outputs__fw57M-tied__42__frequency_64000__.cache__eval_model", |
| "system_instruction": null, |
| "system_instruction_sha": null, |
| "fewshot_as_multiturn": false, |
| "chat_template": null, |
| "chat_template_sha": null, |
| "start_time": 99068.118683634, |
| "end_time": 100062.752246584, |
| "total_evaluation_time_seconds": "994.6335629499954" |
| } |