{ "checkpoint": { "path": "/workspace/runs/minimal-en-125m-4ep/checkpoints/ckpt-benchmark-adapter.pt", "sha256": "47d7dad1b398a023f2cffc1f28e5544e31fb9fb7c3a63d810aa1da8f87a60eeb", "trusted_input_required": true }, "created_at": "2026-08-25T12:20:23.934641+00:00", "execution": { "batch_size": 8, "device": "cuda", "limit_per_benchmark": null, "precision": "bfloat16", "truncated_requests": 0 }, "lm_evaluation_harness": { "config": { "batch_size": 8, "batch_sizes": [], "bootstrap_iters": 0, "checkpoint": "/workspace/runs/minimal-en-125m-4ep/checkpoints/ckpt-benchmark-adapter.pt", "device": "cuda", "fewshot_seed": 1234, "gen_kwargs": null, "limit": null, "max_length": 1024, "model": "NanoGPTHarnessLM", "model_args": null, "numpy_seed": 1234, "random_seed": 0, "tokenizer": "minimal-en-12k", "torch_seed": 1234, "use_cache": null }, "configs": { "arc_challenge": { "dataset_name": "ARC-Challenge", "dataset_path": "allenai/ai2_arc", "description": "", "doc_to_choice": "{{choices.text}}", "doc_to_decontamination_query": "Question: {{question}}\nAnswer:", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{choices.text}}", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/arc/arc_challenge.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "arc_challenge", "test_split": "test", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "arc_easy": { "dataset_name": "ARC-Easy", "dataset_path": "allenai/ai2_arc", "description": "", "doc_to_choice": "{{choices.text}}", "doc_to_decontamination_query": "Question: {{question}}\nAnswer:", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{choices.text}}", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/arc/arc_easy.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "arc_easy", "test_split": "test", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "blimp_adjunct_island": { "dataset_name": "adjunct_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/adjunct_island.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_adjunct_island", "unsafe_code": false, "validation_split": "train" }, "blimp_anaphor_gender_agreement": { "dataset_name": "anaphor_gender_agreement", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/anaphor_gender_agreement.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_anaphor_gender_agreement", "unsafe_code": false, "validation_split": "train" }, "blimp_anaphor_number_agreement": { "dataset_name": "anaphor_number_agreement", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/anaphor_number_agreement.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_anaphor_number_agreement", "unsafe_code": false, "validation_split": "train" }, "blimp_animate_subject_passive": { "dataset_name": "animate_subject_passive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/animate_subject_passive.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_animate_subject_passive", "unsafe_code": false, "validation_split": "train" }, "blimp_animate_subject_trans": { "dataset_name": "animate_subject_trans", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/animate_subject_trans.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_animate_subject_trans", "unsafe_code": false, "validation_split": "train" }, "blimp_causative": { "dataset_name": "causative", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/causative.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_causative", "unsafe_code": false, "validation_split": "train" }, "blimp_complex_NP_island": { "dataset_name": "complex_NP_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/complex_NP_island.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_complex_NP_island", "unsafe_code": false, "validation_split": "train" }, "blimp_coordinate_structure_constraint_complex_left_branch": { "dataset_name": "coordinate_structure_constraint_complex_left_branch", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/coordinate_structure_constraint_complex_left_branch.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_coordinate_structure_constraint_complex_left_branch", "unsafe_code": false, "validation_split": "train" }, "blimp_coordinate_structure_constraint_object_extraction": { "dataset_name": "coordinate_structure_constraint_object_extraction", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/coordinate_structure_constraint_object_extraction.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_coordinate_structure_constraint_object_extraction", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_1": { "dataset_name": "determiner_noun_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/determiner_noun_agreement_1.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_2": { "dataset_name": "determiner_noun_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/determiner_noun_agreement_2.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_irregular_1": { "dataset_name": "determiner_noun_agreement_irregular_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/determiner_noun_agreement_irregular_1.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_irregular_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_irregular_2": { "dataset_name": "determiner_noun_agreement_irregular_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/determiner_noun_agreement_irregular_2.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_irregular_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_2": { "dataset_name": "determiner_noun_agreement_with_adj_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_2.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_irregular_1": { "dataset_name": "determiner_noun_agreement_with_adj_irregular_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_irregular_1.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_irregular_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_irregular_2": { "dataset_name": "determiner_noun_agreement_with_adj_irregular_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_irregular_2.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_irregular_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adjective_1": { "dataset_name": "determiner_noun_agreement_with_adjective_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adjective_1.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adjective_1", "unsafe_code": false, "validation_split": "train" }, "blimp_distractor_agreement_relational_noun": { "dataset_name": "distractor_agreement_relational_noun", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/distractor_agreement_relational_noun.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_distractor_agreement_relational_noun", "unsafe_code": false, "validation_split": "train" }, "blimp_distractor_agreement_relative_clause": { "dataset_name": "distractor_agreement_relative_clause", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/distractor_agreement_relative_clause.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_distractor_agreement_relative_clause", "unsafe_code": false, "validation_split": "train" }, "blimp_drop_argument": { "dataset_name": "drop_argument", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/drop_argument.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_drop_argument", "unsafe_code": false, "validation_split": "train" }, "blimp_ellipsis_n_bar_1": { "dataset_name": "ellipsis_n_bar_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/ellipsis_n_bar_1.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_ellipsis_n_bar_1", "unsafe_code": false, "validation_split": "train" }, "blimp_ellipsis_n_bar_2": { "dataset_name": "ellipsis_n_bar_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/ellipsis_n_bar_2.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_ellipsis_n_bar_2", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_object_raising": { "dataset_name": "existential_there_object_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/existential_there_object_raising.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_object_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_quantifiers_1": { "dataset_name": "existential_there_quantifiers_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/existential_there_quantifiers_1.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_quantifiers_1", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_quantifiers_2": { "dataset_name": "existential_there_quantifiers_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/existential_there_quantifiers_2.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_quantifiers_2", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_subject_raising": { "dataset_name": "existential_there_subject_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/existential_there_subject_raising.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_subject_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_expletive_it_object_raising": { "dataset_name": "expletive_it_object_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/expletive_it_object_raising.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_expletive_it_object_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_inchoative": { "dataset_name": "inchoative", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/inchoative.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_inchoative", "unsafe_code": false, "validation_split": "train" }, "blimp_intransitive": { "dataset_name": "intransitive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/intransitive.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_intransitive", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_past_participle_adjectives": { "dataset_name": "irregular_past_participle_adjectives", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/irregular_past_participle_adjectives.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_past_participle_adjectives", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_past_participle_verbs": { "dataset_name": "irregular_past_participle_verbs", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/irregular_past_participle_verbs.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_past_participle_verbs", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_plural_subject_verb_agreement_1": { "dataset_name": "irregular_plural_subject_verb_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/irregular_plural_subject_verb_agreement_1.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_plural_subject_verb_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_plural_subject_verb_agreement_2": { "dataset_name": "irregular_plural_subject_verb_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/irregular_plural_subject_verb_agreement_2.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_plural_subject_verb_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_left_branch_island_echo_question": { "dataset_name": "left_branch_island_echo_question", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/left_branch_island_echo_question.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_left_branch_island_echo_question", "unsafe_code": false, "validation_split": "train" }, "blimp_left_branch_island_simple_question": { "dataset_name": "left_branch_island_simple_question", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/left_branch_island_simple_question.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_left_branch_island_simple_question", "unsafe_code": false, "validation_split": "train" }, "blimp_matrix_question_npi_licensor_present": { "dataset_name": "matrix_question_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/matrix_question_npi_licensor_present.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_matrix_question_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_npi_present_1": { "dataset_name": "npi_present_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/npi_present_1.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_npi_present_1", "unsafe_code": false, "validation_split": "train" }, "blimp_npi_present_2": { "dataset_name": "npi_present_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/npi_present_2.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_npi_present_2", "unsafe_code": false, "validation_split": "train" }, "blimp_only_npi_licensor_present": { "dataset_name": "only_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/only_npi_licensor_present.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_only_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_only_npi_scope": { "dataset_name": "only_npi_scope", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/only_npi_scope.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_only_npi_scope", "unsafe_code": false, "validation_split": "train" }, "blimp_passive_1": { "dataset_name": "passive_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/passive_1.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_passive_1", "unsafe_code": false, "validation_split": "train" }, "blimp_passive_2": { "dataset_name": "passive_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/passive_2.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_passive_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_c_command": { "dataset_name": "principle_A_c_command", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/principle_A_c_command.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_c_command", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_case_1": { "dataset_name": "principle_A_case_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/principle_A_case_1.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_case_1", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_case_2": { "dataset_name": "principle_A_case_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/principle_A_case_2.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_case_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_1": { "dataset_name": "principle_A_domain_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/principle_A_domain_1.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_1", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_2": { "dataset_name": "principle_A_domain_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/principle_A_domain_2.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_3": { "dataset_name": "principle_A_domain_3", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/principle_A_domain_3.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_3", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_reconstruction": { "dataset_name": "principle_A_reconstruction", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/principle_A_reconstruction.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_reconstruction", "unsafe_code": false, "validation_split": "train" }, "blimp_regular_plural_subject_verb_agreement_1": { "dataset_name": "regular_plural_subject_verb_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/regular_plural_subject_verb_agreement_1.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_regular_plural_subject_verb_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_regular_plural_subject_verb_agreement_2": { "dataset_name": "regular_plural_subject_verb_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/regular_plural_subject_verb_agreement_2.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_regular_plural_subject_verb_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_negation_npi_licensor_present": { "dataset_name": "sentential_negation_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/sentential_negation_npi_licensor_present.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_negation_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_negation_npi_scope": { "dataset_name": "sentential_negation_npi_scope", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/sentential_negation_npi_scope.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_negation_npi_scope", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_subject_island": { "dataset_name": "sentential_subject_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/sentential_subject_island.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_subject_island", "unsafe_code": false, "validation_split": "train" }, "blimp_superlative_quantifiers_1": { "dataset_name": "superlative_quantifiers_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/superlative_quantifiers_1.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_superlative_quantifiers_1", "unsafe_code": false, "validation_split": "train" }, "blimp_superlative_quantifiers_2": { "dataset_name": "superlative_quantifiers_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/superlative_quantifiers_2.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_superlative_quantifiers_2", "unsafe_code": false, "validation_split": "train" }, "blimp_tough_vs_raising_1": { "dataset_name": "tough_vs_raising_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/tough_vs_raising_1.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_tough_vs_raising_1", "unsafe_code": false, "validation_split": "train" }, "blimp_tough_vs_raising_2": { "dataset_name": "tough_vs_raising_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/tough_vs_raising_2.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_tough_vs_raising_2", "unsafe_code": false, "validation_split": "train" }, "blimp_transitive": { "dataset_name": "transitive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/transitive.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_transitive", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_island": { "dataset_name": "wh_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/wh_island.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_island", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_object_gap": { "dataset_name": "wh_questions_object_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/wh_questions_object_gap.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_object_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_subject_gap": { "dataset_name": "wh_questions_subject_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/wh_questions_subject_gap.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_subject_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_subject_gap_long_distance": { "dataset_name": "wh_questions_subject_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/wh_questions_subject_gap_long_distance.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_subject_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_no_gap": { "dataset_name": "wh_vs_that_no_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/wh_vs_that_no_gap.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_no_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_no_gap_long_distance": { "dataset_name": "wh_vs_that_no_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/wh_vs_that_no_gap_long_distance.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_no_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_with_gap": { "dataset_name": "wh_vs_that_with_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/wh_vs_that_with_gap.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_with_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_with_gap_long_distance": { "dataset_name": "wh_vs_that_with_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/blimp/wh_vs_that_with_gap_long_distance.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_with_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "hellaswag": { "dataset_path": "Rowan/hellaswag", "description": "", "doc_to_choice": "choices", "doc_to_target": "{{label}}", "doc_to_text": "{{query}}", "fewshot_config": { "doc_to_choice": "choices", "doc_to_target": "{{label}}", "doc_to_text": "{{query}}", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": "", "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/hellaswag/hellaswag.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n", "repeats": 1, "should_decontaminate": false, "target_delimiter": " ", "task": "hellaswag", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "lambada_openai": { "dataset_name": "default", "dataset_path": "EleutherAI/lambada_openai", "description": "", "doc_to_decontamination_query": "{{text}}", "doc_to_target": "{{' '+text.split(' ')[-1]}}", "doc_to_text": "{{text.split(' ')[:-1]|join(' ')}}", "fewshot_config": { "doc_to_choice": null, "doc_to_target": "{{' '+text.split(' ')[-1]}}", "doc_to_text": "{{text.split(' ')[:-1]|join(' ')}}", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/lambada/lambada_openai.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "perplexity", "higher_is_better": false, "metric": "perplexity" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "loglikelihood", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "lambada_openai", "test_split": "test", "unsafe_code": false }, "piqa": { "dataset_path": "baber/piqa", "description": "", "doc_to_choice": "{{[sol1, sol2]}}", "doc_to_decontamination_query": "goal", "doc_to_target": "label", "doc_to_text": "Question: {{goal}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{[sol1, sol2]}}", "doc_to_target": "label", "doc_to_text": "Question: {{goal}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/piqa/piqa.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "piqa", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "sciq": { "dataset_path": "allenai/sciq", "description": "", "doc_to_choice": "{{[distractor1, distractor2, distractor3, correct_answer]}}", "doc_to_decontamination_query": "{{support}} {{question}}", "doc_to_target": 3, "doc_to_text": "{{support.lstrip()}}\nQuestion: {{question}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{[distractor1, distractor2, distractor3, correct_answer]}}", "doc_to_target": 3, "doc_to_text": "{{support.lstrip()}}\nQuestion: {{question}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/sciq/sciq.yaml", "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "sciq", "test_split": "test", "training_split": "train", "unsafe_code": false, "validation_split": "validation" } }, "date": 1787659951.0137095, "git_hash": "3adf61e", "group_subtasks": { "blimp": [ "blimp_adjunct_island", "blimp_anaphor_gender_agreement", "blimp_anaphor_number_agreement", "blimp_animate_subject_passive", "blimp_animate_subject_trans", "blimp_causative", "blimp_complex_NP_island", "blimp_coordinate_structure_constraint_complex_left_branch", "blimp_coordinate_structure_constraint_object_extraction", "blimp_determiner_noun_agreement_1", "blimp_determiner_noun_agreement_2", "blimp_determiner_noun_agreement_irregular_1", "blimp_determiner_noun_agreement_irregular_2", "blimp_determiner_noun_agreement_with_adj_2", "blimp_determiner_noun_agreement_with_adj_irregular_1", "blimp_determiner_noun_agreement_with_adj_irregular_2", "blimp_determiner_noun_agreement_with_adjective_1", "blimp_distractor_agreement_relational_noun", "blimp_distractor_agreement_relative_clause", "blimp_drop_argument", "blimp_ellipsis_n_bar_1", "blimp_ellipsis_n_bar_2", "blimp_existential_there_object_raising", "blimp_existential_there_quantifiers_1", "blimp_existential_there_quantifiers_2", "blimp_existential_there_subject_raising", "blimp_expletive_it_object_raising", "blimp_inchoative", "blimp_intransitive", "blimp_irregular_past_participle_adjectives", "blimp_irregular_past_participle_verbs", "blimp_irregular_plural_subject_verb_agreement_1", "blimp_irregular_plural_subject_verb_agreement_2", "blimp_left_branch_island_echo_question", "blimp_left_branch_island_simple_question", "blimp_matrix_question_npi_licensor_present", "blimp_npi_present_1", "blimp_npi_present_2", "blimp_only_npi_licensor_present", "blimp_only_npi_scope", "blimp_passive_1", "blimp_passive_2", "blimp_principle_A_c_command", "blimp_principle_A_case_1", "blimp_principle_A_case_2", "blimp_principle_A_domain_1", "blimp_principle_A_domain_2", "blimp_principle_A_domain_3", "blimp_principle_A_reconstruction", "blimp_regular_plural_subject_verb_agreement_1", "blimp_regular_plural_subject_verb_agreement_2", "blimp_sentential_negation_npi_licensor_present", "blimp_sentential_negation_npi_scope", "blimp_sentential_subject_island", "blimp_superlative_quantifiers_1", "blimp_superlative_quantifiers_2", "blimp_tough_vs_raising_1", "blimp_tough_vs_raising_2", "blimp_transitive", "blimp_wh_island", "blimp_wh_questions_object_gap", "blimp_wh_questions_subject_gap", "blimp_wh_questions_subject_gap_long_distance", "blimp_wh_vs_that_no_gap", "blimp_wh_vs_that_no_gap_long_distance", "blimp_wh_vs_that_with_gap", "blimp_wh_vs_that_with_gap_long_distance" ] }, "groups": { "blimp": { "acc,none": 0.7669701492537313, "acc_stderr,none": "N/A", "alias": "blimp", "name": "blimp", "sample_count": { "acc,none": 67000 }, "sample_len": 67000 } }, "higher_is_better": { "arc_challenge": { "acc": true, "acc_norm": true }, "arc_easy": { "acc": true, "acc_norm": true }, "blimp": { "acc": true }, "blimp_adjunct_island": { "acc": true }, "blimp_anaphor_gender_agreement": { "acc": true }, "blimp_anaphor_number_agreement": { "acc": true }, "blimp_animate_subject_passive": { "acc": true }, "blimp_animate_subject_trans": { "acc": true }, "blimp_causative": { "acc": true }, "blimp_complex_NP_island": { "acc": true }, "blimp_coordinate_structure_constraint_complex_left_branch": { "acc": true }, "blimp_coordinate_structure_constraint_object_extraction": { "acc": true }, "blimp_determiner_noun_agreement_1": { "acc": true }, "blimp_determiner_noun_agreement_2": { "acc": true }, "blimp_determiner_noun_agreement_irregular_1": { "acc": true }, "blimp_determiner_noun_agreement_irregular_2": { "acc": true }, "blimp_determiner_noun_agreement_with_adj_2": { "acc": true }, "blimp_determiner_noun_agreement_with_adj_irregular_1": { "acc": true }, "blimp_determiner_noun_agreement_with_adj_irregular_2": { "acc": true }, "blimp_determiner_noun_agreement_with_adjective_1": { "acc": true }, "blimp_distractor_agreement_relational_noun": { "acc": true }, "blimp_distractor_agreement_relative_clause": { "acc": true }, "blimp_drop_argument": { "acc": true }, "blimp_ellipsis_n_bar_1": { "acc": true }, "blimp_ellipsis_n_bar_2": { "acc": true }, "blimp_existential_there_object_raising": { "acc": true }, "blimp_existential_there_quantifiers_1": { "acc": true }, "blimp_existential_there_quantifiers_2": { "acc": true }, "blimp_existential_there_subject_raising": { "acc": true }, "blimp_expletive_it_object_raising": { "acc": true }, "blimp_inchoative": { "acc": true }, "blimp_intransitive": { "acc": true }, "blimp_irregular_past_participle_adjectives": { "acc": true }, "blimp_irregular_past_participle_verbs": { "acc": true }, "blimp_irregular_plural_subject_verb_agreement_1": { "acc": true }, "blimp_irregular_plural_subject_verb_agreement_2": { "acc": true }, "blimp_left_branch_island_echo_question": { "acc": true }, "blimp_left_branch_island_simple_question": { "acc": true }, "blimp_matrix_question_npi_licensor_present": { "acc": true }, "blimp_npi_present_1": { "acc": true }, "blimp_npi_present_2": { "acc": true }, "blimp_only_npi_licensor_present": { "acc": true }, "blimp_only_npi_scope": { "acc": true }, "blimp_passive_1": { "acc": true }, "blimp_passive_2": { "acc": true }, "blimp_principle_A_c_command": { "acc": true }, "blimp_principle_A_case_1": { "acc": true }, "blimp_principle_A_case_2": { "acc": true }, "blimp_principle_A_domain_1": { "acc": true }, "blimp_principle_A_domain_2": { "acc": true }, "blimp_principle_A_domain_3": { "acc": true }, "blimp_principle_A_reconstruction": { "acc": true }, "blimp_regular_plural_subject_verb_agreement_1": { "acc": true }, "blimp_regular_plural_subject_verb_agreement_2": { "acc": true }, "blimp_sentential_negation_npi_licensor_present": { "acc": true }, "blimp_sentential_negation_npi_scope": { "acc": true }, "blimp_sentential_subject_island": { "acc": true }, "blimp_superlative_quantifiers_1": { "acc": true }, "blimp_superlative_quantifiers_2": { "acc": true }, "blimp_tough_vs_raising_1": { "acc": true }, "blimp_tough_vs_raising_2": { "acc": true }, "blimp_transitive": { "acc": true }, "blimp_wh_island": { "acc": true }, "blimp_wh_questions_object_gap": { "acc": true }, "blimp_wh_questions_subject_gap": { "acc": true }, "blimp_wh_questions_subject_gap_long_distance": { "acc": true }, "blimp_wh_vs_that_no_gap": { "acc": true }, "blimp_wh_vs_that_no_gap_long_distance": { "acc": true }, "blimp_wh_vs_that_with_gap": { "acc": true }, "blimp_wh_vs_that_with_gap_long_distance": { "acc": true }, "hellaswag": { "acc": true, "acc_norm": true }, "lambada_openai": { "acc": true, "perplexity": false }, "piqa": { "acc": true, "acc_norm": true }, "sciq": { "acc": true, "acc_norm": true } }, "lm_eval_version": "0.4.12", "n-samples": { "arc_challenge": { "effective": 1172, "original": 1172 }, "arc_easy": { "effective": 2376, "original": 2376 }, "blimp_adjunct_island": { "effective": 1000, "original": 1000 }, "blimp_anaphor_gender_agreement": { "effective": 1000, "original": 1000 }, "blimp_anaphor_number_agreement": { "effective": 1000, "original": 1000 }, "blimp_animate_subject_passive": { "effective": 1000, "original": 1000 }, "blimp_animate_subject_trans": { "effective": 1000, "original": 1000 }, "blimp_causative": { "effective": 1000, "original": 1000 }, "blimp_complex_NP_island": { "effective": 1000, "original": 1000 }, "blimp_coordinate_structure_constraint_complex_left_branch": { "effective": 1000, "original": 1000 }, "blimp_coordinate_structure_constraint_object_extraction": { "effective": 1000, "original": 1000 }, "blimp_determiner_noun_agreement_1": { "effective": 1000, "original": 1000 }, "blimp_determiner_noun_agreement_2": { "effective": 1000, "original": 1000 }, "blimp_determiner_noun_agreement_irregular_1": { "effective": 1000, "original": 1000 }, "blimp_determiner_noun_agreement_irregular_2": { "effective": 1000, "original": 1000 }, "blimp_determiner_noun_agreement_with_adj_2": { "effective": 1000, "original": 1000 }, "blimp_determiner_noun_agreement_with_adj_irregular_1": { "effective": 1000, "original": 1000 }, "blimp_determiner_noun_agreement_with_adj_irregular_2": { "effective": 1000, "original": 1000 }, "blimp_determiner_noun_agreement_with_adjective_1": { "effective": 1000, "original": 1000 }, "blimp_distractor_agreement_relational_noun": { "effective": 1000, "original": 1000 }, "blimp_distractor_agreement_relative_clause": { "effective": 1000, "original": 1000 }, "blimp_drop_argument": { "effective": 1000, "original": 1000 }, "blimp_ellipsis_n_bar_1": { "effective": 1000, "original": 1000 }, "blimp_ellipsis_n_bar_2": { "effective": 1000, "original": 1000 }, "blimp_existential_there_object_raising": { "effective": 1000, "original": 1000 }, "blimp_existential_there_quantifiers_1": { "effective": 1000, "original": 1000 }, "blimp_existential_there_quantifiers_2": { "effective": 1000, "original": 1000 }, "blimp_existential_there_subject_raising": { "effective": 1000, "original": 1000 }, "blimp_expletive_it_object_raising": { "effective": 1000, "original": 1000 }, "blimp_inchoative": { "effective": 1000, "original": 1000 }, "blimp_intransitive": { "effective": 1000, "original": 1000 }, "blimp_irregular_past_participle_adjectives": { "effective": 1000, "original": 1000 }, "blimp_irregular_past_participle_verbs": { "effective": 1000, "original": 1000 }, "blimp_irregular_plural_subject_verb_agreement_1": { "effective": 1000, "original": 1000 }, "blimp_irregular_plural_subject_verb_agreement_2": { "effective": 1000, "original": 1000 }, "blimp_left_branch_island_echo_question": { "effective": 1000, "original": 1000 }, "blimp_left_branch_island_simple_question": { "effective": 1000, "original": 1000 }, "blimp_matrix_question_npi_licensor_present": { "effective": 1000, "original": 1000 }, "blimp_npi_present_1": { "effective": 1000, "original": 1000 }, "blimp_npi_present_2": { "effective": 1000, "original": 1000 }, "blimp_only_npi_licensor_present": { "effective": 1000, "original": 1000 }, "blimp_only_npi_scope": { "effective": 1000, "original": 1000 }, "blimp_passive_1": { "effective": 1000, "original": 1000 }, "blimp_passive_2": { "effective": 1000, "original": 1000 }, "blimp_principle_A_c_command": { "effective": 1000, "original": 1000 }, "blimp_principle_A_case_1": { "effective": 1000, "original": 1000 }, "blimp_principle_A_case_2": { "effective": 1000, "original": 1000 }, "blimp_principle_A_domain_1": { "effective": 1000, "original": 1000 }, "blimp_principle_A_domain_2": { "effective": 1000, "original": 1000 }, "blimp_principle_A_domain_3": { "effective": 1000, "original": 1000 }, "blimp_principle_A_reconstruction": { "effective": 1000, "original": 1000 }, "blimp_regular_plural_subject_verb_agreement_1": { "effective": 1000, "original": 1000 }, "blimp_regular_plural_subject_verb_agreement_2": { "effective": 1000, "original": 1000 }, "blimp_sentential_negation_npi_licensor_present": { "effective": 1000, "original": 1000 }, "blimp_sentential_negation_npi_scope": { "effective": 1000, "original": 1000 }, "blimp_sentential_subject_island": { "effective": 1000, "original": 1000 }, "blimp_superlative_quantifiers_1": { "effective": 1000, "original": 1000 }, "blimp_superlative_quantifiers_2": { "effective": 1000, "original": 1000 }, "blimp_tough_vs_raising_1": { "effective": 1000, "original": 1000 }, "blimp_tough_vs_raising_2": { "effective": 1000, "original": 1000 }, "blimp_transitive": { "effective": 1000, "original": 1000 }, "blimp_wh_island": { "effective": 1000, "original": 1000 }, "blimp_wh_questions_object_gap": { "effective": 1000, "original": 1000 }, "blimp_wh_questions_subject_gap": { "effective": 1000, "original": 1000 }, "blimp_wh_questions_subject_gap_long_distance": { "effective": 1000, "original": 1000 }, "blimp_wh_vs_that_no_gap": { "effective": 1000, "original": 1000 }, "blimp_wh_vs_that_no_gap_long_distance": { "effective": 1000, "original": 1000 }, "blimp_wh_vs_that_with_gap": { "effective": 1000, "original": 1000 }, "blimp_wh_vs_that_with_gap_long_distance": { "effective": 1000, "original": 1000 }, "hellaswag": { "effective": 10042, "original": 10042 }, "lambada_openai": { "effective": 5153, "original": 5153 }, "piqa": { "effective": 1838, "original": 1838 }, "sciq": { "effective": 1000, "original": 1000 } }, "n-shot": { "arc_challenge": 0, "arc_easy": 0, "blimp": 0, "blimp_adjunct_island": 0, "blimp_anaphor_gender_agreement": 0, "blimp_anaphor_number_agreement": 0, "blimp_animate_subject_passive": 0, "blimp_animate_subject_trans": 0, "blimp_causative": 0, "blimp_complex_NP_island": 0, "blimp_coordinate_structure_constraint_complex_left_branch": 0, "blimp_coordinate_structure_constraint_object_extraction": 0, "blimp_determiner_noun_agreement_1": 0, "blimp_determiner_noun_agreement_2": 0, "blimp_determiner_noun_agreement_irregular_1": 0, "blimp_determiner_noun_agreement_irregular_2": 0, "blimp_determiner_noun_agreement_with_adj_2": 0, "blimp_determiner_noun_agreement_with_adj_irregular_1": 0, "blimp_determiner_noun_agreement_with_adj_irregular_2": 0, "blimp_determiner_noun_agreement_with_adjective_1": 0, "blimp_distractor_agreement_relational_noun": 0, "blimp_distractor_agreement_relative_clause": 0, "blimp_drop_argument": 0, "blimp_ellipsis_n_bar_1": 0, "blimp_ellipsis_n_bar_2": 0, "blimp_existential_there_object_raising": 0, "blimp_existential_there_quantifiers_1": 0, "blimp_existential_there_quantifiers_2": 0, "blimp_existential_there_subject_raising": 0, "blimp_expletive_it_object_raising": 0, "blimp_inchoative": 0, "blimp_intransitive": 0, "blimp_irregular_past_participle_adjectives": 0, "blimp_irregular_past_participle_verbs": 0, "blimp_irregular_plural_subject_verb_agreement_1": 0, "blimp_irregular_plural_subject_verb_agreement_2": 0, "blimp_left_branch_island_echo_question": 0, "blimp_left_branch_island_simple_question": 0, "blimp_matrix_question_npi_licensor_present": 0, "blimp_npi_present_1": 0, "blimp_npi_present_2": 0, "blimp_only_npi_licensor_present": 0, "blimp_only_npi_scope": 0, "blimp_passive_1": 0, "blimp_passive_2": 0, "blimp_principle_A_c_command": 0, "blimp_principle_A_case_1": 0, "blimp_principle_A_case_2": 0, "blimp_principle_A_domain_1": 0, "blimp_principle_A_domain_2": 0, "blimp_principle_A_domain_3": 0, "blimp_principle_A_reconstruction": 0, "blimp_regular_plural_subject_verb_agreement_1": 0, "blimp_regular_plural_subject_verb_agreement_2": 0, "blimp_sentential_negation_npi_licensor_present": 0, "blimp_sentential_negation_npi_scope": 0, "blimp_sentential_subject_island": 0, "blimp_superlative_quantifiers_1": 0, "blimp_superlative_quantifiers_2": 0, "blimp_tough_vs_raising_1": 0, "blimp_tough_vs_raising_2": 0, "blimp_transitive": 0, "blimp_wh_island": 0, "blimp_wh_questions_object_gap": 0, "blimp_wh_questions_subject_gap": 0, "blimp_wh_questions_subject_gap_long_distance": 0, "blimp_wh_vs_that_no_gap": 0, "blimp_wh_vs_that_no_gap_long_distance": 0, "blimp_wh_vs_that_with_gap": 0, "blimp_wh_vs_that_with_gap_long_distance": 0, "hellaswag": 0, "lambada_openai": 0, "piqa": 0, "sciq": 0 }, "pretty_env_info": "PyTorch version: 2.8.0+cu128\nIs debug build: False\nCUDA used to build PyTorch: 12.8\nROCM used to build PyTorch: N/A\n\nOS: Ubuntu 24.04.3 LTS (x86_64)\nGCC version: (Ubuntu 13.3.0-6ubuntu2~24.04) 13.3.0\nClang version: Could not collect\nCMake version: version 3.28.3\nLibc version: glibc-2.39\n\nPython version: 3.12.3 (main, Aug 14 2025, 17:47:21) [GCC 13.3.0] (64-bit runtime)\nPython platform: Linux-6.8.0-64-generic-x86_64-with-glibc2.39\nIs CUDA available: True\nCUDA runtime version: 12.8.93\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: \nGPU 0: NVIDIA GeForce RTX 4090\nGPU 1: NVIDIA GeForce RTX 4090\n\nNvidia driver version: 570.158.01\ncuDNN version: Probably one of the following:\n/usr/lib/x86_64-linux-gnu/libcudnn.so.9.8.0\n/usr/lib/x86_64-linux-gnu/libcudnn_adv.so.9.8.0\n/usr/lib/x86_64-linux-gnu/libcudnn_cnn.so.9.8.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_precompiled.so.9.8.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_runtime_compiled.so.9.8.0\n/usr/lib/x86_64-linux-gnu/libcudnn_graph.so.9.8.0\n/usr/lib/x86_64-linux-gnu/libcudnn_heuristic.so.9.8.0\n/usr/lib/x86_64-linux-gnu/libcudnn_ops.so.9.8.0\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 43 bits physical, 48 bits virtual\nByte Order: Little Endian\nCPU(s): 96\nOn-line CPU(s) list: 0-95\nVendor ID: AuthenticAMD\nModel name: AMD EPYC 7K62 48-Core Processor\nCPU family: 23\nModel: 49\nThread(s) per core: 2\nCore(s) per socket: 48\nSocket(s): 1\nStepping: 0\nFrequency boost: enabled\nCPU(s) scaling MHz: 66%\nCPU max MHz: 2600.0000\nCPU min MHz: 1500.0000\nBogoMIPS: 5200.49\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush mmx fxsr sse sse2 ht syscall nx mmxext fxsr_opt pdpe1gb rdtscp lm constant_tsc rep_good nopl nonstop_tsc cpuid extd_apicid aperfmperf rapl pni pclmulqdq monitor ssse3 fma cx16 sse4_1 sse4_2 movbe popcnt aes xsave avx f16c rdrand lahf_lm cmp_legacy svm extapic cr8_legacy abm sse4a misalignsse 3dnowprefetch osvw ibs skinit wdt tce topoext perfctr_core perfctr_nb bpext perfctr_llc mwaitx cpb cat_l3 cdp_l3 hw_pstate ssbd mba ibrs ibpb stibp vmmcall fsgsbase bmi1 avx2 smep bmi2 cqm rdt_a rdseed adx smap clflushopt clwb sha_ni xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local clzero irperf xsaveerptr rdpru wbnoinvd amd_ppin arat npt lbrv svm_lock nrip_save tsc_scale vmcb_clean flushbyasid decodeassists pausefilter pfthreshold avic v_vmsave_vmload vgif v_spec_ctrl umip rdpid overflow_recov succor smca sev sev_es\nVirtualization: AMD-V\nL1d cache: 1.5 MiB (48 instances)\nL1i cache: 1.5 MiB (48 instances)\nL2 cache: 24 MiB (48 instances)\nL3 cache: 192 MiB (12 instances)\nNUMA node(s): 1\nNUMA node0 CPU(s): 0-95\nVulnerability Gather data sampling: Not affected\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Mmio stale data: Not affected\nVulnerability Reg file data sampling: Not affected\nVulnerability Retbleed: Mitigation; untrained return thunk; SMT enabled with STIBP protection\nVulnerability Spec rstack overflow: Mitigation; Safe RET\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Retpolines; IBPB conditional; STIBP always-on; RSB filling; PBRSB-eIBRS Not affected; BHI Not affected\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] numpy==2.1.2\n[pip3] nvidia-cublas-cu12==12.8.4.1\n[pip3] nvidia-cuda-cupti-cu12==12.8.90\n[pip3] nvidia-cuda-nvrtc-cu12==12.8.93\n[pip3] nvidia-cuda-runtime-cu12==12.8.90\n[pip3] nvidia-cudnn-cu12==9.10.2.21\n[pip3] nvidia-cufft-cu12==11.3.3.83\n[pip3] nvidia-curand-cu12==10.3.9.90\n[pip3] nvidia-cusolver-cu12==11.7.3.90\n[pip3] nvidia-cusparse-cu12==12.5.8.93\n[pip3] nvidia-cusparselt-cu12==0.7.1\n[pip3] nvidia-nccl-cu12==2.27.3\n[pip3] nvidia-nvjitlink-cu12==12.8.93\n[pip3] nvidia-nvtx-cu12==12.8.90\n[pip3] torch==2.8.0+cu128\n[pip3] torchaudio==2.8.0+cu128\n[pip3] torchvision==0.23.0+cu128\n[pip3] triton==3.4.0\n[conda] Could not collect", "results": { "arc_challenge": { "acc,none": 0.20819112627986347, "acc_norm,none": 0.24146757679180889, "acc_norm_stderr,none": "N/A", "acc_stderr,none": "N/A", "alias": "arc_challenge", "name": "arc_challenge", "sample_len": 1172 }, "arc_easy": { "acc,none": 0.4659090909090909, "acc_norm,none": 0.42297979797979796, "acc_norm_stderr,none": "N/A", "acc_stderr,none": "N/A", "alias": "arc_easy", "name": "arc_easy", "sample_len": 2376 }, "blimp": { "acc,none": 0.7669701492537313, "acc_stderr,none": "N/A", "alias": "blimp", "name": "blimp", "sample_count": { "acc,none": 67000 }, "sample_len": 67000 }, "blimp_adjunct_island": { "acc,none": 0.834, "acc_stderr,none": "N/A", "alias": "blimp_adjunct_island", "name": "blimp_adjunct_island", "sample_len": 1000 }, "blimp_anaphor_gender_agreement": { "acc,none": 0.971, "acc_stderr,none": "N/A", "alias": "blimp_anaphor_gender_agreement", "name": "blimp_anaphor_gender_agreement", "sample_len": 1000 }, "blimp_anaphor_number_agreement": { "acc,none": 0.989, "acc_stderr,none": "N/A", "alias": "blimp_anaphor_number_agreement", "name": "blimp_anaphor_number_agreement", "sample_len": 1000 }, "blimp_animate_subject_passive": { "acc,none": 0.773, "acc_stderr,none": "N/A", "alias": "blimp_animate_subject_passive", "name": "blimp_animate_subject_passive", "sample_len": 1000 }, "blimp_animate_subject_trans": { "acc,none": 0.864, "acc_stderr,none": "N/A", "alias": "blimp_animate_subject_trans", "name": "blimp_animate_subject_trans", "sample_len": 1000 }, "blimp_causative": { "acc,none": 0.662, "acc_stderr,none": "N/A", "alias": "blimp_causative", "name": "blimp_causative", "sample_len": 1000 }, "blimp_complex_NP_island": { "acc,none": 0.431, "acc_stderr,none": "N/A", "alias": "blimp_complex_NP_island", "name": "blimp_complex_NP_island", "sample_len": 1000 }, "blimp_coordinate_structure_constraint_complex_left_branch": { "acc,none": 0.512, "acc_stderr,none": "N/A", "alias": "blimp_coordinate_structure_constraint_complex_left_branch", "name": "blimp_coordinate_structure_constraint_complex_left_branch", "sample_len": 1000 }, "blimp_coordinate_structure_constraint_object_extraction": { "acc,none": 0.812, "acc_stderr,none": "N/A", "alias": "blimp_coordinate_structure_constraint_object_extraction", "name": "blimp_coordinate_structure_constraint_object_extraction", "sample_len": 1000 }, "blimp_determiner_noun_agreement_1": { "acc,none": 0.977, "acc_stderr,none": "N/A", "alias": "blimp_determiner_noun_agreement_1", "name": "blimp_determiner_noun_agreement_1", "sample_len": 1000 }, "blimp_determiner_noun_agreement_2": { "acc,none": 0.939, "acc_stderr,none": "N/A", "alias": "blimp_determiner_noun_agreement_2", "name": "blimp_determiner_noun_agreement_2", "sample_len": 1000 }, "blimp_determiner_noun_agreement_irregular_1": { "acc,none": 0.92, "acc_stderr,none": "N/A", "alias": "blimp_determiner_noun_agreement_irregular_1", "name": "blimp_determiner_noun_agreement_irregular_1", "sample_len": 1000 }, "blimp_determiner_noun_agreement_irregular_2": { "acc,none": 0.93, "acc_stderr,none": "N/A", "alias": "blimp_determiner_noun_agreement_irregular_2", "name": "blimp_determiner_noun_agreement_irregular_2", "sample_len": 1000 }, "blimp_determiner_noun_agreement_with_adj_2": { "acc,none": 0.909, "acc_stderr,none": "N/A", "alias": "blimp_determiner_noun_agreement_with_adj_2", "name": "blimp_determiner_noun_agreement_with_adj_2", "sample_len": 1000 }, "blimp_determiner_noun_agreement_with_adj_irregular_1": { "acc,none": 0.873, "acc_stderr,none": "N/A", "alias": "blimp_determiner_noun_agreement_with_adj_irregular_1", "name": "blimp_determiner_noun_agreement_with_adj_irregular_1", "sample_len": 1000 }, "blimp_determiner_noun_agreement_with_adj_irregular_2": { "acc,none": 0.905, "acc_stderr,none": "N/A", "alias": "blimp_determiner_noun_agreement_with_adj_irregular_2", "name": "blimp_determiner_noun_agreement_with_adj_irregular_2", "sample_len": 1000 }, "blimp_determiner_noun_agreement_with_adjective_1": { "acc,none": 0.95, "acc_stderr,none": "N/A", "alias": "blimp_determiner_noun_agreement_with_adjective_1", "name": "blimp_determiner_noun_agreement_with_adjective_1", "sample_len": 1000 }, "blimp_distractor_agreement_relational_noun": { "acc,none": 0.641, "acc_stderr,none": "N/A", "alias": "blimp_distractor_agreement_relational_noun", "name": "blimp_distractor_agreement_relational_noun", "sample_len": 1000 }, "blimp_distractor_agreement_relative_clause": { "acc,none": 0.576, "acc_stderr,none": "N/A", "alias": "blimp_distractor_agreement_relative_clause", "name": "blimp_distractor_agreement_relative_clause", "sample_len": 1000 }, "blimp_drop_argument": { "acc,none": 0.749, "acc_stderr,none": "N/A", "alias": "blimp_drop_argument", "name": "blimp_drop_argument", "sample_len": 1000 }, "blimp_ellipsis_n_bar_1": { "acc,none": 0.801, "acc_stderr,none": "N/A", "alias": "blimp_ellipsis_n_bar_1", "name": "blimp_ellipsis_n_bar_1", "sample_len": 1000 }, "blimp_ellipsis_n_bar_2": { "acc,none": 0.89, "acc_stderr,none": "N/A", "alias": "blimp_ellipsis_n_bar_2", "name": "blimp_ellipsis_n_bar_2", "sample_len": 1000 }, "blimp_existential_there_object_raising": { "acc,none": 0.747, "acc_stderr,none": "N/A", "alias": "blimp_existential_there_object_raising", "name": "blimp_existential_there_object_raising", "sample_len": 1000 }, "blimp_existential_there_quantifiers_1": { "acc,none": 0.988, "acc_stderr,none": "N/A", "alias": "blimp_existential_there_quantifiers_1", "name": "blimp_existential_there_quantifiers_1", "sample_len": 1000 }, "blimp_existential_there_quantifiers_2": { "acc,none": 0.376, "acc_stderr,none": "N/A", "alias": "blimp_existential_there_quantifiers_2", "name": "blimp_existential_there_quantifiers_2", "sample_len": 1000 }, "blimp_existential_there_subject_raising": { "acc,none": 0.838, "acc_stderr,none": "N/A", "alias": "blimp_existential_there_subject_raising", "name": "blimp_existential_there_subject_raising", "sample_len": 1000 }, "blimp_expletive_it_object_raising": { "acc,none": 0.748, "acc_stderr,none": "N/A", "alias": "blimp_expletive_it_object_raising", "name": "blimp_expletive_it_object_raising", "sample_len": 1000 }, "blimp_inchoative": { "acc,none": 0.626, "acc_stderr,none": "N/A", "alias": "blimp_inchoative", "name": "blimp_inchoative", "sample_len": 1000 }, "blimp_intransitive": { "acc,none": 0.783, "acc_stderr,none": "N/A", "alias": "blimp_intransitive", "name": "blimp_intransitive", "sample_len": 1000 }, "blimp_irregular_past_participle_adjectives": { "acc,none": 0.938, "acc_stderr,none": "N/A", "alias": "blimp_irregular_past_participle_adjectives", "name": "blimp_irregular_past_participle_adjectives", "sample_len": 1000 }, "blimp_irregular_past_participle_verbs": { "acc,none": 0.879, "acc_stderr,none": "N/A", "alias": "blimp_irregular_past_participle_verbs", "name": "blimp_irregular_past_participle_verbs", "sample_len": 1000 }, "blimp_irregular_plural_subject_verb_agreement_1": { "acc,none": 0.875, "acc_stderr,none": "N/A", "alias": "blimp_irregular_plural_subject_verb_agreement_1", "name": "blimp_irregular_plural_subject_verb_agreement_1", "sample_len": 1000 }, "blimp_irregular_plural_subject_verb_agreement_2": { "acc,none": 0.902, "acc_stderr,none": "N/A", "alias": "blimp_irregular_plural_subject_verb_agreement_2", "name": "blimp_irregular_plural_subject_verb_agreement_2", "sample_len": 1000 }, "blimp_left_branch_island_echo_question": { "acc,none": 0.306, "acc_stderr,none": "N/A", "alias": "blimp_left_branch_island_echo_question", "name": "blimp_left_branch_island_echo_question", "sample_len": 1000 }, "blimp_left_branch_island_simple_question": { "acc,none": 0.591, "acc_stderr,none": "N/A", "alias": "blimp_left_branch_island_simple_question", "name": "blimp_left_branch_island_simple_question", "sample_len": 1000 }, "blimp_matrix_question_npi_licensor_present": { "acc,none": 0.562, "acc_stderr,none": "N/A", "alias": "blimp_matrix_question_npi_licensor_present", "name": "blimp_matrix_question_npi_licensor_present", "sample_len": 1000 }, "blimp_npi_present_1": { "acc,none": 0.422, "acc_stderr,none": "N/A", "alias": "blimp_npi_present_1", "name": "blimp_npi_present_1", "sample_len": 1000 }, "blimp_npi_present_2": { "acc,none": 0.538, "acc_stderr,none": "N/A", "alias": "blimp_npi_present_2", "name": "blimp_npi_present_2", "sample_len": 1000 }, "blimp_only_npi_licensor_present": { "acc,none": 0.929, "acc_stderr,none": "N/A", "alias": "blimp_only_npi_licensor_present", "name": "blimp_only_npi_licensor_present", "sample_len": 1000 }, "blimp_only_npi_scope": { "acc,none": 0.734, "acc_stderr,none": "N/A", "alias": "blimp_only_npi_scope", "name": "blimp_only_npi_scope", "sample_len": 1000 }, "blimp_passive_1": { "acc,none": 0.899, "acc_stderr,none": "N/A", "alias": "blimp_passive_1", "name": "blimp_passive_1", "sample_len": 1000 }, "blimp_passive_2": { "acc,none": 0.88, "acc_stderr,none": "N/A", "alias": "blimp_passive_2", "name": "blimp_passive_2", "sample_len": 1000 }, "blimp_principle_A_c_command": { "acc,none": 0.616, "acc_stderr,none": "N/A", "alias": "blimp_principle_A_c_command", "name": "blimp_principle_A_c_command", "sample_len": 1000 }, "blimp_principle_A_case_1": { "acc,none": 1.0, "acc_stderr,none": "N/A", "alias": "blimp_principle_A_case_1", "name": "blimp_principle_A_case_1", "sample_len": 1000 }, "blimp_principle_A_case_2": { "acc,none": 0.952, "acc_stderr,none": "N/A", "alias": "blimp_principle_A_case_2", "name": "blimp_principle_A_case_2", "sample_len": 1000 }, "blimp_principle_A_domain_1": { "acc,none": 0.965, "acc_stderr,none": "N/A", "alias": "blimp_principle_A_domain_1", "name": "blimp_principle_A_domain_1", "sample_len": 1000 }, "blimp_principle_A_domain_2": { "acc,none": 0.763, "acc_stderr,none": "N/A", "alias": "blimp_principle_A_domain_2", "name": "blimp_principle_A_domain_2", "sample_len": 1000 }, "blimp_principle_A_domain_3": { "acc,none": 0.596, "acc_stderr,none": "N/A", "alias": "blimp_principle_A_domain_3", "name": "blimp_principle_A_domain_3", "sample_len": 1000 }, "blimp_principle_A_reconstruction": { "acc,none": 0.27, "acc_stderr,none": "N/A", "alias": "blimp_principle_A_reconstruction", "name": "blimp_principle_A_reconstruction", "sample_len": 1000 }, "blimp_regular_plural_subject_verb_agreement_1": { "acc,none": 0.913, "acc_stderr,none": "N/A", "alias": "blimp_regular_plural_subject_verb_agreement_1", "name": "blimp_regular_plural_subject_verb_agreement_1", "sample_len": 1000 }, "blimp_regular_plural_subject_verb_agreement_2": { "acc,none": 0.897, "acc_stderr,none": "N/A", "alias": "blimp_regular_plural_subject_verb_agreement_2", "name": "blimp_regular_plural_subject_verb_agreement_2", "sample_len": 1000 }, "blimp_sentential_negation_npi_licensor_present": { "acc,none": 0.989, "acc_stderr,none": "N/A", "alias": "blimp_sentential_negation_npi_licensor_present", "name": "blimp_sentential_negation_npi_licensor_present", "sample_len": 1000 }, "blimp_sentential_negation_npi_scope": { "acc,none": 0.593, "acc_stderr,none": "N/A", "alias": "blimp_sentential_negation_npi_scope", "name": "blimp_sentential_negation_npi_scope", "sample_len": 1000 }, "blimp_sentential_subject_island": { "acc,none": 0.326, "acc_stderr,none": "N/A", "alias": "blimp_sentential_subject_island", "name": "blimp_sentential_subject_island", "sample_len": 1000 }, "blimp_superlative_quantifiers_1": { "acc,none": 0.85, "acc_stderr,none": "N/A", "alias": "blimp_superlative_quantifiers_1", "name": "blimp_superlative_quantifiers_1", "sample_len": 1000 }, "blimp_superlative_quantifiers_2": { "acc,none": 0.794, "acc_stderr,none": "N/A", "alias": "blimp_superlative_quantifiers_2", "name": "blimp_superlative_quantifiers_2", "sample_len": 1000 }, "blimp_tough_vs_raising_1": { "acc,none": 0.593, "acc_stderr,none": "N/A", "alias": "blimp_tough_vs_raising_1", "name": "blimp_tough_vs_raising_1", "sample_len": 1000 }, "blimp_tough_vs_raising_2": { "acc,none": 0.83, "acc_stderr,none": "N/A", "alias": "blimp_tough_vs_raising_2", "name": "blimp_tough_vs_raising_2", "sample_len": 1000 }, "blimp_transitive": { "acc,none": 0.821, "acc_stderr,none": "N/A", "alias": "blimp_transitive", "name": "blimp_transitive", "sample_len": 1000 }, "blimp_wh_island": { "acc,none": 0.772, "acc_stderr,none": "N/A", "alias": "blimp_wh_island", "name": "blimp_wh_island", "sample_len": 1000 }, "blimp_wh_questions_object_gap": { "acc,none": 0.786, "acc_stderr,none": "N/A", "alias": "blimp_wh_questions_object_gap", "name": "blimp_wh_questions_object_gap", "sample_len": 1000 }, "blimp_wh_questions_subject_gap": { "acc,none": 0.929, "acc_stderr,none": "N/A", "alias": "blimp_wh_questions_subject_gap", "name": "blimp_wh_questions_subject_gap", "sample_len": 1000 }, "blimp_wh_questions_subject_gap_long_distance": { "acc,none": 0.931, "acc_stderr,none": "N/A", "alias": "blimp_wh_questions_subject_gap_long_distance", "name": "blimp_wh_questions_subject_gap_long_distance", "sample_len": 1000 }, "blimp_wh_vs_that_no_gap": { "acc,none": 0.962, "acc_stderr,none": "N/A", "alias": "blimp_wh_vs_that_no_gap", "name": "blimp_wh_vs_that_no_gap", "sample_len": 1000 }, "blimp_wh_vs_that_no_gap_long_distance": { "acc,none": 0.978, "acc_stderr,none": "N/A", "alias": "blimp_wh_vs_that_no_gap_long_distance", "name": "blimp_wh_vs_that_no_gap_long_distance", "sample_len": 1000 }, "blimp_wh_vs_that_with_gap": { "acc,none": 0.541, "acc_stderr,none": "N/A", "alias": "blimp_wh_vs_that_with_gap", "name": "blimp_wh_vs_that_with_gap", "sample_len": 1000 }, "blimp_wh_vs_that_with_gap_long_distance": { "acc,none": 0.251, "acc_stderr,none": "N/A", "alias": "blimp_wh_vs_that_with_gap_long_distance", "name": "blimp_wh_vs_that_with_gap_long_distance", "sample_len": 1000 }, "hellaswag": { "acc,none": 0.2818163712407887, "acc_norm,none": 0.29874526986656047, "acc_norm_stderr,none": "N/A", "acc_stderr,none": "N/A", "alias": "hellaswag", "name": "hellaswag", "sample_len": 10042 }, "lambada_openai": { "acc,none": 0.2780904327576169, "acc_stderr,none": "N/A", "alias": "lambada_openai", "name": "lambada_openai", "perplexity,none": 53.66752251060021, "perplexity_stderr,none": "N/A", "sample_len": 5153 }, "piqa": { "acc,none": 0.6137105549510338, "acc_norm,none": 0.6033732317736671, "acc_norm_stderr,none": "N/A", "acc_stderr,none": "N/A", "alias": "piqa", "name": "piqa", "sample_len": 1838 }, "sciq": { "acc,none": 0.737, "acc_norm,none": 0.658, "acc_norm_stderr,none": "N/A", "acc_stderr,none": "N/A", "alias": "sciq", "name": "sciq", "sample_len": 1000 } }, "transformers_version": "5.15.1", "upper_git_hash": null, "versions": { "arc_challenge": 1.0, "arc_easy": 1.0, "blimp": "2.0", "blimp_adjunct_island": 1.0, "blimp_anaphor_gender_agreement": 1.0, "blimp_anaphor_number_agreement": 1.0, "blimp_animate_subject_passive": 1.0, "blimp_animate_subject_trans": 1.0, "blimp_causative": 1.0, "blimp_complex_NP_island": 1.0, "blimp_coordinate_structure_constraint_complex_left_branch": 1.0, "blimp_coordinate_structure_constraint_object_extraction": 1.0, "blimp_determiner_noun_agreement_1": 1.0, "blimp_determiner_noun_agreement_2": 1.0, "blimp_determiner_noun_agreement_irregular_1": 1.0, "blimp_determiner_noun_agreement_irregular_2": 1.0, "blimp_determiner_noun_agreement_with_adj_2": 1.0, "blimp_determiner_noun_agreement_with_adj_irregular_1": 1.0, "blimp_determiner_noun_agreement_with_adj_irregular_2": 1.0, "blimp_determiner_noun_agreement_with_adjective_1": 1.0, "blimp_distractor_agreement_relational_noun": 1.0, "blimp_distractor_agreement_relative_clause": 1.0, "blimp_drop_argument": 1.0, "blimp_ellipsis_n_bar_1": 1.0, "blimp_ellipsis_n_bar_2": 1.0, "blimp_existential_there_object_raising": 1.0, "blimp_existential_there_quantifiers_1": 1.0, "blimp_existential_there_quantifiers_2": 1.0, "blimp_existential_there_subject_raising": 1.0, "blimp_expletive_it_object_raising": 1.0, "blimp_inchoative": 1.0, "blimp_intransitive": 1.0, "blimp_irregular_past_participle_adjectives": 1.0, "blimp_irregular_past_participle_verbs": 1.0, "blimp_irregular_plural_subject_verb_agreement_1": 1.0, "blimp_irregular_plural_subject_verb_agreement_2": 1.0, "blimp_left_branch_island_echo_question": 1.0, "blimp_left_branch_island_simple_question": 1.0, "blimp_matrix_question_npi_licensor_present": 1.0, "blimp_npi_present_1": 1.0, "blimp_npi_present_2": 1.0, "blimp_only_npi_licensor_present": 1.0, "blimp_only_npi_scope": 1.0, "blimp_passive_1": 1.0, "blimp_passive_2": 1.0, "blimp_principle_A_c_command": 1.0, "blimp_principle_A_case_1": 1.0, "blimp_principle_A_case_2": 1.0, "blimp_principle_A_domain_1": 1.0, "blimp_principle_A_domain_2": 1.0, "blimp_principle_A_domain_3": 1.0, "blimp_principle_A_reconstruction": 1.0, "blimp_regular_plural_subject_verb_agreement_1": 1.0, "blimp_regular_plural_subject_verb_agreement_2": 1.0, "blimp_sentential_negation_npi_licensor_present": 1.0, "blimp_sentential_negation_npi_scope": 1.0, "blimp_sentential_subject_island": 1.0, "blimp_superlative_quantifiers_1": 1.0, "blimp_superlative_quantifiers_2": 1.0, "blimp_tough_vs_raising_1": 1.0, "blimp_tough_vs_raising_2": 1.0, "blimp_transitive": 1.0, "blimp_wh_island": 1.0, "blimp_wh_questions_object_gap": 1.0, "blimp_wh_questions_subject_gap": 1.0, "blimp_wh_questions_subject_gap_long_distance": 1.0, "blimp_wh_vs_that_no_gap": 1.0, "blimp_wh_vs_that_no_gap_long_distance": 1.0, "blimp_wh_vs_that_with_gap": 1.0, "blimp_wh_vs_that_with_gap_long_distance": 1.0, "hellaswag": 1.0, "lambada_openai": 1.0, "piqa": 1.0, "sciq": 1.0 } }, "results": { "arc-challenge": { "language": "English", "links": [ { "label": "ARC-Challenge", "url": "https://allenai.org/data/arc" } ], "metrics": { "acc,none": 0.20819112627986347, "acc_norm,none": 0.24146757679180889, "sample_len": 1172.0 }, "name": "ARC-Challenge", "primary_metric": "acc_norm", "primary_value": 0.24146757679180889, "protocol": { "implementation": "lm-evaluation-harness", "num_fewshot": 0, "task": "arc_challenge", "version": "0.4.12" }, "samples": 1172 }, "arc-easy": { "language": "English", "links": [ { "label": "ARC-Easy", "url": "https://allenai.org/data/arc" } ], "metrics": { "acc,none": 0.4659090909090909, "acc_norm,none": 0.42297979797979796, "sample_len": 2376.0 }, "name": "ARC-Easy", "primary_metric": "acc_norm", "primary_value": 0.42297979797979796, "protocol": { "implementation": "lm-evaluation-harness", "num_fewshot": 0, "task": "arc_easy", "version": "0.4.12" }, "samples": 2376 }, "blimp": { "language": "English", "links": [ { "label": "BLiMP", "url": "https://github.com/alexwarstadt/blimp" } ], "metrics": { "acc,none": 0.7669701492537313, "sample_len": 67000.0 }, "name": "BLiMP", "primary_metric": "acc", "primary_value": 0.7669701492537313, "protocol": { "implementation": "lm-evaluation-harness", "num_fewshot": 0, "task": "blimp", "version": "0.4.12" }, "samples": 67000 }, "hellaswag": { "language": "English", "links": [ { "label": "HellaSwag", "url": "https://rowanzellers.com/hellaswag/" } ], "metrics": { "acc,none": 0.2818163712407887, "acc_norm,none": 0.29874526986656047, "sample_len": 10042.0 }, "name": "HellaSwag", "primary_metric": "acc_norm", "primary_value": 0.29874526986656047, "protocol": { "implementation": "lm-evaluation-harness", "num_fewshot": 0, "task": "hellaswag", "version": "0.4.12" }, "samples": 10042 }, "lambada": { "language": "English", "links": [ { "label": "LAMBADA", "url": "https://huggingface.co/datasets/EleutherAI/lambada_openai" } ], "metrics": { "acc,none": 0.2780904327576169, "perplexity,none": 53.66752251060021, "sample_len": 5153.0 }, "name": "LAMBADA", "primary_metric": "acc", "primary_value": 0.2780904327576169, "protocol": { "implementation": "lm-evaluation-harness", "num_fewshot": 0, "task": "lambada_openai", "version": "0.4.12" }, "samples": 5153 }, "piqa": { "language": "English", "links": [ { "label": "PIQA", "url": "https://yonatanbisk.com/piqa/" } ], "metrics": { "acc,none": 0.6137105549510338, "acc_norm,none": 0.6033732317736671, "sample_len": 1838.0 }, "name": "PIQA", "primary_metric": "acc_norm", "primary_value": 0.6033732317736671, "protocol": { "implementation": "lm-evaluation-harness", "num_fewshot": 0, "task": "piqa", "version": "0.4.12" }, "samples": 1838 }, "sciq": { "language": "English", "links": [ { "label": "SciQ", "url": "https://allenai.org/data/sciq" } ], "metrics": { "acc,none": 0.737, "acc_norm,none": 0.658, "sample_len": 1000.0 }, "name": "SciQ", "primary_metric": "acc_norm", "primary_value": 0.658, "protocol": { "implementation": "lm-evaluation-harness", "num_fewshot": 0, "task": "sciq", "version": "0.4.12" }, "samples": 1000 } }, "run_name": "minimal-en-125m-4ep", "schema_version": 1, "selected": [ "blimp", "lambada", "hellaswag", "piqa", "sciq", "arc-easy", "arc-challenge" ], "tokenizer": { "name": "minimal-en-12k", "reference": { "name": "minimal-en-12k", "path": "/workspace/datasets/minimal-en-corpus-2.5b/tokenizer/tokenizer.json" } } }