task: humaneval dataset_path: openai/openai_humaneval unsafe_code: true output_type: generate_until test_split: test doc_to_text: "{{prompt}}" doc_to_target: "{{test}}\ncheck({{entry_point}})" process_docs: !function utils.process_docs metric_list: - metric: !function utils.pass_at_1 aggregation: mean higher_is_better: true generation_kwargs: until: - "\nclass" - "\ndef" - "\n#" - "\nif" - "\nprint" - "\n```\n" max_gen_toks: 2048 do_sample: false repeats: 1 num_fewshot: 0 # filter_list: # - name: "create_test" # filter: # - function: "custom" # filter_fn: !function utils.build_predictions metadata: version: 1.0