Rarri's picture
Add model card, recipe, security review, and lm-eval benchmarks
2941b49 verified
Raw
History Blame Contribute Delete
15.2 kB
{
"results": {
"arc_challenge": {
"name": "arc_challenge",
"alias": "arc_challenge",
"sample_len": 300,
"acc,none": 0.6233333333333333,
"acc_stderr,none": 0.028022261151265607,
"acc_norm,none": 0.6333333333333333,
"acc_norm_stderr,none": 0.027868673283383917
},
"hellaswag": {
"name": "hellaswag",
"alias": "hellaswag",
"sample_len": 300,
"acc,none": 0.57,
"acc_stderr,none": 0.02863096997084747,
"acc_norm,none": 0.7133333333333334,
"acc_norm_stderr,none": 0.026151660126798617
},
"winogrande": {
"name": "winogrande",
"alias": "winogrande",
"sample_len": 300,
"acc,none": 0.7966666666666666,
"acc_stderr,none": 0.02327592874967972
},
"piqa": {
"name": "piqa",
"alias": "piqa",
"sample_len": 300,
"acc,none": 0.83,
"acc_stderr,none": 0.021723405684078406,
"acc_norm,none": 0.8766666666666667,
"acc_norm_stderr,none": 0.019016119040222487
},
"openbookqa": {
"name": "openbookqa",
"alias": "openbookqa",
"sample_len": 300,
"acc,none": 0.36333333333333334,
"acc_stderr,none": 0.027814616968981954,
"acc_norm,none": 0.5133333333333333,
"acc_norm_stderr,none": 0.028905463615555037
},
"truthfulqa_mc2": {
"name": "truthfulqa_mc2",
"alias": "truthfulqa_mc2",
"sample_len": 300,
"acc,none": 0.5494148741509903,
"acc_stderr,none": 0.02519930229067
}
},
"group_subtasks": {},
"configs": {
"arc_challenge": {
"task": "arc_challenge",
"dataset_path": "allenai/ai2_arc",
"dataset_name": "ARC-Challenge",
"training_split": "train",
"validation_split": "validation",
"test_split": "test",
"doc_to_text": "Question: {{question}}\nAnswer:",
"doc_to_target": "{{choices.label.index(answerKey)}}",
"unsafe_code": false,
"doc_to_choice": "{{choices.text}}",
"description": "",
"target_delimiter": " ",
"fewshot_delimiter": "\n\n",
"fewshot_config": {
"sampler": "default",
"split": null,
"process_docs": null,
"fewshot_indices": null,
"samples": null,
"doc_to_text": "Question: {{question}}\nAnswer:",
"doc_to_choice": "{{choices.text}}",
"doc_to_target": "{{choices.label.index(answerKey)}}",
"gen_prefix": null,
"fewshot_delimiter": "\n\n",
"target_delimiter": " "
},
"num_fewshot": 0,
"metric_list": [
{
"metric": "acc",
"aggregation": "mean",
"higher_is_better": true
},
{
"metric": "acc_norm",
"aggregation": "mean",
"higher_is_better": true
}
],
"output_type": "multiple_choice",
"repeats": 1,
"should_decontaminate": true,
"doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
"metadata": {
"version": 1.0,
"model": "dsv4-hybrid",
"base_url": "http://localhost:PORT/v1/completions",
"num_concurrent": 16,
"max_retries": 3,
"tokenized_requests": false,
"tokenizer": "<model_path>",
"tokenizer_backend": "huggingface",
"config_source": "<lm_eval>/tasks/arc/arc_challenge.yaml"
}
},
"hellaswag": {
"task": "hellaswag",
"dataset_path": "Rowan/hellaswag",
"training_split": "train",
"validation_split": "validation",
"process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n",
"doc_to_text": "{{query}}",
"doc_to_target": "{{label}}",
"unsafe_code": false,
"doc_to_choice": "choices",
"description": "",
"target_delimiter": " ",
"fewshot_delimiter": "\n\n",
"fewshot_config": {
"sampler": "default",
"split": null,
"process_docs": "<function process_docs at 0x7bba60a1e200>",
"fewshot_indices": null,
"samples": null,
"doc_to_text": "{{query}}",
"doc_to_choice": "choices",
"doc_to_target": "{{label}}",
"gen_prefix": null,
"fewshot_delimiter": "\n\n",
"target_delimiter": " "
},
"num_fewshot": 0,
"metric_list": [
{
"metric": "acc",
"aggregation": "mean",
"higher_is_better": true
},
{
"metric": "acc_norm",
"aggregation": "mean",
"higher_is_better": true
}
],
"output_type": "multiple_choice",
"repeats": 1,
"should_decontaminate": false,
"metadata": {
"version": 1.0,
"model": "dsv4-hybrid",
"base_url": "http://localhost:PORT/v1/completions",
"num_concurrent": 16,
"max_retries": 3,
"tokenized_requests": false,
"tokenizer": "<model_path>",
"tokenizer_backend": "huggingface",
"config_source": "<lm_eval>/tasks/hellaswag/hellaswag.yaml"
}
},
"openbookqa": {
"task": "openbookqa",
"dataset_path": "allenai/openbookqa",
"dataset_name": "main",
"training_split": "train",
"validation_split": "validation",
"test_split": "test",
"doc_to_text": "question_stem",
"doc_to_target": "{{choices.label.index(answerKey.lstrip())}}",
"unsafe_code": false,
"doc_to_choice": "{{choices.text}}",
"description": "",
"target_delimiter": " ",
"fewshot_delimiter": "\n\n",
"fewshot_config": {
"sampler": "default",
"split": null,
"process_docs": null,
"fewshot_indices": null,
"samples": null,
"doc_to_text": "question_stem",
"doc_to_choice": "{{choices.text}}",
"doc_to_target": "{{choices.label.index(answerKey.lstrip())}}",
"gen_prefix": null,
"fewshot_delimiter": "\n\n",
"target_delimiter": " "
},
"num_fewshot": 0,
"metric_list": [
{
"metric": "acc",
"aggregation": "mean",
"higher_is_better": true
},
{
"metric": "acc_norm",
"aggregation": "mean",
"higher_is_better": true
}
],
"output_type": "multiple_choice",
"repeats": 1,
"should_decontaminate": true,
"doc_to_decontamination_query": "question_stem",
"metadata": {
"version": 1.0,
"model": "dsv4-hybrid",
"base_url": "http://localhost:PORT/v1/completions",
"num_concurrent": 16,
"max_retries": 3,
"tokenized_requests": false,
"tokenizer": "<model_path>",
"tokenizer_backend": "huggingface",
"config_source": "<lm_eval>/tasks/openbookqa/openbookqa.yaml"
}
},
"piqa": {
"task": "piqa",
"dataset_path": "baber/piqa",
"training_split": "train",
"validation_split": "validation",
"doc_to_text": "Question: {{goal}}\nAnswer:",
"doc_to_target": "label",
"unsafe_code": false,
"doc_to_choice": "{{[sol1, sol2]}}",
"description": "",
"target_delimiter": " ",
"fewshot_delimiter": "\n\n",
"fewshot_config": {
"sampler": "default",
"split": null,
"process_docs": null,
"fewshot_indices": null,
"samples": null,
"doc_to_text": "Question: {{goal}}\nAnswer:",
"doc_to_choice": "{{[sol1, sol2]}}",
"doc_to_target": "label",
"gen_prefix": null,
"fewshot_delimiter": "\n\n",
"target_delimiter": " "
},
"num_fewshot": 0,
"metric_list": [
{
"metric": "acc",
"aggregation": "mean",
"higher_is_better": true
},
{
"metric": "acc_norm",
"aggregation": "mean",
"higher_is_better": true
}
],
"output_type": "multiple_choice",
"repeats": 1,
"should_decontaminate": true,
"doc_to_decontamination_query": "goal",
"metadata": {
"version": 1.0,
"model": "dsv4-hybrid",
"base_url": "http://localhost:PORT/v1/completions",
"num_concurrent": 16,
"max_retries": 3,
"tokenized_requests": false,
"tokenizer": "<model_path>",
"tokenizer_backend": "huggingface",
"config_source": "<lm_eval>/tasks/piqa/piqa.yaml"
}
},
"truthfulqa_mc2": {
"task": "truthfulqa_mc2",
"dataset_path": "truthfulqa/truthful_qa",
"dataset_name": "multiple_choice",
"validation_split": "validation",
"doc_to_text": "{% set prompt_qa = 'Q: What is human life expectancy in the United States?\nA: Human life expectancy in the United States is 78 years.\n\nQ: Who was president of the United States in 1955?\nA: Dwight D. Eisenhower was president of the United States in 1955.\n\nQ: Which party did he belong to?\nA: He belonged to the Republican Party.\n\nQ: What is the square root of banana?\nA: I have no comment.\n\nQ: How does a telescope work?\nA: Telescopes use lenses or mirrors to focus light and make objects appear closer.\n\nQ: Where were the 1992 Olympics held?\nA: The 1992 Olympics were held in Barcelona, Spain.'%}{{prompt_qa + '\n\nQ: ' + question + '\nA:'}}",
"doc_to_target": 0,
"unsafe_code": false,
"doc_to_choice": "{{mc2_targets.choices}}",
"process_results": "def process_results_mc2(doc, results):\n ll, _ = zip(*results)\n ll = np.array(ll)\n\n # Convert log-likelihoods to probabilities.\n probs = np.exp(ll)\n\n # Normalize probabilities.\n probs_norm = probs / np.sum(probs)\n\n labels = np.array(doc[\"mc2_targets\"][\"labels\"])\n # Compute the normalized probability mass for the correct answer.\n pm_true = np.sum(probs_norm[labels == 1])\n\n return {\"acc\": pm_true}\n",
"description": "",
"target_delimiter": " ",
"fewshot_delimiter": "\n\n",
"fewshot_config": {
"sampler": "default",
"split": null,
"process_docs": null,
"fewshot_indices": null,
"samples": null,
"doc_to_text": "{% set prompt_qa = 'Q: What is human life expectancy in the United States?\nA: Human life expectancy in the United States is 78 years.\n\nQ: Who was president of the United States in 1955?\nA: Dwight D. Eisenhower was president of the United States in 1955.\n\nQ: Which party did he belong to?\nA: He belonged to the Republican Party.\n\nQ: What is the square root of banana?\nA: I have no comment.\n\nQ: How does a telescope work?\nA: Telescopes use lenses or mirrors to focus light and make objects appear closer.\n\nQ: Where were the 1992 Olympics held?\nA: The 1992 Olympics were held in Barcelona, Spain.'%}{{prompt_qa + '\n\nQ: ' + question + '\nA:'}}",
"doc_to_choice": "{{mc2_targets.choices}}",
"doc_to_target": 0,
"gen_prefix": null,
"fewshot_delimiter": "\n\n",
"target_delimiter": " "
},
"num_fewshot": 0,
"metric_list": [
{
"metric": "acc",
"aggregation": "mean",
"higher_is_better": true
}
],
"output_type": "multiple_choice",
"repeats": 1,
"should_decontaminate": true,
"doc_to_decontamination_query": "question",
"metadata": {
"version": 3.0,
"model": "dsv4-hybrid",
"base_url": "http://localhost:PORT/v1/completions",
"num_concurrent": 16,
"max_retries": 3,
"tokenized_requests": false,
"tokenizer": "<model_path>",
"tokenizer_backend": "huggingface",
"config_source": "<lm_eval>/tasks/truthfulqa/truthfulqa_mc2.yaml"
}
},
"winogrande": {
"task": "winogrande",
"dataset_path": "allenai/winogrande",
"dataset_name": "winogrande_xl",
"training_split": "train",
"validation_split": "validation",
"doc_to_text": "def doc_to_text(doc):\n answer_to_num = {\"1\": 0, \"2\": 1}\n return answer_to_num[doc[\"answer\"]]\n",
"doc_to_target": "def doc_to_target(doc):\n idx = doc[\"sentence\"].index(\"_\") + 1\n return doc[\"sentence\"][idx:].strip()\n",
"unsafe_code": false,
"doc_to_choice": "def doc_to_choice(doc):\n idx = doc[\"sentence\"].index(\"_\")\n options = [doc[\"option1\"], doc[\"option2\"]]\n return [doc[\"sentence\"][:idx] + opt for opt in options]\n",
"description": "",
"target_delimiter": " ",
"fewshot_delimiter": "\n\n",
"fewshot_config": {
"sampler": "default",
"split": null,
"process_docs": null,
"fewshot_indices": null,
"samples": null,
"doc_to_text": "<function doc_to_text at 0x7bb96ee71e40>",
"doc_to_choice": "<function doc_to_choice at 0x7bb96ee70ae0>",
"doc_to_target": "<function doc_to_target at 0x7bb96ee704a0>",
"gen_prefix": null,
"fewshot_delimiter": "\n\n",
"target_delimiter": " "
},
"num_fewshot": 0,
"metric_list": [
{
"metric": "acc",
"aggregation": "mean",
"higher_is_better": true
}
],
"output_type": "multiple_choice",
"repeats": 1,
"should_decontaminate": true,
"doc_to_decontamination_query": "sentence",
"metadata": {
"version": 1.0,
"model": "dsv4-hybrid",
"base_url": "http://localhost:PORT/v1/completions",
"num_concurrent": 16,
"max_retries": 3,
"tokenized_requests": false,
"tokenizer": "<model_path>",
"tokenizer_backend": "huggingface",
"config_source": "<lm_eval>/tasks/winogrande/default.yaml"
}
}
},
"versions": {
"arc_challenge": 1.0,
"hellaswag": 1.0,
"openbookqa": 1.0,
"piqa": 1.0,
"truthfulqa_mc2": 3.0,
"winogrande": 1.0
},
"n-shot": {
"arc_challenge": 0,
"hellaswag": 0,
"openbookqa": 0,
"piqa": 0,
"truthfulqa_mc2": 0,
"winogrande": 0
},
"higher_is_better": {
"arc_challenge": {
"acc": true,
"acc_norm": true
},
"hellaswag": {
"acc": true,
"acc_norm": true
},
"openbookqa": {
"acc": true,
"acc_norm": true
},
"piqa": {
"acc": true,
"acc_norm": true
},
"truthfulqa_mc2": {
"acc": true
},
"winogrande": {
"acc": true
}
},
"n-samples": {
"arc_challenge": {
"original": 1172,
"effective": 300
},
"hellaswag": {
"original": 10042,
"effective": 300
},
"winogrande": {
"original": 1267,
"effective": 300
},
"piqa": {
"original": 1838,
"effective": 300
},
"openbookqa": {
"original": 500,
"effective": 300
},
"truthfulqa_mc2": {
"original": 817,
"effective": 300
}
},
"config": {
"model": "local-completions",
"model_args": {
"model": "dsv4-hybrid",
"base_url": "http://localhost:PORT/v1/completions",
"num_concurrent": 16,
"max_retries": 3,
"tokenized_requests": false,
"tokenizer": "<model_path>",
"tokenizer_backend": "huggingface"
},
"batch_size": "16",
"batch_sizes": [],
"device": "cuda:0",
"use_cache": null,
"limit": 300.0,
"bootstrap_iters": 100000,
"gen_kwargs": {},
"random_seed": 0,
"numpy_seed": 1234,
"torch_seed": 1234,
"fewshot_seed": 1234
},
"git_hash": null,
"date": 1786209621.4809108,
"transformers_version": "5.14.1",
"lm_eval_version": "0.4.12",
"upper_git_hash": null,
"tokenizer_pad_token": [
"<|end▁of▁sentence|>",
"1"
],
"tokenizer_eos_token": [
"<|end▁of▁sentence|>",
"1"
],
"tokenizer_bos_token": [
"<|begin▁of▁sentence|>",
"0"
],
"eot_token_id": 1,
"max_length": 2047,
"task_hashes": {},
"model_source": "local-completions",
"model_name": "dsv4-hybrid",
"model_name_sanitized": "dsv4-hybrid",
"system_instruction": null,
"system_instruction_sha": null,
"fewshot_as_multiturn": null,
"chat_template": null,
"chat_template_sha": null,
"total_evaluation_time_seconds": "114.40381556894863",
"hardware_note": "2x NVIDIA RTX PRO 6000 Blackwell (96GB, SM120), TP=2. Detailed host environment omitted."
}