Update README.md

Browse files

Files changed (1) hide show

README.md +17 -17

README.md CHANGED Viewed

@@ -84,7 +84,7 @@ This model can be deployed efficiently using the [vLLM](https://docs.vllm.ai/en/
 from vllm import LLM, SamplingParams
 from transformers import AutoTokenizer
-model_id = "neuralmagic/Meta-Llama-3.1-8B-Instruct-quantized.w4a16"
 number_gpus = 1
 max_model_len = 8192
@@ -614,7 +614,7 @@ The results were obtained using the following commands:
 ```
 lm_eval \
   --model vllm \
-  --model_args pretrained="neuralmagic/Meta-Llama-3.1-8B-Instruct-quantized.w4a16",dtype=auto,max_model_len=3850,max_gen_toks=10,tensor_parallel_size=1 \
   --tasks mmlu_llama_3.1_instruct \
   --fewshot_as_multiturn \
   --apply_chat_template \
@@ -626,7 +626,7 @@ lm_eval \
 ```
 lm_eval \
   --model vllm \
-  --model_args pretrained="neuralmagic/Meta-Llama-3.1-8B-Instruct-quantized.w4a16",dtype=auto,max_model_len=4064,max_gen_toks=1024,tensor_parallel_size=1 \
   --tasks mmlu_cot_0shot_llama_3.1_instruct \
   --apply_chat_template \
   --num_fewshot 0 \
@@ -637,7 +637,7 @@ lm_eval \
 ```
 lm_eval \
   --model vllm \
-  --model_args pretrained="neuralmagic/Meta-Llama-3.1-8B-Instruct-quantized.w4a16",dtype=auto,max_model_len=3940,max_gen_toks=100,tensor_parallel_size=1 \
   --tasks arc_challenge_llama_3.1_instruct \
   --apply_chat_template \
   --num_fewshot 0 \
@@ -648,7 +648,7 @@ lm_eval \
 ```
 lm_eval \
   --model vllm \
-  --model_args pretrained="neuralmagic/Meta-Llama-3.1-8B-Instruct-quantized.w4a16",dtype=auto,max_model_len=4096,max_gen_toks=1024,tensor_parallel_size=1 \
   --tasks gsm8k_cot_llama_3.1_instruct \
   --fewshot_as_multiturn \
   --apply_chat_template \
@@ -660,7 +660,7 @@ lm_eval \
 ```
 lm_eval \
   --model vllm \
-  --model_args pretrained="neuralmagic/Meta-Llama-3.1-8B-Instruct-quantized.w4a16",dtype=auto,add_bos_token=True,max_model_len=4096,tensor_parallel_size=1 \
   --tasks hellaswag \
   --num_fewshot 10 \
   --batch_size auto
@@ -670,7 +670,7 @@ lm_eval \
 ```
 lm_eval \
   --model vllm \
-  --model_args pretrained="neuralmagic/Meta-Llama-3.1-8B-Instruct-quantized.w4a16",dtype=auto,add_bos_token=True,max_model_len=4096,tensor_parallel_size=1 \
   --tasks winogrande \
   --num_fewshot 5 \
   --batch_size auto
@@ -680,7 +680,7 @@ lm_eval \
 ```
 lm_eval \
   --model vllm \
-  --model_args pretrained="neuralmagic/Meta-Llama-3.1-8B-Instruct-quantized.w4a16",dtype=auto,add_bos_token=True,max_model_len=4096,tensor_parallel_size=1 \
   --tasks truthfulqa \
   --num_fewshot 0 \
   --batch_size auto
@@ -690,7 +690,7 @@ lm_eval \
 ```
 lm_eval \
   --model vllm \
-  --model_args pretrained="neuralmagic/Meta-Llama-3.1-8B-Instruct-quantized.w4a16",dtype=auto,max_model_len=4096,tensor_parallel_size=1,enable_chunked_prefill=True \
   --apply_chat_template \
   --fewshot_as_multiturn \
   --tasks leaderboard \
@@ -701,7 +701,7 @@ lm_eval \
 ```
 lm_eval \
   --model vllm \
-  --model_args pretrained="neuralmagic/Meta-Llama-3.1-8B-Instruct-quantized.w4a16",dtype=auto,max_model_len=3850,max_gen_toks=10,tensor_parallel_size=1 \
   --tasks mmlu_pt_llama_3.1_instruct \
   --fewshot_as_multiturn \
   --apply_chat_template \
@@ -713,7 +713,7 @@ lm_eval \
 ```
 lm_eval \
   --model vllm \
-  --model_args pretrained="neuralmagic/Meta-Llama-3.1-8B-Instruct-quantized.w4a16",dtype=auto,max_model_len=3850,max_gen_toks=10,tensor_parallel_size=1 \
   --tasks mmlu_es_llama_3.1_instruct \
   --fewshot_as_multiturn \
   --apply_chat_template \
@@ -725,7 +725,7 @@ lm_eval \
 ```
 lm_eval \
   --model vllm \
-  --model_args pretrained="neuralmagic/Meta-Llama-3.1-8B-Instruct-quantized.w4a16",dtype=auto,max_model_len=3850,max_gen_toks=10,tensor_parallel_size=1 \
   --tasks mmlu_it_llama_3.1_instruct \
   --fewshot_as_multiturn \
   --apply_chat_template \
@@ -737,7 +737,7 @@ lm_eval \
 ```
 lm_eval \
   --model vllm \
-  --model_args pretrained="neuralmagic/Meta-Llama-3.1-8B-Instruct-quantized.w4a16",dtype=auto,max_model_len=3850,max_gen_toks=10,tensor_parallel_size=1 \
   --tasks mmlu_de_llama_3.1_instruct \
   --fewshot_as_multiturn \
   --apply_chat_template \
@@ -749,7 +749,7 @@ lm_eval \
 ```
 lm_eval \
   --model vllm \
-  --model_args pretrained="neuralmagic/Meta-Llama-3.1-8B-Instruct-quantized.w4a16",dtype=auto,max_model_len=3850,max_gen_toks=10,tensor_parallel_size=1 \
   --tasks mmlu_fr_llama_3.1_instruct \
   --fewshot_as_multiturn \
   --apply_chat_template \
@@ -761,7 +761,7 @@ lm_eval \
 ```
 lm_eval \
   --model vllm \
-  --model_args pretrained="neuralmagic/Meta-Llama-3.1-8B-Instruct-quantized.w4a16",dtype=auto,max_model_len=3850,max_gen_toks=10,tensor_parallel_size=1 \
   --tasks mmlu_hi_llama_3.1_instruct \
   --fewshot_as_multiturn \
   --apply_chat_template \
@@ -773,7 +773,7 @@ lm_eval \
 ```
 lm_eval \
   --model vllm \
-  --model_args pretrained="neuralmagic/Meta-Llama-3.1-8B-Instruct-quantized.w4a16",dtype=auto,max_model_len=3850,max_gen_toks=10,tensor_parallel_size=1 \
   --tasks mmlu_th_llama_3.1_instruct \
   --fewshot_as_multiturn \
   --apply_chat_template \
@@ -785,7 +785,7 @@ lm_eval \
 ##### Generation
 ```
 python3 codegen/generate.py \
-  --model neuralmagic/Meta-Llama-3.1-8B-Instruct-quantized.w4a16 \
   --bs 16 \
   --temperature 0.2 \
   --n_samples 50 \

 from vllm import LLM, SamplingParams
 from transformers import AutoTokenizer
+model_id = "RedHatAI/Meta-Llama-3.1-8B-Instruct-quantized.w4a16"
 number_gpus = 1
 max_model_len = 8192
 ```
 lm_eval \
   --model vllm \
+  --model_args pretrained="RedHatAI/Meta-Llama-3.1-8B-Instruct-quantized.w4a16",dtype=auto,max_model_len=3850,max_gen_toks=10,tensor_parallel_size=1 \
   --tasks mmlu_llama_3.1_instruct \
   --fewshot_as_multiturn \
   --apply_chat_template \
 ```
 lm_eval \
   --model vllm \
+  --model_args pretrained="RedHatAI/Meta-Llama-3.1-8B-Instruct-quantized.w4a16",dtype=auto,max_model_len=4064,max_gen_toks=1024,tensor_parallel_size=1 \
   --tasks mmlu_cot_0shot_llama_3.1_instruct \
   --apply_chat_template \
   --num_fewshot 0 \
 ```
 lm_eval \
   --model vllm \
+  --model_args pretrained="RedHatAI/Meta-Llama-3.1-8B-Instruct-quantized.w4a16",dtype=auto,max_model_len=3940,max_gen_toks=100,tensor_parallel_size=1 \
   --tasks arc_challenge_llama_3.1_instruct \
   --apply_chat_template \
   --num_fewshot 0 \
 ```
 lm_eval \
   --model vllm \
+  --model_args pretrained="RedHatAI/Meta-Llama-3.1-8B-Instruct-quantized.w4a16",dtype=auto,max_model_len=4096,max_gen_toks=1024,tensor_parallel_size=1 \
   --tasks gsm8k_cot_llama_3.1_instruct \
   --fewshot_as_multiturn \
   --apply_chat_template \
 ```
 lm_eval \
   --model vllm \
+  --model_args pretrained="RedHatAI/Meta-Llama-3.1-8B-Instruct-quantized.w4a16",dtype=auto,add_bos_token=True,max_model_len=4096,tensor_parallel_size=1 \
   --tasks hellaswag \
   --num_fewshot 10 \
   --batch_size auto
 ```
 lm_eval \
   --model vllm \
+  --model_args pretrained="RedHatAI/Meta-Llama-3.1-8B-Instruct-quantized.w4a16",dtype=auto,add_bos_token=True,max_model_len=4096,tensor_parallel_size=1 \
   --tasks winogrande \
   --num_fewshot 5 \
   --batch_size auto
 ```
 lm_eval \
   --model vllm \
+  --model_args pretrained="RedHatAI/Meta-Llama-3.1-8B-Instruct-quantized.w4a16",dtype=auto,add_bos_token=True,max_model_len=4096,tensor_parallel_size=1 \
   --tasks truthfulqa \
   --num_fewshot 0 \
   --batch_size auto
 ```
 lm_eval \
   --model vllm \
+  --model_args pretrained="RedHatAI/Meta-Llama-3.1-8B-Instruct-quantized.w4a16",dtype=auto,max_model_len=4096,tensor_parallel_size=1,enable_chunked_prefill=True \
   --apply_chat_template \
   --fewshot_as_multiturn \
   --tasks leaderboard \
 ```
 lm_eval \
   --model vllm \
+  --model_args pretrained="RedHatAI/Meta-Llama-3.1-8B-Instruct-quantized.w4a16",dtype=auto,max_model_len=3850,max_gen_toks=10,tensor_parallel_size=1 \
   --tasks mmlu_pt_llama_3.1_instruct \
   --fewshot_as_multiturn \
   --apply_chat_template \
 ```
 lm_eval \
   --model vllm \
+  --model_args pretrained="RedHatAI/Meta-Llama-3.1-8B-Instruct-quantized.w4a16",dtype=auto,max_model_len=3850,max_gen_toks=10,tensor_parallel_size=1 \
   --tasks mmlu_es_llama_3.1_instruct \
   --fewshot_as_multiturn \
   --apply_chat_template \
 ```
 lm_eval \
   --model vllm \
+  --model_args pretrained="RedHatAI/Meta-Llama-3.1-8B-Instruct-quantized.w4a16",dtype=auto,max_model_len=3850,max_gen_toks=10,tensor_parallel_size=1 \
   --tasks mmlu_it_llama_3.1_instruct \
   --fewshot_as_multiturn \
   --apply_chat_template \
 ```
 lm_eval \
   --model vllm \
+  --model_args pretrained="RedHatAI/Meta-Llama-3.1-8B-Instruct-quantized.w4a16",dtype=auto,max_model_len=3850,max_gen_toks=10,tensor_parallel_size=1 \
   --tasks mmlu_de_llama_3.1_instruct \
   --fewshot_as_multiturn \
   --apply_chat_template \
 ```
 lm_eval \
   --model vllm \
+  --model_args pretrained="RedHatAI/Meta-Llama-3.1-8B-Instruct-quantized.w4a16",dtype=auto,max_model_len=3850,max_gen_toks=10,tensor_parallel_size=1 \
   --tasks mmlu_fr_llama_3.1_instruct \
   --fewshot_as_multiturn \
   --apply_chat_template \
 ```
 lm_eval \
   --model vllm \
+  --model_args pretrained="RedHatAI/Meta-Llama-3.1-8B-Instruct-quantized.w4a16",dtype=auto,max_model_len=3850,max_gen_toks=10,tensor_parallel_size=1 \
   --tasks mmlu_hi_llama_3.1_instruct \
   --fewshot_as_multiturn \
   --apply_chat_template \
 ```
 lm_eval \
   --model vllm \
+  --model_args pretrained="RedHatAI/Meta-Llama-3.1-8B-Instruct-quantized.w4a16",dtype=auto,max_model_len=3850,max_gen_toks=10,tensor_parallel_size=1 \
   --tasks mmlu_th_llama_3.1_instruct \
   --fewshot_as_multiturn \
   --apply_chat_template \
 ##### Generation
 ```
 python3 codegen/generate.py \
+  --model RedHatAI/Meta-Llama-3.1-8B-Instruct-quantized.w4a16 \
   --bs 16 \
   --temperature 0.2 \
   --n_samples 50 \