Update README.md

Files changed (1) hide show

README.md CHANGED Viewed

@@ -109,7 +109,7 @@ lm_eval --model hf --model_args pretrained=microsoft/Phi-4-mini-instruct --tasks
 ## int4wo-hqq
 ```
-lm_eval --model hf --model_args pretrained=jerryzh168/phi4-mini-int4wo-hqq --tasks hellaswag --device cuda:0 --batch_size 8
 ```
 `TODO: more complete eval results`
@@ -162,7 +162,7 @@ python benchmarks/benchmark_latency.py --input-len 256 --output-len 256 --model
 ### int4wo-hqq
 ```
-python benchmarks/benchmark_latency.py --input-len 256 --output-len 256 --model jerryzh168/phi4-mini-int4wo-hqq --batch-size 1
 ```
 ## benchmark_serving
@@ -186,16 +186,16 @@ python benchmarks/benchmark_serving.py --backend vllm --dataset-name sharegpt --
 ### int4wo-hqq
 Server:
 ```
-vllm serve jerryzh168/phi4-mini-int4wo-hqq --tokenizer microsoft/Phi-4-mini-instruct -O3
 ```
 Client:
 ```
-python benchmarks/benchmark_serving.py --backend vllm --dataset-name sharegpt --tokenizer microsoft/Phi-4-mini-instruct --dataset-path ./ShareGPT_V3_unfiltered_cleaned_split.json --model jerryzh168/phi4-mini-int4wo-hqq --num-prompts 1
 ```
 # Serving with vllm
 We can use the same command we used in serving benchmarks to serve the model with vllm
 ```
-vllm serve jerryzh168/phi4-mini-int4wo-hqq --tokenizer microsoft/Phi-4-mini-instruct -O3
 ```

 ## int4wo-hqq
 ```
+lm_eval --model hf --model_args pretrained=pytorch/Phi-4-mini-instruct-int4wo-hqq --tasks hellaswag --device cuda:0 --batch_size 8
 ```
 `TODO: more complete eval results`
 ### int4wo-hqq
 ```
+python benchmarks/benchmark_latency.py --input-len 256 --output-len 256 --model pytorch/Phi-4-mini-instruct-int4wo-hqq --batch-size 1
 ```
 ## benchmark_serving
 ### int4wo-hqq
 Server:
 ```
+vllm serve pytorch/Phi-4-mini-instruct-int4wo-hqq --tokenizer microsoft/Phi-4-mini-instruct -O3
 ```
 Client:
 ```
+python benchmarks/benchmark_serving.py --backend vllm --dataset-name sharegpt --tokenizer microsoft/Phi-4-mini-instruct --dataset-path ./ShareGPT_V3_unfiltered_cleaned_split.json --model pytorch/Phi-4-mini-instruct-int4wo-hqq --num-prompts 1
 ```
 # Serving with vllm
 We can use the same command we used in serving benchmarks to serve the model with vllm
 ```
+vllm serve pytorch/Phi-4-mini-instruct-int4wo-hqq --tokenizer microsoft/Phi-4-mini-instruct -O3
 ```