Robert Shaw 6ac7b874b1 updated
Signed-off-by: Robert Shaw <robshaw@redhat.com>
2025-07-13 15:46:20 +00:00

21 lines
747 B
Makefile

# set this on your machine
vllm-directory := "/home/rshaw/vllm/"
launch_dp_ep MODEL SIZE:
vllm serve {{MODEL}} --data-parallel-size {{SIZE}} --enable-expert-parallel --disable-log-requests
launch_tp MODEL SIZE:
vllm serve {{MODEL}} --tensor-parallel-size {{SIZE}} --disable-log-requests
eval MODEL:
lm_eval --model local-completions --tasks gsm8k \
--model_args model={{MODEL}},base_url=http://127.0.0.1:8000/v1/completions,num_concurrent=100,tokenized_requests=False
benchmark MODEL NUM_PROMPTS:
python {{vllm-directory}}/benchmarks/benchmark_serving.py \
--model {{MODEL}} \
--dataset-name random \
--random-input-len 1000 \
--random-output-len 100 \
--num-prompts {{NUM_PROMPTS}} \
--seed $(date +%s)