Unverified Commit de527e1c authored by Michael Goin's avatar Michael Goin Committed by GitHub
Browse files

[UX] Add `--moe-backend` arg for explicit kernel selection (#33807)


Signed-off-by: default avatarmgoin <mgoin64@gmail.com>
Co-authored-by: default avatarRobert Shaw <114415538+robertgshaw2-redhat@users.noreply.github.com>
parent 1976356e
...@@ -8,5 +8,4 @@ server_args: >- ...@@ -8,5 +8,4 @@ server_args: >-
--tensor-parallel-size 2 --tensor-parallel-size 2
--enable-expert-parallel --enable-expert-parallel
--speculative-config '{"method":"qwen3_next_mtp","num_speculative_tokens":1}' --speculative-config '{"method":"qwen3_next_mtp","num_speculative_tokens":1}'
env: --moe-backend=flashinfer_trtllm
VLLM_USE_FLASHINFER_MOE_FP4: "1"
...@@ -7,5 +7,4 @@ server_args: >- ...@@ -7,5 +7,4 @@ server_args: >-
--tensor-parallel-size 2 --tensor-parallel-size 2
--enable-expert-parallel --enable-expert-parallel
--async-scheduling --async-scheduling
env: --moe-backend=flashinfer_trtllm
VLLM_USE_FLASHINFER_MOE_FP8: "1"
...@@ -2,7 +2,6 @@ model_name: "nvidia/Llama-4-Scout-17B-16E-Instruct-FP8" ...@@ -2,7 +2,6 @@ model_name: "nvidia/Llama-4-Scout-17B-16E-Instruct-FP8"
accuracy_threshold: 0.92 accuracy_threshold: 0.92
num_questions: 1319 num_questions: 1319
num_fewshot: 5 num_fewshot: 5
server_args: "--enforce-eager --max-model-len 8192 --data-parallel-size 2 --enable-expert-parallel" server_args: "--enforce-eager --max-model-len 8192 --data-parallel-size 2 --enable-expert-parallel --moe-backend=triton"
env: env:
VLLM_USE_FLASHINFER_MOE_FP8: "0"
VLLM_USE_DEEP_GEMM: "0" VLLM_USE_DEEP_GEMM: "0"
...@@ -2,7 +2,4 @@ model_name: "RedHatAI/Qwen3-30B-A3B-NVFP4" ...@@ -2,7 +2,4 @@ model_name: "RedHatAI/Qwen3-30B-A3B-NVFP4"
accuracy_threshold: 0.88 accuracy_threshold: 0.88
num_questions: 1319 num_questions: 1319
num_fewshot: 5 num_fewshot: 5
server_args: "--enforce-eager --max-model-len 8192 --data-parallel-size 2 --enable-expert-parallel --all2all-backend deepep_low_latency" server_args: "--enforce-eager --max-model-len 8192 --data-parallel-size 2 --enable-expert-parallel --all2all-backend deepep_low_latency --moe-backend=flashinfer_cutedsl"
env:
VLLM_USE_FLASHINFER_MOE_FP4: "1"
VLLM_FLASHINFER_MOE_BACKEND: "masked_gemm"
...@@ -2,7 +2,4 @@ model_name: "RedHatAI/Qwen3-30B-A3B-NVFP4" ...@@ -2,7 +2,4 @@ model_name: "RedHatAI/Qwen3-30B-A3B-NVFP4"
accuracy_threshold: 0.88 accuracy_threshold: 0.88
num_questions: 1319 num_questions: 1319
num_fewshot: 5 num_fewshot: 5
server_args: "--enforce-eager --max-model-len 8192 --data-parallel-size 2 --enable-expert-parallel" server_args: "--enforce-eager --max-model-len 8192 --data-parallel-size 2 --enable-expert-parallel --moe-backend=flashinfer_cutlass"
env:
VLLM_USE_FLASHINFER_MOE_FP4: "1"
VLLM_FLASHINFER_MOE_BACKEND: "throughput"
...@@ -2,7 +2,4 @@ model_name: "nvidia/Qwen3-30B-A3B-NVFP4" ...@@ -2,7 +2,4 @@ model_name: "nvidia/Qwen3-30B-A3B-NVFP4"
accuracy_threshold: 0.88 accuracy_threshold: 0.88
num_questions: 1319 num_questions: 1319
num_fewshot: 5 num_fewshot: 5
server_args: "--enforce-eager --max-model-len 8192 --data-parallel-size 2 --enable-expert-parallel --all2all-backend deepep_low_latency" server_args: "--enforce-eager --max-model-len 8192 --data-parallel-size 2 --enable-expert-parallel --all2all-backend deepep_low_latency --moe-backend=flashinfer_cutedsl"
env:
VLLM_USE_FLASHINFER_MOE_FP4: "1"
VLLM_FLASHINFER_MOE_BACKEND: "masked_gemm"
...@@ -2,7 +2,4 @@ model_name: "nvidia/Qwen3-30B-A3B-NVFP4" ...@@ -2,7 +2,4 @@ model_name: "nvidia/Qwen3-30B-A3B-NVFP4"
accuracy_threshold: 0.88 accuracy_threshold: 0.88
num_questions: 1319 num_questions: 1319
num_fewshot: 5 num_fewshot: 5
server_args: "--enforce-eager --max-model-len 8192 --data-parallel-size 2 --enable-expert-parallel" server_args: "--enforce-eager --max-model-len 8192 --data-parallel-size 2 --enable-expert-parallel --moe-backend=flashinfer_cutlass"
env:
VLLM_USE_FLASHINFER_MOE_FP4: "1"
VLLM_FLASHINFER_MOE_BACKEND: "throughput"
...@@ -2,7 +2,4 @@ model_name: "nvidia/Qwen3-30B-A3B-NVFP4" ...@@ -2,7 +2,4 @@ model_name: "nvidia/Qwen3-30B-A3B-NVFP4"
accuracy_threshold: 0.88 accuracy_threshold: 0.88
num_questions: 1319 num_questions: 1319
num_fewshot: 5 num_fewshot: 5
server_args: "--enforce-eager --max-model-len 8192 --data-parallel-size 2 --enable-expert-parallel" server_args: "--enforce-eager --max-model-len 8192 --data-parallel-size 2 --enable-expert-parallel --moe-backend=flashinfer_trtllm"
env:
VLLM_USE_FLASHINFER_MOE_FP4: "1"
VLLM_FLASHINFER_MOE_BACKEND: "latency"
...@@ -2,8 +2,4 @@ model_name: "meta-llama/Llama-4-Scout-17B-16E-Instruct" ...@@ -2,8 +2,4 @@ model_name: "meta-llama/Llama-4-Scout-17B-16E-Instruct"
accuracy_threshold: 0.92 accuracy_threshold: 0.92
num_questions: 1319 num_questions: 1319
num_fewshot: 5 num_fewshot: 5
server_args: "--enforce-eager --max-model-len 8192 --tensor-parallel-size 2 --enable-expert-parallel" server_args: "--enforce-eager --max-model-len 8192 --tensor-parallel-size 2 --enable-expert-parallel --moe-backend=flashinfer_cutlass"
env:
VLLM_USE_FLASHINFER_MOE_FP16: "1"
VLLM_FLASHINFER_MOE_BACKEND: "throughput"
...@@ -2,7 +2,4 @@ model_name: "nvidia/Llama-4-Scout-17B-16E-Instruct-FP8" ...@@ -2,7 +2,4 @@ model_name: "nvidia/Llama-4-Scout-17B-16E-Instruct-FP8"
accuracy_threshold: 0.92 accuracy_threshold: 0.92
num_questions: 1319 num_questions: 1319
num_fewshot: 5 num_fewshot: 5
server_args: "--enforce-eager --max-model-len 8192 --tensor-parallel-size 2" server_args: "--enforce-eager --max-model-len 8192 --tensor-parallel-size 2 --moe-backend=flashinfer_cutlass"
env:
VLLM_USE_FLASHINFER_MOE_FP8: "1"
VLLM_FLASHINFER_MOE_BACKEND: "throughput"
...@@ -2,7 +2,4 @@ model_name: "nvidia/Llama-4-Scout-17B-16E-Instruct-FP8" ...@@ -2,7 +2,4 @@ model_name: "nvidia/Llama-4-Scout-17B-16E-Instruct-FP8"
accuracy_threshold: 0.92 accuracy_threshold: 0.92
num_questions: 1319 num_questions: 1319
num_fewshot: 5 num_fewshot: 5
server_args: "--enforce-eager --max-model-len 8192 --tensor-parallel-size 2" server_args: "--enforce-eager --max-model-len 8192 --tensor-parallel-size 2 --moe-backend=flashinfer_trtllm"
env:
VLLM_USE_FLASHINFER_MOE_FP8: "1"
VLLM_FLASHINFER_MOE_BACKEND: "latency"
...@@ -2,6 +2,4 @@ model_name: "nvidia/Llama-4-Scout-17B-16E-Instruct-FP8" ...@@ -2,6 +2,4 @@ model_name: "nvidia/Llama-4-Scout-17B-16E-Instruct-FP8"
accuracy_threshold: 0.92 accuracy_threshold: 0.92
num_questions: 1319 num_questions: 1319
num_fewshot: 5 num_fewshot: 5
server_args: "--enforce-eager --max-model-len 8192 --tensor-parallel-size 2" server_args: "--enforce-eager --max-model-len 8192 --tensor-parallel-size 2 --moe-backend=triton"
env:
VLLM_USE_FLASHINFER_MOE_FP8: "0"
...@@ -2,7 +2,4 @@ model_name: "mistralai/Mixtral-8x7B-v0.1" ...@@ -2,7 +2,4 @@ model_name: "mistralai/Mixtral-8x7B-v0.1"
accuracy_threshold: 0.58 accuracy_threshold: 0.58
num_questions: 1319 num_questions: 1319
num_fewshot: 5 num_fewshot: 5
server_args: "--enforce-eager --max-model-len 8192 --tensor-parallel-size 2 --enable-expert-parallel" server_args: "--enforce-eager --max-model-len 8192 --tensor-parallel-size 2 --enable-expert-parallel --moe-backend=flashinfer_cutlass"
env:
VLLM_USE_FLASHINFER_MOE_FP16: "1"
VLLM_FLASHINFER_MOE_BACKEND: "throughput"
...@@ -3,7 +3,4 @@ ...@@ -3,7 +3,4 @@
# accuracy_threshold: 0.62 # accuracy_threshold: 0.62
# num_questions: 1319 # num_questions: 1319
# num_fewshot: 5 # num_fewshot: 5
# server_args: "--enforce-eager --max-model-len 8192 --tensor-parallel-size 2" # server_args: "--enforce-eager --max-model-len 8192 --tensor-parallel-size 2 --moe-backend=flashinfer_cutlass"
# env:
# VLLM_USE_FLASHINFER_MOE_FP8: "1"
# VLLM_FLASHINFER_MOE_BACKEND: "throughput"
...@@ -2,7 +2,4 @@ model_name: "nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-FP8" ...@@ -2,7 +2,4 @@ model_name: "nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-FP8"
accuracy_threshold: 0.29 accuracy_threshold: 0.29
num_questions: 1319 num_questions: 1319
num_fewshot: 5 num_fewshot: 5
server_args: "--enforce-eager --max-model-len 8192 --tensor-parallel-size 2" server_args: "--enforce-eager --max-model-len 8192 --tensor-parallel-size 2 --moe-backend=flashinfer_trtllm"
env:
VLLM_USE_FLASHINFER_MOE_FP8: "1"
VLLM_FLASHINFER_MOE_BACKEND: "latency"
...@@ -2,7 +2,4 @@ model_name: "nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-NVFP4" ...@@ -2,7 +2,4 @@ model_name: "nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-NVFP4"
accuracy_threshold: 0.29 accuracy_threshold: 0.29
num_questions: 1319 num_questions: 1319
num_fewshot: 5 num_fewshot: 5
server_args: "--enforce-eager --max-model-len 8192 --tensor-parallel-size 2" server_args: "--enforce-eager --max-model-len 8192 --tensor-parallel-size 2 --moe-backend=flashinfer_cutlass"
env:
VLLM_USE_FLASHINFER_MOE_FP4: "1"
VLLM_FLASHINFER_MOE_BACKEND: "throughput"
...@@ -2,6 +2,4 @@ model_name: "Qwen/Qwen3-30B-A3B" ...@@ -2,6 +2,4 @@ model_name: "Qwen/Qwen3-30B-A3B"
accuracy_threshold: 0.88 accuracy_threshold: 0.88
num_questions: 1319 num_questions: 1319
num_fewshot: 5 num_fewshot: 5
server_args: "--enforce-eager --max-model-len 8192 --tensor-parallel-size 2 --enable-expert-parallel" server_args: "--enforce-eager --max-model-len 8192 --tensor-parallel-size 2 --enable-expert-parallel --moe-backend=flashinfer_cutlass"
env:
VLLM_USE_FLASHINFER_MOE_FP16: "1"
...@@ -2,7 +2,4 @@ model_name: "Qwen/Qwen3-Coder-30B-A3B-Instruct-FP8" ...@@ -2,7 +2,4 @@ model_name: "Qwen/Qwen3-Coder-30B-A3B-Instruct-FP8"
accuracy_threshold: 0.88 accuracy_threshold: 0.88
num_questions: 1319 num_questions: 1319
num_fewshot: 5 num_fewshot: 5
server_args: "--enforce-eager --max-model-len 8192 --tensor-parallel-size 2" server_args: "--enforce-eager --max-model-len 8192 --tensor-parallel-size 2 --moe-backend=flashinfer_cutlass"
env:
VLLM_USE_FLASHINFER_MOE_FP8: "1"
VLLM_FLASHINFER_MOE_BACKEND: "throughput"
...@@ -2,7 +2,4 @@ model_name: "Qwen/Qwen3-Coder-30B-A3B-Instruct-FP8" ...@@ -2,7 +2,4 @@ model_name: "Qwen/Qwen3-Coder-30B-A3B-Instruct-FP8"
accuracy_threshold: 0.88 accuracy_threshold: 0.88
num_questions: 1319 num_questions: 1319
num_fewshot: 5 num_fewshot: 5
server_args: "--enforce-eager --max-model-len 8192 --tensor-parallel-size 2" server_args: "--enforce-eager --max-model-len 8192 --tensor-parallel-size 2 --moe-backend=flashinfer_trtllm"
env:
VLLM_USE_FLASHINFER_MOE_FP8: "1"
VLLM_FLASHINFER_MOE_BACKEND: "latency"
...@@ -2,7 +2,6 @@ model_name: "Qwen/Qwen3-Coder-30B-A3B-Instruct-FP8" ...@@ -2,7 +2,6 @@ model_name: "Qwen/Qwen3-Coder-30B-A3B-Instruct-FP8"
accuracy_threshold: 0.88 accuracy_threshold: 0.88
num_questions: 1319 num_questions: 1319
num_fewshot: 5 num_fewshot: 5
server_args: "--enforce-eager --max-model-len 8192 --tensor-parallel-size 2" server_args: "--enforce-eager --max-model-len 8192 --tensor-parallel-size 2 --moe-backend=triton"
env: env:
VLLM_USE_FLASHINFER_MOE_FP8: "0"
VLLM_USE_DEEP_GEMM: "0" VLLM_USE_DEEP_GEMM: "0"
Markdown is supported
0% or .
You are about to add 0 people to the discussion. Proceed with caution.
Finish editing this message first!
Please register or to comment