Use Flashinfer TRT-LLM as Llama 4 compatible MoE backend (#11928)
This commit is contained in:
@@ -971,6 +971,11 @@ class ServerArgs:
|
||||
logger.warning(
|
||||
"Use trtllm_mha as attention backend on sm100 for Llama4 model"
|
||||
)
|
||||
if is_sm100_supported() and self.moe_runner_backend == "auto":
|
||||
self.moe_runner_backend = "flashinfer_trtllm"
|
||||
logger.info(
|
||||
"Use flashinfer_trtllm as MoE runner backend on SM100 for Llama4"
|
||||
)
|
||||
elif model_arch in [
|
||||
"Gemma2ForCausalLM",
|
||||
"Gemma3ForCausalLM",
|
||||
@@ -1336,8 +1341,10 @@ class ServerArgs:
|
||||
|
||||
if self.moe_runner_backend == "flashinfer_trtllm":
|
||||
assert (
|
||||
self.quantization == "modelopt_fp4" or self.quantization == "fp8"
|
||||
), "modelopt_fp4 or fp8 quantization is required for Flashinfer TRTLLM MoE"
|
||||
self.quantization == "modelopt_fp4"
|
||||
or self.quantization == "modelopt_fp8"
|
||||
or self.quantization == "fp8"
|
||||
), "modelopt_fp4, modelopt_fp8 or fp8 quantization is required for Flashinfer TRTLLM MoE"
|
||||
self.disable_shared_experts_fusion = True
|
||||
logger.warning(
|
||||
"FlashInfer TRTLLM MoE is enabled. --disable-shared-experts-fusion is automatically set."
|
||||
|
||||
Reference in New Issue
Block a user