[Doc] Fix outdated --fp4-gemm-backend documentation (#18350)
This commit is contained in:
committed by
GitHub
parent
d792aa7618
commit
fddef76619
@@ -267,7 +267,7 @@ Please consult the documentation below and [server_args.py](https://github.com/s
|
||||
| `--nsa-prefill-backend` | Choose the NSA backend for the prefill stage (overrides `--attention-backend` when running DeepSeek NSA-style attention). | `flashmla_sparse` | `flashmla_sparse`, `flashmla_kv`, `flashmla_auto`, `fa3`, `tilelang`, `aiter` |
|
||||
| `--nsa-decode-backend` | Choose the NSA backend for the decode stage when running DeepSeek NSA-style attention. Overrides `--attention-backend` for decoding. | `fa3` | `flashmla_sparse`, `flashmla_kv`, `fa3`, `tilelang`, `aiter` |
|
||||
| `--fp8-gemm-backend` | Choose the runner backend for Blockwise FP8 GEMM operations. Options: 'auto' (default, auto-selects based on hardware), 'deep_gemm' (JIT-compiled; enabled by default on NVIDIA Hopper (SM90) and Blackwell (SM100) when DeepGEMM is installed), 'flashinfer_trtllm' (optimal for Blackwell and low-latency), 'cutlass' (optimal for Hopper/Blackwell GPUs and high-throughput), 'triton' (fallback, widely compatible), 'aiter' (ROCm only). **NOTE**: This replaces the deprecated environment variables SGLANG_ENABLE_FLASHINFER_FP8_GEMM and SGLANG_SUPPORT_CUTLASS_BLOCK_FP8. | `auto` | `auto`, `deep_gemm`, `flashinfer_trtllm`, `cutlass`, `triton`, `aiter` |
|
||||
| `--fp4-gemm-backend` | Choose the runner backend for NVFP4 GEMM operations. Options: 'auto' (default, auto-selects between flashinfer_cudnn/flashinfer_cutlass based on CUDA/cuDNN version), 'flashinfer_cudnn' (FlashInfer cuDNN backend, optimal on CUDA 13+ with cuDNN 9.15+), 'flashinfer_cutlass' (FlashInfer CUTLASS backend, optimal on CUDA 12), 'flashinfer_trtllm' (FlashInfer TensorRT-LLM backend, requires different weight preparation with shuffling). All backends are from FlashInfer; when FlashInfer is unavailable, sgl-kernel CUTLASS is used as an automatic fallback. **NOTE**: This replaces the deprecated environment variable SGLANG_FLASHINFER_FP4_GEMM_BACKEND. | `auto` | `auto`, `flashinfer_cudnn`, `flashinfer_cutlass`, `flashinfer_trtllm` |
|
||||
| `--fp4-gemm-backend` | Choose the runner backend for NVFP4 GEMM operations. Options: 'flashinfer_cutlass' (default), 'auto' (auto-selects between flashinfer_cudnn/flashinfer_cutlass based on CUDA/cuDNN version), 'flashinfer_cudnn' (FlashInfer cuDNN backend, optimal on CUDA 13+ with cuDNN 9.15+), 'flashinfer_trtllm' (FlashInfer TensorRT-LLM backend, requires different weight preparation with shuffling). All backends are from FlashInfer; when FlashInfer is unavailable, sgl-kernel CUTLASS is used as an automatic fallback. **NOTE**: This replaces the deprecated environment variable SGLANG_FLASHINFER_FP4_GEMM_BACKEND. | `flashinfer_cutlass` | `auto`, `flashinfer_cudnn`, `flashinfer_cutlass`, `flashinfer_trtllm` |
|
||||
| `--disable-flashinfer-autotune` | Flashinfer autotune is enabled by default. Set this flag to disable the autotune. | `False` | bool flag (set to enable) |
|
||||
|
||||
## Speculative decoding
|
||||
|
||||
@@ -133,11 +133,7 @@ def fp4_gemm(
|
||||
fp4_backend = get_fp4_gemm_runner_backend()
|
||||
if enable_flashinfer_fp4_gemm:
|
||||
# Use the remapping logic to convert SGLang backend names to FlashInfer API names
|
||||
backend = (
|
||||
fp4_backend.get_flashinfer_backend()
|
||||
if not fp4_backend.is_auto()
|
||||
else "cutlass"
|
||||
)
|
||||
backend = fp4_backend.get_flashinfer_backend()
|
||||
return flashinfer_fp4_gemm(
|
||||
input, weight, input_sf, weight_sf, alpha, out_dtype, backend=backend
|
||||
)
|
||||
|
||||
@@ -448,7 +448,7 @@ class ServerArgs:
|
||||
grammar_backend: Optional[str] = None
|
||||
mm_attention_backend: Optional[str] = None
|
||||
fp8_gemm_runner_backend: str = "auto"
|
||||
fp4_gemm_runner_backend: str = "auto"
|
||||
fp4_gemm_runner_backend: str = "flashinfer_cutlass"
|
||||
nsa_prefill_backend: Optional[str] = (
|
||||
None # None = auto-detect based on hardware/kv_cache_dtype
|
||||
)
|
||||
@@ -3789,9 +3789,9 @@ class ServerArgs:
|
||||
default=ServerArgs.fp4_gemm_runner_backend,
|
||||
dest="fp4_gemm_runner_backend",
|
||||
help="Choose the runner backend for NVFP4 GEMM operations. "
|
||||
"Options: 'auto' (default, selects between flashinfer_cudnn/flashinfer_cutlass based on CUDA/cuDNN version), "
|
||||
"Options: 'flashinfer_cutlass' (default), "
|
||||
"'auto' (auto-selects between flashinfer_cudnn/flashinfer_cutlass based on CUDA/cuDNN version), "
|
||||
"'flashinfer_cudnn' (FlashInfer cuDNN backend, optimal on CUDA 13+ with cuDNN 9.15+), "
|
||||
"'flashinfer_cutlass' (FlashInfer CUTLASS backend, optimal on CUDA 12), "
|
||||
"'flashinfer_trtllm' (FlashInfer TensorRT-LLM backend, requires different weight preparation with shuffling). "
|
||||
"NOTE: This replaces the deprecated environment variable "
|
||||
"SGLANG_FLASHINFER_FP4_GEMM_BACKEND.",
|
||||
|
||||
Reference in New Issue
Block a user