Support DeepSeek V3.2 Exp (#11061)
Co-authored-by: Stefan He <11166516+hebiao064@users.noreply.github.com> Co-authored-by: Liangsheng Yin <95566987+hnyls2002@users.noreply.github.com> Co-authored-by: Baizhou Zhang <56809903+fridge003@users.noreply.github.com> Co-authored-by: DarkSharpness <76582120+darksharpness@users.noreply.github.com> Co-authored-by: ZhengdQin <46387172+zhengdqin@users.noreply.github.com> Co-authored-by: DarkSharpness <2040703891@qq.com> Co-authored-by: hnyls2002 <lsyincs@gmail.com> Co-authored-by: Zhengda Qin <zhengdqin@gmail.com> Co-authored-by: Liangsheng Yin <hnyls2002@gmail.com> Co-authored-by: HAI <hixiao@gmail.com> Co-authored-by: Baizhou Zhang <sobereddiezhang@gmail.com>
This commit is contained in:
@@ -91,6 +91,7 @@ ATTENTION_BACKEND_CHOICES = [
|
||||
"triton",
|
||||
"torch_native",
|
||||
"flex_attention",
|
||||
"nsa",
|
||||
# NVIDIA specific
|
||||
"cutlass_mla",
|
||||
"fa3",
|
||||
@@ -116,6 +117,8 @@ GRAMMAR_BACKEND_CHOICES = ["xgrammar", "outlines", "llguidance", "none"]
|
||||
|
||||
DETERMINISTIC_ATTENTION_BACKEND_CHOICES = ["flashinfer", "fa3", "triton"]
|
||||
|
||||
NSA_CHOICES = ["flashmla_prefill", "flashmla_decode", "fa3", "tilelang", "aiter"]
|
||||
|
||||
RADIX_EVICTION_POLICY_CHOICES = ["lru", "lfu"]
|
||||
|
||||
|
||||
@@ -284,6 +287,8 @@ class ServerArgs:
|
||||
sampling_backend: Optional[str] = None
|
||||
grammar_backend: Optional[str] = None
|
||||
mm_attention_backend: Optional[str] = None
|
||||
nsa_prefill: str = "flashmla_prefill"
|
||||
nsa_decode: str = "fa3"
|
||||
|
||||
# Speculative decoding
|
||||
speculative_algorithm: Optional[str] = None
|
||||
@@ -719,6 +724,8 @@ class ServerArgs:
|
||||
self.sampling_backend = "pytorch"
|
||||
|
||||
def _handle_model_specific_adjustments(self):
|
||||
from sglang.srt.configs.model_config import is_deepseek_nsa
|
||||
|
||||
if parse_connector_type(self.model_path) == ConnectorType.INSTANCE:
|
||||
return
|
||||
|
||||
@@ -796,6 +803,48 @@ class ServerArgs:
|
||||
)
|
||||
self.disable_hybrid_swa_memory = True
|
||||
|
||||
if is_deepseek_nsa(hf_config):
|
||||
if (
|
||||
self.attention_backend is None
|
||||
and self.prefill_attention_backend is None
|
||||
and self.decode_attention_backend is None
|
||||
):
|
||||
self.attention_backend = "nsa"
|
||||
logger.warning("Set nsa attention backend for DeepSeek NSA.")
|
||||
|
||||
if not is_npu():
|
||||
self.enable_dp_attention = True
|
||||
self.dp_size = self.tp_size
|
||||
logger.warning("DP attention is enabled for DeepSeek NSA.")
|
||||
|
||||
self.page_size = 64
|
||||
logger.warning("Setting page size to 64 for DeepSeek NSA.")
|
||||
|
||||
self.mem_fraction_static = 0.8
|
||||
logger.warning("Setting mem fraction static to 0.8 for DeepSeek NSA.")
|
||||
|
||||
# For Hopper, we support both bf16 and fp8 kv cache; for Blackwell, we support fp8 only currently
|
||||
import torch
|
||||
|
||||
major, _ = torch.cuda.get_device_capability()
|
||||
if major >= 10:
|
||||
self.kv_cache_dtype = "fp8_e4m3"
|
||||
logger.warning("Setting KV cache dtype to fp8.")
|
||||
|
||||
if self.kv_cache_dtype == "fp8_e4m3":
|
||||
self.nsa_prefill = "flashmla_decode"
|
||||
self.nsa_decode = "flashmla_decode"
|
||||
logger.warning(
|
||||
"Setting NSA backend to flashmla_decode for FP8 KV Cache."
|
||||
)
|
||||
|
||||
# Logging env vars for NSA
|
||||
from sglang.srt.layers.attention.nsa.utils import (
|
||||
print_nsa_bool_env_vars,
|
||||
)
|
||||
|
||||
print_nsa_bool_env_vars()
|
||||
|
||||
def _handle_sampling_backend(self):
|
||||
if self.sampling_backend is None:
|
||||
self.sampling_backend = (
|
||||
@@ -1023,6 +1072,7 @@ class ServerArgs:
|
||||
|
||||
model_arch = self.get_hf_config().architectures[0]
|
||||
if model_arch in [
|
||||
"DeepseekV32ForCausalLM",
|
||||
"DeepseekV3ForCausalLM",
|
||||
"Glm4MoeForCausalLM",
|
||||
"BailingMoeForCausalLM",
|
||||
@@ -1974,6 +2024,18 @@ class ServerArgs:
|
||||
default=ServerArgs.mm_attention_backend,
|
||||
help="Set multimodal attention backend.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--nsa-prefill",
|
||||
default=ServerArgs.nsa_prefill,
|
||||
type=str,
|
||||
choices=NSA_CHOICES,
|
||||
)
|
||||
parser.add_argument(
|
||||
"--nsa-decode",
|
||||
default=ServerArgs.nsa_decode,
|
||||
type=str,
|
||||
choices=NSA_CHOICES,
|
||||
)
|
||||
|
||||
# Speculative decoding
|
||||
parser.add_argument(
|
||||
@@ -3251,6 +3313,7 @@ def auto_choose_speculative_params(self: ServerArgs):
|
||||
# The default value for llama
|
||||
return (5, 4, 8)
|
||||
elif arch in [
|
||||
"DeepseekV32ForCausalLM",
|
||||
"DeepseekV3ForCausalLM",
|
||||
"DeepseekV2ForCausalLM",
|
||||
"GptOssForCausalLM",
|
||||
|
||||
Reference in New Issue
Block a user