move all get_stream in sgl_kernel to c++ to reduce the launch overhead (#12521)
This commit is contained in:
@@ -19,7 +19,6 @@ from transformers.configuration_utils import PretrainedConfig
|
||||
from transformers.utils import logging
|
||||
|
||||
from sglang.srt.configs.mamba_utils import Mamba2CacheParams, Mamba2StateShape
|
||||
from sglang.srt.layers.dp_attention import get_tensor_model_parallel_world_size
|
||||
|
||||
logger = logging.get_logger(__name__)
|
||||
|
||||
@@ -297,8 +296,10 @@ class FalconH1Config(PretrainedConfig):
|
||||
|
||||
@property
|
||||
def mamba2_cache_params(self):
|
||||
from sglang.srt.layers.dp_attention import get_attention_tp_size
|
||||
|
||||
shape = Mamba2StateShape.create(
|
||||
tp_world_size=get_tensor_model_parallel_world_size(),
|
||||
tp_world_size=get_attention_tp_size(),
|
||||
intermediate_size=self.mamba_intermediate,
|
||||
n_groups=self.mamba_n_groups,
|
||||
num_heads=self.mamba_n_heads,
|
||||
|
||||
@@ -20,7 +20,6 @@ from transformers.configuration_utils import PretrainedConfig
|
||||
from transformers.utils import logging
|
||||
|
||||
from sglang.srt.configs.mamba_utils import Mamba2CacheParams, Mamba2StateShape
|
||||
from sglang.srt.layers.dp_attention import get_attention_tp_size
|
||||
|
||||
logger = logging.get_logger(__name__)
|
||||
|
||||
@@ -273,6 +272,8 @@ class NemotronHConfig(PretrainedConfig):
|
||||
|
||||
@property
|
||||
def mamba2_cache_params(self) -> Mamba2CacheParams:
|
||||
from sglang.srt.layers.dp_attention import get_attention_tp_size
|
||||
|
||||
shape = Mamba2StateShape.create(
|
||||
tp_world_size=get_attention_tp_size(),
|
||||
intermediate_size=self.mamba_num_heads * self.mamba_head_dim,
|
||||
|
||||
@@ -21,7 +21,6 @@ from transformers.modeling_rope_utils import rope_config_validation
|
||||
from transformers.utils import logging
|
||||
|
||||
from sglang.srt.configs.mamba_utils import Mamba2CacheParams, Mamba2StateShape
|
||||
from sglang.srt.layers.dp_attention import get_attention_tp_size
|
||||
|
||||
logger = logging.get_logger(__name__)
|
||||
|
||||
@@ -277,6 +276,8 @@ class Qwen3NextConfig(PretrainedConfig):
|
||||
|
||||
@property
|
||||
def mamba2_cache_params(self) -> Mamba2CacheParams:
|
||||
from sglang.srt.layers.dp_attention import get_attention_tp_size
|
||||
|
||||
shape = Mamba2StateShape.create(
|
||||
tp_world_size=get_attention_tp_size(),
|
||||
intermediate_size=self.linear_value_head_dim * self.linear_num_value_heads,
|
||||
|
||||
Reference in New Issue
Block a user