Fix spec decoding acc length for dpsk-r1-fp4 tp8 (2nd attempt) (#12915)

This commit is contained in:
Qiaolin Yu
2025-11-11 13:30:33 -08:00
committed by GitHub
parent 9b247f7374
commit 4a2768a86b
2 changed files with 4 additions and 11 deletions

View File

@@ -604,7 +604,7 @@ class CommunicateWithAllReduceAndLayerNormFn:
and _is_flashinfer_available
and hasattr(layernorm, "forward_with_allreduce_fusion")
and get_global_server_args().enable_flashinfer_allreduce_fusion
and hidden_states.shape[0] <= 4096
and hidden_states.shape[0] <= 2048
):
hidden_states, residual = layernorm.forward_with_allreduce_fusion(
hidden_states, residual

View File

@@ -96,7 +96,7 @@ _workspace_manager = FlashInferWorkspaceManager()
def ensure_workspace_initialized(
max_token_num: int = 16384, hidden_dim: int = 4096, use_fp32_lamport: bool = False
max_token_num: int = 2048, hidden_dim: int = 4096, use_fp32_lamport: bool = False
):
"""Ensure workspace is initialized"""
if not is_flashinfer_available() or _flashinfer_comm is None:
@@ -128,7 +128,7 @@ def flashinfer_allreduce_residual_rmsnorm(
residual: torch.Tensor,
weight: torch.Tensor,
eps: float = 1e-6,
max_token_num: int = 16384,
max_token_num: int = 2048,
use_oneshot: Optional[bool] = None,
trigger_completion_at_end: bool = False,
fp32_acc: bool = False,
@@ -160,14 +160,7 @@ def flashinfer_allreduce_residual_rmsnorm(
logger.debug("Single GPU, no need for allreduce fusion")
return None, None
if input_tensor.shape[0] > max_token_num:
logger.debug(
"Input token(%d) is greater than max_token_num(%d), "
"falling back to standard implementation",
input_tensor.shape[0],
max_token_num,
)
return None, None
assert input_tensor.shape[0] <= max_token_num
if not ensure_workspace_initialized(
max_token_num=max_token_num,