Register allgather/reducescatter buffers with symm memory (#12572)

This commit is contained in:
Nicolas Castet
2025-11-04 17:11:36 -08:00
committed by GitHub
parent 1357ab025a
commit 2340798353
19 changed files with 250 additions and 114 deletions
+5 -3
View File
@@ -480,7 +480,8 @@ class DeepseekV2MLP(nn.Module):
gate_up, _ = self.gate_up_proj(x)
x = self.act_fn(gate_up)
x, _ = self.down_proj(
x, skip_all_reduce=should_allreduce_fusion or use_reduce_scatter
x,
skip_all_reduce=should_allreduce_fusion or use_reduce_scatter,
)
return x
@@ -814,7 +815,6 @@ class DeepseekV2MoE(nn.Module):
final_hidden_states *= self.routed_scaling_factor
if shared_output is not None:
final_hidden_states += shared_output
if (
self.tp_size > 1
and not should_allreduce_fusion
@@ -883,7 +883,9 @@ class DeepseekV2MoE(nn.Module):
return final_hidden_states
def forward_deepep(
self, hidden_states: torch.Tensor, forward_batch: ForwardBatch
self,
hidden_states: torch.Tensor,
forward_batch: ForwardBatch,
) -> torch.Tensor:
shared_output = None
if hidden_states.shape[0] > 0: