Optimize Qwen3-moe model by using flashinfer fused allreduce (#9973)

Co-authored-by: luoyuan.luo <luoyuan.luo@antgroup.com>
This commit is contained in:
Yuan Luo
2025-09-04 20:48:53 +08:00
committed by GitHub
co-authored by luoyuan.luo
parent 106c2b31fb
commit ec15c8360e
3 changed files with 52 additions and 12 deletions
+4 -1
View File
@@ -105,11 +105,14 @@ class Qwen2MoeMLP(nn.Module):
def forward(
self,
x,
should_allreduce_fusion: bool = False,
use_reduce_scatter: bool = False,
):
gate_up, _ = self.gate_up_proj(x)
x = self.act_fn(gate_up)
x, _ = self.down_proj(x, skip_all_reduce=use_reduce_scatter)
x, _ = self.down_proj(
x, skip_all_reduce=should_allreduce_fusion or use_reduce_scatter
)
return x