Fuse routed scaling factor in topk_reduce kernel (#6220)

This commit is contained in:
Xiaoyu Zhang
2025-06-08 02:06:50 +08:00
committed by GitHub
parent f5599ef124
commit 515ef4facb
10 changed files with 331 additions and 9 deletions

View File

@@ -328,4 +328,5 @@ class W8A8FP8MoEMethod:
a1_scale=layer.w13_input_scale,
a2_scale=layer.w2_input_scale,
no_combine=no_combine,
routed_scaling_factor=routed_scaling_factor,
)