Optimized deepseek-v3/r1 model performance on mxfp4 run (#9671)

Co-authored-by: wunhuang <wunhuang@amd.com>
Co-authored-by: wghuang <wghuang@amd.com>
This commit is contained in:
kk
2025-09-02 22:26:28 -07:00
committed by GitHub
co-authored by wunhuang wghuang
parent bcbeed714f
commit 0dfd54d11d
7 changed files with 455 additions and 59 deletions
@@ -0,0 +1,13 @@
from aiter.ops.triton.batched_gemm_afp4wfp4_pre_quant import (
batched_gemm_afp4wfp4_pre_quant,
)
from aiter.ops.triton.fused_mxfp4_quant import (
fused_flatten_mxfp4_quant,
fused_rms_mxfp4_quant,
)
__all__ = [
"fused_rms_mxfp4_quant",
"fused_flatten_mxfp4_quant",
"batched_gemm_afp4wfp4_pre_quant",
]