Optimized deepseek-v3/r1 model performance on mxfp4 run (#9671)
Co-authored-by: wunhuang <wunhuang@amd.com> Co-authored-by: wghuang <wghuang@amd.com>
This commit is contained in:
@@ -0,0 +1,13 @@
|
||||
from aiter.ops.triton.batched_gemm_afp4wfp4_pre_quant import (
|
||||
batched_gemm_afp4wfp4_pre_quant,
|
||||
)
|
||||
from aiter.ops.triton.fused_mxfp4_quant import (
|
||||
fused_flatten_mxfp4_quant,
|
||||
fused_rms_mxfp4_quant,
|
||||
)
|
||||
|
||||
__all__ = [
|
||||
"fused_rms_mxfp4_quant",
|
||||
"fused_flatten_mxfp4_quant",
|
||||
"batched_gemm_afp4wfp4_pre_quant",
|
||||
]
|
||||
Reference in New Issue
Block a user