Optimized deepseek-v3/r1 model performance on mxfp4 run (#10008)
Co-authored-by: wunhuang <wunhuang@amd.com> Co-authored-by: HAI <hixiao@gmail.com> Co-authored-by: Hubert Lu <55214931+hubertlu-tw@users.noreply.github.com>
This commit is contained in:
co-authored by
wunhuang
HAI
Hubert Lu
parent
93088b6975
commit
e96973742c
@@ -0,0 +1,13 @@
|
||||
from aiter.ops.triton.batched_gemm_afp4wfp4_pre_quant import (
|
||||
batched_gemm_afp4wfp4_pre_quant,
|
||||
)
|
||||
from aiter.ops.triton.fused_mxfp4_quant import (
|
||||
fused_flatten_mxfp4_quant,
|
||||
fused_rms_mxfp4_quant,
|
||||
)
|
||||
|
||||
__all__ = [
|
||||
"fused_rms_mxfp4_quant",
|
||||
"fused_flatten_mxfp4_quant",
|
||||
"batched_gemm_afp4wfp4_pre_quant",
|
||||
]
|
||||
Reference in New Issue
Block a user