[AMD] change fused rms quant interface for aiter upgrade (#14497)

This commit is contained in:
yctseng0211
2025-12-08 09:09:23 -08:00
committed by GitHub
parent 9a327bdfcf
commit 763888b5a8
2 changed files with 3 additions and 3 deletions
+2 -2
View File
@@ -31,7 +31,7 @@ ENV BUILD_TRITON="0"
ENV BUILD_LLVM="0"
ENV BUILD_AITER_ALL="1"
ENV BUILD_MOONCAKE="1"
ENV AITER_COMMIT="v0.1.7.post1"
ENV AITER_COMMIT="v0.1.7.post5"
ENV NO_DEPS_FLAG=""
# ===============================
@@ -42,7 +42,7 @@ ENV BUILD_TRITON="0"
ENV BUILD_LLVM="0"
ENV BUILD_AITER_ALL="0"
ENV BUILD_MOONCAKE="1"
ENV AITER_COMMIT="v0.1.7.post2"
ENV AITER_COMMIT="v0.1.7.post5"
ENV NO_DEPS_FLAG=""
# ===============================
# Chosen arch and args
+1 -1
View File
@@ -1839,7 +1839,7 @@ class DeepseekV2AttentionMLA(nn.Module):
current_stream.wait_stream(self.alt_stream)
else:
if _use_aiter_gfx95 and self.q_b_proj.weight.dtype == torch.uint8:
q, k_nope, *_ = fused_rms_mxfp4_quant(
q, _, k_nope, *_ = fused_rms_mxfp4_quant(
q,
self.q_a_layernorm.weight,
self.q_a_layernorm.variance_epsilon,