Fix triton head num (#1482)

This commit is contained in:
Ke Bao
2024-09-21 10:25:20 +08:00
committed by GitHub
parent 014982b5e0
commit a68cb201dd
4 changed files with 54 additions and 1 deletions

View File

@@ -346,7 +346,9 @@ class TritonAttnBackend(AttentionBackend):
self.decode_attention_fwd = decode_attention_fwd
self.extend_attention_fwd = extend_attention_fwd
self.num_head = model_runner.model_config.num_attention_heads
self.num_head = (
model_runner.model_config.num_attention_heads // model_runner.tp_size
)
if global_server_args_dict.get("triton_attention_reduce_in_fp32", False):
self.reduce_dtype = torch.float32