add variable TP Decode > Prefill size support (#9960)
Signed-off-by: Shahar Mor <smor@nvidia.com>
This commit is contained in:
@@ -459,7 +459,9 @@ class MooncakeKVManager(BaseKVManager):
|
||||
dst_head_start_offset = local_tp_rank_in_group * src_heads_per_rank
|
||||
else:
|
||||
# Send KVCache from 1 prefill instance to multiple decode instances
|
||||
src_head_start_offset = dst_tp_rank_in_group * dst_heads_per_rank
|
||||
src_head_start_offset = (
|
||||
dst_tp_rank_in_group * dst_heads_per_rank
|
||||
) % src_heads_per_rank
|
||||
num_heads_to_send = dst_heads_per_rank
|
||||
dst_head_start_offset = 0
|
||||
|
||||
|
||||
Reference in New Issue
Block a user