[Bugfix] For cp: Fixed hang problem in prefix cache and kvcache support fp8 in-seq-split mode (#19656)

Co-authored-by: vincent <vincent@vincentdeMacBook-Pro.local>
This commit is contained in:
Baidu-AIAK
2026-03-04 11:19:46 +08:00
committed by GitHub
parent 88290690f0
commit 6851613b93
3 changed files with 4 additions and 5 deletions

View File

@@ -54,7 +54,7 @@ def can_nsa_prefill_cp_round_robin_split(forward_batch: "ForwardBatch"):
return False
cp_size = get_attention_cp_size()
seq_len = sum(forward_batch.extend_seq_lens_cpu)
return is_nsa_prefill_cp_round_robin_split() and seq_len > 0 and cp_size > 1
return is_nsa_prefill_cp_round_robin_split() and seq_len > 0 and seq_len >= cp_size and cp_size > 1
def nsa_cp_round_robin_split_data(input_: Union[torch.Tensor, List]):
@@ -167,6 +167,7 @@ def can_cp_split(seq_len: int, cp_size: int, use_nsa: bool, forward_batch):
and use_nsa
and forward_batch.forward_mode.is_context_parallel_extend()
and is_nsa_enable_prefill_cp()
and sum(forward_batch.extend_seq_lens_cpu) >= cp_size
):
return True
else: