Intoduce cpu tensor as metadata to avoid blocking gpu kernel launch (#10720)
Co-authored-by: hnyls2002 <lsyincs@gmail.com>
This commit is contained in:
@@ -104,14 +104,21 @@ class EagleVerifyInput(SpecInput):
|
||||
end_offset = batch.seq_lens + self.draft_token_num
|
||||
else:
|
||||
prefix_lens = batch.seq_lens
|
||||
prefix_lens_cpu = batch.seq_lens_cpu
|
||||
end_offset = prefix_lens + self.draft_token_num
|
||||
end_offset_cpu = prefix_lens_cpu + self.draft_token_num
|
||||
last_loc = get_last_loc(
|
||||
batch.req_to_token_pool.req_to_token,
|
||||
batch.req_pool_indices,
|
||||
prefix_lens,
|
||||
)
|
||||
batch.out_cache_loc = batch.alloc_paged_token_slots_extend(
|
||||
prefix_lens, end_offset, last_loc, len(batch.input_ids)
|
||||
prefix_lens,
|
||||
prefix_lens_cpu,
|
||||
end_offset,
|
||||
end_offset_cpu,
|
||||
last_loc,
|
||||
len(batch.input_ids),
|
||||
)
|
||||
self.last_loc = last_loc
|
||||
|
||||
@@ -380,6 +387,8 @@ class EagleVerifyInput(SpecInput):
|
||||
verified_id = predict[accept_index]
|
||||
evict_mask = torch.full_like(self.draft_token, True, dtype=torch.bool)
|
||||
evict_mask[accept_index] = False
|
||||
accept_length_cpu = accept_length.cpu()
|
||||
accept_length_list = accept_length_cpu.tolist()
|
||||
|
||||
if page_size == 1:
|
||||
# TODO: boolean array index leads to a device sync. Remove it.
|
||||
@@ -456,13 +465,15 @@ class EagleVerifyInput(SpecInput):
|
||||
else:
|
||||
batch.out_cache_loc = tgt_cache_loc
|
||||
batch.seq_lens.add_(accept_length + 1)
|
||||
batch.seq_lens_cpu.add_(accept_length_cpu + 1)
|
||||
|
||||
draft_input = EagleDraftInput(
|
||||
hidden_states=batch.spec_info.hidden_states[accept_index],
|
||||
verified_id=verified_id,
|
||||
accept_length=accept_length,
|
||||
accept_length_cpu=accept_length.tolist(),
|
||||
accept_length_cpu=accept_length_list,
|
||||
seq_lens_for_draft_extend=batch.seq_lens,
|
||||
seq_lens_for_draft_extend_cpu=batch.seq_lens_cpu,
|
||||
req_pool_indices_for_draft_extend=batch.req_pool_indices,
|
||||
)
|
||||
|
||||
@@ -485,15 +496,15 @@ class EagleVerifyInput(SpecInput):
|
||||
next_power_of_2(bs),
|
||||
)
|
||||
batch.seq_lens.add_(accept_length + 1)
|
||||
batch.seq_lens_cpu.add_(accept_length_cpu + 1)
|
||||
|
||||
accept_length_cpu = accept_length.tolist()
|
||||
if len(unfinished_accept_index) > 0:
|
||||
unfinished_accept_index = torch.cat(unfinished_accept_index)
|
||||
unfinished_index_device = torch.tensor(
|
||||
unfinished_index, dtype=torch.int64, device=predict.device
|
||||
)
|
||||
draft_input_accept_length_cpu = [
|
||||
accept_length_cpu[i] for i in unfinished_index
|
||||
accept_length_list[i] for i in unfinished_index
|
||||
]
|
||||
if page_size == 1 or self.topk == 1:
|
||||
batch.out_cache_loc = batch.out_cache_loc[unfinished_accept_index]
|
||||
@@ -508,6 +519,7 @@ class EagleVerifyInput(SpecInput):
|
||||
unfinished_index_device,
|
||||
batch.seq_lens,
|
||||
)
|
||||
batch.seq_lens_cpu.add_(accept_length_cpu + 1)
|
||||
filter_finished_cache_loc_kernel[(bs,)](
|
||||
batch.out_cache_loc,
|
||||
tgt_cache_loc,
|
||||
@@ -525,6 +537,7 @@ class EagleVerifyInput(SpecInput):
|
||||
accept_length_cpu=draft_input_accept_length_cpu,
|
||||
accept_length=accept_length[unfinished_index_device],
|
||||
seq_lens_for_draft_extend=batch.seq_lens[unfinished_index_device],
|
||||
seq_lens_for_draft_extend_cpu=batch.seq_lens_cpu[unfinished_index],
|
||||
req_pool_indices_for_draft_extend=batch.req_pool_indices[
|
||||
unfinished_index_device
|
||||
],
|
||||
@@ -542,7 +555,7 @@ class EagleVerifyInput(SpecInput):
|
||||
draft_input=draft_input,
|
||||
logits_output=logits_output,
|
||||
verified_id=verified_id,
|
||||
accept_length_per_req_cpu=accept_length_cpu,
|
||||
accept_length_per_req_cpu=accept_length_list,
|
||||
accepted_indices=accept_index,
|
||||
)
|
||||
|
||||
@@ -575,6 +588,7 @@ class EagleDraftInput(SpecInput):
|
||||
# Inputs for draft extend
|
||||
# shape: (b,)
|
||||
seq_lens_for_draft_extend: torch.Tensor = None
|
||||
seq_lens_for_draft_extend_cpu: torch.Tensor = None
|
||||
req_pool_indices_for_draft_extend: torch.Tensor = None
|
||||
|
||||
def __post_init__(self):
|
||||
@@ -631,6 +645,7 @@ class EagleDraftInput(SpecInput):
|
||||
batch.extend_lens = [x + 1 for x in batch.spec_info.accept_length_cpu]
|
||||
batch.extend_num_tokens = sum(batch.extend_lens)
|
||||
batch.seq_lens = batch.spec_info.seq_lens_for_draft_extend
|
||||
batch.seq_lens_cpu = batch.spec_info.seq_lens_for_draft_extend_cpu
|
||||
batch.req_pool_indices = batch.spec_info.req_pool_indices_for_draft_extend
|
||||
batch.return_logprob = False
|
||||
batch.return_hidden_states = False
|
||||
|
||||
Reference in New Issue
Block a user