DP: support piggyback server load report (#11469)
Signed-off-by: Chang Huaixin (OpenAnolis) <changhuaixin@linux.alibaba.com>
This commit is contained in:
@@ -299,6 +299,18 @@ class SchedulerOutputProcessorMixin:
|
||||
|
||||
return predict_tokens
|
||||
|
||||
def process_batch_result_idle(
|
||||
self: Scheduler,
|
||||
batch: ScheduleBatch,
|
||||
result: GenerationBatchResult,
|
||||
):
|
||||
if result.copy_done is not None:
|
||||
result.copy_done.synchronize()
|
||||
|
||||
self.stream_output_generation(
|
||||
batch.reqs, batch.return_logprob, is_idle_batch=True
|
||||
)
|
||||
|
||||
def process_batch_result_dllm(
|
||||
self: Scheduler,
|
||||
batch: ScheduleBatch,
|
||||
@@ -790,6 +802,7 @@ class SchedulerOutputProcessorMixin:
|
||||
reqs: List[Req],
|
||||
return_logprob: bool,
|
||||
skip_req: Optional[Req] = None,
|
||||
is_idle_batch: bool = False,
|
||||
):
|
||||
rids = []
|
||||
http_worker_ipcs = []
|
||||
@@ -810,6 +823,7 @@ class SchedulerOutputProcessorMixin:
|
||||
spec_accepted_tokens = []
|
||||
retraction_counts = []
|
||||
output_hidden_states = None
|
||||
load = self.get_load()
|
||||
output_routed_experts = None
|
||||
|
||||
queue_times = []
|
||||
@@ -1018,7 +1032,7 @@ class SchedulerOutputProcessorMixin:
|
||||
req.log_time_stats()
|
||||
|
||||
# Send to detokenizer
|
||||
if rids:
|
||||
if reqs or is_idle_batch:
|
||||
if self.model_config.is_multimodal_gen:
|
||||
return
|
||||
|
||||
@@ -1062,6 +1076,7 @@ class SchedulerOutputProcessorMixin:
|
||||
placeholder_tokens_idx=None,
|
||||
placeholder_tokens_val=None,
|
||||
retraction_counts=retraction_counts,
|
||||
load=load,
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
Reference in New Issue
Block a user