[Feature] Enable CUDA graph for PD-Multiplexing. (#11595)

This commit is contained in:
ykcombat
2025-11-14 03:39:40 +08:00
committed by GitHub
parent bfe638f7e8
commit dd192a55f4
3 changed files with 103 additions and 52 deletions
@@ -63,6 +63,7 @@ class EAGLEDraftCudaGraphRunner:
self.enable_profile_cuda_graph = (
model_runner.server_args.enable_profile_cuda_graph
)
self.enable_pdmux = False
self.deepep_adapter = DeepEPCudaGraphRunnerAdapter()
server_args = model_runner.server_args
@@ -160,7 +161,9 @@ class EAGLEDraftCudaGraphRunner:
def capture(self):
CudaGraphRunner.capture(self)
def capture_one_batch_size(self, num_seqs: int, forward: Callable):
def capture_one_batch_size(
self, num_seqs: int, forward: Callable, stream_idx: int = 0
):
graph = torch.cuda.CUDAGraph()
stream = self.stream
num_tokens = num_seqs * self.num_tokens_per_bs
@@ -61,6 +61,7 @@ class EAGLEDraftExtendCudaGraphRunner:
self.enable_profile_cuda_graph = (
model_runner.server_args.enable_profile_cuda_graph
)
self.enable_pdmux = False
self.capture_bs, self.compile_bs = get_batch_sizes_to_capture(model_runner)
self.padded_static_len = -1
self.deepep_adapter = DeepEPCudaGraphRunnerAdapter()
@@ -189,7 +190,7 @@ class EAGLEDraftExtendCudaGraphRunner:
def capture(self):
CudaGraphRunner.capture(self)
def capture_one_batch_size(self, bs: int, forward: Callable):
def capture_one_batch_size(self, bs: int, forward: Callable, stream_idx: int = 0):
graph = torch.cuda.CUDAGraph()
stream = self.stream
num_tokens = bs * self.num_tokens_per_bs