From 92ddc468248a5b62c1e78968f98934ebe00fb269 Mon Sep 17 00:00:00 2001 From: Lianmin Zheng Date: Wed, 24 Dec 2025 14:55:14 -0800 Subject: [PATCH] [Auto Sync] Update server_args.py (20251223) (#15700) Co-authored-by: github-actions[bot] Co-authored-by: Stefan He --- python/sglang/srt/server_args.py | 10 ++++++++-- 1 file changed, 8 insertions(+), 2 deletions(-) diff --git a/python/sglang/srt/server_args.py b/python/sglang/srt/server_args.py index af7a05f95..ac6cb90e2 100644 --- a/python/sglang/srt/server_args.py +++ b/python/sglang/srt/server_args.py @@ -549,7 +549,7 @@ class ServerArgs: enable_piecewise_cuda_graph: bool = False enable_torch_compile_debug_mode: bool = False torch_compile_max_bs: int = 32 - piecewise_cuda_graph_max_tokens: int = 4096 + piecewise_cuda_graph_max_tokens: Optional[int] = None piecewise_cuda_graph_tokens: Optional[List[int]] = None piecewise_cuda_graph_compiler: str = "eager" torchao_config: str = "" @@ -895,6 +895,12 @@ class ServerArgs: else: self.cuda_graph_max_bs = max(self.cuda_graph_bs) + if self.piecewise_cuda_graph_max_tokens is None: + # Capture piecewise cuda graph tokens up to the chunked prefill size. Two benefits: + # 1. cuda graph acceleration for all prefill lengths. + # 2. do not need more temporary memory for activations. Less fragmentation. + self.piecewise_cuda_graph_max_tokens = self.chunked_prefill_size + if self.piecewise_cuda_graph_tokens is None: self.piecewise_cuda_graph_tokens = ( self._generate_piecewise_cuda_graph_tokens() @@ -936,7 +942,7 @@ class ServerArgs: # For piecewise cuda graphs if self.enable_piecewise_cuda_graph: - reserved_mem += self.piecewise_cuda_graph_max_tokens // 4 + reserved_mem += self.piecewise_cuda_graph_max_tokens // 8 self.mem_fraction_static = ( round((gpu_mem - reserved_mem) / gpu_mem, 3)