enable csgmv automatically on cuda (#13600)

This commit is contained in:
b8zhong
2025-11-20 12:53:02 -08:00
committed by GitHub
parent 7291c72e57
commit 42028af614
2 changed files with 7 additions and 0 deletions

View File

@@ -113,6 +113,7 @@ SGLang supports various environment variables that can be used to configure its
| --- | --- | --- |
| `SGLANG_WAIT_WEIGHTS_READY_TIMEOUT` | Timeout period for waiting on weights | `120` |
| `SGLANG_DISABLE_OUTLINES_DISK_CACHE` | Disable Outlines disk cache | `true` |
| `SGLANG_USE_CUSTOM_TRITON_KERNEL_CACHE` | Use SGLang's custom Triton kernel cache implementation for lower overheads (automatically enabled on CUDA) | `false` |
## Function Calling / Tool Use

View File

@@ -3618,6 +3618,12 @@ def cached_triton_kernel(key_fn=None):
"""
def decorator(fn):
# Auto-enable the custom kernel cache for CUDA, where it is
# known to be compatible.
if is_cuda() and not envs.SGLANG_USE_CUSTOM_TRITON_KERNEL_CACHE.is_set():
logger.debug("Detected platform CUDA, using custom triton kernel cache.")
return CachedKernel(fn, key_fn)
if envs.SGLANG_USE_CUSTOM_TRITON_KERNEL_CACHE.get():
logger.debug(
f"{envs.SGLANG_USE_CUSTOM_TRITON_KERNEL_CACHE.name} = True. Using custom triton kernel cache."