[diffusion] feat: support sageattn & sageattn3 backend (#14878)
This commit is contained in:
@@ -129,7 +129,7 @@ class CudaPlatformBase(Platform):
|
||||
SlidingTileAttentionBackend,
|
||||
)
|
||||
|
||||
logger.info("Using Sliding Tile Attention backend.")
|
||||
logger.info("Using Sliding Tile Attention backend")
|
||||
|
||||
return "sglang.multimodal_gen.runtime.layers.attention.backends.sliding_tile_attn.SlidingTileAttentionBackend"
|
||||
except ImportError as e:
|
||||
@@ -147,30 +147,27 @@ class CudaPlatformBase(Platform):
|
||||
SageAttentionBackend,
|
||||
)
|
||||
|
||||
logger.info("Using Sage Attention backend.")
|
||||
logger.info("Using Sage Attention backend")
|
||||
|
||||
return "sglang.multimodal_gen.runtime.layers.attention.backends.sage_attn.SageAttentionBackend"
|
||||
except ImportError as e:
|
||||
logger.info(e)
|
||||
logger.info(
|
||||
"Sage Attention backend is not installed. Fall back to Flash Attention."
|
||||
"Sage Attention backend is not installed (To install it, run `pip install sageattention==2.2.0 --no-build-isolation`). Falling back to Flash Attention."
|
||||
)
|
||||
elif selected_backend == AttentionBackendEnum.SAGE_ATTN_THREE:
|
||||
elif selected_backend == AttentionBackendEnum.SAGE_ATTN_3:
|
||||
try:
|
||||
from sglang.multimodal_gen.runtime.layers.attention.backends.sage_attn3 import ( # noqa: F401
|
||||
SageAttention3Backend,
|
||||
)
|
||||
from sglang.multimodal_gen.runtime.layers.attention.backends.sageattn.api import ( # noqa: F401
|
||||
sageattn_blackwell,
|
||||
)
|
||||
|
||||
logger.info("Using Sage Attention 3 backend.")
|
||||
logger.info("Using Sage Attention 3 backend")
|
||||
|
||||
return "sglang.multimodal_gen.runtime.layers.attention.backends.sage_attn3.SageAttention3Backend"
|
||||
except ImportError as e:
|
||||
logger.info(e)
|
||||
logger.info(
|
||||
"Sage Attention 3 backend is not installed. Fall back to Flash Attention."
|
||||
"Sage Attention 3 backend is not installed (To install it, see https://github.com/thu-ml/SageAttention/tree/main/sageattention3_blackwell#installation). Falling back to Flash Attention."
|
||||
)
|
||||
elif selected_backend == AttentionBackendEnum.VIDEO_SPARSE_ATTN:
|
||||
try:
|
||||
@@ -180,7 +177,7 @@ class CudaPlatformBase(Platform):
|
||||
VideoSparseAttentionBackend,
|
||||
)
|
||||
|
||||
logger.info("Using Video Sparse Attention backend.")
|
||||
logger.info("Using Video Sparse Attention backend")
|
||||
|
||||
return "sglang.multimodal_gen.runtime.layers.attention.backends.video_sparse_attn.VideoSparseAttentionBackend"
|
||||
except ImportError as e:
|
||||
@@ -188,7 +185,7 @@ class CudaPlatformBase(Platform):
|
||||
"Failed to import Video Sparse Attention backend: %s", str(e)
|
||||
)
|
||||
raise ImportError(
|
||||
"Video Sparse Attention backend is not installed. "
|
||||
"Video Sparse Attention backend is not installed."
|
||||
) from e
|
||||
elif selected_backend == AttentionBackendEnum.VMOBA_ATTN:
|
||||
try:
|
||||
@@ -198,7 +195,7 @@ class CudaPlatformBase(Platform):
|
||||
VMOBAAttentionBackend,
|
||||
)
|
||||
|
||||
logger.info("Using Video MOBA Attention backend.")
|
||||
logger.info("Using Video MOBA Attention backend")
|
||||
|
||||
return "sglang.multimodal_gen.runtime.layers.attention.backends.vmoba.VMOBAAttentionBackend"
|
||||
except ImportError as e:
|
||||
@@ -209,10 +206,10 @@ class CudaPlatformBase(Platform):
|
||||
"Video MoBA Attention backend is not installed. "
|
||||
) from e
|
||||
elif selected_backend == AttentionBackendEnum.AITER:
|
||||
logger.info("Using AITer backend.")
|
||||
logger.info("Using AITer backend")
|
||||
return "sglang.multimodal_gen.runtime.layers.attention.backends.aiter.AITerBackend"
|
||||
elif selected_backend == AttentionBackendEnum.TORCH_SDPA:
|
||||
logger.info("Using Torch SDPA backend.")
|
||||
logger.info("Using Torch SDPA backend")
|
||||
return "sglang.multimodal_gen.runtime.layers.attention.backends.sdpa.SDPABackend"
|
||||
elif selected_backend in [
|
||||
AttentionBackendEnum.FA,
|
||||
@@ -272,7 +269,7 @@ class CudaPlatformBase(Platform):
|
||||
target_backend = AttentionBackendEnum.TORCH_SDPA
|
||||
|
||||
if target_backend == AttentionBackendEnum.TORCH_SDPA:
|
||||
logger.info("Using Torch SDPA backend.")
|
||||
logger.info("Using Torch SDPA backend")
|
||||
|
||||
return "sglang.multimodal_gen.runtime.layers.attention.backends.sdpa.SDPABackend"
|
||||
|
||||
|
||||
@@ -27,7 +27,7 @@ class AttentionBackendEnum(enum.Enum):
|
||||
SLIDING_TILE_ATTN = enum.auto()
|
||||
TORCH_SDPA = enum.auto()
|
||||
SAGE_ATTN = enum.auto()
|
||||
SAGE_ATTN_THREE = enum.auto()
|
||||
SAGE_ATTN_3 = enum.auto()
|
||||
VIDEO_SPARSE_ATTN = enum.auto()
|
||||
VMOBA_ATTN = enum.auto()
|
||||
AITER = enum.auto()
|
||||
|
||||
Reference in New Issue
Block a user