[diffusion] fix: fix CLIP text encoder attention mask not used (#14364)

Co-authored-by: niehen6174 <niehen.6174@gmail.com>
Co-authored-by: Mick <mickjagger19@icloud.com>
This commit is contained in:
WenhaoZhang
2025-12-05 16:30:10 +08:00
committed by GitHub
co-authored by niehen6174 Mick
parent 498ea41ca6
commit 35ba6fe19e
2 changed files with 57 additions and 10 deletions
@@ -9,6 +9,7 @@ from sglang.multimodal_gen.configs.models.encoders.base import (
TextEncoderArchConfig,
TextEncoderConfig,
)
from sglang.multimodal_gen.runtime.platforms import AttentionBackendEnum
def _is_transformer_layer(n: str, m) -> bool:
@@ -38,6 +39,11 @@ class CLIPTextArchConfig(TextEncoderArchConfig):
bos_token_id: int = 49406
eos_token_id: int = 49407
text_len: int = 77
_supported_attention_backends: set[AttentionBackendEnum] = field(
default_factory=lambda: {
AttentionBackendEnum.TORCH_SDPA, # Force TORCH_SDPA to support attention_mask
}
)
stacked_params_mapping: list[tuple[str, str, str]] = field(
default_factory=lambda: [
# (param_name, shard_name, shard_id)