[BUGFIX] Fix dp size > 1 for qwen3 vl model (#17624)

Co-authored-by: yizhang2077 <1109276519@qq.com>
This commit is contained in:
Zheng Li
2026-01-30 20:44:25 +08:00
committed by GitHub
co-authored by yizhang2077
parent c04efe030a
commit 0c5a81acb8
5 changed files with 48 additions and 19 deletions
+13 -3
View File
@@ -495,11 +495,19 @@ def run_dp_sharded_mrope_vision_model(
```
"""
tp_size = get_tensor_model_parallel_world_size()
from sglang.srt.layers.dp_attention import (
get_attention_tp_group,
get_attention_tp_rank,
get_attention_tp_size,
)
tp_size = get_attention_tp_size()
if tp_size == 1:
return vision_model(pixel_values, grid_thw=torch.tensor(grid_thw_list))
# GPU_0 tp_rank_local = 0
# GPU_1 tp_rank_local = 1
tp_rank_local = get_tensor_model_parallel_rank()
tp_rank_local = get_attention_tp_rank()
# patches_per_image = [1000, 100, 200, 50]
patches_per_image = [math.prod(grid_thw) for grid_thw in grid_thw_list]
@@ -611,7 +619,9 @@ def run_dp_sharded_mrope_vision_model(
image_embeds_local_padded = image_embeds_local
# Do all_gather to collect embeddings from all ranks
gathered_embeds = tensor_model_parallel_all_gather(image_embeds_local_padded, dim=0)
gathered_embeds = get_attention_tp_group().all_gather(
image_embeds_local_padded, dim=0
)
# Remove padding and reconstruct per-rank embeddings
rank_embeddings = list[torch.Tensor]()