[VLM] Fix CUDA IPC OOM (#16118)

Co-authored-by: luoyuan.luo <luoyuan.luo@antgroup.com>
This commit is contained in:
Yuan Luo
2026-01-07 11:30:35 +08:00
committed by GitHub
co-authored by luoyuan.luo
parent 534ac384db
commit 53846746bf
2 changed files with 7 additions and 0 deletions
@@ -1574,6 +1574,9 @@ class ScheduleBatch(ScheduleBatchDisaggregationDecodeMixin):
mm_item.feature = pixel_values.reconstruct_on_target_device(
torch.cuda.current_device()
)
# The reference by CudaIpcTensorTransportProxy was cut off,
# proactively delete to avoid slow gc.
del pixel_values
self.multimodal_inputs = multimodal_inputs
self.token_type_ids = token_type_ids_tensor
self.seq_lens_sum = sum(seq_lens)