[Model] Add PaddleOCR-VL Model Support (#12953)

Co-authored-by: luoyuan.luo <luoyuan.luo@antgroup.com>
This commit is contained in:
yudian0504
2025-12-09 10:16:02 -08:00
committed by GitHub
co-authored by luoyuan.luo
parent ab0048793c
commit 9496f12d00
6 changed files with 806 additions and 6 deletions
@@ -0,0 +1,30 @@
# Reference: ccr-2vdh3abv-pub.cnc.bj.baidubce.com/paddlepaddle/paddleocr-genai-vllm-server:latest
# Copyright (c) 2025 PaddlePaddle Authors. All Rights Reserved.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
from sglang.srt.models.paddleocr_vl import PaddleOCRVLForConditionalGeneration
from sglang.srt.multimodal.processors.base_processor import MultimodalSpecialTokens
from sglang.srt.multimodal.processors.qwen_vl import QwenVLImageProcessor
class PaddleOCRVLImageProcessor(QwenVLImageProcessor):
models = [PaddleOCRVLForConditionalGeneration]
def __init__(self, hf_config, server_args, _processor, *args, **kwargs):
super().__init__(hf_config, server_args, _processor, *args, **kwargs)
self.mm_tokens = MultimodalSpecialTokens(
image_token="<|IMAGE_START|><|IMAGE_PLACEHOLDER|><|IMAGE_END|>",
image_token_id=hf_config.image_token_id,
video_token_id=hf_config.video_token_id,
).build(_processor)
@@ -235,10 +235,8 @@ class QwenVLImageProcessor(SGLangBaseProcessor):
super().__init__(hf_config, server_args, _processor, *args, **kwargs)
self.IM_START_TOKEN_ID = hf_config.vision_start_token_id
self.IM_END_TOKEN_ID = hf_config.vision_end_token_id
self.vision_start_token_id = hf_config.vision_start_token_id
self.vision_end_token_id = hf_config.vision_end_token_id
self.vision_end_token_id = getattr(hf_config, "vision_end_token_id", None)
self.audio_start_token_id = getattr(hf_config, "audio_start_token_id", None)
self.audio_token_id = getattr(hf_config, "audio_token_id", None)
@@ -351,8 +349,8 @@ class QwenVLImageProcessor(SGLangBaseProcessor):
return {
"input_ids": input_ids.tolist(),
"mm_items": mm_items,
"im_start_id": self.IM_START_TOKEN_ID,
"im_end_id": self.IM_END_TOKEN_ID,
"im_start_id": self.vision_start_token_id,
"im_end_id": self.vision_end_token_id,
"im_token_id": self.mm_tokens.image_token_id,
"video_token_id": self.mm_tokens.video_token_id,
"audio_token_id": self.mm_tokens.audio_token_id,