Support nvidia/NVIDIA-Nemotron-Nano-12B-v2-VL-BF16 (and nvidia/C-RADIOv2-H) (#12277)
This commit is contained in:
@@ -12,6 +12,7 @@ from sglang.srt.configs.kimi_linear import KimiLinearConfig
|
||||
from sglang.srt.configs.kimi_vl import KimiVLConfig
|
||||
from sglang.srt.configs.kimi_vl_moonvit import MoonViTConfig
|
||||
from sglang.srt.configs.longcat_flash import LongcatFlashConfig
|
||||
from sglang.srt.configs.nano_nemotron_vl import NemotronH_Nano_VL_V2_Config
|
||||
from sglang.srt.configs.nemotron_h import NemotronHConfig
|
||||
from sglang.srt.configs.olmo3 import Olmo3Config
|
||||
from sglang.srt.configs.qwen3_next import Qwen3NextConfig
|
||||
@@ -40,6 +41,7 @@ __all__ = [
|
||||
"DotsOCRConfig",
|
||||
"FalconH1Config",
|
||||
"NemotronHConfig",
|
||||
"NemotronH_Nano_VL_V2_Config",
|
||||
"JetNemotronConfig",
|
||||
"JetVLMConfig",
|
||||
]
|
||||
|
||||
@@ -938,6 +938,7 @@ multimodal_model_archs = [
|
||||
"Mistral3ForConditionalGeneration",
|
||||
"MultiModalityCausalLM",
|
||||
"MllamaForConditionalGeneration",
|
||||
"NemotronH_Nano_VL_V2",
|
||||
"Qwen2AudioForConditionalGeneration",
|
||||
"Qwen2VLForConditionalGeneration",
|
||||
"Qwen2_5_VLForConditionalGeneration",
|
||||
|
||||
@@ -0,0 +1,114 @@
|
||||
# Copyright 2025 SGLang Team
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
# ==============================================================================
|
||||
# Adapted from https://huggingface.co/nvidia/NVIDIA-Nemotron-Nano-12B-v2-VL-BF16/blob/cb5a65ff10232128389d882d805fa609427544f1/configuration.py
|
||||
|
||||
from typing import Any
|
||||
|
||||
from transformers.configuration_utils import PretrainedConfig
|
||||
|
||||
from sglang.srt.configs.nemotron_h import NemotronHConfig
|
||||
from sglang.srt.configs.radio import RadioConfig
|
||||
from sglang.srt.multimodal.internvl_utils import IMAGENET_MEAN, IMAGENET_STD
|
||||
|
||||
|
||||
def float_triplet(seq: Any):
|
||||
a, b, c = tuple(seq)
|
||||
assert (
|
||||
isinstance(a, float) and isinstance(b, float) and isinstance(c, float)
|
||||
), "expected three floats"
|
||||
return a, b, c
|
||||
|
||||
|
||||
class NemotronH_Nano_VL_V2_Config(PretrainedConfig):
|
||||
model_type = "NemotronH_Nano_VL_V2"
|
||||
is_composition = True
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
vision_config=None,
|
||||
llm_config=None,
|
||||
force_image_size: int = 512,
|
||||
patch_size: int = 16,
|
||||
downsample_ratio=0.5,
|
||||
template=None,
|
||||
ps_version="v2",
|
||||
image_tag_type="internvl",
|
||||
projector_hidden_size=4096,
|
||||
vit_hidden_size=1280,
|
||||
video_pruning_rate: float = 0.0,
|
||||
video_context_token: str = "<video>",
|
||||
img_context_token: str = "<image>",
|
||||
img_start_token: str = "<img>",
|
||||
img_end_token: str = "</img>",
|
||||
norm_mean: tuple[float, float, float] | list[float] = IMAGENET_MEAN,
|
||||
norm_std: tuple[float, float, float] | list[float] = IMAGENET_STD,
|
||||
use_thumbnail: bool = True,
|
||||
**kwargs,
|
||||
):
|
||||
super().__init__(**kwargs)
|
||||
|
||||
# Handle both cases: when loading from JSON (llm_config is dict) and when called internally by transformers (llm_config; vision_config are None)
|
||||
if llm_config is not None:
|
||||
self.llm_config = NemotronHConfig(**llm_config)
|
||||
assert isinstance(vision_config, dict), "vision_config must be a dictionary"
|
||||
self.raw_vision_config = vision_config
|
||||
else:
|
||||
assert vision_config is None
|
||||
self.llm_config = NemotronHConfig()
|
||||
self.raw_vision_config = {}
|
||||
|
||||
# Assign configuration values
|
||||
vision_image_size = self.raw_vision_config.get("image_size", force_image_size)
|
||||
vision_patch_size = self.raw_vision_config.get("patch_size", patch_size)
|
||||
self.image_size = int(
|
||||
vision_image_size[0]
|
||||
if isinstance(vision_image_size, list)
|
||||
else vision_image_size
|
||||
)
|
||||
self.patch_size = int(
|
||||
vision_patch_size[0]
|
||||
if isinstance(vision_patch_size, list)
|
||||
else vision_patch_size
|
||||
)
|
||||
|
||||
self.downsample_ratio = downsample_ratio
|
||||
self.video_context_token = video_context_token
|
||||
self.img_context_token = img_context_token
|
||||
self.template = template # TODO move out of here and into the tokenizer
|
||||
self.ps_version = ps_version # Pixel shuffle version
|
||||
self.image_tag_type = image_tag_type # TODO: into the tokenizer too?
|
||||
self.projector_hidden_size = projector_hidden_size
|
||||
self.vit_hidden_size = vit_hidden_size
|
||||
self.video_pruning_rate = video_pruning_rate
|
||||
|
||||
self.norm_mean = float_triplet(norm_mean)
|
||||
self.norm_std = float_triplet(norm_std)
|
||||
self.use_thumbnail = use_thumbnail
|
||||
self.img_start_token = img_start_token
|
||||
self.img_end_token = img_end_token
|
||||
|
||||
def create_radio_config(self):
|
||||
config = self.raw_vision_config
|
||||
model_name = config["args"]["model"]
|
||||
reg_tokens = config["args"].get("register_multiple")
|
||||
image_size = config.get("preferred_resolution", [224])[0]
|
||||
radio_config = RadioConfig(
|
||||
patch_size=self.patch_size,
|
||||
norm_mean=self.norm_mean,
|
||||
norm_std=self.norm_std,
|
||||
model_name=model_name,
|
||||
reg_tokens=reg_tokens,
|
||||
image_size=image_size,
|
||||
)
|
||||
return radio_config
|
||||
@@ -0,0 +1,106 @@
|
||||
# Copyright 2025 SGLang Team
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
# ==============================================================================
|
||||
# Adapted from https://github.com/vllm-project/vllm/blob/main/vllm/transformers_utils/configs/radio.py
|
||||
|
||||
"""Radio vision model configuration"""
|
||||
|
||||
from transformers.configuration_utils import PretrainedConfig
|
||||
from transformers.utils import logging
|
||||
|
||||
logger = logging.get_logger(__name__)
|
||||
|
||||
VIT_TIMM_DIM_BY_NAME: dict[str, tuple[int, int, int, int]] = {
|
||||
"vit_small_patch16_224": (384, 12, 6, 1536),
|
||||
"vit_base_patch16_224": (768, 12, 12, 3072),
|
||||
"vit_large_patch16_224": (1024, 24, 16, 4096),
|
||||
"vit_huge_patch16_224": (1280, 32, 16, 5120),
|
||||
}
|
||||
|
||||
OPENAI_CLIP_MEAN = (0.48145466, 0.4578275, 0.40821073)
|
||||
OPENAI_CLIP_STD = (0.26862954, 0.26130258, 0.27577711)
|
||||
|
||||
|
||||
class RadioConfig(PretrainedConfig):
|
||||
r"""
|
||||
This is the configuration class to store the configuration of a Radio
|
||||
vision model. It is used to instantiate a Radio model according to the
|
||||
specified arguments, defining the model architecture.
|
||||
|
||||
Args:
|
||||
model_name: Name of the vision transformer model
|
||||
(e.g., "vit_base_patch16_224"). Used to determine architecture
|
||||
dimensions from `VIT_TIMM_DIM_BY_NAME`.
|
||||
image_size: The size (resolution) of each image.
|
||||
patch_size: The size (resolution) of each patch.
|
||||
qkv_bias: Whether to add a bias to the queries, keys and values.
|
||||
qk_normalization: Whether to apply normalization to queries and keys.
|
||||
norm_type: The normalization type to use.
|
||||
layer_norm_eps: The epsilon used by the layer normalization layers.
|
||||
initializer_factor: A factor for initializing all weight matrices.
|
||||
hidden_act: The non-linear activation function in the encoder.
|
||||
max_img_size: Maximum image size for position embeddings.
|
||||
norm_mean: Mean values for image normalization (RGB channels).
|
||||
Defaults to (0.48145466, 0.4578275, 0.40821073)).
|
||||
norm_std: Standard deviation values for image normalization
|
||||
(RGB channels). Defaults to (0.26862954, 0.26130258, 0.27577711)).
|
||||
reg_tokens: Number of register tokens to use.
|
||||
"""
|
||||
|
||||
model_type = "radio"
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
model_name: str,
|
||||
image_size: int = 224,
|
||||
patch_size: int = 16,
|
||||
qkv_bias: bool = True,
|
||||
qk_normalization: bool = False,
|
||||
norm_type: str = "layer_norm",
|
||||
layer_norm_eps: float = 1e-6,
|
||||
initializer_factor: float = 1.0,
|
||||
hidden_act: str = "gelu",
|
||||
max_img_size: int = 2048,
|
||||
norm_mean: tuple[float, float, float] | list = OPENAI_CLIP_MEAN,
|
||||
norm_std: tuple[float, float, float] | list = OPENAI_CLIP_STD,
|
||||
reg_tokens: int | None = None,
|
||||
drop_path_rate: float = 0.0,
|
||||
dropout: float = 0.0,
|
||||
**kwargs,
|
||||
):
|
||||
self.model_name = model_name
|
||||
(
|
||||
self.hidden_size,
|
||||
self.num_hidden_layers,
|
||||
self.num_attention_heads,
|
||||
self.intermediate_size,
|
||||
) = VIT_TIMM_DIM_BY_NAME[model_name]
|
||||
self.image_size = image_size
|
||||
self.patch_size = patch_size
|
||||
self.qkv_bias = qkv_bias
|
||||
self.qk_normalization = qk_normalization
|
||||
self.norm_type = norm_type
|
||||
self.layer_norm_eps = layer_norm_eps
|
||||
self.initializer_factor = initializer_factor
|
||||
self.hidden_act = hidden_act
|
||||
self.max_img_size = max_img_size
|
||||
self.norm_mean = (
|
||||
list(norm_mean) if isinstance(norm_mean, (tuple, list)) else norm_mean
|
||||
)
|
||||
self.norm_std = (
|
||||
list(norm_std) if isinstance(norm_std, (tuple, list)) else norm_std
|
||||
)
|
||||
self.reg_tokens = reg_tokens
|
||||
self.drop_path_rate = drop_path_rate
|
||||
self.dropout = dropout
|
||||
super().__init__(**kwargs)
|
||||
Reference in New Issue
Block a user