Support nvidia/NVIDIA-Nemotron-Nano-12B-v2-VL-BF16 (and nvidia/C-RADIOv2-H) (#12277)

This commit is contained in:
Netanel Haber
2025-11-26 16:28:52 -07:00
committed by GitHub
parent a8ef4d1804
commit 082b54c689
17 changed files with 1334 additions and 17 deletions
+2
View File
@@ -12,6 +12,7 @@ from sglang.srt.configs.kimi_linear import KimiLinearConfig
from sglang.srt.configs.kimi_vl import KimiVLConfig
from sglang.srt.configs.kimi_vl_moonvit import MoonViTConfig
from sglang.srt.configs.longcat_flash import LongcatFlashConfig
from sglang.srt.configs.nano_nemotron_vl import NemotronH_Nano_VL_V2_Config
from sglang.srt.configs.nemotron_h import NemotronHConfig
from sglang.srt.configs.olmo3 import Olmo3Config
from sglang.srt.configs.qwen3_next import Qwen3NextConfig
@@ -40,6 +41,7 @@ __all__ = [
"DotsOCRConfig",
"FalconH1Config",
"NemotronHConfig",
"NemotronH_Nano_VL_V2_Config",
"JetNemotronConfig",
"JetVLMConfig",
]
@@ -938,6 +938,7 @@ multimodal_model_archs = [
"Mistral3ForConditionalGeneration",
"MultiModalityCausalLM",
"MllamaForConditionalGeneration",
"NemotronH_Nano_VL_V2",
"Qwen2AudioForConditionalGeneration",
"Qwen2VLForConditionalGeneration",
"Qwen2_5_VLForConditionalGeneration",
@@ -0,0 +1,114 @@
# Copyright 2025 SGLang Team
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
# ==============================================================================
# Adapted from https://huggingface.co/nvidia/NVIDIA-Nemotron-Nano-12B-v2-VL-BF16/blob/cb5a65ff10232128389d882d805fa609427544f1/configuration.py
from typing import Any
from transformers.configuration_utils import PretrainedConfig
from sglang.srt.configs.nemotron_h import NemotronHConfig
from sglang.srt.configs.radio import RadioConfig
from sglang.srt.multimodal.internvl_utils import IMAGENET_MEAN, IMAGENET_STD
def float_triplet(seq: Any):
a, b, c = tuple(seq)
assert (
isinstance(a, float) and isinstance(b, float) and isinstance(c, float)
), "expected three floats"
return a, b, c
class NemotronH_Nano_VL_V2_Config(PretrainedConfig):
model_type = "NemotronH_Nano_VL_V2"
is_composition = True
def __init__(
self,
vision_config=None,
llm_config=None,
force_image_size: int = 512,
patch_size: int = 16,
downsample_ratio=0.5,
template=None,
ps_version="v2",
image_tag_type="internvl",
projector_hidden_size=4096,
vit_hidden_size=1280,
video_pruning_rate: float = 0.0,
video_context_token: str = "<video>",
img_context_token: str = "<image>",
img_start_token: str = "<img>",
img_end_token: str = "</img>",
norm_mean: tuple[float, float, float] | list[float] = IMAGENET_MEAN,
norm_std: tuple[float, float, float] | list[float] = IMAGENET_STD,
use_thumbnail: bool = True,
**kwargs,
):
super().__init__(**kwargs)
# Handle both cases: when loading from JSON (llm_config is dict) and when called internally by transformers (llm_config; vision_config are None)
if llm_config is not None:
self.llm_config = NemotronHConfig(**llm_config)
assert isinstance(vision_config, dict), "vision_config must be a dictionary"
self.raw_vision_config = vision_config
else:
assert vision_config is None
self.llm_config = NemotronHConfig()
self.raw_vision_config = {}
# Assign configuration values
vision_image_size = self.raw_vision_config.get("image_size", force_image_size)
vision_patch_size = self.raw_vision_config.get("patch_size", patch_size)
self.image_size = int(
vision_image_size[0]
if isinstance(vision_image_size, list)
else vision_image_size
)
self.patch_size = int(
vision_patch_size[0]
if isinstance(vision_patch_size, list)
else vision_patch_size
)
self.downsample_ratio = downsample_ratio
self.video_context_token = video_context_token
self.img_context_token = img_context_token
self.template = template # TODO move out of here and into the tokenizer
self.ps_version = ps_version # Pixel shuffle version
self.image_tag_type = image_tag_type # TODO: into the tokenizer too?
self.projector_hidden_size = projector_hidden_size
self.vit_hidden_size = vit_hidden_size
self.video_pruning_rate = video_pruning_rate
self.norm_mean = float_triplet(norm_mean)
self.norm_std = float_triplet(norm_std)
self.use_thumbnail = use_thumbnail
self.img_start_token = img_start_token
self.img_end_token = img_end_token
def create_radio_config(self):
config = self.raw_vision_config
model_name = config["args"]["model"]
reg_tokens = config["args"].get("register_multiple")
image_size = config.get("preferred_resolution", [224])[0]
radio_config = RadioConfig(
patch_size=self.patch_size,
norm_mean=self.norm_mean,
norm_std=self.norm_std,
model_name=model_name,
reg_tokens=reg_tokens,
image_size=image_size,
)
return radio_config
+106
View File
@@ -0,0 +1,106 @@
# Copyright 2025 SGLang Team
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
# ==============================================================================
# Adapted from https://github.com/vllm-project/vllm/blob/main/vllm/transformers_utils/configs/radio.py
"""Radio vision model configuration"""
from transformers.configuration_utils import PretrainedConfig
from transformers.utils import logging
logger = logging.get_logger(__name__)
VIT_TIMM_DIM_BY_NAME: dict[str, tuple[int, int, int, int]] = {
"vit_small_patch16_224": (384, 12, 6, 1536),
"vit_base_patch16_224": (768, 12, 12, 3072),
"vit_large_patch16_224": (1024, 24, 16, 4096),
"vit_huge_patch16_224": (1280, 32, 16, 5120),
}
OPENAI_CLIP_MEAN = (0.48145466, 0.4578275, 0.40821073)
OPENAI_CLIP_STD = (0.26862954, 0.26130258, 0.27577711)
class RadioConfig(PretrainedConfig):
r"""
This is the configuration class to store the configuration of a Radio
vision model. It is used to instantiate a Radio model according to the
specified arguments, defining the model architecture.
Args:
model_name: Name of the vision transformer model
(e.g., "vit_base_patch16_224"). Used to determine architecture
dimensions from `VIT_TIMM_DIM_BY_NAME`.
image_size: The size (resolution) of each image.
patch_size: The size (resolution) of each patch.
qkv_bias: Whether to add a bias to the queries, keys and values.
qk_normalization: Whether to apply normalization to queries and keys.
norm_type: The normalization type to use.
layer_norm_eps: The epsilon used by the layer normalization layers.
initializer_factor: A factor for initializing all weight matrices.
hidden_act: The non-linear activation function in the encoder.
max_img_size: Maximum image size for position embeddings.
norm_mean: Mean values for image normalization (RGB channels).
Defaults to (0.48145466, 0.4578275, 0.40821073)).
norm_std: Standard deviation values for image normalization
(RGB channels). Defaults to (0.26862954, 0.26130258, 0.27577711)).
reg_tokens: Number of register tokens to use.
"""
model_type = "radio"
def __init__(
self,
model_name: str,
image_size: int = 224,
patch_size: int = 16,
qkv_bias: bool = True,
qk_normalization: bool = False,
norm_type: str = "layer_norm",
layer_norm_eps: float = 1e-6,
initializer_factor: float = 1.0,
hidden_act: str = "gelu",
max_img_size: int = 2048,
norm_mean: tuple[float, float, float] | list = OPENAI_CLIP_MEAN,
norm_std: tuple[float, float, float] | list = OPENAI_CLIP_STD,
reg_tokens: int | None = None,
drop_path_rate: float = 0.0,
dropout: float = 0.0,
**kwargs,
):
self.model_name = model_name
(
self.hidden_size,
self.num_hidden_layers,
self.num_attention_heads,
self.intermediate_size,
) = VIT_TIMM_DIM_BY_NAME[model_name]
self.image_size = image_size
self.patch_size = patch_size
self.qkv_bias = qkv_bias
self.qk_normalization = qk_normalization
self.norm_type = norm_type
self.layer_norm_eps = layer_norm_eps
self.initializer_factor = initializer_factor
self.hidden_act = hidden_act
self.max_img_size = max_img_size
self.norm_mean = (
list(norm_mean) if isinstance(norm_mean, (tuple, list)) else norm_mean
)
self.norm_std = (
list(norm_std) if isinstance(norm_std, (tuple, list)) else norm_std
)
self.reg_tokens = reg_tokens
self.drop_path_rate = drop_path_rate
self.dropout = dropout
super().__init__(**kwargs)