diff --git a/tensorrt_llm/_torch/models/_arch_index.py b/tensorrt_llm/_torch/models/_arch_index.py index f2f278dea788..d6a98a4908ca 100644 --- a/tensorrt_llm/_torch/models/_arch_index.py +++ b/tensorrt_llm/_torch/models/_arch_index.py @@ -89,6 +89,7 @@ def is_builtin_zoo_module(module_name: str) -> bool: "NemotronHPuzzleForCausalLM": "modeling_nemotron_h", "NemotronH_Nano_Omni_Reasoning_V3": "modeling_nemotron_nano", "NemotronH_Nano_VL_V2": "modeling_nemotron_nano", + "NemotronH_Omni_Reasoning_V3": "modeling_nemotron_nano", "Phi3ForCausalLM": "modeling_phi3", "Phi4MMForCausalLM": "modeling_phi4mm", "PixtralForConditionalGeneration": "modeling_mistral", @@ -215,6 +216,7 @@ def is_builtin_zoo_module(module_name: str) -> bool: "mistral3": "modeling_mistral", "mistral_common": "modeling_mistral", "mistral_large_3": "modeling_mistral", + "nemotron_h_omni": "modeling_nemotron_nano", "phi4mm": "modeling_phi4mm", "qwen2_5_vl": "modeling_qwen2vl", "qwen2_vl": "modeling_qwen2vl", diff --git a/tensorrt_llm/_torch/models/modeling_nemotron_nano.py b/tensorrt_llm/_torch/models/modeling_nemotron_nano.py index ed0808b30d2e..ac0ea865e75e 100644 --- a/tensorrt_llm/_torch/models/modeling_nemotron_nano.py +++ b/tensorrt_llm/_torch/models/modeling_nemotron_nano.py @@ -1029,8 +1029,10 @@ def __init__( self.img_context_token = self.config.img_context_token self.video_context_token = self.config.video_context_token self.video_context_token_id = self.config.video_context_token_id - self.img_start_token = self.config.img_start_token - self.img_end_token = self.config.img_end_token + # Nemotron 3.5 Super VL carries only `img_context_token`; fall back to the + # InternVL ``/`` pair its processor uses. + self.img_start_token = getattr(self.config, "img_start_token", "") + self.img_end_token = getattr(self.config, "img_end_token", "") # Pre-tokenize special tokens for video EVS processing (following vLLM). # These may be multi-token under BPE, so we store the full ID list. self._img_start_token_ids = self.tokenizer.encode( @@ -1068,8 +1070,13 @@ def __init__( "Dynamic resolution (enabled via `vision_config.min_num_patches`) only supports " f"`config.ps_version='v2'. Got {pixel_shuffle_version=}." ) + # Nemotron 3.5 Super VL states the context length only on the inner + # `llm_config`, with no top-level `max_sequence_length`. + max_model_len = getattr(config, "max_sequence_length", None) + if max_model_len is None: + max_model_len = config.llm_config.max_position_embeddings self.dynamic_tiler = DynamicResolutionImageTiler( - max_model_len=config.max_sequence_length, + max_model_len=max_model_len, patch_size=self.patch_size, downsample_ratio=self.downsample_ratio, min_num_patches=vision_args["min_num_patches"], @@ -1082,12 +1089,17 @@ def __init__( # Video temporal compression and sizing config. vision_config = getattr(config, "vision_config", config) self.video_temporal_patch_size = getattr(vision_config, "video_temporal_patch_size", 1) + # Tile-based (InternVL-style) checkpoints state the video sizing on + # `vision_config`. Patch-based ones (Nemotron 3.5 Super VL) state it on the + # image processor instead, and that processor has no `max_num_tiles`, so the + # tile fallback in `_process_videos_frames` cannot serve them. + video_sizing = vision_config if hasattr(self.processor, "max_num_tiles") else self.processor self.video_maintain_aspect_ratio = getattr( - vision_config, "video_maintain_aspect_ratio", False + video_sizing, "video_maintain_aspect_ratio", False ) # Resolve video target size: video_target_num_patches or video_target_img_size. - target_num_patches = getattr(vision_config, "video_target_num_patches", None) - target_img_size = getattr(vision_config, "video_target_img_size", None) + target_num_patches = getattr(video_sizing, "video_target_num_patches", None) + target_img_size = getattr(video_sizing, "video_target_img_size", None) if target_num_patches is not None and target_img_size is not None: raise ValueError( "Exactly one of video_target_num_patches or " @@ -2613,6 +2625,9 @@ def _resample_audios( @register_vision_encoder(NanoV2VLMultimodalEncoder) @register_auto_model("NemotronH_Nano_Omni_Reasoning_V3") @register_auto_model("NemotronH_Nano_VL_V2") +# Nemotron 3.5 Super VL: same class, no "Nano" infix, and vision-only (its +# `sound_config` is null) despite the "Omni" in the architecture name. +@register_auto_model("NemotronH_Omni_Reasoning_V3") @register_input_processor( NanoV2VLInputProcessor, model_type="NemotronH_Nano_VL_V2", @@ -2623,6 +2638,13 @@ def _resample_audios( model_type="NemotronH_Nano_Omni_Reasoning_V3", placeholder_metadata=_NANO_VL_PLACEHOLDER_METADATA, ) +# `model_type`, not architecture: the Nano checkpoints set both to the same +# string, this one does not. +@register_input_processor( + NanoV2VLInputProcessor, + model_type="nemotron_h_omni", + placeholder_metadata=_NANO_VL_PLACEHOLDER_METADATA, +) class NemotronH_Nano_VL_V2(MultimodalModelMixin, transformers.PreTrainedModel): _supports_flash_attn = True diff --git a/tensorrt_llm/_torch/models/modeling_radio.py b/tensorrt_llm/_torch/models/modeling_radio.py index 4bccf16a7841..724854bf33aa 100644 --- a/tensorrt_llm/_torch/models/modeling_radio.py +++ b/tensorrt_llm/_torch/models/modeling_radio.py @@ -1214,11 +1214,12 @@ def __init__(self, img_size = VIT_TIMM_CONFIG_BY_NAME[model_name].img_size mlp_ratio = intermediate_size / embed_dim - # Build the model. + # Build the model. C-RADIOv4-H omits `in_chans`, `input_size` and `drop` + # from the RADIO argument bag, so these reads tolerate their absence. in_chans = 3 - if args.in_chans is not None: + if getattr(args, 'in_chans', None) is not None: in_chans = args.in_chans - elif args.input_size is not None: + elif getattr(args, 'input_size', None) is not None: in_chans = args.input_size[0] vit_model = VisionTransformer( img_size=img_size, @@ -1228,7 +1229,7 @@ def __init__(self, depth=depth, num_heads=num_attention_heads, mlp_ratio=mlp_ratio, - drop_rate=args.drop, + drop_rate=getattr(args, 'drop', 0.0), special_args=args, model_config=self.model_config, ) @@ -1242,30 +1243,41 @@ def __init__(self, input_conditioner = nn.Identity() input_conditioner.dtype = config.torch_dtype - adaptor_names = config.adaptor_names or [] + # C-RADIOv4 configs omit these three fields; absent means "not configured", + # which is the only case these guards accept anyway. + adaptor_names = getattr(config, 'adaptor_names', None) or [] if len(adaptor_names) > 0: raise ValueError( "Adaptor names are not supported for RADIO models.") adaptors = dict() feature_normalizer = None - if config.feature_normalizer_config is not None: + if getattr(config, 'feature_normalizer_config', None) is not None: raise ValueError( "Feature normalizer is not supported for RADIO models.") inter_feature_normalizer = None - if config.inter_feature_normalizer_config is not None: + if getattr(config, 'inter_feature_normalizer_config', None) is not None: raise ValueError( "Intermediate feature normalizer is not supported for RADIO models." ) + # C-RADIOv4 renames both: `max_img_size` is `max_resolution`, and + # `image_size` the preferred (square) resolution. `vitdet_window_size` + # has no v4 counterpart; None is the constructor's own default. + max_resolution = (config.max_resolution if hasattr( + config, 'max_resolution') else config.max_img_size) + preferred_resolution = (config.preferred_resolution if hasattr( + config, 'preferred_resolution') else Resolution( + config.image_size, config.image_size)) + self.radio_model = RADIOVisionModelBase( vit_model, input_conditioner, patch_size=config.patch_size, - max_resolution=config.max_resolution, - window_size=config.vitdet_window_size, - preferred_resolution=config.preferred_resolution, + max_resolution=max_resolution, + window_size=getattr(config, 'vitdet_window_size', None), + preferred_resolution=preferred_resolution, adaptors=adaptors, feature_normalizer=feature_normalizer, inter_feature_normalizer=inter_feature_normalizer, diff --git a/tests/unittest/_torch/modeling/test_modeling_nemotron_nano_v2_vl.py b/tests/unittest/_torch/modeling/test_modeling_nemotron_nano_v2_vl.py index 993a52f3fdcb..6cc708b4b946 100644 --- a/tests/unittest/_torch/modeling/test_modeling_nemotron_nano_v2_vl.py +++ b/tests/unittest/_torch/modeling/test_modeling_nemotron_nano_v2_vl.py @@ -66,8 +66,12 @@ def _make_minimal_nano_model_config(): @pytest.mark.cpu_only def test_nemotron_nano_registers_native_multimodal_epd_components(): - """Native Nano VL/Omni classes advertise MM EPD support.""" - for arch in ("NemotronH_Nano_VL_V2", "NemotronH_Nano_Omni_Reasoning_V3"): + """Every arch served by the native Nano VL class advertises MM EPD support.""" + for arch in ( + "NemotronH_Nano_VL_V2", + "NemotronH_Nano_Omni_Reasoning_V3", + "NemotronH_Omni_Reasoning_V3", + ): vision_encoder_cls, vlm_base_model = MODEL_CLASS_VISION_ENCODER_MAPPING[arch] assert vision_encoder_cls is NanoV2VLMultimodalEncoder assert vlm_base_model is None