From 2d49523062341cdab210b5246b4722086587685a Mon Sep 17 00:00:00 2001 From: David-Wu1119 <133224895+David-Wu1119@users.noreply.github.com> Date: Sat, 26 Sep 2026 16:50:19 -0400 Subject: [PATCH] [None][fix] Pass max_seq_len to the Qwen-VL attention helper in the MiniCPM-V 4.6 vision encoder MiniCPMV4_6VisionModel._make_attn_metadata calls the shared _prepare_qwen_vl_vision_attn_metadata helper. #16568 made max_seq_len a required keyword-only argument of that helper and updated the Qwen2-VL and Qwen3-VL callers, but not this one, so every MiniCPM-V 4.6 image or video request raised TypeError in the vision encoder. Pass max(seq_lens), which is the value the helper used before #16568. The metadata here is built per call, so this keeps the vision encoder's behavior as it was when MiniCPM-V 4.6 support landed. The new unit test replaces the helper with one that has the same signature and checks the call. Co-Authored-By: Claude Opus 5.5 Signed-off-by: David-Wu1119 <133224895+David-Wu1119@users.noreply.github.com> --- .../_torch/models/modeling_minicpmv4_6.py | 4 +++- .../modeling/test_modeling_minicpmv4_6.py | 22 +++++++++++++++++++ 2 files changed, 25 insertions(+), 1 deletion(-) diff --git a/tensorrt_llm/_torch/models/modeling_minicpmv4_6.py b/tensorrt_llm/_torch/models/modeling_minicpmv4_6.py index 223262940a30..be73da2a8504 100644 --- a/tensorrt_llm/_torch/models/modeling_minicpmv4_6.py +++ b/tensorrt_llm/_torch/models/modeling_minicpmv4_6.py @@ -549,7 +549,9 @@ def _make_attn_metadata(self, seq_lens: List[int]) -> AttentionMetadata: max_num_tokens=sum(seq_lens) + 1, kv_cache_manager=None, ) - return _prepare_qwen_vl_vision_attn_metadata(seq_lens, attn_metadata) + return _prepare_qwen_vl_vision_attn_metadata( + seq_lens, attn_metadata, max_seq_len=max(seq_lens) + ) @staticmethod def _grid_seq_lens(target_sizes: torch.Tensor) -> List[int]: diff --git a/tests/unittest/_torch/modeling/test_modeling_minicpmv4_6.py b/tests/unittest/_torch/modeling/test_modeling_minicpmv4_6.py index ab41a41fb8e8..f264bf74d2b7 100644 --- a/tests/unittest/_torch/modeling/test_modeling_minicpmv4_6.py +++ b/tests/unittest/_torch/modeling/test_modeling_minicpmv4_6.py @@ -26,6 +26,7 @@ with ``None.itemsize``). * The self-contained ``MiniCPMV4_6VisionConfig`` window helpers. * The runtime ``transformers>=5.7.0`` guard used by the input processor. +* The vision encoder's call into the shared Qwen-VL attention-metadata helper. A single ``transformers>=5.7.0``-gated test asserts the native config is present once the pin is bumped (at which point the local shim can be removed). @@ -33,6 +34,7 @@ import copy import json +from types import SimpleNamespace import pytest import torch @@ -254,6 +256,26 @@ def test_passes_on_supported_transformers(self, monkeypatch, version): mod._ensure_transformers_supports_minicpmv4_6() +# --------------------------------------------------------------------------- +# Vision encoder attention metadata +# --------------------------------------------------------------------------- +def test_vision_attn_metadata_passes_max_seq_len(monkeypatch): + from tensorrt_llm._torch.models import modeling_minicpmv4_6 as mod + + calls = [] + + def capture_prepare_attn_metadata(seq_lens, attn_metadata, *, max_seq_len): + calls.append((seq_lens, max_seq_len)) + return attn_metadata + + monkeypatch.setattr(mod, "_prepare_qwen_vl_vision_attn_metadata", capture_prepare_attn_metadata) + vision_model = SimpleNamespace(metadata_cls=lambda **kwargs: SimpleNamespace(**kwargs)) + + mod.MiniCPMV4_6VisionModel._make_attn_metadata(vision_model, [64, 16]) + + assert calls == [([64, 16], 64)] + + # --------------------------------------------------------------------------- # Native transformers config (only once the pin is bumped to >=5.7.0) # ---------------------------------------------------------------------------