Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 3 additions & 1 deletion tensorrt_llm/_torch/models/modeling_minicpmv4_6.py
Original file line number Diff line number Diff line change
Expand Up @@ -549,7 +549,9 @@ def _make_attn_metadata(self, seq_lens: List[int]) -> AttentionMetadata:
max_num_tokens=sum(seq_lens) + 1,
kv_cache_manager=None,
)
return _prepare_qwen_vl_vision_attn_metadata(seq_lens, attn_metadata)
return _prepare_qwen_vl_vision_attn_metadata(
seq_lens, attn_metadata, max_seq_len=max(seq_lens)
)

@staticmethod
def _grid_seq_lens(target_sizes: torch.Tensor) -> List[int]:
Expand Down
22 changes: 22 additions & 0 deletions tests/unittest/_torch/modeling/test_modeling_minicpmv4_6.py
Original file line number Diff line number Diff line change
Expand Up @@ -26,13 +26,15 @@
with ``None.itemsize``).
* The self-contained ``MiniCPMV4_6VisionConfig`` window helpers.
* The runtime ``transformers>=5.7.0`` guard used by the input processor.
* The vision encoder's call into the shared Qwen-VL attention-metadata helper.

A single ``transformers>=5.7.0``-gated test asserts the native config is present
once the pin is bumped (at which point the local shim can be removed).
"""

import copy
import json
from types import SimpleNamespace

import pytest
import torch
Expand Down Expand Up @@ -254,6 +256,26 @@ def test_passes_on_supported_transformers(self, monkeypatch, version):
mod._ensure_transformers_supports_minicpmv4_6()


# ---------------------------------------------------------------------------
# Vision encoder attention metadata
# ---------------------------------------------------------------------------
def test_vision_attn_metadata_passes_max_seq_len(monkeypatch):
from tensorrt_llm._torch.models import modeling_minicpmv4_6 as mod

calls = []

def capture_prepare_attn_metadata(seq_lens, attn_metadata, *, max_seq_len):
calls.append((seq_lens, max_seq_len))
return attn_metadata

monkeypatch.setattr(mod, "_prepare_qwen_vl_vision_attn_metadata", capture_prepare_attn_metadata)
vision_model = SimpleNamespace(metadata_cls=lambda **kwargs: SimpleNamespace(**kwargs))

mod.MiniCPMV4_6VisionModel._make_attn_metadata(vision_model, [64, 16])

assert calls == [([64, 16], 64)]


# ---------------------------------------------------------------------------
# Native transformers config (only once the pin is bumped to >=5.7.0)
# ---------------------------------------------------------------------------
Expand Down
Loading