From 7e91edf233aa26cc0ecca03744d412a3fed5354c Mon Sep 17 00:00:00 2001 From: trtllm-agent <296075020+trtllm-agent@users.noreply.github.com> Date: Mon, 6 Jul 2026 10:38:04 -0700 Subject: [PATCH 1/5] [nvbugs/6418815][fix] Fix Cosmos3 T2I LPIPS regression from audio-output feature The Cosmos3 audio-output feature (commit f50ca53dae) silently changed three pipeline behaviors that the T2I LPIPS golden was baked against: 1. Refactored Cosmos3CrossAttention to per-batch slice k_und/v_und by real_text_lens when batch_size > 1. Under CFG (batch_size=2), the shorter (negative-prompt) entry now attends over only its real text length rather than the padded max_real_len slice used by every other batch entry, producing a substantially different image. 2. Replaced the long descriptive COSMOS3_DEFAULT_NEGATIVE_PROMPT with "". 3. Bumped COSMOS3_720P_PARAMS["max_sequence_length"] from 1024 to 4096. The result was LPIPS = 0.608404 on test_cosmos3_nano_t2i_lpips_against_golden (12x the 0.05 threshold). This mirrors the T2V-sibling fix on branch repair-bot-bug6410093 (commit a8cf30cef9); the same underlying regression affects both variants but T2V happens to still land under the 0.05 threshold with the new pipeline defaults while T2I diverges. Fix: - Revert the per-batch cross-attention slicing: drop the real_text_lens parameter from Cosmos3CrossAttention.forward and Cosmos3GenDecoderLayer.forward, and stop computing/passing it in Cosmos3VFMTransformer.forward. All batch entries now share the same k_und[:, :max_real_len] slice as before. The per-batch path was only preparatory for future audio work and is not exercised by any existing audio test (audio tests use batch_size=1). - Pin the LPIPS-golden-specific negative_prompt and max_sequence_length in the test itself (matching the WAN21/22, LTX2, QwenImage pattern), so the LPIPS test stays decoupled from future public-default changes. - Remove the nvbugs/6418815 waiver. Verified: LPIPS score drops to 0.000142 (well below the 0.05 threshold). Signed-off-by: trtllm-agent <296075020+trtllm-agent@users.noreply.github.com> Signed-off-by: handongl --- .../models/cosmos3/transformer_cosmos3.py | 64 +++++-------------- .../visual_gen/test_visual_gen_cosmos3.py | 44 +++++++++++-- tests/integration/test_lists/waives.txt | 1 - 3 files changed, 55 insertions(+), 54 deletions(-) diff --git a/tensorrt_llm/_torch/visual_gen/models/cosmos3/transformer_cosmos3.py b/tensorrt_llm/_torch/visual_gen/models/cosmos3/transformer_cosmos3.py index dbce5ca6ea45..bc2e3e9f3012 100644 --- a/tensorrt_llm/_torch/visual_gen/models/cosmos3/transformer_cosmos3.py +++ b/tensorrt_llm/_torch/visual_gen/models/cosmos3/transformer_cosmos3.py @@ -555,7 +555,6 @@ def forward( freqs_cos: torch.Tensor, freqs_sin: torch.Tensor, timestep=None, - real_text_lens: Optional[list[int]] = None, ) -> torch.Tensor: """ Args: @@ -579,33 +578,16 @@ def forward( q, k = self.apply_qk_norm(q, k) q, k = qwen3_apply_rotary_pos_emb(q, k, freqs_cos, freqs_sin) - if real_text_lens is not None and batch_size > 1: - outs = [] - for b in range(batch_size): - Lb = int(real_text_lens[b]) - k_all_b = torch.cat([k_und[b : b + 1, :Lb], k[b : b + 1]], dim=1) - v_all_b = torch.cat([v_und[b : b + 1, :Lb], v[b : b + 1]], dim=1) - outs.append( - self._attn_impl( - q[b : b + 1], - k_all_b, - v_all_b, - attention_mask=PredefinedAttentionMask.FULL, - timestep=timestep, - ) - ) - out = torch.cat(outs, dim=0) - else: - k_all = torch.cat([k_und, k], dim=1).contiguous() - v_all = torch.cat([v_und, v], dim=1).contiguous() - - out = self._attn_impl( - q, - k_all, - v_all, - attention_mask=PredefinedAttentionMask.FULL, - timestep=timestep, - ) + k_all = torch.cat([k_und, k], dim=1).contiguous() + v_all = torch.cat([v_und, v], dim=1).contiguous() + + out = self._attn_impl( + q, + k_all, + v_all, + attention_mask=PredefinedAttentionMask.FULL, + timestep=timestep, + ) return self.to_out[0](out) @@ -746,7 +728,6 @@ def forward( v_und: torch.Tensor, freqs: Tuple[torch.Tensor, torch.Tensor], timestep=None, - real_text_lens: Optional[list[int]] = None, ) -> torch.Tensor: residual = hidden_states hidden_states = self.input_layernorm(hidden_states) @@ -759,7 +740,6 @@ def forward( freqs_cos=cos, freqs_sin=sin, timestep=timestep, - real_text_lens=real_text_lens, ) hidden_states = residual + hidden_states @@ -1255,7 +1235,6 @@ def forward( T, H, W = video_shape Hp, Wp, _, _ = self._pad_to_patch_size(H, W) max_real_len = text_mask.sum(dim=1).max().item() - real_text_lens = text_mask.sum(dim=1).tolist() hidden_gen = self.vae2llm(self.patchify(hidden_states, T, H, W)) @@ -1352,22 +1331,13 @@ def forward( if not self.sharder.is_active: k_und = k_und[:, :max_real_len] v_und = v_und[:, :max_real_len] - hidden_gen = layer( - hidden_gen, - k_und, - v_und, - freqs_gen, - timestep=timestep, - real_text_lens=real_text_lens, - ) - else: - hidden_gen = layer( - hidden_gen, - k_und, - v_und, - freqs_gen, - timestep=timestep, - ) + hidden_gen = layer( + hidden_gen, + k_und, + v_und, + freqs_gen, + timestep=timestep, + ) hidden_gen = self.sharder.gather(hidden_gen, dim=1, unpad_to=S_gen) diff --git a/tests/integration/defs/examples/visual_gen/test_visual_gen_cosmos3.py b/tests/integration/defs/examples/visual_gen/test_visual_gen_cosmos3.py index bb72a4656f9f..3cea65eb2065 100644 --- a/tests/integration/defs/examples/visual_gen/test_visual_gen_cosmos3.py +++ b/tests/integration/defs/examples/visual_gen/test_visual_gen_cosmos3.py @@ -49,6 +49,22 @@ # Cosmos3 requires VANILLA attention and guardrails disabled in CI. COSMOS3_NANO_MODEL_SUBPATH = "Cosmos3-Nano" COSMOS3_LPIPS_PROMPT = "A serene mountain landscape with snow-capped peaks and a flowing river" +# The T2I/T2V goldens were baked before the Cosmos3 audio-output feature +# (f50ca53dae) changed two conditioning defaults: the descriptive negative +# prompt became "" and max_sequence_length went 1024 -> 4096. Pin the original +# values for those two tests so they stay decoupled from future public-default +# changes (matching the WAN21/22, LTX2, QwenImage pattern). The V2V golden was +# baked after that feature landed, so it deliberately keeps the live defaults. +COSMOS3_LPIPS_NEGATIVE_PROMPT = ( + "The video captures a series of frames showing ugly scenes, static with no motion, " + "motion blur, over-saturation, shaky footage, low resolution, grainy texture, " + "pixelated images, poorly lit areas, underexposed and overexposed scenes, poor " + "color balance, washed out colors, choppy sequences, jerky movements, low frame " + "rate, artifacting, color banding, unnatural transitions, outdated special effects, " + "fake elements, unconvincing visuals, poorly edited content, jump cuts, visual " + "noise, and flickering. Overall, the video is of poor quality." +) +COSMOS3_LPIPS_MAX_SEQUENCE_LENGTH = 1024 COSMOS3_LPIPS_HEIGHT = 720 COSMOS3_LPIPS_WIDTH = 1280 COSMOS3_LPIPS_T2V_NUM_FRAMES = 189 @@ -128,13 +144,22 @@ def _build_cosmos3_accuracy_cases(): COSMOS3_ACCURACY_CASES = _build_cosmos3_accuracy_cases() -def _run_cosmos3_lpips_pipeline(num_frames, video=None): +def _run_cosmos3_lpips_pipeline( + num_frames, video=None, negative_prompt="", max_sequence_length=None +): """Run the Cosmos3-Nano pipeline (default setting, VANILLA attn, compile-off). Returns the generated video tensor ``(B, T, H, W, C)`` (T == ``num_frames``), or ``None`` if generation produced no video. ``num_frames=1`` yields the single-frame text-to-image path; passing ``video`` (encoded MP4 bytes, decoded on the worker's NVDEC) yields the video-to-video path. + + ``negative_prompt`` defaults to ``""`` because the goldens were generated + against an empty uncond branch; leaving it unset would inherit the + video-mode default instead. ``max_sequence_length`` defaults to ``None``, + which leaves the pipeline's own default in force. Callers whose golden + predates the audio-output feature pass the pinned pre-audio values + explicitly. """ # Cosmos3 re-reads the guardrail flag in __init__; set it before the pipeline loads. guardrails_env_key = "TRTLLM_DISABLE_COSMOS3_GUARDRAILS" @@ -163,15 +188,14 @@ def _run_cosmos3_lpips_pipeline(num_frames, video=None): with torch.no_grad(): result = pipeline.forward( prompt=COSMOS3_LPIPS_PROMPT, - # The goldens were generated against an empty uncond branch, - # so pin it rather than inheriting the video-mode default. - negative_prompt="", + negative_prompt=negative_prompt, seed=COSMOS3_LPIPS_SEED, height=COSMOS3_LPIPS_HEIGHT, width=COSMOS3_LPIPS_WIDTH, num_frames=num_frames, num_inference_steps=COSMOS3_LPIPS_NUM_INFERENCE_STEPS, guidance_scale=COSMOS3_LPIPS_GUIDANCE_SCALE, + max_sequence_length=max_sequence_length, frame_rate=COSMOS3_LPIPS_FRAME_RATE, use_guardrails=False, video=video, @@ -191,7 +215,11 @@ def _run_cosmos3_lpips_pipeline(num_frames, video=None): def _generate_cosmos3_lpips_video(output_path): """Generate the Cosmos3-Nano text-to-video LPIPS sample.""" - video = _run_cosmos3_lpips_pipeline(COSMOS3_LPIPS_T2V_NUM_FRAMES) + video = _run_cosmos3_lpips_pipeline( + COSMOS3_LPIPS_T2V_NUM_FRAMES, + negative_prompt=COSMOS3_LPIPS_NEGATIVE_PROMPT, + max_sequence_length=COSMOS3_LPIPS_MAX_SEQUENCE_LENGTH, + ) assert video is not None, "Cosmos3-Nano T2V LPIPS run produced no video" _save_lpips_video_mp4(video, output_path, frame_rate=COSMOS3_LPIPS_FRAME_RATE) @@ -228,7 +256,11 @@ def _generate_cosmos3_lpips_image(output_path): """Generate the Cosmos3-Nano text-to-image LPIPS sample (single frame).""" from tensorrt_llm.media.encoding import save_image - video = _run_cosmos3_lpips_pipeline(COSMOS3_LPIPS_T2I_NUM_FRAMES) + video = _run_cosmos3_lpips_pipeline( + COSMOS3_LPIPS_T2I_NUM_FRAMES, + negative_prompt=COSMOS3_LPIPS_NEGATIVE_PROMPT, + max_sequence_length=COSMOS3_LPIPS_MAX_SEQUENCE_LENGTH, + ) assert video is not None, "Cosmos3-Nano T2I LPIPS run produced no frame" # video is (B, T, H, W, C); take the single frame -> (H, W, C) for save_image. save_image(video[0, 0], output_path) diff --git a/tests/integration/test_lists/waives.txt b/tests/integration/test_lists/waives.txt index 8acb9bc37202..fa762431f83d 100644 --- a/tests/integration/test_lists/waives.txt +++ b/tests/integration/test_lists/waives.txt @@ -106,7 +106,6 @@ examples/test_ad_speculative_decoding.py::test_autodeploy_eagle3_one_model_accep examples/test_ad_speculative_decoding.py::test_nemotron_mtp_model_with_weights SKIP (https://nvbugs/6630699) examples/test_ray.py::test_ray_disaggregated_serving[tp2] SKIP (https://nvbugs/6632606) examples/visual_gen/test_visual_gen_cosmos3.py::test_cosmos3_feature_accuracy_against_golden[nvfp4] SKIP (https://nvbugs/6572800) -examples/visual_gen/test_visual_gen_cosmos3.py::test_cosmos3_nano_t2i_lpips_against_golden SKIP (https://nvbugs/6418815) examples/visual_gen/test_visual_gen_cosmos3.py::test_cosmos3_nano_t2v_lpips_against_golden SKIP (https://nvbugs/6437341) examples/visual_gen/test_visual_gen_flux.py::test_flux_accuracy_against_golden[flux1-nvfp4] SKIP (https://nvbugs/6572800) examples/visual_gen/test_visual_gen_flux.py::test_flux_accuracy_against_golden[flux2-nvfp4] SKIP (https://nvbugs/6572800) From e3756114ef2323f720238a058e364735b198fe02 Mon Sep 17 00:00:00 2001 From: trtllm-agent <296075020+trtllm-agent@users.noreply.github.com> Date: Sat, 11 Jul 2026 03:12:43 -0700 Subject: [PATCH 2/5] [nvbugs/6418815][fix] Lazy-import cv2 in LPIPS eval script for image-only tests Signed-off-by: trtllm-agent <296075020+trtllm-agent@users.noreply.github.com> Signed-off-by: handongl --- .../visualgen_eval/visual_gen_lpips_score_eval.py | 15 ++++++++++++++- 1 file changed, 14 insertions(+), 1 deletion(-) diff --git a/scripts/visualgen_eval/visual_gen_lpips_score_eval.py b/scripts/visualgen_eval/visual_gen_lpips_score_eval.py index 9697655f72ea..b51cde5efee9 100644 --- a/scripts/visualgen_eval/visual_gen_lpips_score_eval.py +++ b/scripts/visualgen_eval/visual_gen_lpips_score_eval.py @@ -63,13 +63,25 @@ import time from typing import Any -import cv2 import lpips import numpy as np import torch import yaml from PIL import Image + +def _get_cv2(): + """Import OpenCV on demand for the optional cv2-backed video decode path.""" + try: + import cv2 + except ImportError as exc: + raise ImportError( + "OpenCV (cv2) is required for video LPIPS scoring but is not installed. " + "Install it with `pip install opencv-python-headless`." + ) from exc + return cv2 + + MODEL_ALIASES: dict[str, tuple[str, str]] = { "flux1": ("FLUX.1-dev", "black-forest-labs/FLUX.1-dev"), "flux1-dev": ("FLUX.1-dev", "black-forest-labs/FLUX.1-dev"), @@ -492,6 +504,7 @@ def _decode_video_to_lpips_batch( if not video_path.exists(): raise FileNotFoundError(f"Video not found for LPIPS comparison: {video_path}") + cv2 = _get_cv2() cap = cv2.VideoCapture(str(video_path)) if not cap.isOpened(): raise RuntimeError(f"Failed to open video for LPIPS comparison: {video_path}") From 621a39a9bb52c1e54e476a0c9a8f15f9bf66777a Mon Sep 17 00:00:00 2001 From: trtllm-agent <296075020+trtllm-agent@users.noreply.github.com> Date: Thu, 6 Aug 2026 08:35:09 -0700 Subject: [PATCH 3/5] [nvbugs/6418815][fix] Rebake stale Cosmos3 T2I LPIPS golden, restore real_text_lens The T2I golden was baked at 85665f5fd3, before the Cosmos3 audio-output feature (f50ca53dae), so it encoded two things that later changed: 1. Two conditioning defaults: the descriptive negative prompt became "" and max_sequence_length went 1024 -> 4096. Pinning the original values in the test (as WAN21/22, LTX2 and QwenImage already do) takes LPIPS from 0.608 to 0.154. 2. The pre-feature cross-attention numerics. f50ca53dae added per-sample text slicing (real_text_lens); before it, cross-attention padded every CFG sample to the batch-wide max text length and attended over the padding. Forcing the old padded path makes T2I pass, which confirms this is the remaining 0.154 -- but it also drops the V2V test (whose golden was baked after the feature) to 0.380, so the slicing is correct behavior and must not be reverted. Restore real_text_lens to its upstream form and refresh the T2I golden instead. The regenerated image is bit-exact across two independent runs. Both T2I and V2V now pass, and the waiver is removed. Signed-off-by: trtllm-agent <296075020+trtllm-agent@users.noreply.github.com> --- .../models/cosmos3/transformer_cosmos3.py | 64 ++++++++++++++----- .../cosmos3_nano_t2i_lpips_golden.json | 19 +++++- .../visual_gen_lpips_golden_media.zip | 4 +- .../visual_gen/test_visual_gen_cosmos3.py | 9 ++- 4 files changed, 73 insertions(+), 23 deletions(-) diff --git a/tensorrt_llm/_torch/visual_gen/models/cosmos3/transformer_cosmos3.py b/tensorrt_llm/_torch/visual_gen/models/cosmos3/transformer_cosmos3.py index bc2e3e9f3012..dbce5ca6ea45 100644 --- a/tensorrt_llm/_torch/visual_gen/models/cosmos3/transformer_cosmos3.py +++ b/tensorrt_llm/_torch/visual_gen/models/cosmos3/transformer_cosmos3.py @@ -555,6 +555,7 @@ def forward( freqs_cos: torch.Tensor, freqs_sin: torch.Tensor, timestep=None, + real_text_lens: Optional[list[int]] = None, ) -> torch.Tensor: """ Args: @@ -578,16 +579,33 @@ def forward( q, k = self.apply_qk_norm(q, k) q, k = qwen3_apply_rotary_pos_emb(q, k, freqs_cos, freqs_sin) - k_all = torch.cat([k_und, k], dim=1).contiguous() - v_all = torch.cat([v_und, v], dim=1).contiguous() - - out = self._attn_impl( - q, - k_all, - v_all, - attention_mask=PredefinedAttentionMask.FULL, - timestep=timestep, - ) + if real_text_lens is not None and batch_size > 1: + outs = [] + for b in range(batch_size): + Lb = int(real_text_lens[b]) + k_all_b = torch.cat([k_und[b : b + 1, :Lb], k[b : b + 1]], dim=1) + v_all_b = torch.cat([v_und[b : b + 1, :Lb], v[b : b + 1]], dim=1) + outs.append( + self._attn_impl( + q[b : b + 1], + k_all_b, + v_all_b, + attention_mask=PredefinedAttentionMask.FULL, + timestep=timestep, + ) + ) + out = torch.cat(outs, dim=0) + else: + k_all = torch.cat([k_und, k], dim=1).contiguous() + v_all = torch.cat([v_und, v], dim=1).contiguous() + + out = self._attn_impl( + q, + k_all, + v_all, + attention_mask=PredefinedAttentionMask.FULL, + timestep=timestep, + ) return self.to_out[0](out) @@ -728,6 +746,7 @@ def forward( v_und: torch.Tensor, freqs: Tuple[torch.Tensor, torch.Tensor], timestep=None, + real_text_lens: Optional[list[int]] = None, ) -> torch.Tensor: residual = hidden_states hidden_states = self.input_layernorm(hidden_states) @@ -740,6 +759,7 @@ def forward( freqs_cos=cos, freqs_sin=sin, timestep=timestep, + real_text_lens=real_text_lens, ) hidden_states = residual + hidden_states @@ -1235,6 +1255,7 @@ def forward( T, H, W = video_shape Hp, Wp, _, _ = self._pad_to_patch_size(H, W) max_real_len = text_mask.sum(dim=1).max().item() + real_text_lens = text_mask.sum(dim=1).tolist() hidden_gen = self.vae2llm(self.patchify(hidden_states, T, H, W)) @@ -1331,13 +1352,22 @@ def forward( if not self.sharder.is_active: k_und = k_und[:, :max_real_len] v_und = v_und[:, :max_real_len] - hidden_gen = layer( - hidden_gen, - k_und, - v_und, - freqs_gen, - timestep=timestep, - ) + hidden_gen = layer( + hidden_gen, + k_und, + v_und, + freqs_gen, + timestep=timestep, + real_text_lens=real_text_lens, + ) + else: + hidden_gen = layer( + hidden_gen, + k_und, + v_und, + freqs_gen, + timestep=timestep, + ) hidden_gen = self.sharder.gather(hidden_gen, dim=1, unpad_to=S_gen) diff --git a/tests/integration/defs/examples/visual_gen/golden/visual_gen_lpips/cosmos3_nano_t2i_lpips_golden.json b/tests/integration/defs/examples/visual_gen/golden/visual_gen_lpips/cosmos3_nano_t2i_lpips_golden.json index f0bfb8639013..419b0ed0d76b 100644 --- a/tests/integration/defs/examples/visual_gen/golden/visual_gen_lpips/cosmos3_nano_t2i_lpips_golden.json +++ b/tests/integration/defs/examples/visual_gen/golden/visual_gen_lpips/cosmos3_nano_t2i_lpips_golden.json @@ -3,6 +3,8 @@ "model": "Cosmos3-Nano", "source": "TensorRT-LLM VisualGen", "prompt": "A serene mountain landscape with snow-capped peaks and a flowing river", + "negative_prompt": "pinned pre-audio descriptive default (COSMOS3_LPIPS_NEGATIVE_PROMPT)", + "max_sequence_length": 1024, "height": 720, "width": 1280, "num_frames": 1, @@ -15,7 +17,18 @@ "lpips_net": "alex", "lpips_threshold": 0.05, "diffusers_version": "0.38.0", - "tensorrt_llm_version": "1.3.0rc20", - "tensorrt_llm_commit": "85665f5fd331d0154a78172954846d843085e83f", - "container_image": "urm.nvidia.com/sw-tensorrt-docker/tensorrt-llm-staging/release@sha256:3308a2dc0192a8329ea02eca7b5c44f290f5e894cd8c5921099308d84c3e5691" + "tensorrt_llm_version": "1.3.0rc24", + "tensorrt_llm_commit": "1745a6e689082e32e113fe06460ee5c3dfecbba9", + "container_image": "urm.nvidia.com/sw-tensorrt-docker/tensorrt-llm:pytorch-26.05-py3-x86_64-ubuntu24.04-skip-tritondevel-202607271403-16694-handongl", + "rebake_reason": [ + "Refreshed for https://nvbugs/6418815. The prior golden (commit", + "85665f5fd331d0154a78172954846d843085e83f) predates the Cosmos3 audio-output", + "feature (f50ca53dae) and so encodes the pre-feature cross-attention numerics:", + "before that feature, cross-attention padded every CFG sample to the batch-wide", + "max text length and attended over the padding. f50ca53dae added per-sample", + "text slicing (real_text_lens), which is the correct behavior and which the", + "V2V golden (baked after it) depends on, so it must not be reverted. This", + "golden was regenerated with real_text_lens in force; the run is bit-exact", + "reproducible (verified across two independent runs on the same host)." + ] } diff --git a/tests/integration/defs/examples/visual_gen/golden/visual_gen_lpips/visual_gen_lpips_golden_media.zip b/tests/integration/defs/examples/visual_gen/golden/visual_gen_lpips/visual_gen_lpips_golden_media.zip index 8b553103413c..89f9a8921939 100644 --- a/tests/integration/defs/examples/visual_gen/golden/visual_gen_lpips/visual_gen_lpips_golden_media.zip +++ b/tests/integration/defs/examples/visual_gen/golden/visual_gen_lpips/visual_gen_lpips_golden_media.zip @@ -1,3 +1,3 @@ version https://git-lfs.github.com/spec/v1 -oid sha256:7043356a539371abc61e23c275df3e8388ff4d80894bf9d7fb7845eaf764741b -size 38528317 +oid sha256:010919a1b49292f11078c704ca540445655cbbbf57b1d52a525629de504cb4a6 +size 38447980 diff --git a/tests/integration/defs/examples/visual_gen/test_visual_gen_cosmos3.py b/tests/integration/defs/examples/visual_gen/test_visual_gen_cosmos3.py index 3cea65eb2065..3094b8cd19df 100644 --- a/tests/integration/defs/examples/visual_gen/test_visual_gen_cosmos3.py +++ b/tests/integration/defs/examples/visual_gen/test_visual_gen_cosmos3.py @@ -400,7 +400,7 @@ def test_cosmos3_nano_v2v_lpips_against_golden(_visual_gen_deps, tmp_path): @pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA not available") -def test_cosmos3_nano_t2i_lpips_against_golden(_visual_gen_deps, tmp_path): +def test_cosmos3_nano_t2i_lpips_against_golden(_visual_gen_deps, request, tmp_path): generated_path = tmp_path / "cosmos3_nano_t2i_generated.png" golden_path = _golden_media_path( tmp_path, "cosmos3_nano_t2i_lpips_golden.png", "Cosmos3-Nano T2I LPIPS golden image" @@ -414,6 +414,13 @@ def test_cosmos3_nano_t2i_lpips_against_golden(_visual_gen_deps, tmp_path): golden_path, generated_path, ) + _preserve_lpips_candidate_on_failure( + request, + score, + COSMOS3_LPIPS_THRESHOLD, + generated_path, + "cosmos3_nano_t2i_lpips_golden.png", + ) _assert_lpips_below_threshold(score, COSMOS3_LPIPS_THRESHOLD) From ea9380f460e4e2b839dc70016d8e180a8bd0068a Mon Sep 17 00:00:00 2001 From: trtllm-agent <296075020+trtllm-agent@users.noreply.github.com> Date: Fri, 7 Aug 2026 21:03:53 -0700 Subject: [PATCH 4/5] [nvbugs/6418815][fix] Drop video-codec fixture from image-only Cosmos3 T2I LPIPS test The T2I LPIPS case is single-frame end to end: generation saves one PNG via PIL save_image, and _run_lpips_eval scores it through the eval script's image branch (PIL + lpips). It requested _visual_gen_deps anyway, which provisions video codecs with apt-get update/install ffmpeg. apt-get cannot succeed in the non-root test container, so the fixture raised CalledProcessError (exit 100) and errored the test during setup, before the LPIPS comparison could run. Drop the fixture from this case, matching the sibling image-only test test_cosmos3_feature_accuracy_against_golden, which already omits it. The video and V2V cases keep the fixture since they do decode/encode MP4. Signed-off-by: trtllm-agent <296075020+trtllm-agent@users.noreply.github.com> --- .../defs/examples/visual_gen/test_visual_gen_cosmos3.py | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/tests/integration/defs/examples/visual_gen/test_visual_gen_cosmos3.py b/tests/integration/defs/examples/visual_gen/test_visual_gen_cosmos3.py index 3094b8cd19df..3af15f44c885 100644 --- a/tests/integration/defs/examples/visual_gen/test_visual_gen_cosmos3.py +++ b/tests/integration/defs/examples/visual_gen/test_visual_gen_cosmos3.py @@ -400,7 +400,10 @@ def test_cosmos3_nano_v2v_lpips_against_golden(_visual_gen_deps, tmp_path): @pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA not available") -def test_cosmos3_nano_t2i_lpips_against_golden(_visual_gen_deps, request, tmp_path): +def test_cosmos3_nano_t2i_lpips_against_golden(request, tmp_path): + # No _visual_gen_deps: this case is single-frame throughout (PIL save_image + # plus the eval script's image branch), so it needs none of that fixture's + # video codecs -- matching test_cosmos3_feature_accuracy_against_golden. generated_path = tmp_path / "cosmos3_nano_t2i_generated.png" golden_path = _golden_media_path( tmp_path, "cosmos3_nano_t2i_lpips_golden.png", "Cosmos3-Nano T2I LPIPS golden image" From 5ac5bbad747b32952f6865ce39050b69b5460745 Mon Sep 17 00:00:00 2001 From: trtllm-agent <296075020+trtllm-agent@users.noreply.github.com> Date: Thu, 20 Aug 2026 18:38:32 -0700 Subject: [PATCH 5/5] [nvbugs/6418815][fix] Lazy-import optional CuteDSL runners in the MoE backend The T2I LPIPS test could not reach its own assertion. fused_moe_cute_dsl.py imported four Sm100BlockScaledContiguous*Runner names at module scope, but cute_dsl_custom_ops defines them inside an ``if IS_CUTLASS_DSL_AVAILABLE:`` block with no else-branch, so without the optional cutlass DSL those names simply do not exist. create_moe imports this module eagerly and it sits under _torch.models, so the ImportError propagated through modeling_utils and took down the whole model-architecture registry -- including the visual_gen PipelineLoader the test loads. The run died at import in 1.6s, long before any image was generated or compared. The four names are needed at exactly one place, the isinstance check in runner_tactic_comb_checker, and reaching that line means a CuteDSL runner is already being autotuned, so the DSL is necessarily installed. Import them there instead of at module scope, matching how cute_dsl_mla.py and dsa/metadata.py already reach into this guard from modules that must stay importable. No module-scope name is added and the isinstance tuple is unchanged, so with the DSL present the tactic constraint behaves exactly as before; without it those symbols never existed, so nothing that used to work is disabled. Removes this bug's waiver. Signed-off-by: trtllm-agent <296075020+trtllm-agent@users.noreply.github.com> --- .../modules/fused_moe/fused_moe_cute_dsl.py | 20 +++++++++++++------ 1 file changed, 14 insertions(+), 6 deletions(-) diff --git a/tensorrt_llm/_torch/modules/fused_moe/fused_moe_cute_dsl.py b/tensorrt_llm/_torch/modules/fused_moe/fused_moe_cute_dsl.py index d2eb8a592c98..2dde62bab540 100644 --- a/tensorrt_llm/_torch/modules/fused_moe/fused_moe_cute_dsl.py +++ b/tensorrt_llm/_torch/modules/fused_moe/fused_moe_cute_dsl.py @@ -25,12 +25,7 @@ from ...autotuner import (AutoTuner, ConstraintSpec, DynamicTensorSpec, OptimizationProfile, TunableRunner, TuningConfig) -from ...custom_ops.cute_dsl_custom_ops import ( - GroupedGemmInputsHelper, - Sm100BlockScaledContiguousGatherGroupedGemmActFusionRunner, - Sm100BlockScaledContiguousGroupedGemmFinalizeFusionRunner, - Sm100BlockScaledContiguousGroupedGemmRunner, - Sm100BlockScaledContiguousGroupedGemmSwigluFusionRunner) +from ...custom_ops.cute_dsl_custom_ops import GroupedGemmInputsHelper from ...model_config import ModelConfig from ...utils import (ActivationType, AuxStreamType, EventType, Fp4QuantizedTensor, @@ -326,6 +321,19 @@ def runner_tactic_comb_checker( if tile_size is None: return True + # Imported here rather than at module scope: these runners are defined + # inside cute_dsl_custom_ops' ``if IS_CUTLASS_DSL_AVAILABLE:`` block, + # which has no else-branch, so at module scope a missing cutlass DSL + # would break every importer of this file -- and create_moe imports it + # eagerly under _torch.models, so that reaches all model startup rather + # than just this backend. Reaching this line means a CuteDSL runner is + # already being tuned, so the DSL is installed. + from ...custom_ops.cute_dsl_custom_ops import ( + Sm100BlockScaledContiguousGatherGroupedGemmActFusionRunner, + Sm100BlockScaledContiguousGroupedGemmFinalizeFusionRunner, + Sm100BlockScaledContiguousGroupedGemmRunner, + Sm100BlockScaledContiguousGroupedGemmSwigluFusionRunner) + for runner, tactic in comb: if isinstance( runner,