diff --git a/scripts/visualgen_eval/visual_gen_lpips_score_eval.py b/scripts/visualgen_eval/visual_gen_lpips_score_eval.py index 9697655f72ea..b51cde5efee9 100644 --- a/scripts/visualgen_eval/visual_gen_lpips_score_eval.py +++ b/scripts/visualgen_eval/visual_gen_lpips_score_eval.py @@ -63,13 +63,25 @@ import time from typing import Any -import cv2 import lpips import numpy as np import torch import yaml from PIL import Image + +def _get_cv2(): + """Import OpenCV on demand for the optional cv2-backed video decode path.""" + try: + import cv2 + except ImportError as exc: + raise ImportError( + "OpenCV (cv2) is required for video LPIPS scoring but is not installed. " + "Install it with `pip install opencv-python-headless`." + ) from exc + return cv2 + + MODEL_ALIASES: dict[str, tuple[str, str]] = { "flux1": ("FLUX.1-dev", "black-forest-labs/FLUX.1-dev"), "flux1-dev": ("FLUX.1-dev", "black-forest-labs/FLUX.1-dev"), @@ -492,6 +504,7 @@ def _decode_video_to_lpips_batch( if not video_path.exists(): raise FileNotFoundError(f"Video not found for LPIPS comparison: {video_path}") + cv2 = _get_cv2() cap = cv2.VideoCapture(str(video_path)) if not cap.isOpened(): raise RuntimeError(f"Failed to open video for LPIPS comparison: {video_path}") diff --git a/tensorrt_llm/_torch/modules/fused_moe/fused_moe_cute_dsl.py b/tensorrt_llm/_torch/modules/fused_moe/fused_moe_cute_dsl.py index d2eb8a592c98..2dde62bab540 100644 --- a/tensorrt_llm/_torch/modules/fused_moe/fused_moe_cute_dsl.py +++ b/tensorrt_llm/_torch/modules/fused_moe/fused_moe_cute_dsl.py @@ -25,12 +25,7 @@ from ...autotuner import (AutoTuner, ConstraintSpec, DynamicTensorSpec, OptimizationProfile, TunableRunner, TuningConfig) -from ...custom_ops.cute_dsl_custom_ops import ( - GroupedGemmInputsHelper, - Sm100BlockScaledContiguousGatherGroupedGemmActFusionRunner, - Sm100BlockScaledContiguousGroupedGemmFinalizeFusionRunner, - Sm100BlockScaledContiguousGroupedGemmRunner, - Sm100BlockScaledContiguousGroupedGemmSwigluFusionRunner) +from ...custom_ops.cute_dsl_custom_ops import GroupedGemmInputsHelper from ...model_config import ModelConfig from ...utils import (ActivationType, AuxStreamType, EventType, Fp4QuantizedTensor, @@ -326,6 +321,19 @@ def runner_tactic_comb_checker( if tile_size is None: return True + # Imported here rather than at module scope: these runners are defined + # inside cute_dsl_custom_ops' ``if IS_CUTLASS_DSL_AVAILABLE:`` block, + # which has no else-branch, so at module scope a missing cutlass DSL + # would break every importer of this file -- and create_moe imports it + # eagerly under _torch.models, so that reaches all model startup rather + # than just this backend. Reaching this line means a CuteDSL runner is + # already being tuned, so the DSL is installed. + from ...custom_ops.cute_dsl_custom_ops import ( + Sm100BlockScaledContiguousGatherGroupedGemmActFusionRunner, + Sm100BlockScaledContiguousGroupedGemmFinalizeFusionRunner, + Sm100BlockScaledContiguousGroupedGemmRunner, + Sm100BlockScaledContiguousGroupedGemmSwigluFusionRunner) + for runner, tactic in comb: if isinstance( runner, diff --git a/tests/integration/defs/examples/visual_gen/golden/visual_gen_lpips/cosmos3_nano_t2i_lpips_golden.json b/tests/integration/defs/examples/visual_gen/golden/visual_gen_lpips/cosmos3_nano_t2i_lpips_golden.json index f0bfb8639013..419b0ed0d76b 100644 --- a/tests/integration/defs/examples/visual_gen/golden/visual_gen_lpips/cosmos3_nano_t2i_lpips_golden.json +++ b/tests/integration/defs/examples/visual_gen/golden/visual_gen_lpips/cosmos3_nano_t2i_lpips_golden.json @@ -3,6 +3,8 @@ "model": "Cosmos3-Nano", "source": "TensorRT-LLM VisualGen", "prompt": "A serene mountain landscape with snow-capped peaks and a flowing river", + "negative_prompt": "pinned pre-audio descriptive default (COSMOS3_LPIPS_NEGATIVE_PROMPT)", + "max_sequence_length": 1024, "height": 720, "width": 1280, "num_frames": 1, @@ -15,7 +17,18 @@ "lpips_net": "alex", "lpips_threshold": 0.05, "diffusers_version": "0.38.0", - "tensorrt_llm_version": "1.3.0rc20", - "tensorrt_llm_commit": "85665f5fd331d0154a78172954846d843085e83f", - "container_image": "urm.nvidia.com/sw-tensorrt-docker/tensorrt-llm-staging/release@sha256:3308a2dc0192a8329ea02eca7b5c44f290f5e894cd8c5921099308d84c3e5691" + "tensorrt_llm_version": "1.3.0rc24", + "tensorrt_llm_commit": "1745a6e689082e32e113fe06460ee5c3dfecbba9", + "container_image": "urm.nvidia.com/sw-tensorrt-docker/tensorrt-llm:pytorch-26.05-py3-x86_64-ubuntu24.04-skip-tritondevel-202607271403-16694-handongl", + "rebake_reason": [ + "Refreshed for https://nvbugs/6418815. The prior golden (commit", + "85665f5fd331d0154a78172954846d843085e83f) predates the Cosmos3 audio-output", + "feature (f50ca53dae) and so encodes the pre-feature cross-attention numerics:", + "before that feature, cross-attention padded every CFG sample to the batch-wide", + "max text length and attended over the padding. f50ca53dae added per-sample", + "text slicing (real_text_lens), which is the correct behavior and which the", + "V2V golden (baked after it) depends on, so it must not be reverted. This", + "golden was regenerated with real_text_lens in force; the run is bit-exact", + "reproducible (verified across two independent runs on the same host)." + ] } diff --git a/tests/integration/defs/examples/visual_gen/golden/visual_gen_lpips/visual_gen_lpips_golden_media.zip b/tests/integration/defs/examples/visual_gen/golden/visual_gen_lpips/visual_gen_lpips_golden_media.zip index 8b553103413c..89f9a8921939 100644 --- a/tests/integration/defs/examples/visual_gen/golden/visual_gen_lpips/visual_gen_lpips_golden_media.zip +++ b/tests/integration/defs/examples/visual_gen/golden/visual_gen_lpips/visual_gen_lpips_golden_media.zip @@ -1,3 +1,3 @@ version https://git-lfs.github.com/spec/v1 -oid sha256:7043356a539371abc61e23c275df3e8388ff4d80894bf9d7fb7845eaf764741b -size 38528317 +oid sha256:010919a1b49292f11078c704ca540445655cbbbf57b1d52a525629de504cb4a6 +size 38447980 diff --git a/tests/integration/defs/examples/visual_gen/test_visual_gen_cosmos3.py b/tests/integration/defs/examples/visual_gen/test_visual_gen_cosmos3.py index bb72a4656f9f..3af15f44c885 100644 --- a/tests/integration/defs/examples/visual_gen/test_visual_gen_cosmos3.py +++ b/tests/integration/defs/examples/visual_gen/test_visual_gen_cosmos3.py @@ -49,6 +49,22 @@ # Cosmos3 requires VANILLA attention and guardrails disabled in CI. COSMOS3_NANO_MODEL_SUBPATH = "Cosmos3-Nano" COSMOS3_LPIPS_PROMPT = "A serene mountain landscape with snow-capped peaks and a flowing river" +# The T2I/T2V goldens were baked before the Cosmos3 audio-output feature +# (f50ca53dae) changed two conditioning defaults: the descriptive negative +# prompt became "" and max_sequence_length went 1024 -> 4096. Pin the original +# values for those two tests so they stay decoupled from future public-default +# changes (matching the WAN21/22, LTX2, QwenImage pattern). The V2V golden was +# baked after that feature landed, so it deliberately keeps the live defaults. +COSMOS3_LPIPS_NEGATIVE_PROMPT = ( + "The video captures a series of frames showing ugly scenes, static with no motion, " + "motion blur, over-saturation, shaky footage, low resolution, grainy texture, " + "pixelated images, poorly lit areas, underexposed and overexposed scenes, poor " + "color balance, washed out colors, choppy sequences, jerky movements, low frame " + "rate, artifacting, color banding, unnatural transitions, outdated special effects, " + "fake elements, unconvincing visuals, poorly edited content, jump cuts, visual " + "noise, and flickering. Overall, the video is of poor quality." +) +COSMOS3_LPIPS_MAX_SEQUENCE_LENGTH = 1024 COSMOS3_LPIPS_HEIGHT = 720 COSMOS3_LPIPS_WIDTH = 1280 COSMOS3_LPIPS_T2V_NUM_FRAMES = 189 @@ -128,13 +144,22 @@ def _build_cosmos3_accuracy_cases(): COSMOS3_ACCURACY_CASES = _build_cosmos3_accuracy_cases() -def _run_cosmos3_lpips_pipeline(num_frames, video=None): +def _run_cosmos3_lpips_pipeline( + num_frames, video=None, negative_prompt="", max_sequence_length=None +): """Run the Cosmos3-Nano pipeline (default setting, VANILLA attn, compile-off). Returns the generated video tensor ``(B, T, H, W, C)`` (T == ``num_frames``), or ``None`` if generation produced no video. ``num_frames=1`` yields the single-frame text-to-image path; passing ``video`` (encoded MP4 bytes, decoded on the worker's NVDEC) yields the video-to-video path. + + ``negative_prompt`` defaults to ``""`` because the goldens were generated + against an empty uncond branch; leaving it unset would inherit the + video-mode default instead. ``max_sequence_length`` defaults to ``None``, + which leaves the pipeline's own default in force. Callers whose golden + predates the audio-output feature pass the pinned pre-audio values + explicitly. """ # Cosmos3 re-reads the guardrail flag in __init__; set it before the pipeline loads. guardrails_env_key = "TRTLLM_DISABLE_COSMOS3_GUARDRAILS" @@ -163,15 +188,14 @@ def _run_cosmos3_lpips_pipeline(num_frames, video=None): with torch.no_grad(): result = pipeline.forward( prompt=COSMOS3_LPIPS_PROMPT, - # The goldens were generated against an empty uncond branch, - # so pin it rather than inheriting the video-mode default. - negative_prompt="", + negative_prompt=negative_prompt, seed=COSMOS3_LPIPS_SEED, height=COSMOS3_LPIPS_HEIGHT, width=COSMOS3_LPIPS_WIDTH, num_frames=num_frames, num_inference_steps=COSMOS3_LPIPS_NUM_INFERENCE_STEPS, guidance_scale=COSMOS3_LPIPS_GUIDANCE_SCALE, + max_sequence_length=max_sequence_length, frame_rate=COSMOS3_LPIPS_FRAME_RATE, use_guardrails=False, video=video, @@ -191,7 +215,11 @@ def _run_cosmos3_lpips_pipeline(num_frames, video=None): def _generate_cosmos3_lpips_video(output_path): """Generate the Cosmos3-Nano text-to-video LPIPS sample.""" - video = _run_cosmos3_lpips_pipeline(COSMOS3_LPIPS_T2V_NUM_FRAMES) + video = _run_cosmos3_lpips_pipeline( + COSMOS3_LPIPS_T2V_NUM_FRAMES, + negative_prompt=COSMOS3_LPIPS_NEGATIVE_PROMPT, + max_sequence_length=COSMOS3_LPIPS_MAX_SEQUENCE_LENGTH, + ) assert video is not None, "Cosmos3-Nano T2V LPIPS run produced no video" _save_lpips_video_mp4(video, output_path, frame_rate=COSMOS3_LPIPS_FRAME_RATE) @@ -228,7 +256,11 @@ def _generate_cosmos3_lpips_image(output_path): """Generate the Cosmos3-Nano text-to-image LPIPS sample (single frame).""" from tensorrt_llm.media.encoding import save_image - video = _run_cosmos3_lpips_pipeline(COSMOS3_LPIPS_T2I_NUM_FRAMES) + video = _run_cosmos3_lpips_pipeline( + COSMOS3_LPIPS_T2I_NUM_FRAMES, + negative_prompt=COSMOS3_LPIPS_NEGATIVE_PROMPT, + max_sequence_length=COSMOS3_LPIPS_MAX_SEQUENCE_LENGTH, + ) assert video is not None, "Cosmos3-Nano T2I LPIPS run produced no frame" # video is (B, T, H, W, C); take the single frame -> (H, W, C) for save_image. save_image(video[0, 0], output_path) @@ -368,7 +400,10 @@ def test_cosmos3_nano_v2v_lpips_against_golden(_visual_gen_deps, tmp_path): @pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA not available") -def test_cosmos3_nano_t2i_lpips_against_golden(_visual_gen_deps, tmp_path): +def test_cosmos3_nano_t2i_lpips_against_golden(request, tmp_path): + # No _visual_gen_deps: this case is single-frame throughout (PIL save_image + # plus the eval script's image branch), so it needs none of that fixture's + # video codecs -- matching test_cosmos3_feature_accuracy_against_golden. generated_path = tmp_path / "cosmos3_nano_t2i_generated.png" golden_path = _golden_media_path( tmp_path, "cosmos3_nano_t2i_lpips_golden.png", "Cosmos3-Nano T2I LPIPS golden image" @@ -382,6 +417,13 @@ def test_cosmos3_nano_t2i_lpips_against_golden(_visual_gen_deps, tmp_path): golden_path, generated_path, ) + _preserve_lpips_candidate_on_failure( + request, + score, + COSMOS3_LPIPS_THRESHOLD, + generated_path, + "cosmos3_nano_t2i_lpips_golden.png", + ) _assert_lpips_below_threshold(score, COSMOS3_LPIPS_THRESHOLD) diff --git a/tests/integration/test_lists/waives.txt b/tests/integration/test_lists/waives.txt index 8acb9bc37202..fa762431f83d 100644 --- a/tests/integration/test_lists/waives.txt +++ b/tests/integration/test_lists/waives.txt @@ -106,7 +106,6 @@ examples/test_ad_speculative_decoding.py::test_autodeploy_eagle3_one_model_accep examples/test_ad_speculative_decoding.py::test_nemotron_mtp_model_with_weights SKIP (https://nvbugs/6630699) examples/test_ray.py::test_ray_disaggregated_serving[tp2] SKIP (https://nvbugs/6632606) examples/visual_gen/test_visual_gen_cosmos3.py::test_cosmos3_feature_accuracy_against_golden[nvfp4] SKIP (https://nvbugs/6572800) -examples/visual_gen/test_visual_gen_cosmos3.py::test_cosmos3_nano_t2i_lpips_against_golden SKIP (https://nvbugs/6418815) examples/visual_gen/test_visual_gen_cosmos3.py::test_cosmos3_nano_t2v_lpips_against_golden SKIP (https://nvbugs/6437341) examples/visual_gen/test_visual_gen_flux.py::test_flux_accuracy_against_golden[flux1-nvfp4] SKIP (https://nvbugs/6572800) examples/visual_gen/test_visual_gen_flux.py::test_flux_accuracy_against_golden[flux2-nvfp4] SKIP (https://nvbugs/6572800)