diff --git a/tests/integration/test_lists/waives.txt b/tests/integration/test_lists/waives.txt index d4f2d2e24421..982b8dbad0b6 100644 --- a/tests/integration/test_lists/waives.txt +++ b/tests/integration/test_lists/waives.txt @@ -170,6 +170,7 @@ full:GB300/accuracy/test_llm_api_pytorch_multimodal.py::TestExaone4_5_33B::test_ full:GB300/disaggregated/test_disaggregated.py::test_disaggregated_deepseek_v3_lite_fp8_attention_dp_overlap[DeepSeek-V3-Lite-fp8] SKIP (https://nvbugs/6581064) full:GB300/disaggregated/test_disaggregated.py::test_disaggregated_logprobs_serving[llama-3.1-8b-instruct] SKIP (https://nvbugs/6275959) full:GB300/unittest/_torch/modeling/test_modeling_gpt_oss.py::test_gpt_oss_trtllmgen[CUTLASS] SKIP (https://nvbugs/6633932) +full:GB300/unittest/_torch/modules/tests_lora_modules/test_qwen3_sanity.py::TestQwen3LoRA::test_qwen3_fp8_lora SKIP (https://nvbugs/6668777) full:H100/accuracy/test_disaggregated_serving.py::TestDeepSeekV3Lite::test_auto_dtype[mtp_nextn=2-overlap_scheduler=False] SKIP (https://nvbugs/6313072) full:H100/accuracy/test_disaggregated_serving.py::TestDeepSeekV3Lite::test_auto_dtype[mtp_nextn=2-overlap_scheduler=True] SKIP (https://nvbugs/6313072) full:H100/accuracy/test_llm_api_pytorch.py::TestGPTOSS::test_dummy_load_format SKIP (https://nvbugs/6626640) @@ -266,7 +267,6 @@ unittest/_torch/modeling/test_modeling_nemotron_h.py::test_nemotron_h_breakable_ unittest/_torch/modeling/test_modeling_nemotron_nano_v2_vl.py::test_nemotron_nano_v2_vl_image_batch_equivalence SKIP (https://nvbugs/6758946) unittest/_torch/modeling/test_modeling_nemotron_nano_v2_vl.py::test_nemotron_nano_v2_vl_video_batch_equivalence SKIP (https://nvbugs/6625695) unittest/_torch/modules/tests_lora_modules/test_nemotron_h_lora_sanity.py::TestNemotronHLoRA::test_lora_pp2_sanity SKIP (https://nvbugs/6428124) -unittest/_torch/modules/tests_lora_modules/test_qwen3_sanity.py::TestQwen3LoRA::test_qwen3_fp8_lora SKIP (https://nvbugs/6668777) unittest/_torch/moe/test_kimi_k3_situ_moe.py::test_kimi_k3_trtllm_accepts_nvfp4_routed_experts SKIP (https://nvbugs/6765038) unittest/_torch/moe/test_moe_backend.py::test_moe_backend[act=Relu2-e60_k4_h2048_i1408-seq=8-dtype=torch.bfloat16-backend=TRTLLM-quant=NVFP4-routing=Renormalize] SKIP (https://nvbugs/5989912) unittest/_torch/multi_gpu/test_linear.py::test_row_linear_norm_fusion[2-hidden:16-seqlen:2] SKIP (https://nvbugs/6501404) diff --git a/tests/unittest/_torch/modules/tests_lora_modules/test_qwen3_sanity.py b/tests/unittest/_torch/modules/tests_lora_modules/test_qwen3_sanity.py index cc3ecb3e4e81..4f5b2d95f416 100644 --- a/tests/unittest/_torch/modules/tests_lora_modules/test_qwen3_sanity.py +++ b/tests/unittest/_torch/modules/tests_lora_modules/test_qwen3_sanity.py @@ -31,6 +31,7 @@ from tensorrt_llm._torch.peft.lora.config import LoraConfig from tensorrt_llm._torch.peft.lora.layer import LoraLayer, LoraModuleType from tensorrt_llm.executor.request import LoRARequest +from tensorrt_llm.llmapi import KvCacheConfig # HF module name -> block path relative to layers.{idx}. # Attention targets work on all architectures. MLP targets only apply to @@ -110,12 +111,15 @@ def _create_lora_adapter( return output_dir -def _run_with_and_without_lora(model_path, lora_config, lora_dir, prompts): +def _run_with_and_without_lora( + model_path, lora_config, lora_dir, prompts, kv_cache_config: KvCacheConfig | None = None +): """Run inference with and without LoRA, return (lora_outputs, base_outputs).""" with LLM( model=model_path, backend="pytorch", lora_config=lora_config, + kv_cache_config=kv_cache_config or KvCacheConfig(), tensor_parallel_size=1, max_batch_size=4, max_num_tokens=256, @@ -177,6 +181,7 @@ def _run_lora_test( dtype=torch.bfloat16, overlap=False, specialize_cuda_graph=False, + kv_cache_config: KvCacheConfig | None = None, ): """End-to-end helper: create adapter, run inference, assert output differs.""" with tempfile.TemporaryDirectory() as tmpdir: @@ -200,11 +205,14 @@ def _run_lora_test( lora_config, lora_dir, ["The capital of France is", "Hello, how are you"], + kv_cache_config=kv_cache_config, ) _assert_lora_changes_output(out_lora, out_base) -def _run_mixed_lora_cuda_graph_test(model_path, target_modules, trtllm_modules): +def _run_mixed_lora_cuda_graph_test( + model_path, target_modules, trtllm_modules, kv_cache_config: KvCacheConfig +): """Verify base rows remain unchanged when sharing a LoRA-specialized graph.""" with tempfile.TemporaryDirectory() as tmpdir: lora_dir = _create_lora_adapter( @@ -223,11 +231,14 @@ def _run_mixed_lora_cuda_graph_test(model_path, target_modules, trtllm_modules): model=model_path, backend="pytorch", lora_config=lora_config, + kv_cache_config=kv_cache_config, tensor_parallel_size=1, max_batch_size=4, max_num_tokens=256, ) as llm: - sampling = SamplingParams(max_tokens=20, temperature=0.0, logprobs=0) + # Prevent adapter EOS from shrinking the mixed batch and changing + # BF16 GEMM rounding relative to the all-base reference. + sampling = SamplingParams(max_tokens=20, temperature=0.0, logprobs=0, ignore_eos=True) prompts = [ "The capital of France is", "The capital of France is", @@ -242,6 +253,11 @@ def _run_mixed_lora_cuda_graph_test(model_path, target_modules, trtllm_modules): ) base_outputs = llm.generate(prompts, sampling) + # Keep both calls on full prefill: partial reuse can select a different + # attention kernel, whose numerical drift can fail this strict token and + # logprob comparison even when LoRA correctly leaves base rows unchanged. + assert all(output.cached_tokens == 0 for output in mixed_outputs) + assert all(output.cached_tokens == 0 for output in base_outputs) for index in (1, 3): _assert_outputs_match(mixed_outputs[index], base_outputs[index]) _assert_lora_changes_output( @@ -286,6 +302,7 @@ class TestQwen3LoRA: @pytest.fixture(autouse=True) def setup(self): self.model_path = f"{llm_models_root()}/Qwen3/Qwen3-0.6B" + self.kv_cache_config = KvCacheConfig(use_kv_cache_manager_v2=True, enable_block_reuse=True) if not os.path.exists(self.model_path): pytest.skip(f"Model not found: {self.model_path}") @@ -294,6 +311,7 @@ def test_qwen3_bf16_lora(self): self.model_path, {**_ATTN_LORA_MODULES, **_MLP_LORA_MODULES}, _ATTN_TRTLLM_MODULES + _MLP_TRTLLM_MODULES, + kv_cache_config=self.kv_cache_config, ) def test_qwen3_fp8_lora(self): @@ -302,6 +320,7 @@ def test_qwen3_fp8_lora(self): {**_ATTN_LORA_MODULES, **_MLP_LORA_MODULES}, _ATTN_TRTLLM_MODULES + _MLP_TRTLLM_MODULES, dtype=torch.float8_e4m3fn, + kv_cache_config=self.kv_cache_config, ) def test_qwen3_bf16_lora_overlap(self): @@ -310,6 +329,7 @@ def test_qwen3_bf16_lora_overlap(self): {**_ATTN_LORA_MODULES, **_MLP_LORA_MODULES}, _ATTN_TRTLLM_MODULES + _MLP_TRTLLM_MODULES, overlap=True, + kv_cache_config=self.kv_cache_config, ) def test_qwen3_fp8_lora_overlap(self): @@ -319,6 +339,7 @@ def test_qwen3_fp8_lora_overlap(self): _ATTN_TRTLLM_MODULES + _MLP_TRTLLM_MODULES, dtype=torch.float8_e4m3fn, overlap=True, + kv_cache_config=self.kv_cache_config, ) def test_qwen3_bf16_lora_cuda_graph_specialization(self): @@ -327,13 +348,17 @@ def test_qwen3_bf16_lora_cuda_graph_specialization(self): {**_ATTN_LORA_MODULES, **_MLP_LORA_MODULES}, _ATTN_TRTLLM_MODULES + _MLP_TRTLLM_MODULES, specialize_cuda_graph=True, + kv_cache_config=self.kv_cache_config, ) def test_qwen3_bf16_lora_cuda_graph_specialization_mixed_batch(self): + # Keep these short prompts on the same context attention path so the + # strict comparison measures LoRA isolation without cold/warm drift. _run_mixed_lora_cuda_graph_test( self.model_path, {**_ATTN_LORA_MODULES, **_MLP_LORA_MODULES}, _ATTN_TRTLLM_MODULES + _MLP_TRTLLM_MODULES, + kv_cache_config=self.kv_cache_config.model_copy(update={"enable_partial_reuse": False}), )